20 Commits

Author SHA1 Message Date
bd6cffc9d0 Prepare documentation for Promptkit v0.4.0 2026-07-30 23:48:25 +00:00
e40c4f182b Document structured capacity errors 2026-07-30 23:26:44 +00:00
7428e50c2c Expose structured capacity errors 2026-07-30 23:24:04 +00:00
63c67a4520 Add internal capacity error identity 2026-07-30 23:21:27 +00:00
25a7052a3d Clarify prompt input requirement documentation 2026-07-30 22:10:33 +00:00
fc3255967e Document prompt inspection API 2026-07-30 21:06:33 +00:00
e920168b30 Expose prompt inspection through the engine 2026-07-30 21:03:47 +00:00
272b6a4bc1 Add internal prompt inspection 2026-07-30 21:00:05 +00:00
dde48a31fc Document profile inspection API 2026-07-30 19:57:07 +00:00
242eace4a7 Expose profile inspection through the engine 2026-07-30 19:52:00 +00:00
0bf5f88136 Add internal profile inspection resolution 2026-07-30 19:48:09 +00:00
369ab5392d Add feature roadmap and implementation plan for profile inspection API 2026-07-30 19:42:51 +00:00
2ba0146e5d Tighten prepared execution credential handling 2026-07-30 19:06:17 +00:00
6112c2af0c Document prepared execution workflow 2026-07-30 18:27:03 +00:00
f5e12c00f5 Expose prepared execution handles 2026-07-30 18:20:35 +00:00
49fe402dd2 Add prepared execution lifecycle to the runner 2026-07-30 18:10:28 +00:00
c301eb8d55 Add frozen validation preparation plans 2026-07-30 17:59:45 +00:00
c13e9710d9 Add feature roadmap and implementation plan for downstream consumer wishlist items 2026-07-30 17:54:44 +00:00
87b5ec3d75 Organize downstream feature requests in the future roadmap 2026-07-30 17:13:26 +00:00
cb4028a637 Add feature roadmaps with wishlists from downstream consumers 2026-07-30 16:59:37 +00:00
43 changed files with 4441 additions and 961 deletions

View File

@@ -33,6 +33,9 @@ boundary and constraints that framework work must preserve.
## Release Guidance ## Release Guidance
Consumers upgrading from `v0.3.0` to `v0.4.0` should read the
[v0.4.0 changelog and adoption guide](docs/releases/v0.4.0.md).
Consumers upgrading from `v0.2.0` to `v0.3.0` should read the Consumers upgrading from `v0.2.0` to `v0.3.0` should read the
[v0.3.0 changelog](docs/releases/v0.3.0.md). [v0.3.0 changelog](docs/releases/v0.3.0.md).

View File

@@ -40,12 +40,12 @@ type Backend struct {
// calls allowed for this backend within one Engine. Zero leaves the backend // calls allowed for this backend within one Engine. Zero leaves the backend
// unlimited. A negative value makes NewEngine fail with ErrInvalidConfig. // unlimited. A negative value makes NewEngine fail with ErrInvalidConfig.
ConcurrencyLimit int ConcurrencyLimit int
// QueueCapacity controls how many additional Run calls may be admitted // QueueCapacity controls how many additional Run or RunPrepared calls may
// beyond ConcurrencyLimit. Nil uses 1024 when ConcurrencyLimit is positive; // be admitted beyond ConcurrencyLimit. Nil uses 1024 when ConcurrencyLimit
// a pointer uses its exact value, including zero. The pointed-to value must // is positive; a pointer uses its exact value, including zero. The pointed-to
// be non-negative, and QueueCapacity must be nil when ConcurrencyLimit is // value must be non-negative, and QueueCapacity must be nil when
// zero. Their sum must fit in an int. WithBackend copies the value and does // ConcurrencyLimit is zero. Their sum must fit in an int. WithBackend copies
// not retain the pointer. // the value and does not retain the pointer.
QueueCapacity *int QueueCapacity *int
} }

View File

@@ -60,7 +60,18 @@ func TestEngineRejectsRunBeforeCompletionWhenAdmissionIsFull(t *testing.T) {
awaitCapacitySignal(t, reader.entered, "first artifact read") awaitCapacitySignal(t, reader.entered, "first artifact read")
result, err := engine.Run(context.Background(), capacityInputRequest("http://second.example/v1")) canceledContext, cancel := context.WithCancel(context.Background())
cancel()
result, err := engine.Run(canceledContext, capacityInputRequest("http://canceled.example/v1"))
if result != nil || !errors.Is(err, context.Canceled) {
t.Fatalf("canceled capacity admission=(%+v, %v), want context cancellation", result, err)
}
var canceledCapacityErr *promptkit.CapacityError
if errors.Is(err, promptkit.ErrCapacityExceeded) || errors.As(err, &canceledCapacityErr) {
t.Fatalf("canceled admission exposed capacity rejection: %v", err)
}
result, err = engine.Run(context.Background(), capacityInputRequest("http://second.example/v1"))
if result != nil { if result != nil {
t.Fatalf("capacity rejection returned partial result: %+v", result) t.Fatalf("capacity rejection returned partial result: %+v", result)
} }
@@ -70,6 +81,21 @@ func TestEngineRejectsRunBeforeCompletionWhenAdmissionIsFull(t *testing.T) {
if errors.Is(err, promptkit.ErrInvalidRequest) || errors.Is(err, promptkit.ErrLLMGenerate) { if errors.Is(err, promptkit.ErrInvalidRequest) || errors.Is(err, promptkit.ErrLLMGenerate) {
t.Fatalf("capacity rejection had an unrelated category: %v", err) t.Fatalf("capacity rejection had an unrelated category: %v", err)
} }
var capacityErr *promptkit.CapacityError
if !errors.As(err, &capacityErr) || capacityErr == nil {
t.Fatalf("capacity rejection=%v, want CapacityError", err)
}
if capacityErr.BackendID != "limited" {
t.Fatalf("capacity backend ID=%q, want limited", capacityErr.BackendID)
}
capacityErr.BackendID = "changed"
result, err = engine.Run(context.Background(), capacityInputRequest("http://third.example/v1"))
var subsequentCapacityErr *promptkit.CapacityError
if result != nil || !errors.As(err, &subsequentCapacityErr) ||
subsequentCapacityErr == nil || subsequentCapacityErr.BackendID != "limited" {
t.Fatalf("subsequent capacity rejection=(%+v, %v), want independent limited CapacityError", result, err)
}
if calls := reader.callCount(); calls != 1 { if calls := reader.callCount(); calls != 1 {
t.Fatalf("artifact calls=%d, want only the admitted run", calls) t.Fatalf("artifact calls=%d, want only the admitted run", calls)
} }
@@ -174,6 +200,19 @@ func TestCapacityExceededSentinelContract(t *testing.T) {
if promptkit.ErrCapacityExceeded == nil { if promptkit.ErrCapacityExceeded == nil {
t.Fatal("ErrCapacityExceeded is nil") t.Fatal("ErrCapacityExceeded is nil")
} }
var nilCapacityErr *promptkit.CapacityError
zeroCapacityErr := &promptkit.CapacityError{}
populatedCapacityErr := &promptkit.CapacityError{BackendID: "limited"}
for _, capacityErr := range []error{nilCapacityErr, zeroCapacityErr} {
if !errors.Is(capacityErr, promptkit.ErrCapacityExceeded) {
t.Fatalf("capacity error=%v, want ErrCapacityExceeded", capacityErr)
}
}
var discoveredCapacityErr *promptkit.CapacityError
if !errors.As(populatedCapacityErr, &discoveredCapacityErr) || discoveredCapacityErr != populatedCapacityErr {
t.Fatalf("populated capacity error is not discoverable: %v", populatedCapacityErr)
}
for _, unrelated := range []error{ for _, unrelated := range []error{
promptkit.ErrInvalidConfig, promptkit.ErrInvalidConfig,
promptkit.ErrInvalidRequest, promptkit.ErrInvalidRequest,
@@ -181,7 +220,8 @@ func TestCapacityExceededSentinelContract(t *testing.T) {
promptkit.ErrValidation, promptkit.ErrValidation,
} { } {
if errors.Is(promptkit.ErrCapacityExceeded, unrelated) || if errors.Is(promptkit.ErrCapacityExceeded, unrelated) ||
errors.Is(unrelated, promptkit.ErrCapacityExceeded) { errors.Is(unrelated, promptkit.ErrCapacityExceeded) ||
errors.Is(populatedCapacityErr, unrelated) {
t.Fatalf("ErrCapacityExceeded aliases unrelated sentinel %v", unrelated) t.Fatalf("ErrCapacityExceeded aliases unrelated sentinel %v", unrelated)
} }
} }

40
capacity_error.go Normal file
View File

@@ -0,0 +1,40 @@
package promptkit
import (
"fmt"
"strings"
)
// CapacityError reports bounded admission rejected for a selected backend.
//
// Engine-produced values identify only rejection at Promptkit's bounded
// [Engine.Run] or [Engine.RunPrepared] admission boundary. BackendID is the
// normalized registered backend ID used for routing and capacity; endpoint
// overrides do not change it. Every engine-produced value is nonnil and has a
// nonblank BackendID. Provider errors, active-generation waiting, and caller
// cancellation are not represented by this type.
//
// Callers own returned values and may mutate BackendID without affecting engine
// state or another error. CapacityError and its default Go encoding have no
// stable JSON contract. Consumer-constructed values do not establish that an
// engine rejected work.
type CapacityError struct {
// BackendID is the normalized registered backend ID whose admission was
// rejected.
BackendID string
}
// Error returns diagnostic wording that is not a parsing contract. It is safe
// to call on a nil receiver or a value with a blank BackendID.
func (e *CapacityError) Error() string {
if e == nil || strings.TrimSpace(e.BackendID) == "" {
return ErrCapacityExceeded.Error()
}
return fmt.Sprintf("backend %q admission: %v", e.BackendID, ErrCapacityExceeded)
}
// Unwrap returns ErrCapacityExceeded so errors.Is and errors.As can be used
// together. It is safe to call on a nil receiver or a zero value.
func (e *CapacityError) Unwrap() error {
return ErrCapacityExceeded
}

View File

@@ -170,6 +170,40 @@ func fromDomainExecutionTarget(target domain.ExecutionTarget) ExecutionTarget {
} }
} }
func fromDomainProfileInspection(inspection *domain.ProfileInspection) *ProfileInspection {
if inspection == nil {
return nil
}
return &ProfileInspection{
ProfileID: inspection.ProfileID,
EffectiveModelParams: fromDomainExecutionTarget(inspection.EffectiveModelParams),
APIKeyRequired: inspection.APIKeyRequired,
}
}
func fromDomainPromptInspection(inspection *domain.PromptInspection) *PromptInspection {
if inspection == nil {
return nil
}
inputs := make([]PromptInputDefinition, len(inspection.Inputs))
for i, input := range inspection.Inputs {
inputs[i] = PromptInputDefinition{
Name: input.Name,
Required: input.Required,
ContentType: input.ContentType,
Description: input.Description,
}
}
return &PromptInspection{
PromptID: inspection.PromptID,
PromptVersion: inspection.PromptVersion,
PromptHash: inspection.PromptHash,
DefaultProfileID: inspection.DefaultProfileID,
Inputs: inputs,
OutputContract: fromDomainOutputContract(inspection.OutputContract),
}
}
func fromDomainExecutionTargetPresence(presence domain.ExecutionTargetPresence) ExecutionTargetPresence { func fromDomainExecutionTargetPresence(presence domain.ExecutionTargetPresence) ExecutionTargetPresence {
return ExecutionTargetPresence{ return ExecutionTargetPresence{
Temperature: presence.Temperature, Temperature: presence.Temperature,

41
doc.go
View File

@@ -3,23 +3,28 @@
// //
// Applications construct an [Engine] with [NewEngine], select filesystem or // Applications construct an [Engine] with [NewEngine], select filesystem or
// in-memory sources and optional engine-scoped [Backend] registrations, and // in-memory sources and optional engine-scoped [Backend] registrations, and
// call [Engine.Prepare] or [Engine.Run]. Concrete registries, repositories, // call [Engine.InspectPrompt], [Engine.InspectProfile], [Engine.Prepare],
// validators, and the built-in OpenAI-compatible client remain internal // [Engine.PrepareExecution], [Engine.Run], or [Engine.RunPrepared]. Concrete
// implementation details. // registries, repositories, validators, and the built-in OpenAI-compatible
// client remain internal implementation details.
// //
// # Concurrency and ownership // # Concurrency and ownership
// //
// An Engine supports concurrent Prepare and Run calls. Engine-local backend // An Engine supports concurrent InspectPrompt, InspectProfile, Prepare,
// policies bound admitted Run calls and model generations where configured, // PrepareExecution, Run, and RunPrepared calls. Engine-local backend policies
// while different backend pools and unlimited backends continue independently. // bound admitted Run and RunPrepared calls and model generations where
// An injected [LLMClient] or [ArtifactReader] can therefore still receive // configured, while different backend pools and unlimited backends continue
// concurrent calls and must be safe for that use. // independently. An injected [LLMClient] or [ArtifactReader] can therefore
// still receive concurrent calls and must be safe for that use.
// //
// NewEngine copies in-memory profiles and backend definitions. Prepare and Run // NewEngine copies in-memory profiles and backend definitions. Prepare,
// copy request maps, slices, pointer values, and JSON-compatible extra // PrepareExecution, and Run copy request maps, slices, pointer values, and
// parameters before using them. Returned values and values passed to extension // JSON-compatible extra parameters before using them. InspectPrompt and
// interfaces are likewise isolated from engine state. Callers own those copies // InspectProfile return copied inspection values. Returned values and values
// and may mutate them after the call that supplied or returned them. // passed to extension interfaces are likewise isolated from engine state.
// Callers own those copies and may mutate them after the call that supplied or
// returned them. Returned structured errors are likewise caller-owned and may
// be mutated without affecting engine state or another error.
// //
// # Security and sensitive data // # Security and sensitive data
// //
@@ -45,10 +50,12 @@
// [GenerateResponse], [ExecutionTargetPresence], and the string value types // [GenerateResponse], [ExecutionTargetPresence], and the string value types
// used by those values. // used by those values.
// //
// Construction values, including [Config], [Backend], [RunRequest], // Construction, inspection, handle, and error values, including [Config],
// [ArtifactRef], [ExecutionTargetOverride], [Profile], and // [Backend], [RunRequest], [ArtifactRef], [ExecutionTargetOverride], [Profile],
// [OpenAICompatibleProfileConfig], do not have stable JSON representations. // [OpenAICompatibleProfileConfig], [ProfileInspection],
// Direct API keys are nevertheless excluded from JSON for every public value. // [PromptInputDefinition], [PromptInspection], [PreparedExecution], and
// [CapacityError], do not have stable JSON representations. Direct API keys
// are nevertheless excluded from JSON for every public value.
// //
// JSON timestamps use time.Time's RFC 3339 encoding and are omitted when zero. // JSON timestamps use time.Time's RFC 3339 encoding and are omitted when zero.
// PreparedRun and RunResult durations are encoded as integer milliseconds in // PreparedRun and RunResult durations are encoded as integer milliseconds in

View File

@@ -40,11 +40,37 @@ validation, and default transport behavior. Source discovery, format
validation, and profile precedence are defined by the validation, and profile precedence are defined by the
[framework format reference](../formats.md). [framework format reference](../formats.md).
## Inspect A Prompt Before Preparation
Use [`Engine.InspectPrompt`](../../engine.go) to check one configured prompt's
declared inputs and output workflow without creating placeholder inputs or
resolving a profile:
```go
inspection, err := engine.InspectPrompt(ctx, "meeting.summary", "")
if err != nil {
return err
}
for _, input := range inspection.Inputs {
// Compare the declared input with application configuration.
}
```
Use this configuration-time boundary when the application needs only the
declared prompt interface. Use `InspectProfile` separately when it must also
check a configured profile. Use `Prepare` when it needs inputs, schemas, or
rendered messages, and use prepared execution when that work must remain tied
to later execution. The method's [GoDoc](../../engine.go) owns exact fields,
hash, ownership, and error semantics.
## Prepare Without Model Execution ## Prepare Without Model Execution
[`Engine.Prepare`](../../engine.go) resolves the selected prompt and profile, [`Engine.Prepare`](../../engine.go) resolves the selected prompt and profile,
loads inputs and any structured-output schema, and renders messages without loads inputs and any structured-output schema, and renders messages without
calling a model client: calling a model client. Choose it when the prepared value is the final
inspection or persistence result and no later execution must be tied to that
exact snapshot:
```go ```go
prepared, err := engine.Prepare(ctx, promptkit.RunRequest{ prepared, err := engine.Prepare(ctx, promptkit.RunRequest{
@@ -61,12 +87,48 @@ shows a complete runnable setup with a prompt file, in-memory profile, and
inline input. Exact request requirements and prepared-result fields belong to inline input. Exact request requirements and prepared-result fields belong to
the [`RunRequest` and `PreparedRun` GoDoc](../../types.go). the [`RunRequest` and `PreparedRun` GoDoc](../../types.go).
## Prepare Now And Execute The Same Snapshot Later
Use [`Engine.PrepareExecution`](../../engine.go) when an application must
inspect or persist preflight details before deciding whether to start model
work, while ensuring that later execution uses those exact rendered messages,
inputs, target settings, and validation resources:
```go
preparedExecution, err := engine.PrepareExecution(ctx, promptkit.RunRequest{
PromptID: "meeting.summary",
Inputs: map[string]promptkit.ArtifactRef{
"note": promptkit.Inline("Synthetic meeting notes"),
},
})
if err != nil {
return err
}
defer preparedExecution.Discard()
details := preparedExecution.Details()
// Inspect or persist an application-selected safe subset of details.
result, err := engine.RunPrepared(ctx, preparedExecution)
```
Preparation does not call the model or reserve backend capacity.
`RunPrepared` executes from the retained snapshot rather than reloading
consumer sources. The handle is opaque in-process state, while `Details`
contains rendered content and remains subject to the application's data
handling policy. The
[`PreparedExecution` and method GoDoc](../../prepared_execution.go) and
[engine operation GoDoc](../../engine.go) own exact lifecycle, engine-binding,
credential, cancellation, timing, and error semantics.
## Execute And Validate ## Execute And Validate
[`Engine.Run`](../../engine.go) performs the same preparation, invokes the [`Engine.Run`](../../engine.go) performs the same preparation, invokes the
configured model client, classifies the generated artifact, and validates the configured model client, classifies the generated artifact, and validates the
content. A completed content check may return `ValidationFailed` in the result; content in one call. Choose it when the application does not need a preflight
an operational inability to validate returns an error. boundary tied to the eventual execution. A completed content check may return
`ValidationFailed` in the result; an operational inability to validate returns
an error.
The maintained The maintained
[offline execution example](../../examples/go-library/run/main.go) injects a [offline execution example](../../examples/go-library/run/main.go) injects a
@@ -97,6 +159,34 @@ For programmatic profiles,
[`OpenAICompatibleProfile`](../../profiles.go) converts ordinary [`OpenAICompatibleProfile`](../../profiles.go) converts ordinary
OpenAI-compatible settings into a value accepted by `WithProfiles`. OpenAI-compatible settings into a value accepted by `WithProfiles`.
### Inspect A Profile Before Prompt Work
Use [`Engine.InspectProfile`](../../engine.go) to validate one configured
profile without constructing a synthetic prompt or placeholder inputs. It
resolves the profile's effective target but does not prepare or execute a
prompt:
```go
inspection, err := engine.InspectProfile(ctx, profileID)
if err != nil {
return err
}
target := inspection.EffectiveModelParams
if target.APIKeyEnv != "" {
// Apply application policy for the named environment variable.
} else if inspection.APIKeyRequired {
// Arrange a direct credential before later execution.
}
```
Use this configuration-time boundary when only the profile and its target need
checking. Use `Prepare` when the application also needs prompt, input, schema,
or rendering work; use prepared execution when that work must remain tied to a
later execution. Inspection reports credential requirements but leaves the
timing of credential enforcement to the application. The method's
[GoDoc](../../engine.go) owns its exact result and error contract.
### Set A Per-Run Session And Reasoning ### Set A Per-Run Session And Reasoning
Supply a direct session ID when one prompt should be correlated with a Supply a direct session ID when one prompt should be correlated with a
@@ -287,6 +377,11 @@ When a limited backend has admitted all active and waiting calls, handle
```go ```go
result, err := engine.Run(ctx, request) result, err := engine.Run(ctx, request)
if errors.Is(err, promptkit.ErrCapacityExceeded) { if errors.Is(err, promptkit.ErrCapacityExceeded) {
var capacityErr *promptkit.CapacityError
if errors.As(err, &capacityErr) {
// Record capacityErr.BackendID using application-owned diagnostics.
}
// Apply application policy: shed work, report overload, or retry later. // Apply application policy: shed work, report overload, or retry later.
} }
``` ```
@@ -294,8 +389,9 @@ if errors.Is(err, promptkit.ErrCapacityExceeded) {
A rejected call returns no partial result and does not invoke the model A rejected call returns no partial result and does not invoke the model
client. Promptkit does not prescribe retries or map this error to an HTTP client. Promptkit does not prescribe retries or map this error to an HTTP
status; those choices remain with the consuming application. The status; those choices remain with the consuming application. The
[`Engine.Run` and error GoDoc](../../engine.go) owns exact error and [`CapacityError` GoDoc](../../capacity_error.go) owns the exact typed-error
cancellation identities. contract, while the [`Engine.Run` and error GoDoc](../../engine.go) owns broad
error and cancellation identities.
## Application Boundary ## Application Boundary

View File

@@ -58,6 +58,11 @@ When a request omits a version, the selected prompt ID must identify exactly
one definition. When it supplies a version, the ID and version pair must be one definition. When it supplies a version, the ID and version pair must be
unique. unique.
Exact prompt inspection uses this same configured source, strict decoding,
referenced content-file resolution, and ID/version selection. It reports the
selected definition's declared metadata without changing the prompt format or
executing the definition.
### Inputs ### Inputs
Each `inputs` item has these fields: Each `inputs` item has these fields:
@@ -152,7 +157,7 @@ extra_params:
| Field | Required | Meaning | | Field | Required | Meaning |
| --- | --- | --- | | --- | --- | --- |
| `id` | yes | Non-empty profile identifier. IDs must be unique within one source. | | `id` | yes | Non-empty profile identifier. IDs must be unique within one source. |
| `backend` | unless `endpoint` is present | Backend registry ID. It is trimmed and registry membership is checked when the profile is prepared. | | `backend` | unless `endpoint` is present | Backend registry ID. It is trimmed and registry membership is checked when the profile is prepared or inspected. |
| `endpoint` | unless `backend` is present | Non-empty OpenAI-compatible base URL, including an API version path when required. When both connection fields are present, this overrides the backend endpoint without changing backend identity. | | `endpoint` | unless `backend` is present | Non-empty OpenAI-compatible base URL, including an API version path when required. When both connection fields are present, this overrides the backend endpoint without changing backend identity. |
| `model` | yes | Non-empty provider model name. | | `model` | yes | Non-empty provider model name. |
| `temperature` | no | Number from 0 through 2. | | `temperature` | no | Number from 0 through 2. |
@@ -219,7 +224,9 @@ defines how the effective settings are serialized.
### Source And Profile Precedence ### Source And Profile Precedence
An explicit request profile ID takes precedence over the prompt's An explicit request profile ID takes precedence over the prompt's
`default_profile`. If neither is present, preparation fails. `default_profile`. If neither is present, preparation fails. Exact profile
inspection instead takes one explicit profile ID and does not use a prompt
default.
Profile sources resolve matching IDs in this order: Profile sources resolve matching IDs in this order:
@@ -231,6 +238,7 @@ A higher-precedence source falls back only when the profile is absent. An
invalid matching profile is an error and does not fall back. In-memory invalid matching profile is an error and does not fall back. In-memory
`Profile` values follow the same ranges as YAML profiles. They use `Profile` values follow the same ranges as YAML profiles. They use
`APIKeyRequired` for request-scoped credentials instead of `api_key_env`. `APIKeyRequired` for request-scoped credentials instead of `api_key_env`.
Preparation and exact profile inspection use this same source precedence.
## Built-In Profile Catalog ## Built-In Profile Catalog

View File

@@ -29,21 +29,28 @@ One pool owns immutable active and total limits plus mutex-protected admission
count, active count, and ordered waiter list. Pool state exists only for the count, active count, and ordered waiter list. Pool state exists only for the
lifetime of its engine. lifetime of its engine.
## Bounded Run Admission ## Bounded Execution Admission
The runner asks the manager to admit a run after resolving the prompt, profile, For ordinary `Run`, the runner asks the manager to admit after resolving the
selected backend, effective execution target, credentials, and output contract, prompt, profile, selected backend, effective execution target, credentials, and
but before schema loading, artifact loading, or rendering. Admission is output contract, but before schema loading, artifact loading, or rendering.
immediate: a limited pool either reserves a slot or returns the internal `PrepareExecution` performs no admission. `RunPrepared` claims its handle,
`ErrCapacityExceeded` identity. The root facade maps that identity to the rechecks credential availability, and then asks the manager to admit the
public error without treating it as an invalid request or generation failure. frozen backend before generation.
Admission is immediate: a limited pool either reserves a slot or returns only
the internal `ErrCapacityExceeded` identity. The runner attaches the selected
backend identity at its use-case boundary, and the root facade translates that
typed value without treating it as an invalid request or generation failure.
The total admitted bound is the active-generation limit plus its configured The total admitted bound is the active-generation limit plus its configured
waiting capacity. The returned release function is idempotent. The runner waiting capacity. The returned release function is idempotent. The runner
defers it as soon as admission succeeds and holds the lease across remaining defers it as soon as admission succeeds. An ordinary run holds the lease across
preparation, initial generation, validation, every repair attempt, and all remaining preparation, initial generation, validation, every repair attempt,
failure or cancellation exits. A repair is part of its original admission and and all failure or cancellation exits. Prepared execution holds the normal
does not reserve another bounded slot. lease across generation, validation, every internal repair attempt, and all
execution exits. A repair is part of its original admission and does not
reserve another bounded slot.
## FIFO Generation Permits ## FIFO Generation Permits
@@ -90,11 +97,17 @@ unlimited admission. The
FIFO transfer, canceled-waiter removal, grant/cancel races, independent pools, FIFO transfer, canceled-waiter removal, grant/cancel races, independent pools,
unlimited calls, passthrough behavior, and panic release. unlimited calls, passthrough behavior, and panic release.
The [runner tests](../../internal/usecase/runner_test.go) own early admission, The [runner tests](../../internal/usecase/runner_test.go) own ordinary early
lease lifetime, failure release, and shared initial/repair scheduling. The admission, lease lifetime, failure release, and shared initial/repair
scheduling. The
[prepared-execution use-case tests](../../internal/usecase/prepared_execution_test.go)
own deferred admission, credential ordering, and prepared-execution lease
release. The
[external package capacity tests](../../capacity_contract_test.go) own the [external package capacity tests](../../capacity_contract_test.go) own the
assembled public-engine behavior for configured limits, capacity errors, assembled public-engine behavior for configured limits, capacity errors,
endpoint identity, engine independence, and injected clients. The endpoint identity, engine independence, and injected clients. The
[prepared-execution contract tests](../../prepared_execution_contract_test.go)
own the public prepared-capacity boundary. The
[root error-boundary tests](../../errors_internal_test.go) own preservation of [root error-boundary tests](../../errors_internal_test.go) own preservation of
the public generation category and context identity when generation is the public generation category and context identity when generation is
canceled. canceled.

View File

@@ -43,6 +43,21 @@ without making the model client depend on registry configuration.
The implementation has no retry loop, tool-call support, provider catalog, The implementation has no retry loop, tool-call support, provider catalog,
inbound HTTP behavior, or durable session store. inbound HTTP behavior, or durable session store.
## Prepared Generation
For [`RunPrepared`](../../engine.go), the runner supplies the model client with
the target, rendered messages, and structured-output constraint retained by
executable preparation. Execution does not reopen or rerender consumer
sources.
Before backend admission, the runner rechecks that the frozen credential
environment-variable name is available. The handle does not retain the
environment value; the model client resolves the value visible when generation
begins. A direct request key remains in private execution state only until the
claimed execution finishes or an unclaimed handle is discarded. Exact public
ownership and redaction semantics belong to the
[`PreparedExecution` GoDoc](../../prepared_execution.go).
## Failure Categories ## Failure Categories
The package preserves distinct error identities for invalid client The package preserves distinct error identities for invalid client

View File

@@ -11,23 +11,23 @@ contributor workflow and validation.
| Component | Implemented responsibility | References | | Component | Implemented responsibility | References |
| --- | --- | --- | | --- | --- | --- |
| Root `promptkit` package | Provides the supported engine facade, source, backend-registration, and injection options, public request and result values, profile construction, extension interfaces, value conversion, redacted formatting, public error mapping, and engine-local assembly. | [Package GoDoc](../../doc.go), [backend API](../../backends.go), [engine assembly](../../engine.go) | | Root `promptkit` package | Provides the supported engine facade, source, backend-registration, and injection options, public request, result, prompt-inspection, and profile-inspection values, opaque prepared-execution handles, profile construction, extension interfaces, value conversion, redacted formatting, typed capacity errors, public error mapping, and engine-local assembly. | [Package GoDoc](../../doc.go), [prepared execution](../../prepared_execution.go), [backend API](../../backends.go), [engine assembly](../../engine.go) |
| `examples/go-library/prepare` | Demonstrates an offline downstream consumer using a prompt file, in-memory profile, inline input, and `Prepare`. It is not a public library package. | [Example program](../../examples/go-library/prepare/main.go) | | `examples/go-library/prepare` | Demonstrates an offline downstream consumer using a prompt file, in-memory profile, inline input, and `Prepare`. It is not a public library package. | [Example program](../../examples/go-library/prepare/main.go) |
| `examples/go-library/run` | Demonstrates an offline downstream consumer using a prompt file, in-memory profile, inline input, an injected deterministic model client, and `Run`. It is not a public library package. | [Example program](../../examples/go-library/run/main.go) | | `examples/go-library/run` | Demonstrates an offline downstream consumer using a prompt file, in-memory profile, inline input, an injected deterministic model client, and `Run`. It is not a public library package. | [Example program](../../examples/go-library/run/main.go) |
| `internal/backend` | Constructs each engine's immutable registry from the built-in OpenRouter definition and consumer additions, validates and defensively copies definitions through the shared JSON-value package, and consumes the LLM-owned OpenAI-compatible reserved request-field rule. | [Backend registry](../../internal/backend/registry.go) | | `internal/backend` | Constructs each engine's immutable registry from the built-in OpenRouter definition and consumer additions, validates and defensively copies definitions through the shared JSON-value package, and consumes the LLM-owned OpenAI-compatible reserved request-field rule. | [Backend registry](../../internal/backend/registry.go) |
| `internal/capacity` | Owns engine-local bounded run admission and FIFO model-generation permits for limited backend IDs, including cancellation-safe waiter removal and client wrapping. | [Internal capacity management](capacity.md) | | `internal/capacity` | Owns engine-local bounded execution admission and FIFO model-generation permits for limited backend IDs, including cancellation-safe waiter removal and client wrapping. | [Internal capacity management](capacity.md) |
| `internal/domain` | Defines internal framework values for requests, artifacts, prompt definitions, profiles, execution targets, rendering, generation, and validation. | [Domain declarations](../../internal/domain/domain.go) | | `internal/domain` | Defines internal framework values for requests, artifacts, prompt definitions, profiles, execution targets, rendering, generation, and validation. | [Domain declarations](../../internal/domain/domain.go) |
| `internal/defaults` | Defines application-neutral framework constants and constructs the default execution target. It contains no CLI, server, or inbound HTTP limits. | [Framework defaults](../../internal/defaults/defaults.go) | | `internal/defaults` | Defines application-neutral framework constants and constructs the default execution target. It contains no CLI, server, or inbound HTTP limits. | [Framework defaults](../../internal/defaults/defaults.go) |
| `internal/filecatalog` | Provides deterministic YAML discovery and path helpers for operating-system filesystems and `fs.FS` sources. | [File catalog](../../internal/filecatalog/catalog.go) | | `internal/filecatalog` | Provides deterministic YAML discovery and path helpers for operating-system filesystems and `fs.FS` sources. | [File catalog](../../internal/filecatalog/catalog.go) |
| `internal/jsonvalue` | Validates and deeply copies JSON-compatible extra-parameter trees while preserving supported concrete value types. | [JSON values](../../internal/jsonvalue/jsonvalue.go) | | `internal/jsonvalue` | Validates and deeply copies JSON-compatible extra-parameter and prepared-schema trees while preserving supported concrete value types. | [JSON values](../../internal/jsonvalue/jsonvalue.go) |
| `internal/promptdef` | Loads strictly decoded, validated prompt definitions from filesystem and `fs.FS` sources, including version selection and contained file-backed message content. | [Framework formats](../formats.md), [prompt-definition repository](../../internal/promptdef/filesystem_repository.go) | | `internal/promptdef` | Loads strictly decoded, validated prompt definitions from filesystem and `fs.FS` sources, including version selection and contained file-backed message content. | [Framework formats](../formats.md), [prompt-definition repository](../../internal/promptdef/filesystem_repository.go) |
| `internal/profile` | Loads strictly decoded, validated execution profiles, including backend selection, from filesystem and `fs.FS` sources and composes repositories with error-preserving fallback. | [Framework formats](../formats.md), [profile repositories](../../internal/profile/filesystem_repository.go) | | `internal/profile` | Loads strictly decoded, validated execution profiles, including backend selection, from filesystem and `fs.FS` sources and composes repositories with error-preserving fallback. | [Framework formats](../formats.md), [profile repositories](../../internal/profile/filesystem_repository.go) |
| `internal/profile/builtin` | Embeds the built-in profile catalog, whose entries select OpenRouter, and combines it with an optional primary repository. | [Built-in catalog](../formats.md#built-in-profile-catalog), [repository](../../internal/profile/builtin/repository.go) | | `internal/profile/builtin` | Embeds the built-in profile catalog, whose entries select OpenRouter, and combines it with an optional primary repository. | [Built-in catalog](../formats.md#built-in-profile-catalog), [repository](../../internal/profile/builtin/repository.go) |
| `internal/prompt` | Renders prompt messages from Go templates with artifact, variable, session, and cache-control data. | [Go-template renderer](../../internal/prompt/go_renderer.go) | | `internal/prompt` | Renders prompt messages from Go templates with artifact, variable, session, and cache-control data. | [Go-template renderer](../../internal/prompt/go_renderer.go) |
| `internal/artifact` | Resolves ordinary inline and unrestricted caller-selected file references into copied artifacts with metadata and hashes. | [Internal sources and validation](sources.md) | | `internal/artifact` | Resolves ordinary inline and unrestricted caller-selected file references into copied artifacts with metadata and hashes. | [Internal sources and validation](sources.md) |
| `internal/validate` | Validates basic, JSON, and JSON Schema output using operating-system filesystem or `fs.FS` schema sources. | [Framework formats](../formats.md#schemas), [internal sources and validation](sources.md) | | `internal/validate` | Validates basic, JSON, and JSON Schema output using operating-system filesystem or `fs.FS` schema sources and creates frozen validation plans for prepared execution. | [Framework formats](../formats.md#schemas), [internal sources and validation](sources.md) |
| `internal/llm` | Defines the internal generation boundary and implements outbound OpenAI-compatible chat requests from resolved execution targets, including response decoding, authentication, deadline handling, and ownership of the OpenAI-compatible reserved request-field policy. | [Internal model client](llm.md) | | `internal/llm` | Defines the internal generation boundary and implements outbound OpenAI-compatible chat requests from resolved execution targets, including response decoding, authentication, deadline handling, and ownership of the OpenAI-compatible reserved request-field policy. | [Internal model client](llm.md) |
| `internal/usecase` | Resolves backend, profile, and request settings and coordinates preparation and execution across internal sources, rendering, artifact loading, generation, validation, and optional repair. | [Internal runner](runner.md) | | `internal/usecase` | Resolves prompt definitions and hashes, profiles, backends, and targets for exact inspection and request settings for preparation, and coordinates ordinary execution and one-attempt prepared execution across internal sources, rendering, artifact loading, generation, validation, capacity, and optional repair. | [Internal runner](runner.md), [prepared-execution implementation](../../internal/usecase/prepared_execution.go) |
The root package assembles these internal components without exposing their The root package assembles these internal components without exposing their
representations. Consumers depend only on the root facade. representations. Consumers depend only on the root facade.

View File

@@ -28,6 +28,34 @@ constructor does not enable one.
Each invocation carries its state in request, prepared-run, and result values. Each invocation carries its state in request, prepared-run, and result values.
The runner has no durable run or session store. The runner has no durable run or session store.
## Shared Prompt Selection
The runner uses one prompt-selection and hashing boundary for ordinary
preparation and exact prompt inspection. Preparation retains its early
request-ID check before direct-session normalization; both operations then use
the configured prompt repository to select one definition, load referenced
message content, and calculate the same prompt hash.
Inspection stops after that structural lookup. It does not parse templates or
touch profile, artifact, schema, renderer, validator, admission, or model
collaborators. The root [`Engine.InspectPrompt`](../../engine.go) GoDoc owns
the public operation's exact contract.
## Shared Profile Selection
The runner uses one profile-selection and target-resolution boundary for
ordinary preparation and exact profile inspection. Preparation first selects a
request profile or a prompt default; inspection begins with its required
explicit profile ID. Both then apply the ordinary source precedence, resolve a
named backend, and construct the effective target from framework, backend, and
profile values.
Inspection stops after the resulting endpoint and model are structurally
validated. It does not check credential availability or perform prompt,
artifact, schema, rendering, admission, or model-client work. The root
[`Engine.InspectProfile`](../../engine.go) GoDoc owns the public operation's
exact contract.
## Shared Preparation Pipeline ## Shared Preparation Pipeline
`Prepare` and `Run` share one private preparation pipeline split at the point `Prepare` and `Run` share one private preparation pipeline split at the point
@@ -124,15 +152,17 @@ validation failures. Wrapping preserves the package identities mapped by the
public facade and retains collaborator identities where they are part of the public facade and retains collaborator identities where they are part of the
internal contract. internal contract.
Admission capacity exhaustion retains the internal capacity identity and adds Admission capacity exhaustion retains the internal capacity identity. At the
the selected backend ID as context. It is not recategorized as an invalid use-case boundary, the runner attaches the selected backend ID in an internal
request or generation failure, and no partial result is returned. A context typed error, and the root facade copies that value into the public
already done at admission retains its context identity directly. Cancellation [`CapacityError`](../../capacity_error.go) without parsing diagnostic text. It
while waiting for an active generation permit prevents client invocation when is not recategorized as an invalid request or generation failure, and no
it wins the grant race; the model-client boundary then preserves the context partial result is returned. A context already done at admission retains its
error through the generation-failure category. Deferred release restores the context identity directly. Cancellation while waiting for an active generation
admission lease on preparation, generation, validation, repair, and permit prevents client invocation when it wins the grant race; the model-client
cancellation failures. boundary then preserves the context error through the generation-failure
category. Deferred release restores the admission lease on preparation,
generation, validation, repair, and cancellation failures.
Other context cancellation propagates through the invoked collaborator and is Other context cancellation propagates through the invoked collaborator and is
classified by the owning operation. classified by the owning operation.

View File

@@ -16,6 +16,11 @@ validation modes, built-in catalog, and source precedence.
definitions, selects an ID and optional version, and resolves file-backed definitions, selects an ID and optional version, and resolves file-backed
message content within the selected operating-system or `fs.FS` source. message content within the selected operating-system or `fs.FS` source.
Exact prompt inspection performs one point-in-time lookup through that same
repository and validates referenced message content before returning declared
metadata. It does not parse templates or read profile, input, or schema
sources, and it does not retain the definition for a later execution.
Its package tests own prompt selection, strict decoding, definition validation, Its package tests own prompt selection, strict decoding, definition validation,
duplicate detection, and source containment: duplicate detection, and source containment:
[prompt-definition repository tests](../../internal/promptdef/repository_test.go). [prompt-definition repository tests](../../internal/promptdef/repository_test.go).
@@ -28,7 +33,12 @@ with fallback only when the primary reports that a profile is absent. Strict
YAML decoding recognizes the optional `backend` field, trims its value, and YAML decoding recognizes the optional `backend` field, trims its value, and
requires a model plus at least one non-blank backend or endpoint. Loading does requires a model plus at least one non-blank backend or endpoint. Loading does
not check registry membership because the available registry belongs to the not check registry membership because the available registry belongs to the
assembled engine; the runner checks membership during preparation. assembled engine; the runner checks membership during preparation and exact
profile inspection.
Exact profile inspection performs one point-in-time lookup through those
profile sources and checks the resolved target without reading prompt, input,
or schema sources. It does not retain that lookup for a later execution.
`internal/profile/builtin` embeds the maintained built-in profile catalog and `internal/profile/builtin` embeds the maintained built-in profile catalog and
can place a caller-selected repository ahead of that catalog. Every embedded can place a caller-selected repository ahead of that catalog. Every embedded
@@ -68,6 +78,22 @@ filesystem or an `fs.FS`. Invalid generated content is returned as a validation
result; inability to load, register, or compile a schema is an operational result; inability to load, register, or compile a schema is an operational
error. error.
For executable preparation, the built-in validators create a frozen validation
plan. None, basic, and JSON modes retain the effective output contract without
source access. JSON Schema mode loads the root document, resolves and compiles
every transitive reference during preparation, and retains the compiled
validator. The provider-facing structured-output metadata uses that same
captured root document.
`PrepareExecution` also completes prompt and profile selection, artifact
loading and hashing, session and message rendering, and target resolution.
`RunPrepared` uses the retained source-derived state and validation plan; it
does not reopen prompt, profile, input, or schema sources and does not rerender
the request. By contrast, ordinary `Prepare` produces a preparation value only:
a later `Run` performs its own source resolution and preparation.
The [validator tests](../../internal/validate/standard_validator_test.go) own The [validator tests](../../internal/validate/standard_validator_test.go) own
basic, JSON, JSON Schema, source resolution, schema loading, compilation, and basic, JSON, JSON Schema, source resolution, schema loading, compilation,
content-failure behavior. frozen-reference behavior, and content-failure behavior. Prepared execution
orchestration is owned by the
[use-case tests](../../internal/usecase/prepared_execution_test.go).

189
docs/releases/v0.4.0.md Normal file
View File

@@ -0,0 +1,189 @@
# Promptkit v0.4.0
This supplemental changelog and adoption guide summarizes the consumer-facing
changes from `v0.3.0` to `v0.4.0`. The annotated `v0.4.0` tag is the
authoritative release record. Exact current contracts belong to the linked
GoDoc and durable documentation.
## Summary
`v0.4.0` adds four complementary capabilities:
- opaque prepared-execution handles for preparing once, inspecting safe
details, and executing the same frozen snapshot;
- exact prompt-definition inspection without profile resolution or execution;
- exact profile inspection without selecting a prompt or checking credential
availability; and
- structured backend identity on engine admission-capacity rejection.
These APIs let consumers perform more precise preflight work and retain useful
operational context without reproducing Promptkit's internal resolution logic.
## Compatibility
The release is additive for `v0.3.0` consumers. Existing uses of `Prepare`,
`Run`, backend registration, endpoint-only profiles, local-backend helpers,
runtime overrides, public JSON values, and error sentinels continue to work
without migration.
Capacity rejection now returns a structured error while continuing to match
`ErrCapacityExceeded` through `errors.Is`. Error-string wording and direct
sentinel equality were not public contracts.
The new inspection values, capacity error, and prepared-execution handle do not
have stable JSON representations. `PreparedExecution.Details` returns the
existing stable `PreparedRun` value.
## Upgrade
Update the module dependency with:
```sh
go get gitea.maximumdirect.net/eric/promptkit@v0.4.0
go mod tidy
```
Run the consuming project's ordinary and race-enabled tests after upgrading.
No source migration is required.
## Prepare Once And Execute The Same Snapshot
Consumers that need to persist preparation details before generation can now
prepare an opaque, engine-bound execution:
```go
prepared, err := engine.PrepareExecution(ctx, request)
if err != nil {
// Handle preparation failure.
}
defer prepared.Discard()
details := prepared.Details()
// Persist a consumer-selected, appropriately protected preparation record.
result, err := engine.RunPrepared(ctx, prepared)
```
Preparation freezes the selected sources, rendered messages, effective
settings, input content, structured-output metadata, and validation resources
needed by execution. `Details` returns a fresh, caller-owned,
credential-redacted `PreparedRun`.
A handle belongs to its creating engine and permits one execution attempt.
`RunPrepared` consumes that attempt on success and on operational failure.
`Discard` is idempotent and releases an unclaimed handle's execution-only
state. Consumers should discard handles they will not execute, particularly
when a direct request API key may be retained privately until claim or
discard.
Prepared execution does not reserve backend admission during preparation.
Credential availability and backend admission are checked when execution
begins. The execution context is independent of the preparation context.
See the
[prepared-execution consumer guide](../consumers/pkg-promptkit.md#prepare-now-and-execute-the-same-snapshot-later),
the [`PreparedExecution` GoDoc](../../prepared_execution.go), and the
[`Engine.PrepareExecution` and `Engine.RunPrepared` GoDoc](../../engine.go)
for the exact lifecycle, ownership, cancellation, capacity, timing, and
failure contracts.
## Inspect A Prompt
`Engine.InspectPrompt` resolves one prompt ID and optional version through the
engine's configured prompt source:
```go
inspection, err := engine.InspectPrompt(ctx, "report.summary", "")
```
The result includes prompt identity, the opaque prompt hash, declared default
profile ID, declared input metadata, and normalized output contract. It
structurally loads the selected definition and referenced message content but
does not resolve a profile, load schemas or artifacts, render templates,
reserve capacity, or contact a model.
Use inspection for exact configuration checks and metadata discovery. Use
`PrepareExecution` rather than relying on a prior inspection when later
execution must freeze one exact source state, because filesystem-backed
inspection is only a point-in-time lookup.
See the
[prompt-inspection consumer guide](../consumers/pkg-promptkit.md#inspect-a-prompt-before-preparation)
and [`Engine.InspectPrompt` GoDoc](../../engine.go) for exact selection,
ownership, and error behavior.
## Inspect A Profile
`Engine.InspectProfile` resolves one explicit profile independently of a
prompt:
```go
inspection, err := engine.InspectProfile(ctx, "report-production")
```
The result includes the resolved effective execution target and whether a
later request must provide a direct credential. Environment-variable names may
be reported, but inspection does not read credential values or require the
named variable to be populated.
Inspection applies the ordinary configured and built-in profile precedence and
resolves any selected backend. It does not load a prompt, render content,
reserve capacity, or contact a model.
See the
[profile-inspection consumer guide](../consumers/pkg-promptkit.md#inspect-a-profile-before-prompt-work)
and [`Engine.InspectProfile` GoDoc](../../engine.go) for the exact resolution,
credential, ownership, and error contracts.
## Identify Capacity-Rejected Backends
Calls rejected at Promptkit's bounded engine admission boundary continue to
match `ErrCapacityExceeded`. Consumers can additionally obtain the selected
registered backend ID without parsing diagnostic text:
```go
result, err := engine.Run(ctx, request)
if errors.Is(err, promptkit.ErrCapacityExceeded) {
var capacityErr *promptkit.CapacityError
if errors.As(err, &capacityErr) {
// Record capacityErr.BackendID using application-owned diagnostics.
}
// Apply application-owned overload or retry policy.
}
```
The structured error applies to `Run` and `RunPrepared` admission rejection.
It does not represent provider throttling, quota exhaustion, cancellation
while waiting for generation capacity, or another model-client failure.
Promptkit does not prescribe retry timing or transport status mapping.
See the
[error-handling consumer guide](../consumers/pkg-promptkit.md#handle-errors),
the [`CapacityError` GoDoc](../../capacity_error.go), and the
[`ErrCapacityExceeded` GoDoc](../../engine.go) for the canonical contracts.
## Public API Additions
The release adds:
- `Engine.PrepareExecution`;
- `Engine.RunPrepared`;
- `PreparedExecution`, including `Details`, `Discard`, `String`, and
`GoString`;
- `Engine.InspectPrompt`;
- `PromptInspection`;
- `PromptInputDefinition`;
- `Engine.InspectProfile`;
- `ProfileInspection`; and
- `CapacityError`.
No public API was removed.
## Consumer Action
None. Existing `v0.3.0` workflows may upgrade without adopting the new APIs.
Consumers that adopt prepared execution should discard unused handles.
Consumers that need backend-specific capacity diagnostics may add an
`errors.As` check while retaining their existing `errors.Is` classification.

View File

@@ -1,237 +0,0 @@
# Backend-Specific Concurrency Management
**Status:** Complete.
## Purpose
This roadmap defines the scope and target end state for engine-local,
backend-specific concurrency management. It records the intended capability,
consumer value, and important policy choices.
This document is planning material, not a description of current behavior.
Current exported contracts remain owned by Go declarations and GoDoc, backend
registration guidance by the
[consumer guide](../consumers/pkg-promptkit.md#configure-a-local-openai-compatible-endpoint),
and
implemented orchestration by the
[internal runner document](../internal/runner.md).
## Motivation
Different model backends can sustain very different request loads. A local
network endpoint may need a small concurrency limit, while OpenRouter can
usually accept substantially more simultaneous work. Requiring every consumer
to build its own semaphores and queues would duplicate routing knowledge,
create inconsistent cancellation behavior, and make it easy for one caller to
bypass the intended backend limit.
Promptkit should own this coordination because it already resolves each run to
an engine-scoped backend identity and owns every model-generation call made by
the runner. Consumers should continue submitting ready-to-run requests through
the synchronous API, including concurrently from multiple goroutines, without
implementing their own backend scheduler.
The buffered queue is a safety boundary, not an ordinary throughput
restriction. Its primary purpose is to prevent a bug or unintended submission
loop from creating an unbounded in-memory backlog.
## Scope
The feature will add optional concurrency policy to registered backends and
coordinate `Run` calls against independent per-backend capacity pools.
Each policy has two distinct controls:
- an active-generation limit, which protects the backend from too many
simultaneous model requests; and
- a bounded waiting capacity, which protects the process from admitting an
unbounded backlog.
Concurrency policy belongs to a backend registration. It is not a profile
model parameter and cannot be overridden per run. Profiles select the policy
through their backend ID, while a profile or request endpoint override remains
in the selected backend's pool.
`Prepare` does not call a model and will remain outside concurrency admission.
## Defaults And Configuration
The built-in OpenRouter backend will use:
- an active-generation limit of 16; and
- a waiting capacity of 1024.
The waiting default is intentionally generous. Reaching it should indicate
abnormal submission pressure rather than normal application behavior.
Consumer-registered backends will remain unlimited unless the consumer
configures an active-generation limit. When a consumer enables a limit and
does not specify waiting capacity, the waiting capacity will default to 1024.
Consumers may configure a different bounded capacity, including zero when
they want no admitted backlog beyond the active-limit-sized run set.
The public representation must distinguish an omitted waiting capacity from
an explicit zero.
Endpoint-only profiles have no backend registration from which to obtain
policy and will remain unlimited. A future engine-wide or endpoint-keyed
policy can be considered separately if consumers demonstrate that need.
Invalid limits or capacities will fail engine construction as invalid
configuration. Policy values will be copied into engine-owned immutable state
along with the rest of the backend registration.
## Admission And Execution Behavior
`Run` remains a synchronous, wait-for-result operation. Concurrent callers may
block inside `Run` while waiting for their selected backend, then receive the
ordinary result or error from that invocation.
For a configured pool, the active-generation limit plus the waiting capacity
defines the maximum number of concurrent `Run` invocations that Promptkit will
accept for that backend. A waiting capacity of zero therefore accepts no more
runs than the active limit. Admission is immediate: a call either reserves one
of those bounded slots or receives the capacity error. An accepted run may
then wait internally for active-generation capacity.
For a limited backend, Promptkit will bound the number of accepted runs before
expensive artifact loading, prompt rendering, and large defensive copies where
practical. Lightweight prompt, profile, and backend resolution may occur first
when it is required to identify the correct capacity pool. This pre-admission
resolution must not become a second execution-precedence path with behavior
that can drift from `Prepare`.
An accepted run retains its admission until it completes or fails. Every
actual model-generation call for that run must separately observe the
backend's active-generation limit. This includes:
- the initial generation;
- every output-repair generation; and
- calls made through either the built-in or an injected model client.
Preparation and output validation should not hold an active-generation permit.
A repair remains part of its already-admitted run, but reacquires active
generation capacity so repairs cannot exceed the backend limit. It must not be
rejected merely because new runs filled the waiting queue after its initial
generation.
Within one backend pool, waiting generation calls should be served in FIFO
order, subject to canceled calls being removed. Different backend pools make
progress independently; a saturated local backend must not consume
OpenRouter's active or waiting capacity.
The feature will not promise ordering across backend pools or completion order
among admitted runs.
## Capacity Failure And Cancellation
When a backend's bounded waiting capacity is full, a new `Run` call will fail
promptly rather than waiting outside the bounded admission system. The public
API will expose a recognizable capacity-exhaustion error identity distinct
from invalid configuration, invalid requests, and model-client failures.
Rejected calls return no partial result and do not invoke the model client.
Waiting within the admitted backlog or for active-generation capacity must
honor the caller's context. Cancellation or deadline expiry will:
- stop waiting promptly;
- release any admission or generation capacity held by that invocation;
- preserve the applicable context error identity; and
- avoid invoking the model client if cancellation wins before generation
starts.
Capacity must also be released after preparation, generation, validation,
repair, or collaborator failure. One failed or canceled run must not reduce
the backend's future usable capacity.
Elapsed `Run` timing will include time spent waiting after the call is
accepted. `PreparedRun` timing will continue to describe preparation rather
than queue waiting.
## Engine And Client Boundaries
All pools and queued state belong to one `Engine`. Separate engines do not
share capacity, even when they register the same backend ID or endpoint. The
feature introduces no process-global scheduler.
The engine will apply policy consistently to the built-in model client and an
injected `LLMClient`. Consumers calling their own client outside Promptkit are
outside this boundary. Injected clients remain responsible for their internal
thread safety and cancellation behavior.
Backend policy is keyed by the resolved backend ID rather than endpoint text.
This preserves stable routing when a selected backend's endpoint is overridden
and avoids accidentally combining unrelated registrations that happen to use
the same URL.
## Queue Lifetime And Observability
Admission state is buffered, ephemeral, and in-process. It is not persisted
and has no survival guarantee across engine disposal or process termination.
Promptkit will not introduce background job ownership or require consumers to
start or stop workers.
The initial feature does not require public queue-depth metrics, callbacks, or
inspection APIs. Capacity errors and ordinary call timing provide the
consumer-visible behavior. Operational observability can be added later
without coupling the scheduling mechanism to an application logging or
metrics system.
## Compatibility
Consumer-registered backends and endpoint-only profiles remain unlimited
unless concurrency is explicitly configured, preserving their existing
behavior.
The built-in OpenRouter backend will change from unlimited concurrency to a
limit of 16 with a bounded waiting capacity of 1024. Ordinary synchronous
calls remain unchanged, while unusually high concurrent use may now wait or
return the capacity error. This behavioral change must be identified in the
release notes for the version that publishes it.
Adding backend policy fields and a public capacity error is otherwise
additive. The change will use a pre-`v1` minor release under Promptkit's
[release policy](../release.md#release-model).
## Non-Goals
This scope does not include:
- asynchronous job handles, polling, or detached result delivery;
- durable or cross-process queues;
- persistence or recovery across engine or process shutdown;
- priorities, scheduling weights, or consumer-defined fairness classes;
- automatic retries, backoff, rate-limit interpretation, or provider quota
discovery;
- token-per-minute or request-per-minute rate limiting;
- dynamic reconfiguration after engine construction;
- per-profile or per-run concurrency overrides;
- endpoint-keyed pooling for profiles without a backend ID;
- process-global coordination across engines;
- application worker lifecycle, logging, tracing, or metrics policy; or
- changes to prompt, profile, schema, or model-provider wire formats.
## Target End State
This roadmap reaches its target end state when:
- each engine independently coordinates configured backend capacity;
- the built-in OpenRouter backend allows 16 active generations and up to 1024
waiting runs;
- consumer backends can opt into their own active and waiting limits while
remaining unlimited by default;
- endpoint overrides retain the selected backend's capacity pool and
endpoint-only profiles remain unlimited;
- synchronous `Run` callers wait for and receive their ordinary result;
- admission is bounded before expensive preparation work where practical;
- every initial and repair generation observes the backend's active limit
without serializing preparation or validation;
- a full waiting queue returns a recognizable capacity error without invoking
the model client;
- cancellation and all failure paths promptly release capacity and preserve
context error identity;
- built-in and injected model clients receive the same scheduling behavior;
- pools remain ephemeral, engine-scoped, and independent across backend IDs;
and
- current-state GoDoc, consumer, internal, and release documentation describe
the implemented behavior once it lands.

View File

@@ -33,9 +33,7 @@ consumers.
## Ideas ## Ideas
No ideas are currently cataloged. Backend-specific concurrency management has No ideas currently await selection.
been selected for active planning in the
[focused concurrency roadmap](concurrency.md).
## Entry Format ## Entry Format

View File

@@ -1,386 +0,0 @@
# Local Backend Convenience Implementation Plan
**Status:** Complete.
## Purpose
This document is the decision-complete implementation plan for
[local backend convenience](local-backend.md). It is written for a coding
agent that will implement each stage in order.
The feature roadmap owns the motivation, consumer paths, policy choices,
compatibility requirements, non-goals, and target end state. This document
owns the exact proposed API, file-level changes, implementation sequence, test
ownership, documentation work, validation commands, and completion gates.
## Implementation Rules
- Complete the stages in order. Keep the root package compiling and its
focused tests passing at every stage boundary.
- Preserve unrelated working-tree changes. The feature roadmap may already be
uncommitted when implementation begins; retain it.
- Follow every policy under `docs/policy/`, the task-specific reading guide in
`docs/development.md`, and the accepted behavior in
`local-backend.md`.
- Keep the API in the root `promptkit` package. Do not add a public or internal
package for this feature.
- Implement the helper as a transparent constructor for the existing
`Backend` type. Do not add a second backend representation or bypass
`WithBackend`.
- Leave normalization, validation, copying, registry construction, capacity
defaulting, and duplicate detection in their existing owners.
- Do not add environment-variable discovery, package-global registration,
implicit local defaults, model selection, profile construction, or support
for non-OpenAI-compatible transports.
- Keep tests lean and behavior-focused. Do not duplicate the existing
registry and concurrency test matrices merely because the constructor
reaches those mechanisms.
- Update exact GoDoc with the exported declarations. Update current-state
consumer guidance only after the corresponding API is implemented.
- Do not update release notes, create a release, change a module version, or
tag a commit as part of this work.
## Fixed Design
### Exported API
Add these declarations to `backends.go` in package `promptkit`:
```go
// BackendLocal is the conventional ID used by LocalBackend. It is not a
// built-in or reserved backend and must be registered with WithBackend.
const BackendLocal = "local"
// LocalBackend returns a Backend for a conventional local OpenAI-compatible
// endpoint.
func LocalBackend(endpoint string, concurrencyLimit int) Backend {
return Backend{
ID: BackendLocal,
Endpoint: endpoint,
ConcurrencyLimit: concurrencyLimit,
}
}
```
The exact GoDoc may be wrapped or expanded for clarity, but it must own and
communicate all of these contract points:
- `BackendLocal` is the case-sensitive conventional ID `"local"`;
- it is neither built in nor reserved;
- calling `LocalBackend` does not register anything;
- the returned value must be supplied through `WithBackend`;
- `endpoint` and `concurrencyLimit` are copied into the corresponding fields
without normalization or validation;
- `APIKeyEnv`, `ExtraParams`, and `QueueCapacity` retain their zero values; and
- normal `NewEngine` backend validation and concurrency semantics apply after
registration.
Keep `BackendOpenRouter` unchanged. `BackendLocal` must not alias an internal
registry constant because the internal registry has no special local-backend
identity or behavior.
Place `BackendLocal` near `BackendOpenRouter`, and place `LocalBackend` after
the `Backend` declaration and before `WithBackend`. This keeps the conventional
IDs, configured value, convenience constructor, and registration option
discoverable in one file.
### Constructor Semantics
`LocalBackend` is a pure struct constructor. Its complete behavior is
equivalent to the keyed literal shown above.
In particular, the constructor must not:
- trim or parse the endpoint;
- reject blank endpoints or negative limits;
- choose a default limit;
- assign an API-key environment variable;
- allocate an empty `ExtraParams` map;
- assign a queue-capacity pointer;
- read process environment;
- mutate package-global or engine state; or
- call `WithBackend` itself.
Deferred validation is intentional. It keeps one validation path for all
`Backend` values: `WithBackend` copies the public value into engine options,
and `NewEngine` constructs the validated immutable registry. A positive limit
with a nil queue continues to select the existing default queue capacity of
1024; zero continues to mean unlimited; a negative value continues to make
`NewEngine` fail with `ErrInvalidConfig`.
The returned `Backend` remains an ordinary caller-owned value. Consumers can
modify it before passing it to `WithBackend`, although task-oriented
documentation should direct materially customized configurations to an
explicit keyed `Backend` literal.
### Identity And Compatibility
Do not add `"local"` to the internal built-in or reserved ID set. A consumer
must be able to register either:
```go
promptkit.LocalBackend(endpoint, limit)
```
or:
```go
promptkit.Backend{
ID: promptkit.BackendLocal,
Endpoint: endpoint,
}
```
through the existing `WithBackend` option. Both forms participate in ordinary
duplicate-ID detection. Existing consumers that already use the literal ID
`"local"` remain source- and behavior-compatible.
No existing `Backend`, `WithBackend`, profile, registry, execution-target, or
capacity semantics change. Do not modify `internal/backend`,
`internal/capacity`, `internal/domain`, `engine.go`, `profiles.go`, or
`types.go` for this feature.
### Test Ownership
The root external-package contract suite in `public_contract_test.go` owns the
new public behavior. Add one focused test named:
```go
func TestLocalBackendConstructsAndRegistersConventionalBackend(t *testing.T)
```
The test must:
1. call `promptkit.LocalBackend` with a test endpoint and a positive,
test-owned concurrency limit;
2. compare the returned value with this complete expected value:
```go
promptkit.Backend{
ID: promptkit.BackendLocal,
Endpoint: endpoint,
ConcurrencyLimit: limit,
}
```
A whole-value comparison is appropriate here because the exact zero-value
fields are part of this small public constructor's contract.
3. register that returned value with `WithBackend`;
4. add an in-memory profile whose `BackendID` is
`promptkit.BackendLocal`;
5. construct an engine through the existing contract-test prompt fixture;
6. call `Prepare`; and
7. assert that the prepared result exposes `BackendLocal` as the selected
backend and the supplied endpoint as the effective endpoint.
This single test protects the realistic compatibility risks: accidental field
defaults, a changed conventional ID, failure to compose with `WithBackend`,
and accidental treatment of `"local"` as reserved. It also demonstrates that
the helper uses the existing backend/profile path.
Do not add separate tests for blank endpoints, malformed endpoints, negative
limits, queue defaulting, duplicate IDs, engine isolation, runtime capacity,
or caller mutation. Those mechanisms are unchanged and already have tests at
their owning boundaries. Do not add internal-package tests for this root
facade constructor.
### Consumer Documentation
Update `docs/consumers/pkg-promptkit.md` after the API exists. Keep Go
declarations and GoDoc as the exact API owner; the guide should help consumers
choose a workflow and link to `backends.go` for precise semantics.
Restructure the backend guidance to present these paths in increasing order of
configuration:
1. **Endpoint-only profile.** Show a small in-memory `Profile` with
`Endpoint` and `Model`. Explain that this is the simplest choice when only
one profile needs the endpoint and shared backend identity or capacity
policy is unnecessary.
2. **Local convenience constructor.** Show
`WithBackend(promptkit.LocalBackend("http://localhost:8000/v1", 2))`
together with a profile using
`BackendID: promptkit.BackendLocal`. Explain briefly that the helper is
explicit, is not pre-registered, does not read environment variables, and
leaves the queue capacity at the existing default for a positive limit.
3. **Complete backend value.** Preserve an advanced example using a keyed
`Backend` literal for needs such as `APIKeyEnv`, an explicit
`QueueCapacity`, extra parameters, a custom ID, or multiple local
endpoints. Use a custom ID other than `"local"` in that example so the
distinction from the conventional helper is clear.
Keep the existing backend-routing, selected-backend identity, concurrency,
credential, and error guidance unless a small wording adjustment is required
to make the new decision path coherent. Avoid repeating the complete field
contract or registry validation rules from GoDoc.
Do not add a new maintained example, README section, format-reference entry,
integration-contract change, internal-document change, or release note. The
consumer-guide snippets are sufficient for this small convenience API, and
none of those other documents owns the affected task or contract.
## Stage 1: Add The Public Constructor And Contract Test
**Status:** Complete.
### Objective
Add the smallest public API that expresses the accepted local-backend
convention and protect its compatibility through the root public boundary.
### Implementation Prompt
1. Re-read `docs/development.md`, all files under `docs/policy/`,
`local-backend.md`, `backends.go`, the backend-related portion of
`public_contract_test.go`, and the existing backend/concurrency GoDoc before
editing.
2. Confirm the working tree and preserve the uncommitted roadmaps and any
unrelated consumer changes.
3. Add the untyped exported string constant `BackendLocal = "local"` to
`backends.go` without changing `BackendOpenRouter`.
4. Add `LocalBackend(endpoint string, concurrencyLimit int) Backend` to
`backends.go` using the exact keyed-literal implementation in the fixed
design.
5. Write complete GoDoc for both declarations. Make their conventional,
explicit, non-built-in, non-reserved, and deferred-validation semantics
unambiguous.
6. Add
`TestLocalBackendConstructsAndRegistersConventionalBackend` to
`public_contract_test.go` exactly as specified under Test Ownership. Reuse
the existing contract prompt fixture rather than adding a fixture or test
helper.
7. Do not edit internal packages. If the constructor appears to require an
internal change, stop and reconcile the implementation with the fixed
transparent-constructor design instead.
### Focused Validation
Run:
```sh
gofmt -w backends.go public_contract_test.go
go test . -run '^TestLocalBackendConstructsAndRegistersConventionalBackend$'
go test .
go vet .
go build .
git diff --check
```
Inspect the diff and confirm that this stage changes only `backends.go`,
`public_contract_test.go`, and the already-present roadmap files.
### Completion Gate
Stage 1 is complete when:
- the exported constant and constructor match the fixed API;
- the constructor returns only the three specified non-zero fields;
- the helper registers through the ordinary `WithBackend` path;
- `"local"` remains a valid consumer registration rather than a reserved
built-in;
- the focused public-contract test passes; and
- the root package test, vet, build, formatting, and whitespace checks pass.
## Stage 2: Publish Consumer Guidance And Validate The Repository
**Status:** Complete.
### Objective
Make the simplest suitable local-endpoint configuration easy to discover,
confirm the complete change across the repository, and close the temporary
roadmaps.
### Implementation Prompt
1. Re-read the implemented declarations and GoDoc before describing them.
2. Update `docs/consumers/pkg-promptkit.md` according to the three-path
structure under Consumer Documentation.
3. Keep examples illustrative, minimal, secret-free, and consistent with the
implemented declarations. Link precise semantics to `backends.go` rather
than duplicating its field-by-field contract.
4. Follow every added or changed Markdown link and confirm its target exists.
Confirm that all local repository-relative links in
`local-backend.md`, this plan, and the changed consumer guide resolve.
5. Run the complete maintainer validation sequence below.
6. Inspect the final diff for accidental internal behavior changes,
generated artifacts, credentials, local workspace files, or module
replacements.
7. After every completion gate passes, set both stage statuses, this plan's
status, and the status in `local-backend.md` to `Complete`. Do not retire or
remove the roadmaps in the implementation change; roadmap retirement
follows implementation review.
### Full Validation
Run the complete sequence from `docs/development.md`:
```sh
go test ./...
go test -race ./...
go vet ./...
go build ./...
go run ./examples/go-library/prepare
gofmt -l $(git ls-files '*.go')
git diff --check
```
The formatting command must produce no paths. Follow every added or changed
Markdown link and confirm its target and heading exist.
Also inspect:
```sh
git status --short
git diff --stat
git diff
```
Confirm that:
- production changes are limited to the root convenience API;
- no internal backend, registry, capacity, profile, or engine behavior
changed;
- `BackendLocal` is conventional and consumer-registerable, not built in or
reserved;
- `LocalBackend` performs no validation, normalization, environment lookup,
registration, allocation, or hidden default selection;
- the returned `Backend` leaves `QueueCapacity` nil so existing positive-limit
defaulting remains owned by the registry;
- existing endpoint-only profiles and complete custom backends remain
documented and supported;
- current-state documentation describes only the now-implemented API and
links to the canonical GoDoc for exact semantics;
- no `go.work`, `go.work.sum`, local module replacement, credential,
generated binary, or unrelated change was introduced; and
- the feature and implementation roadmaps contain no unresolved work marked
complete.
### Completion Gate
Stage 2 is complete when every target-end-state item in
`local-backend.md` is implemented, the consumer guide clearly presents all
three configuration paths, all complete validation commands pass, all changed
links resolve, and the roadmap statuses accurately report completion.
## Implementation Handoff
The implementation handoff should report:
- the new `BackendLocal` and `LocalBackend` public API;
- that the helper remains explicit and composes with the ordinary backend
registry;
- the focused public-contract coverage added;
- the consumer-guide decision path added;
- the complete validation commands and results; and
- any unrelated working-tree changes that were preserved.
Do not claim that a local backend is built in, pre-registered, configured from
environment variables, or assigned a default model or concurrency limit.
## Open Questions
None. The feature roadmap and this plan fix the exported API, constructor
semantics, identity treatment, compatibility behavior, test boundary,
documentation ownership, implementation sequence, validation gates, and
non-goals required for implementation.

View File

@@ -1,163 +0,0 @@
# Local Backend Convenience
**Status:** Complete.
## Purpose
Make the common case of using a local OpenAI-compatible endpoint concise and
easy to discover without introducing implicit configuration or a separate
backend abstraction.
The existing `Backend` type and registry remain the canonical, fully
configurable interface. A small convenience constructor will cover the usual
local-network case, while improved consumer documentation will make it clear
when an endpoint-only profile, the convenience constructor, or a complete
`Backend` value is appropriate.
## Motivation
Consumers can already use a local endpoint by setting `Profile.Endpoint`, or
register one as a backend with `WithBackend`. The first option is concise but
does not provide shared backend-level concurrency control. The second supports
the complete backend feature set but requires consumers to understand and
populate several fields for a common configuration.
Most consumers adding a local backend need only:
- a stable backend ID;
- an OpenAI-compatible endpoint; and
- a concurrency limit appropriate for the local server.
Promptkit should provide a direct path for that case while keeping all
configuration explicit and preserving the full registry interface for
advanced needs.
## Consumer Paths
Documentation should present three progressively more configurable paths:
1. Set `Profile.Endpoint` when a profile only needs to target a local endpoint
and does not need shared backend policy.
2. Use the local-backend convenience constructor when profiles should share a
named local endpoint and its concurrency limit.
3. Construct a complete `Backend` value when the consumer needs a custom
backend ID, authentication, extra request parameters, an explicit queue
capacity, or multiple local backends.
These are complementary interfaces. The convenience constructor must return an
ordinary `Backend`, so it does not create a second configuration model.
## Public Convenience API
The public package should expose:
```go
const BackendLocal = "local"
func LocalBackend(endpoint string, concurrencyLimit int) Backend
```
`LocalBackend` should return a `Backend` with:
- `ID` set to `BackendLocal`;
- `Endpoint` set to the supplied endpoint;
- `ConcurrencyLimit` set to the supplied limit; and
- all other fields left at their zero values.
The returned value is passed to `WithBackend` and follows the same copying,
normalization, validation, and registration rules as any consumer-constructed
`Backend`.
The constructor should be a transparent value constructor. It should not read
environment variables, mutate global state, register the backend, validate
arguments independently, or create profiles. Consumers may inspect or modify
the returned value before registration, although documentation should direct
substantially customized configurations to the full `Backend` form.
## Identity and Registration
`BackendLocal` is a conventional ID used by the convenience constructor. It is
not pre-registered and should not become a specially reserved registry ID.
Consumers remain responsible for registering the returned backend with
`WithBackend` and naming it from profiles through `BackendID`.
This distinction preserves compatibility with consumers that may already
register their own backend using the ID `"local"`. Normal duplicate-ID rules
still apply if a consumer attempts to register more than one backend with that
ID.
Consumers that need multiple local endpoints should choose distinct IDs and
use complete `Backend` values rather than the single conventional helper ID.
## Concurrency and Queue Semantics
The constructor must preserve the existing backend concurrency contract:
- a positive concurrency limit bounds simultaneous requests and uses the
existing default queue capacity because `QueueCapacity` remains `nil`;
- a zero concurrency limit leaves the backend unconstrained; and
- a negative concurrency limit is rejected through the existing engine
configuration validation path.
The constructor should not select a hidden default concurrency limit. Local
servers vary substantially in capacity, so the consumer should make this
choice explicitly.
## Documentation
The final documentation state has two canonical surfaces:
- Public Go documentation describes the exact contract of
`BackendLocal` and `LocalBackend`, including their conventional,
non-pre-registered nature.
- The [promptkit consumer guide](../consumers/pkg-promptkit.md) includes
a task-oriented local-endpoint section that shows the three consumer paths,
explains the decision between them, and provides concise examples of the
endpoint-only and convenience-constructor forms.
The consumer guide continues to document the full `Backend` interface as
the advanced path rather than attempting to reproduce every configuration
variation through convenience APIs.
## Compatibility
This feature is additive:
- existing endpoint-only profiles continue to work unchanged;
- existing `Backend` values and `WithBackend` registrations remain the
canonical general-purpose interface;
- existing registrations using the literal ID `"local"` remain valid; and
- OpenRouter defaults and all other backend behavior remain unchanged.
No consumer is required to adopt the convenience constructor.
## Non-Goals
This work does not include:
- pre-registering or implicitly enabling a local backend;
- discovering a local endpoint, API key, or concurrency limit from environment
variables;
- adding local-backend fields to `Config`;
- selecting a default local model or creating a profile automatically;
- adding a combined backend-and-profile constructor;
- adding convenience parameters for API keys, extra request parameters, or
queue capacity;
- replacing or redesigning the backend registry;
- adding support for non-OpenAI-compatible local APIs; or
- changing backend routing, scheduling, or queue behavior.
## Target End State
After this work:
- consumers with a simple one-profile local endpoint can continue to configure
it directly on the profile;
- consumers needing a shared local endpoint and concurrency policy can express
it with one `LocalBackend` call and register the returned value normally;
- consumers with advanced or multiple-local-backend requirements have a clear
path to the complete `Backend` interface;
- all local configuration remains explicit, inspectable, and compatible with
dependency injection; and
- canonical documentation makes the simplest suitable interface easy to find
without obscuring the underlying registry model.

View File

@@ -0,0 +1,245 @@
# Notarius PromptKit Wishlist
## Purpose
This document records features and interface changes that would be useful
additions to PromptKit from the perspective of the maintainers of Notarius, a
downstream application that consumes PromptKit.
PromptKit now provides the capabilities Notarius currently needs. The
remaining deferred ideas are optional opportunities to improve checkpointing
and operational observability.
The examples are API sketches intended to communicate the desired capability,
not prescriptive names or finalized Go contracts.
## Priority 1: Atomic Execution With Prepared Details
**Disposition:** Implemented through [`Engine.PrepareExecution` and
`Engine.RunPrepared`](../../engine.go). See the
[consumer guidance](../consumers/pkg-promptkit.md#prepare-now-and-execute-the-same-snapshot-later).
A separate `RunDetailed` method is not cataloged.
### Downstream need
Notarius needs both:
- the completed `RunResult`; and
- the rendered messages, effective output contract, hashes, and other
preparation details exposed by `PreparedRun`.
Notarius uses the prepared details to construct redaction-aware debug bundles
and retain enough information to diagnose model behavior.
### Implemented behavior
Notarius can prepare one frozen execution snapshot, retain a caller-owned and
credential-redacted `Details` value for its debug bundle, and execute the same
snapshot through `RunPrepared`. The opaque handle is engine-bound and
single-use; an unused handle can be released with `Discard`. The consumer
guide and exported GoDoc own the exact lifecycle and failure contracts.
### Value to Notarius
This removes duplicate work from PromptKit-backed calls and ensures that
retained debug material corresponds atomically to the actual execution.
## Priority 2: Prompt-Independent Profile Inspection
**Disposition:** Implemented as
[`Engine.InspectProfile`](../../engine.go). See the
[consumer guidance](../consumers/pkg-promptkit.md#inspect-a-profile-before-prompt-work).
### Downstream need
Notarius validates configured pipeline profile IDs before beginning a run. It
needs to determine whether:
- a profile exists;
- its referenced backend is registered;
- its execution target can be resolved; and
- it declares a credential requirement that the application may need to
enforce.
This validation should not require model generation.
### Previous integration
Before profile inspection was available, Notarius constructed a synthetic
prompt using `testing/fstest.MapFS`, supplied a dummy transcript, and called
`Engine.Prepare` solely to exercise profile and backend resolution.
### Value to Notarius
The implemented interface eliminates a synthetic production-only prompt
fixture and establishes a direct, supported contract for configuration-time
profile and backend validation.
## Priority 3: Semantic Execution-Target Fingerprints
**Disposition:** Deferred pending a separate semantic-equality design for
resolved execution targets.
### Downstream need
Notarius checkpoints model-backed pipeline stages. A checkpoint must not be
reused when generation-affecting PromptKit configuration changes.
Notarius therefore needs a stable equality signal for the effective profile
and backend target used by a pipeline.
### Current integration
Notarius currently constructs this identity itself from:
- a manually maintained marker for the PromptKit release and built-in profile
catalog;
- raw hashes of configured profile files; and
- a separate hash of the configured conventional local-backend endpoint.
This is safe but conservative and coupled to PromptKit details. Raw file
hashing also invalidates checkpoints for semantically irrelevant YAML changes,
such as comments or formatting.
### Requested capability
Expose an opaque semantic digest for a resolved profile and its effective
generation target. It could be returned by the proposed profile-resolution
API:
```go
type ResolvedProfile struct {
ProfileID string
BackendID string
EffectiveTarget ExecutionTarget
ExecutionDigest string
}
```
Alternatively, PromptKit could expose a dedicated method such as
`ProfileExecutionDigest(profileID)`.
### Desired equality semantics
The digest should change when generation-affecting state changes, including:
- resolved model and endpoint;
- backend routing identity;
- backend request defaults and extra parameters;
- profile generation parameters; and
- the semantic identity of any selected built-in profile.
The digest should not incorporate:
- credential values;
- concurrency or queue capacity;
- filesystem source paths;
- YAML comments or formatting; or
- other settings that affect scheduling or source representation without
changing the generation target.
The credential environment-variable name may need to participate if changing
it can select a materially different provider account or target. PromptKit
should define this deliberately while continuing to exclude the resolved
secret value.
### Design considerations
- Treat the digest as an opaque equality value rather than a public encoding
of internal structures.
- Document which categories of change affect equality.
- Include a versioned semantic marker internally so PromptKit can deliberately
invalidate old digests when its resolution semantics change.
- Prefer a per-profile digest over a digest of every profile known to an
engine. Notarius generally knows which profiles a resolved pipeline uses.
- Do not require consumers to know PromptKit's built-in catalog version.
### Value to Notarius
This would let Notarius remove its PromptKit release marker and raw
profile-source fingerprinting, reduce unnecessary checkpoint invalidation, and
delegate execution-target equality to the component that owns target
resolution.
## Priority 4: Structured Capacity Errors
**Disposition:** Implemented behavior. See the consumer guide's
[Handle Errors](../consumers/pkg-promptkit.md#handle-errors) section.
### Downstream need
Notarius translates PromptKit backend-capacity rejection into a
provider-neutral application error. When multiple backends are active,
operators would benefit from knowing which backend rejected admission without
parsing an error string or exposing endpoint details.
### Implemented behavior
PromptKit now retains the broad capacity classification while allowing
Notarius to obtain the selected backend ID without parsing diagnostic text.
The consumer guide owns the application workflow, including retry and backoff
policy.
### Value to Notarius
This improves operational diagnostics and future metrics while preserving
the provider-neutral error boundary used by Notarius.
## Capabilities PromptKit Already Provides Well
The current PromptKit boundary is sufficient for Notarius's implemented
behavior. In particular, PromptKit already provides:
- filesystem, `fs.FS`, and programmatic prompt, profile, and schema sources;
- offline preparation without model execution;
- structured output and content validation;
- direct session propagation;
- tri-state per-run reasoning overrides;
- selected profile, backend, model, endpoint, effective parameters, hashes, and
token-usage provenance;
- endpoint-only profiles;
- the conventional `local` backend helper;
- arbitrary engine-scoped `Backend` registrations;
- backend authentication environment names, extra parameters, concurrency
limits, and queue-capacity policies;
- provider-client and artifact-reader extension interfaces;
- context cancellation; and
- useful public error sentinels, including profile absence and capacity
exhaustion.
The wishlist does not imply that Notarius needs PromptKit to broaden its core
responsibilities. It primarily asks for more direct access to information and
operations that PromptKit already computes internally.
## Responsibilities That Should Remain In Notarius
The following concerns belong to the downstream application and should not
move into PromptKit for the sake of Notarius:
- pipeline staging, dependencies, and generated references;
- application-wide scheduling across providers and backends;
- module and validation retry policy;
- checkpoints, resume, and recomputation;
- durable run artifacts and manifests;
- D&D prompts, schemas, extractors, validators, and normalizers;
- Notarius configuration-file parsing and precedence;
- domain-specific prompt-cache prefix policy; and
- application-specific redaction, retention, and debug-bundle policy.
PromptKit's complete `Backend` API already supports custom IDs, multiple local
endpoints, authentication, extra parameters, and explicit queue policies.
Whether Notarius exposes those capabilities in its own configuration is an
application-policy decision, not an upstream PromptKit gap.
## Suggested Upstream Sequence
For downstream adoption and any remaining upstream work, the useful order is:
1. Adopt prepared execution for atomic details and results.
2. Add a semantic execution-target digest, preferably alongside profile
inspection.
3. Use the implemented typed capacity error where backend admission diagnostics
are needed.
The first removes the concrete execution workaround. The second would improve
checkpoint correctness and reduce coupling. The third is operational polish.

View File

@@ -0,0 +1,331 @@
# Weatherreporter PromptKit Wishlist
## Purpose
This document records features and interface changes that would be useful
additions to PromptKit from the perspective of the maintainers of
Weatherreporter, a downstream application planning to replace its Scriptorium
CLI integration with PromptKit.
PromptKit now provides the capabilities Weatherreporter needs for the
migration. The remaining deferred ideas are optional opportunities to validate
configuration earlier and improve durable failure diagnostics.
The examples are API sketches intended to communicate the desired capability,
not prescriptive names or finalized Go contracts. The related
[Notarius PromptKit wishlist](notarius-promptkit-wishlist.md) proposes several
overlapping features from another downstream consumer's perspective.
## Priority 1: Executable Preparation Handles
**Disposition:** Implemented as [`Engine.PrepareExecution` and
`Engine.RunPrepared`](../../engine.go). See the
[consumer guidance](../consumers/pkg-promptkit.md#prepare-now-and-execute-the-same-snapshot-later).
### Downstream need
Weatherreporter treats prompt preparation as a durable preflight boundary. It
needs to:
1. prepare the exact request that will be executed;
2. persist a safe preparation record before starting the provider call; and
3. execute without reloading or rerendering prompt, profile, schema, or input
sources.
Persisting preflight before generation leaves useful evidence when a provider
call fails or the process is interrupted during generation.
### Implemented behavior
Weatherreporter can prepare one frozen execution snapshot, persist a
caller-owned and credential-redacted `Details` value, and execute that same
snapshot through `RunPrepared`. The opaque handle is engine-bound and
single-use; an unused handle can be released with `Discard`. The consumer
guide and exported GoDoc own the exact lifecycle, credential, cancellation,
and capacity contracts.
### Value to Weatherreporter
This preserves Weatherreporter's durable preflight behavior, removes duplicate
work, eliminates the source-consistency window, and ensures that persisted
provenance describes the actual execution.
## Priority 2: Prompt-Definition Inspection
**Disposition:** Implemented as
[`Engine.InspectPrompt`](../../engine.go). See the
[consumer guidance](../consumers/pkg-promptkit.md#inspect-a-prompt-before-preparation).
### Downstream need
Weatherreporter has a fixed registry of seven report definitions. Each report
selects a prompt ID and one of two output workflows:
- direct Markdown; or
- structured generated text followed by application-owned domain validation
and Markdown template rendering.
Weatherreporter will embed the PromptKit prompt definitions and private
response schemas that implement those reports. It needs to validate that the
report registry and embedded prompt corpus agree before weather collection or
provider execution.
### Previous integration option
Before prompt inspection was available, Weatherreporter could maintain
synthetic data-package fixtures and call `Engine.Prepare` for every report
prompt during tests. Runtime validation could also occur through the ordinary
per-report preparation stage.
This required complete placeholder inputs and profile resolution when the
application primarily wanted to inspect prompt identity and declared contracts.
### Value to Weatherreporter
The implemented interface lets Weatherreporter directly verify that every
report prompt exists, requires the curated `data_package` input, and declares
the expected Markdown or JSON Schema output contract. It reduces synthetic
test setup and moves failures ahead of weather collection.
## Priority 3: Prompt-Independent Profile Inspection
**Disposition:** Implemented as
[`Engine.InspectProfile`](../../engine.go). See the
[consumer guidance](../consumers/pkg-promptkit.md#inspect-a-profile-before-prompt-work).
### Downstream need
Weatherreporter will allow operators to select an external PromptKit profile
source and may allow an explicit profile override. It needs to reject a missing
profile, unknown backend, or malformed execution target before collecting
weather data or writing report artifacts, and to apply its own policy to a
reported credential requirement.
### Value to Weatherreporter
The implemented interface improves fail-fast configuration validation and gives
operator-facing errors direct profile and backend context. It remains optional
for the initial migration.
## Priority 4: Eager Source Validation
**Disposition:** Deferred until prompt and profile inspection have been used
to determine whether a broader engine-wide validation operation is still
needed.
### Downstream need
PromptKit deliberately defers reading and validating filesystem and `fs.FS`
prompt, profile, and schema content until a request needs it. Weatherreporter
has a small fixed embedded prompt corpus and one optional external profile
source. It would benefit from an explicit offline validation operation for
tests, startup diagnostics, and configuration checks.
### Current integration option
Weatherreporter can prepare every report prompt with fixture inputs and inspect
any explicit profiles individually. That provides strong coverage but requires
consumer-maintained traversal and synthetic material.
### Requested capability
Consider an opt-in source-validation operation:
```go
type SourceValidationOptions struct {
RequireCredentials bool
}
func (e *Engine) ValidateSources(
ctx context.Context,
opts SourceValidationOptions,
) error
```
The operation should eagerly discover and structurally validate the configured
prompt, profile, and schema sources without model generation.
### Design considerations
- Keep deferred validation as the normal `NewEngine` behavior.
- Make eager validation an explicit consumer choice.
- Validate duplicate IDs and versions, strict YAML decoding, referenced content
files, profile/backend membership, schema syntax, and schema references.
- Distinguish structural credential declarations from current environment
availability.
- Do not read or expose credential values when credential availability is not
requested.
- Preserve source-specific public error identities and useful path context.
- Respect context cancellation during filesystem discovery and schema work.
- Consider whether exact prompt and profile inspection APIs already provide a
smaller sufficient surface before adding an engine-wide operation.
### Value to Weatherreporter
This would simplify offline corpus checks and catch malformed operator profile
sources before report work begins. It is helpful but lower priority than exact
prompt and profile inspection.
## Priority 5: Structured Generation Errors
**Disposition:** Deferred pending stronger downstream demand and a narrower
design that does not duplicate prepared provenance or impose HTTP-specific
fields on injected model clients.
### Downstream need
Weatherreporter preserves redacted, inspectable failure receipts for report
runs. When model generation fails operationally, it needs to classify the
failure and retain safe execution context without parsing error prose.
Prompt preparation already supplies selected profile, backend, and model
identity. Provider status classification would add useful operator context,
especially when the built-in OpenAI-compatible client receives a non-success
HTTP status.
### Current integration option
PromptKit exposes `ErrLLMGenerate` and preserves injected client errors through
`errors.Is`. Weatherreporter can reliably classify generation failure and use
its preparation record for profile, backend, and model provenance. Any further
diagnostic detail remains a redacted error string.
### Requested capability
Consider a typed generation error that continues to match `ErrLLMGenerate`:
```go
type GenerationError struct {
BackendID string
Model string
StatusCode int
}
```
The exact fields may differ. The useful contract is safe structured context
available through `errors.As`, while `errors.Is(err, ErrLLMGenerate)` remains
compatible.
### Design considerations
- Include only fields that PromptKit knows reliably and can expose safely.
- Treat an HTTP status as optional because injected model clients may not use
HTTP.
- Do not expose provider response bodies, endpoints, credential environment
names, credential values, request content, or generated content.
- Do not make a structured error a second source of prompt/profile provenance
already present in a prepared execution.
- Preserve injected client error identity.
- Keep retry and backoff policy with the consuming application.
### Value to Weatherreporter
This would improve durable failure receipts and troubleshooting, particularly
for built-in transport failures. It is not required if preparation details and
the existing sentinel remain available.
## Lower-Priority Shared Wishlist Items
### Structured Capacity Errors
**Disposition:** Implemented behavior. See the consumer guide's
[Handle Errors](../consumers/pkg-promptkit.md#handle-errors) section.
PromptKit now exposes the stable backend ID for capacity rejection without
requiring Weatherreporter to parse error text.
Weatherreporter currently generates batch reports sequentially and constructs
one engine per invocation, so engine-local capacity exhaustion is unlikely in
the initial design. The typed error becomes more valuable if report generation
later becomes concurrent or PromptKit engines become longer-lived.
It should not block adoption.
### Semantic Execution-Target Fingerprints
**Disposition:** Deferred pending a separate semantic-equality design for
resolved execution targets.
The semantic target digest proposed by the
[Notarius wishlist](notarius-promptkit-wishlist.md#priority-3-semantic-execution-target-fingerprints)
would provide a compact equality signal for audit metadata.
Weatherreporter does not currently reuse LLM-dependent checkpoints. Its Recent
Changes behavior compares deterministic module snapshots rather than generated
reports, so the digest has no immediate cache-correctness role. Existing
PromptKit result metadata is sufficient for the initial integration. A digest
would still be useful provenance and future-proofing, but it is not a
migration priority.
## Capabilities PromptKit Already Provides Well
PromptKit already provides the essential Weatherreporter integration surface:
- importable in-process engine construction;
- filesystem, `fs.FS`, and programmatic prompt, profile, and schema sources;
- offline preparation without model execution;
- prepared execution handles for a durable preflight boundary;
- exact prompt and profile inspection;
- versioned prompt selection;
- text, Markdown, JSON, and JSON Schema output contracts;
- single-pass output validation with raw output retained after completed
validation failure;
- inline input artifacts with provenance URIs and input hashes;
- selected profile, backend, model, effective target, prompt hashes, timing,
and token-usage provenance;
- endpoint-only profiles and engine-scoped backend registration;
- injected model-client and artifact-reader interfaces;
- caller cancellation, generation timeout, and transport timeout behavior; and
- public error sentinels for configuration, prompt, profile, artifact,
validation, capacity, and generation failures; and
- structured backend identity for capacity rejection.
These capabilities are sufficient for Weatherreporter to adopt PromptKit
without waiting for new upstream work.
## Responsibilities That Should Remain In Weatherreporter
The following concerns belong to Weatherreporter and should not move into
PromptKit:
- report definitions, valid periods, batches, and output naming;
- application prompt content and private report response schemas;
- deterministic weather facts, modules, and Recent Changes;
- curated `data_package` construction and persistence;
- generated-text domain validation and Markdown template rendering;
- managed artifact paths, atomic writes, metadata, and inspection commands;
- preparation, execution, raw-output, and failure-receipt schemas;
- CLI configuration loading and precedence;
- debug enablement, redaction, placement, sensitivity, and retention;
- distributor notification;
- batch continuation and any future retry policy; and
- application-level compatibility and migration policy.
## Suggested Upstream Sequence
For downstream adoption and any remaining upstream work, the useful order is:
1. Adopt the implemented executable preparation handles.
2. Consider eager source validation after evaluating whether the two exact
inspection APIs are sufficient.
3. Add structured generation errors.
4. Use structured capacity errors and consider semantic execution-target
fingerprints as lower-priority operational improvements.
The first item removes the material integration workaround. Prompt and profile
inspection improve fail-fast validation. The remaining items are optional
ergonomic and diagnostic improvements.
## Adoption Sequencing
Weatherreporter should not wait for the deferred wishlist items. The current
PromptKit interface is sufficient when Weatherreporter:
- embeds immutable prompt and schema assets;
- supplies immutable inline data-package bytes;
- constructs one engine per CLI invocation;
- prepares an execution, persists selected `Details`, and calls
`RunPrepared`; and
- keeps PromptKit behind a weatherreporter-owned adapter contract.
Prompt inspection, profile inspection, source validation, structured errors,
capacity details, and semantic fingerprints should not gate adoption.

191
engine.go
View File

@@ -63,9 +63,10 @@ var (
// ErrPromptRender identifies a failure to render prompt messages or the // ErrPromptRender identifies a failure to render prompt messages or the
// session ID from the resolved inputs and variables. // session ID from the resolved inputs and variables.
ErrPromptRender = errors.New("failed to render prompt") ErrPromptRender = errors.New("failed to render prompt")
// ErrCapacityExceeded identifies a Run rejected because the selected backend // ErrCapacityExceeded identifies a Run or RunPrepared rejected because the
// already admitted ConcurrencyLimit + QueueCapacity calls. It is not an // selected backend already admitted ConcurrencyLimit + QueueCapacity calls.
// invalid request, an LLM or provider rate-limit response, or ErrLLMGenerate. // A [CapacityError] reports the selected backend ID. It is not an invalid
// request, an LLM or provider rate-limit response, or ErrLLMGenerate.
ErrCapacityExceeded = errors.New("backend capacity exceeded") ErrCapacityExceeded = errors.New("backend capacity exceeded")
// ErrLLMGenerate identifies a model-client failure or a nil successful // ErrLLMGenerate identifies a model-client failure or a nil successful
// response. Errors returned by an injected LLMClient remain available // response. Errors returned by an injected LLMClient remain available
@@ -77,12 +78,15 @@ var (
ErrValidation = errors.New("failed to validate output") ErrValidation = errors.New("failed to validate output")
) )
// Engine prepares and runs Promptkit prompt requests. // Engine inspects prompts and profiles and prepares and runs Promptkit prompt
// requests.
// //
// An Engine is safe for concurrent calls to [Engine.Prepare] and [Engine.Run]. // An Engine is safe for concurrent calls to [Engine.InspectPrompt],
// Each Engine owns independent backend-capacity pools that coordinate Run // [Engine.InspectProfile], [Engine.Prepare], [Engine.PrepareExecution],
// admission and model generation. Injected collaborators may still be invoked // [Engine.Run], and [Engine.RunPrepared]. Each Engine owns independent
// concurrently across different backend pools or for unlimited backends. // backend-capacity pools that coordinate Run and RunPrepared admission and
// model generation. Injected collaborators may still be invoked concurrently
// across different backend pools or for unlimited backends.
type Engine struct { type Engine struct {
runner *usecase.Runner runner *usecase.Runner
} }
@@ -146,12 +150,14 @@ type engineOptions struct {
artifactSource bool artifactSource bool
} }
// WithLLMClient replaces the built-in model client used by [Engine.Run]. // WithLLMClient replaces the built-in model client used by [Engine.Run] and
// [Engine.RunPrepared].
// //
// A nil client makes NewEngine fail with ErrInvalidConfig. The Engine schedules // A nil client makes NewEngine fail with ErrInvalidConfig. The Engine schedules
// Generate calls according to the selected backend's capacity policy, but the // Generate calls according to the selected backend's capacity policy, but the
// client may still be called concurrently across different backend pools or for // client may still be called concurrently across different backend pools or for
// unlimited backends. The client is not used by [Engine.Prepare]. // unlimited backends. The client is not used by [Engine.Prepare] or
// [Engine.PrepareExecution].
func WithLLMClient(client LLMClient) Option { func WithLLMClient(client LLMClient) Option {
return optionFunc(func(options *engineOptions) error { return optionFunc(func(options *engineOptions) error {
if client == nil { if client == nil {
@@ -423,6 +429,93 @@ func fileSource(name string) (fs.FS, string, error) {
return os.DirFS(dir), filepath.ToSlash(base), nil return os.DirFS(dir), filepath.ToSlash(base), nil
} }
// InspectPrompt resolves one explicit prompt definition without selecting a
// profile or starting execution work.
//
// InspectPrompt requires a nonblank promptID. It passes nonblank promptID and
// promptVersion values unchanged to the engine's ordinary, case-sensitive
// prompt selection. An empty version succeeds only when that source has one
// selected ID; a nonempty version selects one exact ID/version pair. The
// configured prompt source is used without merging, fallback, or enumeration.
//
// A successful result proves that the selected definition and any referenced
// message content files were structurally loaded. Inputs are returned in
// definition order. DefaultProfileID is declared metadata only and is not
// resolved. OutputContract is the normalized declared contract, with a JSON
// Schema path when declared but without loading or compiling that schema.
// PromptHash is the same opaque equality value as PreparedRun.PromptHash for
// the selected definition and observed source state; its spelling, length,
// encoding, algorithm, and security properties are not contracts.
//
// This method does not return prompt bodies, templates, source paths, schemas,
// rendered messages, or execution settings. It does not resolve a profile or
// credential, read artifacts or schemas, render, validate, admit capacity,
// contact a provider, or generate model output. The returned PromptInspection
// and its input slice are caller-owned. Filesystem-backed inspection is a
// point-in-time lookup and does not freeze a definition for later execution.
//
// A nil Engine returns an error matching ErrInvalidConfig. A blank prompt ID
// matches ErrInvalidRequest. An absent exact ID or version matches
// ErrPromptNotFound and not ErrPromptLoad. Malformed, unreadable, duplicate,
// ambiguous, referenced-content, or hashing failures match ErrPromptLoad.
// Cancellation during lookup matches ErrPromptLoad while preserving the
// context error. InspectPrompt returns no partial result on error.
func (e *Engine) InspectPrompt(
ctx context.Context,
promptID string,
promptVersion string,
) (*PromptInspection, error) {
if e == nil || e.runner == nil {
return nil, fmt.Errorf("%w: engine is nil", ErrInvalidConfig)
}
inspection, err := e.runner.InspectPrompt(ctx, promptID, promptVersion)
if err != nil {
return nil, mapPublicError(err)
}
return fromDomainPromptInspection(inspection), nil
}
// InspectProfile resolves one explicit profile without selecting a prompt or
// starting execution work.
//
// InspectProfile trims surrounding whitespace from profileID and looks up the
// resulting nonblank ID exactly and case-sensitively through the engine's
// ordinary in-memory, configured-source, and built-in profile precedence. It
// applies framework defaults, the selected backend, and then the selected
// profile to EffectiveModelParams without a request override. BackendID is
// empty for an endpoint-only profile.
//
// APIKeyEnv in the returned target is an environment-variable name, never its
// value. APIKeyRequired instead reports a direct credential requirement and is
// mutually exclusive with a nonblank APIKeyEnv. InspectProfile neither derives
// an ID from a prompt default_profile nor checks credential availability, so an
// absent or blank named environment variable is not an error.
//
// The returned ProfileInspection and all nested mutable values are
// caller-owned. Filesystem-backed inspection is a point-in-time lookup and
// does not freeze the profile for a later execution. This method does not load
// a prompt, render, read artifacts or schemas, admit backend capacity, contact
// a provider, or generate model output.
//
// A nil Engine returns an error matching ErrInvalidConfig. A blank profile ID
// matches ErrInvalidRequest. An absent exact ID matches ErrProfileNotFound and
// not ErrProfileLoad. Malformed or unreadable profile data, an unknown backend,
// or an invalid resolved target matches ErrProfileLoad. Cancellation during
// profile loading matches ErrProfileLoad while preserving the context error.
// InspectProfile returns no partial result on error.
func (e *Engine) InspectProfile(ctx context.Context, profileID string) (*ProfileInspection, error) {
if e == nil || e.runner == nil {
return nil, fmt.Errorf("%w: engine is nil", ErrInvalidConfig)
}
inspection, err := e.runner.InspectProfile(ctx, profileID)
if err != nil {
return nil, mapPublicError(err)
}
return fromDomainProfileInspection(inspection), nil
}
// Prepare resolves and renders a prompt request without calling an LLM. // Prepare resolves and renders a prompt request without calling an LLM.
// //
// Prepare selects the prompt and profile, resolves any selected backend and // Prepare selects the prompt and profile, resolves any selected backend and
@@ -457,6 +550,38 @@ func (e *Engine) Prepare(ctx context.Context, req RunRequest) (*PreparedRun, err
return fromDomainPreparedRun(prepared), nil return fromDomainPreparedRun(prepared), nil
} }
// PrepareExecution completely prepares a prompt request without calling the
// configured LLMClient or reserving backend admission capacity.
//
// The returned opaque handle is bound to this Engine and permits one
// [Engine.RunPrepared] invocation. Preparation freezes the selected sources,
// rendered messages, effective settings, inputs, provider structured-output
// metadata, and validation resources needed by that invocation. The handle
// retains a direct RunRequest.APIKey only in private execution state;
// [PreparedExecution.Details] is credential-redacted.
//
// The context governs preparation only. Cancellation after this method
// returns does not invalidate the handle or propagate to RunPrepared.
// PrepareExecution returns the same error categories as [Engine.Prepare] and
// returns no handle on error. A nil Engine returns an error matching
// ErrInvalidConfig.
func (e *Engine) PrepareExecution(ctx context.Context, req RunRequest) (*PreparedExecution, error) {
if e == nil || e.runner == nil {
return nil, fmt.Errorf("%w: engine is nil", ErrInvalidConfig)
}
domainReq, err := toDomainRunRequest(req)
if err != nil {
return nil, fmt.Errorf("%w: %v", ErrInvalidRequest, err)
}
prepared, err := e.runner.PrepareExecution(ctx, domainReq)
if err != nil {
return nil, mapPublicError(err)
}
return &PreparedExecution{internal: prepared}, nil
}
// Run prepares a request, invokes the configured LLMClient, and validates the // Run prepares a request, invokes the configured LLMClient, and validates the
// generated output. // generated output.
// //
@@ -467,9 +592,10 @@ func (e *Engine) Prepare(ctx context.Context, req RunRequest) (*PreparedRun, err
// single-pass even when OutputContract.RepairAttempts is positive. // single-pass even when OutputContract.RepairAttempts is positive.
// //
// Run can return every error category documented by [Engine.Prepare], plus // Run can return every error category documented by [Engine.Prepare], plus
// ErrCapacityExceeded and ErrLLMGenerate. ErrCapacityExceeded identifies // ErrCapacityExceeded and ErrLLMGenerate. An engine admission rejection is
// rejection before artifacts, schemas, rendering, or model generation because // discoverable as [CapacityError] and still matches ErrCapacityExceeded. It
// the selected backend's admission capacity is full; it does not match // occurs before artifacts, schemas, rendering, or model generation because the
// selected backend's admission capacity is full; it does not match
// ErrInvalidRequest or ErrLLMGenerate. Errors from injected clients remain // ErrInvalidRequest or ErrLLMGenerate. Errors from injected clients remain
// available through errors.Is. Cancellation while waiting for model-generation // available through errors.Is. Cancellation while waiting for model-generation
// capacity matches both ErrLLMGenerate and the context error. Cancellation // capacity matches both ErrLLMGenerate and the context error. Cancellation
@@ -491,3 +617,42 @@ func (e *Engine) Run(ctx context.Context, req RunRequest) (*RunResult, error) {
} }
return fromDomainRunResult(result), nil return fromDomainRunResult(result), nil
} }
// RunPrepared atomically claims and executes a handle created by
// [Engine.PrepareExecution].
//
// A valid owning-Engine invocation consumes the handle's one attempt before
// credential revalidation, backend admission, generation, or validation.
// Cancellation, capacity rejection, generation failure, operational
// validation failure, and success all leave the handle unusable. A nil,
// zero-value, foreign-Engine, discarded, claimed, or used handle returns an
// error matching ErrInvalidRequest; a nil Engine returns ErrInvalidConfig and
// does not claim the handle.
//
// The supplied context governs this execution attempt independently of the
// preparation context. It covers credential revalidation, admission,
// generation, validation, and any internal repair. Result timing begins after
// the claim and excludes preparation and consumer-held delay.
//
// RunPrepared can return ErrInvalidRequest, ErrAPIKeyEnvMissing,
// ErrCapacityExceeded, ErrLLMGenerate, or ErrValidation as applicable while
// preserving documented collaborator and context identities. An engine
// admission rejection is discoverable as [CapacityError] and still matches
// ErrCapacityExceeded. A completed content-validation rejection is returned
// in RunResult, not as an operational error. An operational error returns no
// partial RunResult.
func (e *Engine) RunPrepared(ctx context.Context, prepared *PreparedExecution) (*RunResult, error) {
if e == nil || e.runner == nil {
return nil, fmt.Errorf("%w: engine is nil", ErrInvalidConfig)
}
var internal *usecase.PreparedExecution
if prepared != nil {
internal = prepared.internal
}
result, err := e.runner.RunPrepared(ctx, internal)
if err != nil {
return nil, mapPublicError(err)
}
return fromDomainRunResult(result), nil
}

View File

@@ -3,6 +3,7 @@ package promptkit
import ( import (
"errors" "errors"
"fmt" "fmt"
"strings"
"gitea.maximumdirect.net/eric/promptkit/internal/capacity" "gitea.maximumdirect.net/eric/promptkit/internal/capacity"
"gitea.maximumdirect.net/eric/promptkit/internal/profile" "gitea.maximumdirect.net/eric/promptkit/internal/profile"
@@ -14,6 +15,11 @@ func mapPublicError(err error) error {
if err == nil { if err == nil {
return nil return nil
} }
var internalCapacityError *usecase.CapacityError
if errors.As(err, &internalCapacityError) && internalCapacityError != nil &&
strings.TrimSpace(internalCapacityError.BackendID) != "" {
return &CapacityError{BackendID: internalCapacityError.BackendID}
}
publicErr := publicErrorFor(err) publicErr := publicErrorFor(err)
if publicErr == nil { if publicErr == nil {
return err return err

View File

@@ -20,3 +20,31 @@ func TestMapPublicErrorPreservesGenerationCancellation(t *testing.T) {
t.Fatalf("mapped error=%v, want context.Canceled", err) t.Fatalf("mapped error=%v, want context.Canceled", err)
} }
} }
func TestMapPublicErrorTranslatesCapacityError(t *testing.T) {
internalErr := &usecase.CapacityError{BackendID: "limited"}
err := mapPublicError(internalErr)
var publicErr *CapacityError
if !errors.As(err, &publicErr) || publicErr == nil {
t.Fatalf("mapped error=%v, want public CapacityError", err)
}
if publicErr.BackendID != "limited" {
t.Fatalf("mapped backend ID=%q, want limited", publicErr.BackendID)
}
if !errors.Is(err, ErrCapacityExceeded) {
t.Fatalf("mapped error=%v, want ErrCapacityExceeded", err)
}
if errors.Is(err, ErrInvalidRequest) || errors.Is(err, ErrLLMGenerate) {
t.Fatalf("mapped capacity error has an unrelated category: %v", err)
}
var leakedInternalErr *usecase.CapacityError
if errors.As(err, &leakedInternalErr) {
t.Fatalf("mapped error exposes internal CapacityError: %v", err)
}
internalErr.BackendID = "changed"
if publicErr.BackendID != "limited" {
t.Fatalf("mapped backend ID changed with source error: %q", publicErr.BackendID)
}
}

View File

@@ -124,6 +124,11 @@ func TestManagerAdmissionHonorsContextAndUnlimitedBackends(t *testing.T) {
if err != nil { if err != nil {
t.Fatalf("construct manager: %v", err) t.Fatalf("construct manager: %v", err)
} }
release, err := manager.Admit(context.Background(), "limited")
if err != nil {
t.Fatalf("fill limited pool: %v", err)
}
defer release()
ctx, cancel := context.WithCancel(context.Background()) ctx, cancel := context.WithCancel(context.Background())
cancel() cancel()

View File

@@ -145,6 +145,16 @@ type PromptDefinition struct {
Validation OutputContract `yaml:"validation"` Validation OutputContract `yaml:"validation"`
} }
// PromptInspection is the resolved result of exact prompt inspection.
type PromptInspection struct {
PromptID string
PromptVersion string
PromptHash string
DefaultProfileID string
Inputs []PromptInput
OutputContract OutputContract
}
// PromptInput describes one named input expected by a prompt definition. // PromptInput describes one named input expected by a prompt definition.
type PromptInput struct { type PromptInput struct {
Name string `yaml:"name"` Name string `yaml:"name"`
@@ -236,6 +246,13 @@ type ExecutionTarget struct {
ExtraParams map[string]any `yaml:"extra_params" json:"extra_params"` ExtraParams map[string]any `yaml:"extra_params" json:"extra_params"`
} }
// ProfileInspection is the resolved result of exact profile inspection.
type ProfileInspection struct {
ProfileID string
EffectiveModelParams ExecutionTarget
APIKeyRequired bool
}
// OutputContract defines the requirements for the output artifact. // OutputContract defines the requirements for the output artifact.
type OutputContract struct { type OutputContract struct {
Format OutputFormat `yaml:"format"` Format OutputFormat `yaml:"format"`

View File

@@ -1,5 +1,5 @@
// Package jsonvalue validates and defensively copies JSON-compatible value // Package jsonvalue validates and defensively copies JSON-compatible value
// trees used by public configuration and request boundaries. // trees used by configuration, request, and prepared-state boundaries.
package jsonvalue package jsonvalue
import ( import (
@@ -18,13 +18,19 @@ type visit struct {
ptr uintptr ptr uintptr
} }
// Copy validates and deeply copies a JSON-compatible value while preserving
// compatible concrete map, slice, array, scalar, and number types.
func Copy(src any) (any, error) {
return copyValue(reflect.ValueOf(src), "value", make(map[visit]struct{}), true)
}
// CopyMap validates and deeply copies an extra-parameter map while preserving // CopyMap validates and deeply copies an extra-parameter map while preserving
// compatible concrete map, slice, array, scalar, and number types. // compatible concrete map, slice, array, scalar, and number types.
func CopyMap(src map[string]any) (map[string]any, error) { func CopyMap(src map[string]any) (map[string]any, error) {
if src == nil { if src == nil {
return nil, nil return nil, nil
} }
copied, err := copyValue(reflect.ValueOf(src), "extra_params", make(map[visit]struct{})) copied, err := copyValue(reflect.ValueOf(src), "extra_params", make(map[visit]struct{}), false)
if err != nil { if err != nil {
return nil, err return nil, err
} }
@@ -35,7 +41,12 @@ func CopyMap(src map[string]any) (map[string]any, error) {
return out, nil return out, nil
} }
func copyValue(value reflect.Value, path string, seen map[visit]struct{}) (any, error) { func copyValue(
value reflect.Value,
path string,
seen map[visit]struct{},
allowEmptyMapKeys bool,
) (any, error) {
if !value.IsValid() { if !value.IsValid() {
return nil, nil return nil, nil
} }
@@ -43,7 +54,7 @@ func copyValue(value reflect.Value, path string, seen map[visit]struct{}) (any,
if value.IsNil() { if value.IsNil() {
return nil, nil return nil, nil
} }
return copyValue(value.Elem(), path, seen) return copyValue(value.Elem(), path, seen, allowEmptyMapKeys)
} }
if !value.CanInterface() { if !value.CanInterface() {
return nil, fmt.Errorf("%s: value cannot be copied", path) return nil, fmt.Errorf("%s: value cannot be copied", path)
@@ -88,22 +99,27 @@ func copyValue(value reflect.Value, path string, seen map[visit]struct{}) (any,
} }
seen[current] = struct{}{} seen[current] = struct{}{}
defer delete(seen, current) defer delete(seen, current)
return copyValue(value.Elem(), path, seen) return copyValue(value.Elem(), path, seen, allowEmptyMapKeys)
case reflect.Map: case reflect.Map:
return copyMapValue(value, path, seen) return copyMapValue(value, path, seen, allowEmptyMapKeys)
case reflect.Slice: case reflect.Slice:
if value.IsNil() { if value.IsNil() {
return nil, nil return nil, nil
} }
return copySequenceValue(value, path, seen) return copySequenceValue(value, path, seen, allowEmptyMapKeys)
case reflect.Array: case reflect.Array:
return copySequenceValue(value, path, seen) return copySequenceValue(value, path, seen, allowEmptyMapKeys)
default: default:
return nil, fmt.Errorf("%s: unsupported JSON value type %s", path, value.Type()) return nil, fmt.Errorf("%s: unsupported JSON value type %s", path, value.Type())
} }
} }
func copyMapValue(value reflect.Value, path string, seen map[visit]struct{}) (any, error) { func copyMapValue(
value reflect.Value,
path string,
seen map[visit]struct{},
allowEmptyMapKeys bool,
) (any, error) {
if value.IsNil() { if value.IsNil() {
return nil, nil return nil, nil
} }
@@ -133,10 +149,10 @@ func copyMapValue(value reflect.Value, path string, seen map[visit]struct{}) (an
elementType := value.Type().Elem() elementType := value.Type().Elem()
for _, key := range keys { for _, key := range keys {
name := key.String() name := key.String()
if name == "" { if name == "" && !allowEmptyMapKeys {
return nil, fmt.Errorf("%s: map key must not be empty", path) return nil, fmt.Errorf("%s: map key must not be empty", path)
} }
copied, err := copyValue(value.MapIndex(key), path+"."+name, seen) copied, err := copyValue(value.MapIndex(key), path+"."+name, seen, allowEmptyMapKeys)
if err != nil { if err != nil {
return nil, err return nil, err
} }
@@ -171,7 +187,12 @@ func copyMapValue(value reflect.Value, path string, seen map[visit]struct{}) (an
return out, nil return out, nil
} }
func copySequenceValue(value reflect.Value, path string, seen map[visit]struct{}) (any, error) { func copySequenceValue(
value reflect.Value,
path string,
seen map[visit]struct{},
allowEmptyMapKeys bool,
) (any, error) {
var current visit var current visit
if value.Kind() == reflect.Slice { if value.Kind() == reflect.Slice {
current = visit{typ: value.Type(), ptr: value.Pointer()} current = visit{typ: value.Type(), ptr: value.Pointer()}
@@ -186,7 +207,12 @@ func copySequenceValue(value reflect.Value, path string, seen map[visit]struct{}
preserveType := true preserveType := true
elementType := value.Type().Elem() elementType := value.Type().Elem()
for i := 0; i < value.Len(); i++ { for i := 0; i < value.Len(); i++ {
copied, err := copyValue(value.Index(i), fmt.Sprintf("%s[%d]", path, i), seen) copied, err := copyValue(
value.Index(i),
fmt.Sprintf("%s[%d]", path, i),
seen,
allowEmptyMapKeys,
)
if err != nil { if err != nil {
return nil, err return nil, err
} }

View File

@@ -44,6 +44,21 @@ func TestCopyMapPreservesTypesAndIsolatesMutations(t *testing.T) {
} }
} }
func TestCopyAllowsEmptyObjectKeysAndIsolatesMutations(t *testing.T) {
nested := map[string]any{"": []any{"original"}}
copiedValue, err := jsonvalue.Copy(nested)
if err != nil {
t.Fatalf("copy value: %v", err)
}
nested[""].([]any)[0] = "changed"
copied := copiedValue.(map[string]any)
if got := copied[""].([]any)[0]; got != "original" {
t.Fatalf("copied value was not isolated: %v", got)
}
}
func TestCopyMapRejectsInvalidValues(t *testing.T) { func TestCopyMapRejectsInvalidValues(t *testing.T) {
cyclicMap := map[string]any{} cyclicMap := map[string]any{}
cyclicMap["self"] = cyclicMap cyclicMap["self"] = cyclicMap

View File

@@ -0,0 +1,24 @@
package usecase
import (
"fmt"
"strings"
"gitea.maximumdirect.net/eric/promptkit/internal/capacity"
)
// CapacityError identifies bounded admission rejected for one selected backend.
type CapacityError struct {
BackendID string
}
func (e *CapacityError) Error() string {
if e == nil || strings.TrimSpace(e.BackendID) == "" {
return capacity.ErrCapacityExceeded.Error()
}
return fmt.Sprintf("backend %q admission: %v", e.BackendID, capacity.ErrCapacityExceeded)
}
func (e *CapacityError) Unwrap() error {
return capacity.ErrCapacityExceeded
}

View File

@@ -0,0 +1,298 @@
package usecase
import (
"context"
"fmt"
"sync"
"time"
"gitea.maximumdirect.net/eric/promptkit/internal/domain"
"gitea.maximumdirect.net/eric/promptkit/internal/jsonvalue"
"gitea.maximumdirect.net/eric/promptkit/internal/validate"
)
type preparedExecutionState uint8
const (
preparedExecutionReady preparedExecutionState = iota
preparedExecutionClaimed
preparedExecutionDiscarded
)
// PreparedExecution owns one frozen, single-use runner execution.
type PreparedExecution struct {
owner *Runner
mu sync.Mutex
state preparedExecutionState
details *domain.PreparedRun
payload *preparedExecutionPayload
}
type preparedExecutionPayload struct {
prepared *domain.PreparedRun
validation validate.PreparedValidation
directKey string
}
// PrepareExecution completes preparation without generation or admission and
// returns a runner-bound, single-use execution.
func (r *Runner) PrepareExecution(ctx context.Context, req domain.RunRequest) (*PreparedExecution, error) {
state, err := r.resolvePreparation(ctx, req, time.Now().UTC())
if err != nil {
return nil, err
}
validationPlan, err := r.prepareValidation(ctx, state.effectiveContract)
if err != nil {
return nil, err
}
structuredOutput, err := r.structuredOutputFromValidationPlan(
state.definition,
state.effectiveContract,
validationPlan,
)
if err != nil {
return nil, err
}
prepared, err := r.completePreparationWithStructuredOutput(ctx, req, state, structuredOutput)
if err != nil {
return nil, err
}
executionSnapshot, err := clonePreparedRun(prepared)
if err != nil {
return nil, fmt.Errorf("%w: failed to copy prepared execution: %v", ErrInvalidRequest, err)
}
details, err := clonePreparedRun(executionSnapshot)
if err != nil {
return nil, fmt.Errorf("%w: failed to copy prepared execution details: %v", ErrInvalidRequest, err)
}
return &PreparedExecution{
owner: r,
state: preparedExecutionReady,
details: details,
payload: &preparedExecutionPayload{
prepared: executionSnapshot,
validation: validationPlan,
directKey: state.effectiveModel.APIKey,
},
}, nil
}
func (r *Runner) prepareValidation(
ctx context.Context,
contract domain.OutputContract,
) (validate.PreparedValidation, error) {
if r.validator == nil {
return noOpPreparedValidation{contract: contract}, nil
}
preparer, ok := r.validator.(validate.ValidationPreparer)
if !ok {
return nil, fmt.Errorf("%w: validator does not support prepared validation", ErrValidation)
}
plan, err := preparer.PrepareValidation(ctx, contract)
if err != nil {
return nil, fmt.Errorf("%w: %w", ErrValidation, err)
}
if plan == nil {
return nil, fmt.Errorf("%w: validator returned nil prepared validation", ErrValidation)
}
return plan, nil
}
func (r *Runner) structuredOutputFromValidationPlan(
def *domain.PromptDefinition,
contract domain.OutputContract,
plan validate.PreparedValidation,
) (*domain.StructuredOutputSpec, error) {
if contract.ValidationMode != domain.ValidationJSONSchema {
return nil, nil
}
schemaDocument := plan.SchemaDocument()
if schemaDocument == nil {
if r.validator == nil {
return nil, nil
}
return nil, fmt.Errorf("%w: prepared json_schema validation has no schema document", ErrValidation)
}
return structuredOutputSpec(def, schemaDocument), nil
}
// Details returns a fresh credential-redacted copy of the prepared run.
func (p *PreparedExecution) Details() *domain.PreparedRun {
if p == nil {
return nil
}
p.mu.Lock()
detailsSnapshot := p.details
p.mu.Unlock()
details, err := clonePreparedRun(detailsSnapshot)
if err != nil {
return nil
}
return details
}
// Discard invalidates an unclaimed execution and drops its private payload.
func (p *PreparedExecution) Discard() {
if p == nil {
return
}
p.mu.Lock()
if p.state != preparedExecutionReady {
p.mu.Unlock()
return
}
p.state = preparedExecutionDiscarded
payload := p.payload
p.payload = nil
p.mu.Unlock()
payload.clear()
}
// RunPrepared claims and executes one prepared execution owned by this runner.
func (r *Runner) RunPrepared(ctx context.Context, prepared *PreparedExecution) (*domain.RunResult, error) {
payload, err := prepared.claim(r)
if err != nil {
return nil, err
}
defer payload.clear()
runID, err := newRunID()
if err != nil {
return nil, fmt.Errorf("failed to create run id: %w", err)
}
start := time.Now().UTC()
target := payload.prepared.EffectiveModelParams
if err := validateAPIKey(target.APIKeyEnv, payload.directKey, target.APIKeyRequired); err != nil {
return nil, fmt.Errorf("%w: %w", ErrInvalidRequest, err)
}
release, err := r.admitRun(ctx, target.BackendID)
if err != nil {
return nil, err
}
defer release()
return r.executePreparedRun(ctx, payload.prepared, payload.directKey, runID, start, func(
ctx context.Context,
artifact *domain.Artifact,
attemptsUsed int,
) (domain.ValidationResult, error) {
result, validationErr := payload.validation.Validate(ctx, artifact)
if validationErr != nil {
return domain.ValidationResult{}, validationErr
}
result.RepairAttempts = attemptsUsed
return result, nil
})
}
func (p *PreparedExecution) claim(owner *Runner) (*preparedExecutionPayload, error) {
if p == nil || owner == nil || p.owner != owner {
return nil, fmt.Errorf("%w: prepared execution does not belong to this runner", ErrInvalidRequest)
}
p.mu.Lock()
defer p.mu.Unlock()
if p.state != preparedExecutionReady || p.payload == nil {
return nil, fmt.Errorf("%w: prepared execution is not ready", ErrInvalidRequest)
}
p.state = preparedExecutionClaimed
payload := p.payload
p.payload = nil
return payload, nil
}
func (p *preparedExecutionPayload) clear() {
if p == nil {
return
}
if p.prepared != nil {
p.prepared.EffectiveModelParams.APIKey = ""
}
p.prepared = nil
p.validation = nil
p.directKey = ""
}
type noOpPreparedValidation struct {
contract domain.OutputContract
}
func (p noOpPreparedValidation) Validate(
ctx context.Context,
_ *domain.Artifact,
) (domain.ValidationResult, error) {
if err := ctx.Err(); err != nil {
return domain.ValidationResult{}, err
}
return domain.ValidationResult{
Status: domain.ValidationSkipped,
Mode: p.contract.ValidationMode,
SchemaPath: p.contract.SchemaPath,
RepairAttempts: p.contract.RepairAttempts,
IsValid: true,
}, nil
}
func (noOpPreparedValidation) SchemaDocument() any {
return nil
}
func clonePreparedRun(source *domain.PreparedRun) (*domain.PreparedRun, error) {
if source == nil {
return nil, nil
}
copied := *source
extraParams, err := jsonvalue.CopyMap(source.EffectiveModelParams.ExtraParams)
if err != nil {
return nil, err
}
copied.EffectiveModelParams.ExtraParams = extraParams
if source.InputHashes != nil {
copied.InputHashes = make(map[string]string, len(source.InputHashes))
for name, hash := range source.InputHashes {
copied.InputHashes[name] = hash
}
}
if source.Messages != nil {
copied.Messages = make([]domain.RenderedMessage, len(source.Messages))
for i, message := range source.Messages {
copied.Messages[i] = message
if message.CacheControl != nil {
cacheControl := *message.CacheControl
copied.Messages[i].CacheControl = &cacheControl
}
}
}
if source.StructuredOutput != nil {
structuredOutput := *source.StructuredOutput
copied.StructuredOutput = &structuredOutput
if source.StructuredOutput.JSONSchema != nil {
jsonSchema := *source.StructuredOutput.JSONSchema
copied.StructuredOutput.JSONSchema = &jsonSchema
schema, copyErr := cloneJSONValue(source.StructuredOutput.JSONSchema.Schema)
if copyErr != nil {
return nil, copyErr
}
copied.StructuredOutput.JSONSchema.Schema = schema
}
}
return &copied, nil
}
func cloneJSONValue(source any) (any, error) {
return jsonvalue.Copy(source)
}

View File

@@ -0,0 +1,510 @@
package usecase
import (
"context"
"errors"
"os"
"reflect"
"testing"
"gitea.maximumdirect.net/eric/promptkit/internal/domain"
"gitea.maximumdirect.net/eric/promptkit/internal/validate"
)
type recordingPreparedValidation struct {
contract domain.OutputContract
schemaDocument any
results []domain.ValidationResult
errs []error
artifacts []string
}
func (p *recordingPreparedValidation) Validate(
_ context.Context,
artifact *domain.Artifact,
) (domain.ValidationResult, error) {
p.artifacts = append(p.artifacts, string(artifact.Body))
index := len(p.artifacts) - 1
if index < len(p.errs) && p.errs[index] != nil {
return domain.ValidationResult{}, p.errs[index]
}
if len(p.results) == 0 {
return domain.ValidationResult{
Status: domain.ValidationPassed,
Mode: p.contract.ValidationMode,
IsValid: true,
}, nil
}
if index >= len(p.results) {
index = len(p.results) - 1
}
return p.results[index], nil
}
func (p *recordingPreparedValidation) SchemaDocument() any {
return p.schemaDocument
}
type recordingValidationPreparer struct {
plan *recordingPreparedValidation
prepareErr error
prepareCalls int
directValidateCalls int
}
func (v *recordingValidationPreparer) Validate(
context.Context,
*domain.Artifact,
domain.OutputContract,
) (domain.ValidationResult, error) {
v.directValidateCalls++
return domain.ValidationResult{}, errors.New("live validation must not be used")
}
func (v *recordingValidationPreparer) PrepareValidation(
_ context.Context,
contract domain.OutputContract,
) (validate.PreparedValidation, error) {
v.prepareCalls++
if v.prepareErr != nil {
return nil, v.prepareErr
}
v.plan.contract = contract
return v.plan, nil
}
func TestRunnerPrepareExecutionCompletesWithoutAdmissionOrGeneration(t *testing.T) {
schemaDocument := map[string]any{
"type": "object",
"properties": map[string]any{
"value": map[string]any{"type": "string"},
"": map[string]any{"type": "boolean"},
},
}
def := promptDef(domain.FormatJSON, domain.ValidationJSONSchema, 0)
def.Validation.SchemaPath = "schema.json"
reader := defaultArtifactReader()
renderer := &fakeRenderer{rendered: &domain.RenderedPrompt{
SessionID: "prepared-session",
Messages: []domain.RenderedMessage{{
Role: "user",
Content: "original message",
}},
}}
llmClient := &fakeLLM{forbid: true}
validator := &recordingValidationPreparer{
plan: &recordingPreparedValidation{schemaDocument: schemaDocument},
}
admitter := &fakeRunAdmitter{}
profile := defaultExecutionProfile()
profile.ExtraParams = map[string]any{
"metadata": map[string]any{"source": "original"},
}
runner := NewRunner(
&fakePromptRepo{def: def},
&fakeExecutionProfileRepo{profiles: map[string]*domain.ExecutionProfile{"exec": profile}},
nil,
reader,
renderer,
llmClient,
validator,
admitter,
)
prepared, err := runner.PrepareExecution(context.Background(), domain.RunRequest{
PromptID: "p",
ProfileID: "exec",
APIKey: "direct-test-key",
Inputs: singleInputRef(),
})
if err != nil {
t.Fatalf("prepare execution: %v", err)
}
defer prepared.Discard()
if validator.prepareCalls != 1 || validator.directValidateCalls != 0 {
t.Fatalf(
"validation calls=(prepare=%d direct=%d), want (1, 0)",
validator.prepareCalls,
validator.directValidateCalls,
)
}
if reader.calls != 1 || renderer.calls != 1 {
t.Fatalf("completion calls=(artifact=%d render=%d), want (1, 1)", reader.calls, renderer.calls)
}
if len(admitter.backendIDs) != 0 || llmClient.calls != 0 {
t.Fatalf("prepare invoked execution collaborators: admission=%v generation=%d", admitter.backendIDs, llmClient.calls)
}
first := prepared.Details()
if first == nil {
t.Fatal("prepared details are nil")
}
if first.EffectiveModelParams.APIKey != "" {
t.Fatal("prepared details retained the direct API key")
}
if first.StructuredOutput == nil ||
first.StructuredOutput.JSONSchema == nil ||
!reflect.DeepEqual(first.StructuredOutput.JSONSchema.Schema, schemaDocument) {
t.Fatalf("prepared details have unexpected structured output: %#v", first.StructuredOutput)
}
first.Messages[0].Content = "caller mutation"
first.InputHashes["input"] = "caller mutation"
first.EffectiveModelParams.ExtraParams["metadata"].(map[string]any)["source"] = "caller mutation"
first.StructuredOutput.JSONSchema.Schema.(map[string]any)["type"] = "string"
renderer.rendered.Messages[0].Content = "source mutation"
second := prepared.Details()
if second.Messages[0].Content != "original message" ||
second.InputHashes["input"] == "caller mutation" ||
second.EffectiveModelParams.ExtraParams["metadata"].(map[string]any)["source"] != "original" ||
second.StructuredOutput.JSONSchema.Schema.(map[string]any)["type"] != "object" {
t.Fatalf("details did not preserve an independent snapshot: %#v", second)
}
}
func TestRunnerRunPreparedRechecksEnvironmentCredentialBeforeAdmission(t *testing.T) {
const environmentName = "PROMPTKIT_PREPARED_EXECUTION_TEST_KEY"
t.Setenv(environmentName, "available-during-preparation")
profile := defaultExecutionProfile()
profile.APIKeyEnv = environmentName
validator := &recordingValidationPreparer{plan: &recordingPreparedValidation{}}
admitter := &fakeRunAdmitter{}
llmClient := &fakeLLM{resp: &domain.GenerateResponse{Content: "unexpected"}}
runner := NewRunner(
&fakePromptRepo{def: promptDef(domain.FormatText, domain.ValidationNone, 0)},
&fakeExecutionProfileRepo{profiles: map[string]*domain.ExecutionProfile{"exec": profile}},
nil,
defaultArtifactReader(),
defaultRenderer(),
llmClient,
validator,
admitter,
)
prepared, err := runner.PrepareExecution(context.Background(), domain.RunRequest{
PromptID: "p",
ProfileID: "exec",
Inputs: singleInputRef(),
})
if err != nil {
t.Fatalf("prepare execution: %v", err)
}
if err := os.Unsetenv(environmentName); err != nil {
t.Fatalf("unset credential environment: %v", err)
}
result, err := runner.RunPrepared(context.Background(), prepared)
if result != nil {
t.Fatalf("credential failure returned partial result: %+v", result)
}
if !errors.Is(err, ErrInvalidRequest) || !errors.Is(err, ErrAPIKeyEnvMissing) {
t.Fatalf("credential error identities are missing: %v", err)
}
if len(admitter.backendIDs) != 0 || llmClient.calls != 0 {
t.Fatalf("credential failure reached admission or generation: admission=%v generation=%d", admitter.backendIDs, llmClient.calls)
}
if _, err := runner.RunPrepared(context.Background(), prepared); !errors.Is(err, ErrInvalidRequest) {
t.Fatalf("credential failure did not consume execution: %v", err)
}
}
func TestRunnerRunPreparedKeepsDirectCredentialOutOfMetadata(t *testing.T) {
const directKey = "direct-prepared-test-key"
profile := defaultExecutionProfile()
profile.APIKeyRequired = true
client := &fakeLLM{resp: &domain.GenerateResponse{Content: "ok"}}
runner := NewRunner(
&fakePromptRepo{def: promptDef(domain.FormatText, domain.ValidationNone, 0)},
&fakeExecutionProfileRepo{profiles: map[string]*domain.ExecutionProfile{"exec": profile}},
nil,
defaultArtifactReader(),
defaultRenderer(),
client,
&recordingValidationPreparer{plan: &recordingPreparedValidation{}},
nil,
)
prepared, err := runner.PrepareExecution(context.Background(), domain.RunRequest{
PromptID: "p",
ProfileID: "exec",
APIKey: directKey,
Inputs: singleInputRef(),
})
if err != nil {
t.Fatalf("prepare execution: %v", err)
}
if details := prepared.Details(); details.EffectiveModelParams.APIKey != "" {
t.Fatal("prepared details retained direct credential")
}
result, err := runner.RunPrepared(context.Background(), prepared)
if err != nil {
t.Fatalf("run prepared: %v", err)
}
if client.lastReq.Target.APIKey != directKey {
t.Fatal("generation did not receive direct credential")
}
if result.EffectiveModelParams.APIKey != "" {
t.Fatal("run result retained direct credential")
}
}
func TestRunnerRunPreparedUsesFrozenValidationForInitialAndRepairOutputs(t *testing.T) {
validator := &recordingValidationPreparer{
plan: &recordingPreparedValidation{
results: []domain.ValidationResult{
{
Status: domain.ValidationFailed,
Mode: domain.ValidationJSON,
Errors: []string{"invalid"},
IsValid: false,
},
{
Status: domain.ValidationPassed,
Mode: domain.ValidationJSON,
IsValid: true,
},
},
},
}
repairer := &fakeRepairer{
responses: []*domain.GenerateResponse{{Content: `{"repaired":true}`}},
}
admitter := &fakeRunAdmitter{}
reader := defaultArtifactReader()
renderer := defaultRenderer()
runner := NewRunnerWithRepairer(
&fakePromptRepo{def: promptDef(domain.FormatJSON, domain.ValidationJSON, 1)},
&fakeExecutionProfileRepo{profiles: map[string]*domain.ExecutionProfile{"exec": defaultExecutionProfile()}},
nil,
reader,
renderer,
&fakeLLM{resp: &domain.GenerateResponse{Content: `{"broken":true}`}},
validator,
repairer,
admitter,
)
prepared, err := runner.PrepareExecution(context.Background(), domain.RunRequest{
PromptID: "p",
ProfileID: "exec",
Inputs: singleInputRef(),
})
if err != nil {
t.Fatalf("prepare execution: %v", err)
}
result, err := runner.RunPrepared(context.Background(), prepared)
if err != nil {
t.Fatalf("run prepared: %v", err)
}
if !reflect.DeepEqual(validator.plan.artifacts, []string{`{"broken":true}`, `{"repaired":true}`}) {
t.Fatalf("prepared validation artifacts=%#v", validator.plan.artifacts)
}
if validator.directValidateCalls != 0 || repairer.calls != 1 {
t.Fatalf("validation/repair calls=(direct=%d repair=%d), want (0, 1)", validator.directValidateCalls, repairer.calls)
}
if result.Validation.Status != domain.ValidationPassed || result.Validation.RepairAttempts != 1 {
t.Fatalf("unexpected repaired validation result: %+v", result.Validation)
}
if admitter.releaseCalls != 1 {
t.Fatalf("admission releases=%d, want 1", admitter.releaseCalls)
}
if reader.calls != 1 || renderer.calls != 1 {
t.Fatalf("execution reopened preparation sources: artifact=%d render=%d", reader.calls, renderer.calls)
}
}
func TestRunnerRunPreparedReleasesAdmissionAcrossExecutionErrors(t *testing.T) {
generationFailure := errors.New("generation failed")
validationFailure := errors.New("validation failed")
repairFailure := errors.New("repair failed")
tests := []struct {
name string
generationErr error
validation *recordingPreparedValidation
repairer *fakeRepairer
wantError error
}{
{
name: "generation failure",
generationErr: generationFailure,
validation: &recordingPreparedValidation{},
wantError: ErrLLMGenerate,
},
{
name: "validation failure",
validation: &recordingPreparedValidation{
errs: []error{validationFailure},
},
wantError: ErrValidation,
},
{
name: "repair failure",
validation: &recordingPreparedValidation{
results: []domain.ValidationResult{{
Status: domain.ValidationFailed,
Mode: domain.ValidationJSON,
Errors: []string{"invalid"},
IsValid: false,
}},
},
repairer: &fakeRepairer{err: repairFailure},
wantError: ErrValidation,
},
}
for _, test := range tests {
t.Run(test.name, func(t *testing.T) {
def := promptDef(domain.FormatJSON, domain.ValidationJSON, 1)
if test.repairer == nil {
def.Validation.RepairAttempts = 0
}
validator := &recordingValidationPreparer{plan: test.validation}
admitter := &fakeRunAdmitter{}
runner := NewRunnerWithRepairer(
&fakePromptRepo{def: def},
&fakeExecutionProfileRepo{profiles: map[string]*domain.ExecutionProfile{"exec": defaultExecutionProfile()}},
nil,
defaultArtifactReader(),
defaultRenderer(),
&fakeLLM{
resp: &domain.GenerateResponse{Content: `{"value":true}`},
err: test.generationErr,
},
validator,
test.repairer,
admitter,
)
prepared, err := runner.PrepareExecution(context.Background(), domain.RunRequest{
PromptID: "p",
ProfileID: "exec",
Inputs: singleInputRef(),
})
if err != nil {
t.Fatalf("prepare execution: %v", err)
}
result, err := runner.RunPrepared(context.Background(), prepared)
if result != nil || !errors.Is(err, test.wantError) {
t.Fatalf("run prepared=(%+v, %v), want %v", result, err, test.wantError)
}
if len(admitter.backendIDs) != 1 || admitter.releaseCalls != 1 {
t.Fatalf(
"admission calls=%#v releases=%d, want one each",
admitter.backendIDs,
admitter.releaseCalls,
)
}
})
}
}
func TestRunnerPreparedExecutionOwnershipUseAndDiscard(t *testing.T) {
newRunner := func() *Runner {
return NewRunner(
&fakePromptRepo{def: promptDef(domain.FormatText, domain.ValidationNone, 0)},
&fakeExecutionProfileRepo{profiles: map[string]*domain.ExecutionProfile{"exec": defaultExecutionProfile()}},
nil,
defaultArtifactReader(),
defaultRenderer(),
&fakeLLM{resp: &domain.GenerateResponse{Content: "ok"}},
&recordingValidationPreparer{plan: &recordingPreparedValidation{}},
nil,
)
}
request := domain.RunRequest{
PromptID: "p",
ProfileID: "exec",
Inputs: singleInputRef(),
}
owner := newRunner()
prepared, err := owner.PrepareExecution(context.Background(), request)
if err != nil {
t.Fatalf("prepare execution: %v", err)
}
if _, err := newRunner().RunPrepared(context.Background(), prepared); !errors.Is(err, ErrInvalidRequest) {
t.Fatalf("foreign runner error=%v, want ErrInvalidRequest", err)
}
if _, err := owner.RunPrepared(context.Background(), prepared); err != nil {
t.Fatalf("owner run prepared: %v", err)
}
if _, err := owner.RunPrepared(context.Background(), prepared); !errors.Is(err, ErrInvalidRequest) {
t.Fatalf("second owner run error=%v, want ErrInvalidRequest", err)
}
if prepared.Details() == nil {
t.Fatal("details unavailable after execution")
}
discarded, err := owner.PrepareExecution(context.Background(), request)
if err != nil {
t.Fatalf("prepare discarded execution: %v", err)
}
discarded.Discard()
discarded.Discard()
if _, err := owner.RunPrepared(context.Background(), discarded); !errors.Is(err, ErrInvalidRequest) {
t.Fatalf("discarded execution error=%v, want ErrInvalidRequest", err)
}
if discarded.Details() == nil {
t.Fatal("details unavailable after discard")
}
}
func TestRunnerPreparedExecutionWithoutValidatorSkipsValidation(t *testing.T) {
runner := NewRunner(
&fakePromptRepo{def: promptDef(domain.FormatJSON, domain.ValidationJSON, 0)},
&fakeExecutionProfileRepo{profiles: map[string]*domain.ExecutionProfile{"exec": defaultExecutionProfile()}},
nil,
defaultArtifactReader(),
defaultRenderer(),
&fakeLLM{resp: &domain.GenerateResponse{Content: `{}`}},
nil,
nil,
)
prepared, err := runner.PrepareExecution(context.Background(), domain.RunRequest{
PromptID: "p",
ProfileID: "exec",
Inputs: singleInputRef(),
})
if err != nil {
t.Fatalf("prepare execution: %v", err)
}
result, err := runner.RunPrepared(context.Background(), prepared)
if err != nil {
t.Fatalf("run prepared: %v", err)
}
if result.Validation.Status != domain.ValidationSkipped || !result.Validation.IsValid {
t.Fatalf("unexpected no-validator result: %+v", result.Validation)
}
}
func TestRunnerPrepareExecutionRequiresValidationPreparer(t *testing.T) {
reader := defaultArtifactReader()
runner := NewRunner(
&fakePromptRepo{def: promptDef(domain.FormatText, domain.ValidationBasic, 0)},
&fakeExecutionProfileRepo{profiles: map[string]*domain.ExecutionProfile{"exec": defaultExecutionProfile()}},
nil,
reader,
defaultRenderer(),
&fakeLLM{forbid: true},
&fakeValidator{},
nil,
)
prepared, err := runner.PrepareExecution(context.Background(), domain.RunRequest{
PromptID: "p",
ProfileID: "exec",
Inputs: singleInputRef(),
})
if prepared != nil || !errors.Is(err, ErrValidation) {
t.Fatalf("prepare execution=(%+v, %v), want ErrValidation", prepared, err)
}
if reader.calls != 0 {
t.Fatalf("unsupported validator allowed completion, artifact calls=%d", reader.calls)
}
}

View File

@@ -0,0 +1,103 @@
package usecase
import (
"context"
"errors"
"fmt"
"strings"
"gitea.maximumdirect.net/eric/promptkit/internal/domain"
)
type resolvedProfileSelection struct {
id string
profile *domain.ExecutionProfile
backend *domain.Backend
}
func (r *Runner) resolveProfileSelection(
ctx context.Context,
profileID string,
) (*resolvedProfileSelection, error) {
normalizedID := strings.TrimSpace(profileID)
if normalizedID == "" {
return nil, fmt.Errorf("%w: profile id is required", ErrInvalidRequest)
}
if r == nil || r.profiles == nil {
return nil, fmt.Errorf("%w: profile repository is not configured", ErrProfileLoad)
}
selectedProfile, err := r.profiles.GetProfile(ctx, normalizedID)
if err != nil {
return nil, fmt.Errorf("%w: %w", ErrProfileLoad, err)
}
if selectedProfile == nil {
return nil, fmt.Errorf("%w: profile repository returned nil profile", ErrProfileLoad)
}
profileValue := *selectedProfile
profileValue.BackendID = strings.TrimSpace(profileValue.BackendID)
var selectedBackend *domain.Backend
if profileValue.BackendID != "" {
if r.backends == nil {
return nil, fmt.Errorf("%w: backend %q cannot be resolved", ErrProfileLoad, profileValue.BackendID)
}
backendValue, err := r.backends.GetBackend(profileValue.BackendID)
if err != nil {
return nil, fmt.Errorf("%w: backend %q: %w", ErrProfileLoad, profileValue.BackendID, err)
}
selectedBackend = &backendValue
}
return &resolvedProfileSelection{
id: normalizedID,
profile: &profileValue,
backend: selectedBackend,
}, nil
}
func validateResolvedExecutionTarget(target domain.ExecutionTarget) error {
if strings.TrimSpace(target.Endpoint) == "" {
return errors.New("execution endpoint is required")
}
if strings.TrimSpace(target.Model) == "" {
return errors.New("execution model is required")
}
return nil
}
// InspectProfile resolves one explicit profile without prompt or execution work.
func (r *Runner) InspectProfile(
ctx context.Context,
profileID string,
) (*domain.ProfileInspection, error) {
normalizedID := strings.TrimSpace(profileID)
if normalizedID == "" {
return nil, fmt.Errorf("%w: profile id is required", ErrInvalidRequest)
}
select {
case <-ctx.Done():
return nil, fmt.Errorf("%w: %w", ErrProfileLoad, ctx.Err())
default:
}
selection, err := r.resolveProfileSelection(ctx, normalizedID)
if err != nil {
return nil, err
}
target, _, err := resolveExecutionTarget(selection.backend, selection.profile, nil)
if err != nil {
return nil, fmt.Errorf("%w: %w", ErrProfileLoad, err)
}
if err := validateResolvedExecutionTarget(target); err != nil {
return nil, fmt.Errorf("%w: %w", ErrProfileLoad, err)
}
target.APIKey = ""
return &domain.ProfileInspection{
ProfileID: selection.id,
EffectiveModelParams: target,
APIKeyRequired: target.APIKeyRequired,
}, nil
}

View File

@@ -0,0 +1,213 @@
package usecase
import (
"context"
"errors"
"reflect"
"testing"
"gitea.maximumdirect.net/eric/promptkit/internal/defaults"
"gitea.maximumdirect.net/eric/promptkit/internal/domain"
"gitea.maximumdirect.net/eric/promptkit/internal/profile"
)
type inspectionProfileRepository struct {
profile *domain.ExecutionProfile
err error
calls int
id string
}
func (r *inspectionProfileRepository) GetProfile(
_ context.Context,
id string,
) (*domain.ExecutionProfile, error) {
r.calls++
r.id = id
if r.err != nil {
return nil, r.err
}
return r.profile, nil
}
type inspectionBackendResolver struct {
backend domain.Backend
err error
calls int
id string
}
func (r *inspectionBackendResolver) GetBackend(id string) (domain.Backend, error) {
r.calls++
r.id = id
if r.err != nil {
return domain.Backend{}, r.err
}
return r.backend, nil
}
func TestRunnerInspectProfileResolvesProfileAndBackendOnce(t *testing.T) {
profiles := &inspectionProfileRepository{profile: &domain.ExecutionProfile{
ID: "profile",
BackendID: " backend ",
Model: "profile-model",
Temperature: 0.4,
MaxTokens: 32,
TimeoutSeconds: 45,
ServiceTier: "priority",
ReasoningEffort: "high",
ExtraParams: map[string]any{
"profile": "value",
},
}}
backends := &inspectionBackendResolver{backend: domain.Backend{
ID: "backend",
Endpoint: "https://backend.example/v1",
APIKeyEnv: "BACKEND_KEY",
ExtraParams: map[string]any{
"backend": "value",
},
}}
runner := &Runner{profiles: profiles, backends: backends}
inspection, err := runner.InspectProfile(context.Background(), " profile ")
if err != nil {
t.Fatalf("inspect profile: %v", err)
}
if profiles.calls != 1 || profiles.id != "profile" {
t.Fatalf("profile lookup=(calls=%d id=%q), want one exact lookup", profiles.calls, profiles.id)
}
if backends.calls != 1 || backends.id != "backend" {
t.Fatalf("backend lookup=(calls=%d id=%q), want one exact lookup", backends.calls, backends.id)
}
if profiles.profile.BackendID != " backend " {
t.Fatalf("inspection mutated repository profile backend: %q", profiles.profile.BackendID)
}
wantTarget := domain.ExecutionTarget{
BackendID: "backend",
Endpoint: "https://backend.example/v1",
Model: "profile-model",
Temperature: 0.4,
MaxTokens: 32,
TopP: defaults.ExecutionTargetDefault().TopP,
TimeoutSeconds: 45,
ServiceTier: "priority",
ReasoningEffort: "high",
APIKeyEnv: "BACKEND_KEY",
ExtraParams: map[string]any{
"profile": "value",
},
}
if inspection.ProfileID != "profile" || inspection.APIKeyRequired ||
!reflect.DeepEqual(inspection.EffectiveModelParams, wantTarget) {
t.Fatalf("inspection=%#v, want profile=%q target=%#v", inspection, "profile", wantTarget)
}
}
func TestRunnerInspectProfileDoesNotNeedExecutionCollaboratorsOrCredentials(t *testing.T) {
t.Setenv("PROMPTKIT_INSPECTION_TEST_KEY", "")
profiles := &inspectionProfileRepository{profile: &domain.ExecutionProfile{
ID: "endpoint-only",
Endpoint: "https://profile.example/v1",
Model: "profile-model",
APIKeyEnv: "PROMPTKIT_INSPECTION_TEST_KEY",
}}
runner := &Runner{profiles: profiles}
inspection, err := runner.InspectProfile(context.Background(), "endpoint-only")
if err != nil {
t.Fatalf("inspect endpoint-only profile: %v", err)
}
if inspection.EffectiveModelParams.BackendID != "" ||
inspection.EffectiveModelParams.APIKeyEnv != "PROMPTKIT_INSPECTION_TEST_KEY" ||
inspection.APIKeyRequired {
t.Fatalf("unexpected endpoint-only inspection: %#v", inspection)
}
}
func TestRunnerInspectProfileDirectCredentialRequirementClearsBackendEnvironment(t *testing.T) {
profiles := &inspectionProfileRepository{profile: &domain.ExecutionProfile{
ID: "direct-key",
BackendID: "backend",
Model: "profile-model",
APIKeyRequired: true,
}}
backends := &inspectionBackendResolver{backend: domain.Backend{
ID: "backend",
Endpoint: "https://backend.example/v1",
APIKeyEnv: "BACKEND_KEY",
}}
inspection, err := (&Runner{profiles: profiles, backends: backends}).InspectProfile(
context.Background(),
"direct-key",
)
if err != nil {
t.Fatalf("inspect direct-key profile: %v", err)
}
if !inspection.APIKeyRequired || inspection.EffectiveModelParams.APIKeyEnv != "" {
t.Fatalf("credential requirement was not resolved exclusively: %#v", inspection)
}
}
func TestRunnerInspectProfileClassifiesFailuresWithoutRepositoryWorkAfterCancellation(t *testing.T) {
t.Run("blank ID", func(t *testing.T) {
profiles := &inspectionProfileRepository{}
_, err := (&Runner{profiles: profiles}).InspectProfile(context.Background(), " \t ")
if !errors.Is(err, ErrInvalidRequest) || profiles.calls != 0 {
t.Fatalf("blank inspection=(%v, calls=%d), want invalid request without lookup", err, profiles.calls)
}
})
t.Run("canceled context", func(t *testing.T) {
ctx, cancel := context.WithCancel(context.Background())
cancel()
profiles := &inspectionProfileRepository{}
_, err := (&Runner{profiles: profiles}).InspectProfile(ctx, "profile")
if !errors.Is(err, ErrProfileLoad) || !errors.Is(err, context.Canceled) || profiles.calls != 0 {
t.Fatalf("canceled inspection=(%v, calls=%d), want profile load and context identities without lookup", err, profiles.calls)
}
})
t.Run("missing profile", func(t *testing.T) {
profiles := &inspectionProfileRepository{err: profile.ErrProfileNotFound}
_, err := (&Runner{profiles: profiles}).InspectProfile(context.Background(), "missing")
if !errors.Is(err, ErrProfileLoad) || !errors.Is(err, profile.ErrProfileNotFound) {
t.Fatalf("missing profile error=%v, want profile load and not-found identities", err)
}
})
t.Run("unknown backend", func(t *testing.T) {
backendErr := errors.New("unknown backend")
profiles := &inspectionProfileRepository{profile: &domain.ExecutionProfile{
ID: "profile", BackendID: "backend", Model: "profile-model",
}}
backends := &inspectionBackendResolver{err: backendErr}
_, err := (&Runner{profiles: profiles, backends: backends}).InspectProfile(context.Background(), "profile")
if !errors.Is(err, ErrProfileLoad) || !errors.Is(err, backendErr) {
t.Fatalf("unknown backend error=%v, want profile load and backend identities", err)
}
})
t.Run("defensive invalid dependencies", func(t *testing.T) {
cases := []struct {
name string
runner *Runner
}{
{name: "nil repository", runner: &Runner{}},
{name: "nil profile", runner: &Runner{profiles: &inspectionProfileRepository{}}},
{name: "invalid target", runner: &Runner{profiles: &inspectionProfileRepository{
profile: &domain.ExecutionProfile{ID: "profile", Endpoint: "https://profile.example/v1"},
}}},
}
for _, tc := range cases {
t.Run(tc.name, func(t *testing.T) {
_, err := tc.runner.InspectProfile(context.Background(), "profile")
if !errors.Is(err, ErrProfileLoad) {
t.Fatalf("inspection error=%v, want profile load", err)
}
})
}
})
}

View File

@@ -0,0 +1,73 @@
package usecase
import (
"context"
"fmt"
"strings"
"gitea.maximumdirect.net/eric/promptkit/internal/domain"
)
type resolvedPromptDefinition struct {
definition *domain.PromptDefinition
hash string
}
func (r *Runner) resolvePromptDefinition(
ctx context.Context,
promptID string,
promptVersion string,
) (*resolvedPromptDefinition, error) {
if strings.TrimSpace(promptID) == "" {
return nil, fmt.Errorf("%w: prompt id is required", ErrInvalidRequest)
}
if r == nil || r.promptDefs == nil {
return nil, fmt.Errorf("%w: prompt repository is not configured", ErrPromptLoad)
}
definition, err := r.promptDefs.GetPromptDefinition(ctx, promptID, promptVersion)
if err != nil {
return nil, fmt.Errorf("%w: %w", ErrPromptLoad, err)
}
if definition == nil {
return nil, fmt.Errorf("%w: prompt repository returned nil definition", ErrPromptLoad)
}
hash, err := hashPromptDefinition(definition)
if err != nil {
return nil, fmt.Errorf("%w: failed to hash prompt definition: %v", ErrPromptLoad, err)
}
return &resolvedPromptDefinition{definition: definition, hash: hash}, nil
}
// InspectPrompt resolves one explicit prompt without execution work.
func (r *Runner) InspectPrompt(
ctx context.Context,
promptID string,
promptVersion string,
) (*domain.PromptInspection, error) {
if strings.TrimSpace(promptID) == "" {
return nil, fmt.Errorf("%w: prompt id is required", ErrInvalidRequest)
}
select {
case <-ctx.Done():
return nil, fmt.Errorf("%w: %w", ErrPromptLoad, ctx.Err())
default:
}
selection, err := r.resolvePromptDefinition(ctx, promptID, promptVersion)
if err != nil {
return nil, err
}
inputs := make([]domain.PromptInput, len(selection.definition.Inputs))
copy(inputs, selection.definition.Inputs)
return &domain.PromptInspection{
PromptID: selection.definition.ID,
PromptVersion: selection.definition.Version,
PromptHash: selection.hash,
DefaultProfileID: selection.definition.DefaultProfile,
Inputs: inputs,
OutputContract: selection.definition.Validation,
}, nil
}

View File

@@ -0,0 +1,157 @@
package usecase
import (
"context"
"errors"
"reflect"
"testing"
"gitea.maximumdirect.net/eric/promptkit/internal/domain"
"gitea.maximumdirect.net/eric/promptkit/internal/promptdef"
)
type inspectionPromptRepository struct {
definition *domain.PromptDefinition
err error
calls int
id string
version string
}
func (r *inspectionPromptRepository) GetPromptDefinition(
_ context.Context,
id string,
version string,
) (*domain.PromptDefinition, error) {
r.calls++
r.id = id
r.version = version
if r.err != nil {
return nil, r.err
}
return r.definition, nil
}
func TestRunnerInspectPromptResolvesOneDefinitionWithoutExecutionCollaborators(t *testing.T) {
definition := &domain.PromptDefinition{
ID: "normalized.prompt",
Version: "1.2.3",
DefaultProfile: "not-resolved",
Inputs: []domain.PromptInput{
{Name: "document", Required: true, ContentType: "text/plain", Description: "Source document."},
{Name: "audience", ContentType: "text/plain", Description: "Intended reader."},
},
Validation: domain.OutputContract{
Format: domain.FormatJSON,
ValidationMode: domain.ValidationJSONSchema,
SchemaPath: "schemas/result.json",
RepairAttempts: 2,
},
}
repository := &inspectionPromptRepository{definition: definition}
runner := &Runner{promptDefs: repository}
inspection, err := runner.InspectPrompt(context.Background(), " prompt-id ", " version ")
if err != nil {
t.Fatalf("inspect prompt: %v", err)
}
wantHash, err := hashPromptDefinition(definition)
if err != nil {
t.Fatalf("hash prompt definition: %v", err)
}
if repository.calls != 1 || repository.id != " prompt-id " || repository.version != " version " {
t.Fatalf("prompt lookup=(calls=%d id=%q version=%q), want one unchanged lookup", repository.calls, repository.id, repository.version)
}
if inspection.PromptID != definition.ID ||
inspection.PromptVersion != definition.Version ||
inspection.PromptHash != wantHash ||
inspection.DefaultProfileID != definition.DefaultProfile ||
!reflect.DeepEqual(inspection.Inputs, definition.Inputs) ||
inspection.OutputContract != definition.Validation {
t.Fatalf("inspection=%#v, want definition metadata", inspection)
}
inspection.Inputs[0].Name = "changed"
second, err := runner.InspectPrompt(context.Background(), " prompt-id ", " version ")
if err != nil {
t.Fatalf("inspect prompt again: %v", err)
}
if definition.Inputs[0].Name != "document" || second.Inputs[0].Name != "document" {
t.Fatalf("inspection input mutation escaped caller result: definition=%#v next=%#v", definition.Inputs, second.Inputs)
}
}
func TestRunnerInspectPromptClassifiesFailuresWithoutRepositoryWorkAfterCancellation(t *testing.T) {
t.Run("blank ID", func(t *testing.T) {
repository := &inspectionPromptRepository{}
_, err := (&Runner{promptDefs: repository}).InspectPrompt(context.Background(), " \t ", "version")
if !errors.Is(err, ErrInvalidRequest) || repository.calls != 0 {
t.Fatalf("blank inspection=(%v, calls=%d), want invalid request without lookup", err, repository.calls)
}
})
t.Run("canceled context", func(t *testing.T) {
ctx, cancel := context.WithCancel(context.Background())
cancel()
repository := &inspectionPromptRepository{}
_, err := (&Runner{promptDefs: repository}).InspectPrompt(ctx, "prompt", "version")
if !errors.Is(err, ErrPromptLoad) || !errors.Is(err, context.Canceled) || repository.calls != 0 {
t.Fatalf("canceled inspection=(%v, calls=%d), want prompt load and context identities without lookup", err, repository.calls)
}
})
t.Run("missing prompt", func(t *testing.T) {
repository := &inspectionPromptRepository{err: promptdef.ErrPromptDefinitionNotFound}
_, err := (&Runner{promptDefs: repository}).InspectPrompt(context.Background(), "missing", "version")
if !errors.Is(err, ErrPromptLoad) || !errors.Is(err, promptdef.ErrPromptDefinitionNotFound) {
t.Fatalf("missing prompt error=%v, want prompt load and not-found identities", err)
}
})
t.Run("defensive prompt dependencies", func(t *testing.T) {
var nilRunner *Runner
cases := []struct {
name string
runner *Runner
}{
{name: "nil runner", runner: nilRunner},
{name: "nil repository", runner: &Runner{}},
{name: "nil definition", runner: &Runner{promptDefs: &inspectionPromptRepository{}}},
}
for _, tc := range cases {
t.Run(tc.name, func(t *testing.T) {
_, err := tc.runner.InspectPrompt(context.Background(), "prompt", "version")
if !errors.Is(err, ErrPromptLoad) {
t.Fatalf("inspection error=%v, want prompt load", err)
}
})
}
})
}
func TestRunnerPrepareUsesThePromptInspectionSelectionAndHash(t *testing.T) {
definition := promptDef(domain.FormatMarkdown, domain.ValidationBasic, 0)
repository := &fakePromptRepo{def: definition}
runner := NewRunner(
repository,
&fakeExecutionProfileRepo{profiles: map[string]*domain.ExecutionProfile{"exec": defaultExecutionProfile()}},
nil,
&fakeArtifactReader{},
&fakeRenderer{rendered: &domain.RenderedPrompt{Messages: []domain.RenderedMessage{{Role: "user", Content: "hello"}}}},
nil,
nil,
nil,
)
inspection, err := runner.InspectPrompt(context.Background(), definition.ID, definition.Version)
if err != nil {
t.Fatalf("inspect prompt: %v", err)
}
prepared, err := runner.Prepare(context.Background(), domain.RunRequest{PromptID: definition.ID, PromptVersion: definition.Version, ProfileID: "exec"})
if err != nil {
t.Fatalf("prepare prompt: %v", err)
}
if inspection.PromptHash != prepared.PromptHash {
t.Fatalf("inspection hash=%q, preparation hash=%q", inspection.PromptHash, prepared.PromptHash)
}
}

View File

@@ -131,29 +131,46 @@ func (r *Runner) Run(ctx context.Context, req domain.RunRequest) (*domain.RunRes
return nil, err return nil, err
} }
if r.admitter != nil { release, err := r.admitRun(ctx, state.effectiveModel.BackendID)
release, admitErr := r.admitter.Admit(ctx, state.effectiveModel.BackendID) if err != nil {
if admitErr != nil { return nil, err
if errors.Is(admitErr, capacity.ErrCapacityExceeded) {
return nil, fmt.Errorf(
"backend %q admission: %w",
state.effectiveModel.BackendID,
admitErr,
)
}
return nil, admitErr
}
defer release()
} }
defer release()
prepared, err := r.completePreparation(ctx, req, state) prepared, err := r.completePreparation(ctx, req, state)
if err != nil { if err != nil {
return nil, err return nil, err
} }
directAPIKey := state.effectiveModel.APIKey
return r.executePreparedRun(ctx, prepared, directAPIKey, runID, start, func(
ctx context.Context,
artifact *domain.Artifact,
attemptsUsed int,
) (domain.ValidationResult, error) {
return r.validateOutput(ctx, artifact, prepared.OutputContract, attemptsUsed)
})
}
type preparedValidationFunc func(
context.Context,
*domain.Artifact,
int,
) (domain.ValidationResult, error)
func (r *Runner) executePreparedRun(
ctx context.Context,
prepared *domain.PreparedRun,
directAPIKey string,
runID string,
start time.Time,
validateArtifact preparedValidationFunc,
) (*domain.RunResult, error) {
executionTarget := prepared.EffectiveModelParams
executionTarget.APIKey = directAPIKey
genResp, err := r.llm.Generate(ctx, domain.GenerateRequest{ genResp, err := r.llm.Generate(ctx, domain.GenerateRequest{
Prompt: domain.RenderedPrompt{SessionID: prepared.SessionID, Messages: prepared.Messages}, Prompt: domain.RenderedPrompt{SessionID: prepared.SessionID, Messages: prepared.Messages},
Target: prepared.EffectiveModelParams, Target: executionTarget,
TargetPresence: prepared.TargetPresence, TargetPresence: prepared.TargetPresence,
StructuredOutput: prepared.StructuredOutput, StructuredOutput: prepared.StructuredOutput,
}) })
@@ -165,7 +182,7 @@ func (r *Runner) Run(ctx context.Context, req domain.RunRequest) (*domain.RunRes
} }
outputArtifact := buildOutputArtifact(genResp.Content, prepared.OutputContract.Format) outputArtifact := buildOutputArtifact(genResp.Content, prepared.OutputContract.Format)
validationResult, err := r.validateOutput(ctx, &outputArtifact, prepared.OutputContract, 0) validationResult, err := validateArtifact(ctx, &outputArtifact, 0)
if err != nil { if err != nil {
return nil, fmt.Errorf("%w: %w", ErrValidation, err) return nil, fmt.Errorf("%w: %w", ErrValidation, err)
} }
@@ -179,7 +196,7 @@ func (r *Runner) Run(ctx context.Context, req domain.RunRequest) (*domain.RunRes
PreviousOutput: genResp.Content, PreviousOutput: genResp.Content,
ValidationErrors: validationResult.Errors, ValidationErrors: validationResult.Errors,
SessionID: prepared.SessionID, SessionID: prepared.SessionID,
Target: prepared.EffectiveModelParams, Target: executionTarget,
StructuredOutput: prepared.StructuredOutput, StructuredOutput: prepared.StructuredOutput,
Attempt: attemptsUsed, Attempt: attemptsUsed,
MaxAttempts: prepared.OutputContract.RepairAttempts, MaxAttempts: prepared.OutputContract.RepairAttempts,
@@ -195,7 +212,7 @@ func (r *Runner) Run(ctx context.Context, req domain.RunRequest) (*domain.RunRes
genResp = repairResp genResp = repairResp
outputArtifact = buildOutputArtifact(genResp.Content, prepared.OutputContract.Format) outputArtifact = buildOutputArtifact(genResp.Content, prepared.OutputContract.Format)
validationResult, err = r.validateOutput(ctx, &outputArtifact, prepared.OutputContract, attemptsUsed) validationResult, err = validateArtifact(ctx, &outputArtifact, attemptsUsed)
if err != nil { if err != nil {
return nil, fmt.Errorf("%w: %w", ErrValidation, err) return nil, fmt.Errorf("%w: %w", ErrValidation, err)
} }
@@ -203,6 +220,7 @@ func (r *Runner) Run(ctx context.Context, req domain.RunRequest) (*domain.RunRes
} }
end := time.Now().UTC() end := time.Now().UTC()
executionTarget.APIKey = ""
return &domain.RunResult{ return &domain.RunResult{
RunID: runID, RunID: runID,
@@ -218,7 +236,7 @@ func (r *Runner) Run(ctx context.Context, req domain.RunRequest) (*domain.RunRes
SelectedBackendID: prepared.SelectedBackendID, SelectedBackendID: prepared.SelectedBackendID,
ModelName: prepared.EffectiveModelParams.Model, ModelName: prepared.EffectiveModelParams.Model,
Endpoint: prepared.EffectiveModelParams.Endpoint, Endpoint: prepared.EffectiveModelParams.Endpoint,
EffectiveModelParams: prepared.EffectiveModelParams, EffectiveModelParams: executionTarget,
InputHashes: prepared.InputHashes, InputHashes: prepared.InputHashes,
Usage: genResp.Usage, Usage: genResp.Usage,
StartTime: start, StartTime: start,
@@ -248,14 +266,12 @@ func (r *Runner) resolvePreparation(
return nil, fmt.Errorf("%w: session_id: %v", ErrInvalidRequest, err) return nil, fmt.Errorf("%w: session_id: %v", ErrInvalidRequest, err)
} }
def, err := r.promptDefs.GetPromptDefinition(ctx, req.PromptID, req.PromptVersion) promptSelection, err := r.resolvePromptDefinition(ctx, req.PromptID, req.PromptVersion)
if err != nil { if err != nil {
return nil, fmt.Errorf("%w: %w", ErrPromptLoad, err) return nil, err
}
promptDefinitionHash, err := hashPromptDefinition(def)
if err != nil {
return nil, fmt.Errorf("%w: failed to hash prompt definition: %v", ErrPromptLoad, err)
} }
def := promptSelection.definition
promptDefinitionHash := promptSelection.hash
selectedProfileID := strings.TrimSpace(req.ProfileID) selectedProfileID := strings.TrimSpace(req.ProfileID)
if selectedProfileID == "" { if selectedProfileID == "" {
@@ -265,34 +281,18 @@ func (r *Runner) resolvePreparation(
return nil, fmt.Errorf("%w: %w: profile id is required either in request or prompt default_profile", ErrInvalidRequest, ErrProfileRequired) return nil, fmt.Errorf("%w: %w: profile id is required either in request or prompt default_profile", ErrInvalidRequest, ErrProfileRequired)
} }
execProfile, err := r.profiles.GetProfile(ctx, selectedProfileID) selection, err := r.resolveProfileSelection(ctx, selectedProfileID)
if err != nil { if err != nil {
return nil, fmt.Errorf("%w: %w", ErrProfileLoad, err) return nil, err
} }
var selectedBackend *domain.Backend effectiveModel, targetPresence, err := resolveExecutionTarget(selection.backend, selection.profile, req.Execution)
if backendID := strings.TrimSpace(execProfile.BackendID); backendID != "" {
execProfile.BackendID = backendID
if r.backends == nil {
return nil, fmt.Errorf("%w: backend %q cannot be resolved", ErrProfileLoad, backendID)
}
resolvedBackend, resolveErr := r.backends.GetBackend(backendID)
if resolveErr != nil {
return nil, fmt.Errorf("%w: backend %q: %w", ErrProfileLoad, backendID, resolveErr)
}
selectedBackend = &resolvedBackend
}
effectiveModel, targetPresence, err := resolveExecutionTarget(selectedBackend, execProfile, req.Execution)
if err != nil { if err != nil {
return nil, fmt.Errorf("%w: %w", ErrInvalidRequest, err) return nil, fmt.Errorf("%w: %w", ErrInvalidRequest, err)
} }
effectiveModel.APIKey = req.APIKey effectiveModel.APIKey = req.APIKey
if strings.TrimSpace(effectiveModel.Endpoint) == "" { if err := validateResolvedExecutionTarget(effectiveModel); err != nil {
return nil, fmt.Errorf("%w: execution endpoint is required", ErrInvalidRequest) return nil, fmt.Errorf("%w: %w", ErrInvalidRequest, err)
}
if strings.TrimSpace(effectiveModel.Model) == "" {
return nil, fmt.Errorf("%w: execution model is required", ErrInvalidRequest)
} }
if err := validateAPIKey(effectiveModel.APIKeyEnv, effectiveModel.APIKey, effectiveModel.APIKeyRequired); err != nil { if err := validateAPIKey(effectiveModel.APIKeyEnv, effectiveModel.APIKey, effectiveModel.APIKeyRequired); err != nil {
return nil, fmt.Errorf("%w: %w", ErrInvalidRequest, err) return nil, fmt.Errorf("%w: %w", ErrInvalidRequest, err)
@@ -303,7 +303,7 @@ func (r *Runner) resolvePreparation(
definition: def, definition: def,
directSessionID: directSessionID, directSessionID: directSessionID,
promptDefinitionHash: promptDefinitionHash, promptDefinitionHash: promptDefinitionHash,
selectedProfileID: selectedProfileID, selectedProfileID: selection.id,
effectiveModel: effectiveModel, effectiveModel: effectiveModel,
targetPresence: targetPresence, targetPresence: targetPresence,
effectiveContract: effectiveContract, effectiveContract: effectiveContract,
@@ -324,7 +324,15 @@ func (r *Runner) completePreparation(
if err != nil { if err != nil {
return nil, err return nil, err
} }
return r.completePreparationWithStructuredOutput(ctx, req, state, structuredOutput)
}
func (r *Runner) completePreparationWithStructuredOutput(
ctx context.Context,
req domain.RunRequest,
state *preparationState,
structuredOutput *domain.StructuredOutputSpec,
) (*domain.PreparedRun, error) {
resolvedInputs := make(map[string]*domain.Artifact, len(req.Inputs)) resolvedInputs := make(map[string]*domain.Artifact, len(req.Inputs))
inputHashes := make(map[string]string, len(req.Inputs)) inputHashes := make(map[string]string, len(req.Inputs))
for name, ref := range req.Inputs { for name, ref := range req.Inputs {
@@ -354,13 +362,15 @@ func (r *Runner) completePreparation(
} }
end := time.Now().UTC() end := time.Now().UTC()
effectiveModel := state.effectiveModel
effectiveModel.APIKey = ""
return &domain.PreparedRun{ return &domain.PreparedRun{
PromptID: state.definition.ID, PromptID: state.definition.ID,
PromptVersion: state.definition.Version, PromptVersion: state.definition.Version,
PromptHash: state.promptDefinitionHash, PromptHash: state.promptDefinitionHash,
SelectedProfileID: state.selectedProfileID, SelectedProfileID: state.selectedProfileID,
SelectedBackendID: state.effectiveModel.BackendID, SelectedBackendID: state.effectiveModel.BackendID,
EffectiveModelParams: state.effectiveModel, EffectiveModelParams: effectiveModel,
TargetPresence: state.targetPresence, TargetPresence: state.targetPresence,
OutputContract: state.effectiveContract, OutputContract: state.effectiveContract,
StructuredOutput: structuredOutput, StructuredOutput: structuredOutput,
@@ -374,6 +384,20 @@ func (r *Runner) completePreparation(
}, nil }, nil
} }
func (r *Runner) admitRun(ctx context.Context, backendID string) (func(), error) {
if r.admitter == nil {
return func() {}, nil
}
release, err := r.admitter.Admit(ctx, backendID)
if err != nil {
if errors.Is(err, capacity.ErrCapacityExceeded) && strings.TrimSpace(backendID) != "" {
return nil, &CapacityError{BackendID: backendID}
}
return nil, err
}
return release, nil
}
func (r *Runner) resolveStructuredOutput(ctx context.Context, def *domain.PromptDefinition, contract domain.OutputContract) (*domain.StructuredOutputSpec, error) { func (r *Runner) resolveStructuredOutput(ctx context.Context, def *domain.PromptDefinition, contract domain.OutputContract) (*domain.StructuredOutputSpec, error) {
if contract.ValidationMode != domain.ValidationJSONSchema { if contract.ValidationMode != domain.ValidationJSONSchema {
return nil, nil return nil, nil
@@ -389,14 +413,18 @@ func (r *Runner) resolveStructuredOutput(ctx context.Context, def *domain.Prompt
return nil, fmt.Errorf("%w: failed to load json schema for structured output: %v", ErrValidation, err) return nil, fmt.Errorf("%w: failed to load json schema for structured output: %v", ErrValidation, err)
} }
return structuredOutputSpec(def, schemaDoc), nil
}
func structuredOutputSpec(def *domain.PromptDefinition, schemaDocument any) *domain.StructuredOutputSpec {
return &domain.StructuredOutputSpec{ return &domain.StructuredOutputSpec{
Type: domain.StructuredOutputJSONSchema, Type: domain.StructuredOutputJSONSchema,
JSONSchema: &domain.StructuredOutputJSONSpec{ JSONSchema: &domain.StructuredOutputJSONSpec{
Name: deriveStructuredSchemaName(def.ID, def.Version), Name: deriveStructuredSchemaName(def.ID, def.Version),
Strict: true, Strict: true,
Schema: schemaDoc, Schema: schemaDocument,
}, },
}, nil }
} }
func deriveStructuredSchemaName(promptID string, promptVersion string) string { func deriveStructuredSchemaName(promptID string, promptVersion string) string {

View File

@@ -1357,14 +1357,14 @@ func TestRunnerAdmissionUsesResolvedBackendIdentity(t *testing.T) {
func TestRunnerAdmissionFailureSkipsCompletionCollaborators(t *testing.T) { func TestRunnerAdmissionFailureSkipsCompletionCollaborators(t *testing.T) {
tests := []struct { tests := []struct {
name string name string
admissionError error admissionError error
wantBackendContext bool wantCapacityType bool
}{ }{
{ {
name: "capacity exhausted", name: "capacity exhausted",
admissionError: capacity.ErrCapacityExceeded, admissionError: capacity.ErrCapacityExceeded,
wantBackendContext: true, wantCapacityType: true,
}, },
{ {
name: "context canceled", name: "context canceled",
@@ -1414,8 +1414,16 @@ func TestRunnerAdmissionFailureSkipsCompletionCollaborators(t *testing.T) {
if errors.Is(err, ErrInvalidRequest) || errors.Is(err, ErrLLMGenerate) { if errors.Is(err, ErrInvalidRequest) || errors.Is(err, ErrLLMGenerate) {
t.Fatalf("admission error was recategorized: %v", err) t.Fatalf("admission error was recategorized: %v", err)
} }
if tc.wantBackendContext && !strings.Contains(err.Error(), "custom") { var capacityErr *CapacityError
t.Fatalf("capacity error lacks backend context: %v", err) if tc.wantCapacityType {
if !errors.As(err, &capacityErr) {
t.Fatalf("capacity error=%v, want internal typed identity", err)
}
if capacityErr.BackendID != "custom" {
t.Fatalf("capacity backend ID=%q, want custom", capacityErr.BackendID)
}
} else if errors.As(err, &capacityErr) {
t.Fatalf("non-capacity admission error exposed typed capacity identity: %v", err)
} }
if !reflect.DeepEqual(admitter.backendIDs, []string{"custom"}) { if !reflect.DeepEqual(admitter.backendIDs, []string{"custom"}) {
t.Fatalf("admitted backend IDs=%#v, want custom", admitter.backendIDs) t.Fatalf("admitted backend IDs=%#v, want custom", admitter.backendIDs)
@@ -1757,7 +1765,7 @@ func TestRunnerRunDirectAPIKeyBypassesMissingEnvAndReachesLLM(t *testing.T) {
llmClient := &fakeLLM{resp: &domain.GenerateResponse{Content: "ok"}} llmClient := &fakeLLM{resp: &domain.GenerateResponse{Content: "ok"}}
runner := NewRunner(promptRepo, execRepo, nil, defaultArtifactReader(), defaultRenderer(), llmClient, nil, nil) runner := NewRunner(promptRepo, execRepo, nil, defaultArtifactReader(), defaultRenderer(), llmClient, nil, nil)
_, err := runner.Run(context.Background(), domain.RunRequest{ result, err := runner.Run(context.Background(), domain.RunRequest{
PromptID: "p", PromptID: "p",
ProfileID: "exec", ProfileID: "exec",
APIKey: directKey, APIKey: directKey,
@@ -1769,6 +1777,9 @@ func TestRunnerRunDirectAPIKeyBypassesMissingEnvAndReachesLLM(t *testing.T) {
if llmClient.lastReq.Target.APIKey != directKey { if llmClient.lastReq.Target.APIKey != directKey {
t.Fatalf("expected direct API key to reach LLM request") t.Fatalf("expected direct API key to reach LLM request")
} }
if result.EffectiveModelParams.APIKey != "" {
t.Fatal("run result retained direct API key")
}
if llmClient.lastReq.Target.APIKeyEnv != "PROMPTKIT_MISSING_KEY" { if llmClient.lastReq.Target.APIKeyEnv != "PROMPTKIT_MISSING_KEY" {
t.Fatalf("expected api_key_env name to remain on target, got %q", llmClient.lastReq.Target.APIKeyEnv) t.Fatalf("expected api_key_env name to remain on target, got %q", llmClient.lastReq.Target.APIKeyEnv)
} }

View File

@@ -45,6 +45,101 @@ func (v *FSValidator) Validate(ctx context.Context, artifact *domain.Artifact, c
return validateArtifact(ctx, artifact, contract, v.validateJSONSchema) return validateArtifact(ctx, artifact, contract, v.validateJSONSchema)
} }
type preparedValidation struct {
contract domain.OutputContract
schemaDocument any
schema *jsonschema.Schema
}
func (p *preparedValidation) Validate(ctx context.Context, artifact *domain.Artifact) (domain.ValidationResult, error) {
return validateArtifact(ctx, artifact, p.contract, p.validateJSONSchema)
}
func (p *preparedValidation) SchemaDocument() any {
return p.schemaDocument
}
func (p *preparedValidation) validateJSONSchema(instance any, _ string) ([]string, error) {
if p.schema == nil {
return nil, errors.New("prepared JSON schema is unavailable")
}
if err := p.schema.Validate(instance); err != nil {
return []string{fmt.Sprintf("json schema validation failed: %v", err)}, nil
}
return nil, nil
}
func (v *StandardValidator) PrepareValidation(ctx context.Context, contract domain.OutputContract) (PreparedValidation, error) {
if err := ctx.Err(); err != nil {
return nil, err
}
prepared := &preparedValidation{contract: contract}
if contract.ValidationMode != domain.ValidationJSONSchema {
return prepared, nil
}
resolvedSchemaPath, err := v.resolveSchemaPath(contract.SchemaPath)
if err != nil {
return nil, err
}
schemaDocument, err := loadJSONSchemaFile(resolvedSchemaPath)
if err != nil {
return nil, fmt.Errorf("failed to compile JSON schema %q: %w", resolvedSchemaPath, err)
}
schemaRoot, err := v.schemaRoot()
if err != nil {
return nil, err
}
compiler := newSchemaCompiler(standardSchemaLoader{root: schemaRoot})
if err := compiler.AddResource(resolvedSchemaPath, schemaDocument); err != nil {
return nil, fmt.Errorf("failed to register JSON schema %q: %w", resolvedSchemaPath, err)
}
schema, err := compiler.Compile(resolvedSchemaPath)
if err != nil {
return nil, fmt.Errorf("failed to compile JSON schema %q: %w", resolvedSchemaPath, err)
}
if err := ctx.Err(); err != nil {
return nil, err
}
prepared.schemaDocument = schemaDocument
prepared.schema = schema
return prepared, nil
}
func (v *FSValidator) PrepareValidation(ctx context.Context, contract domain.OutputContract) (PreparedValidation, error) {
if err := ctx.Err(); err != nil {
return nil, err
}
prepared := &preparedValidation{contract: contract}
if contract.ValidationMode != domain.ValidationJSONSchema {
return prepared, nil
}
schemaName, schemaDocument, err := v.loadSchemaDocument(contract.SchemaPath)
if err != nil {
return nil, err
}
resourceURL := fsSchemaResourceURL(schemaName)
compiler := newSchemaCompiler(fsSchemaLoader{fsys: v.fsys, root: filecatalog.CleanFSRoot(v.root)})
if err := compiler.AddResource(resourceURL, schemaDocument); err != nil {
return nil, fmt.Errorf("failed to register JSON schema %q: %w", schemaName, err)
}
schema, err := compiler.Compile(resourceURL)
if err != nil {
return nil, fmt.Errorf("failed to compile JSON schema %q: %w", schemaName, err)
}
if err := ctx.Err(); err != nil {
return nil, err
}
prepared.schemaDocument = schemaDocument
prepared.schema = schema
return prepared, nil
}
type schemaValidatorFunc func(instance any, schemaPath string) ([]string, error) type schemaValidatorFunc func(instance any, schemaPath string) ([]string, error)
func validateArtifact(ctx context.Context, artifact *domain.Artifact, contract domain.OutputContract, validateSchema schemaValidatorFunc) (domain.ValidationResult, error) { func validateArtifact(ctx context.Context, artifact *domain.Artifact, contract domain.OutputContract, validateSchema schemaValidatorFunc) (domain.ValidationResult, error) {

View File

@@ -5,6 +5,7 @@ import (
"encoding/json" "encoding/json"
"os" "os"
"path/filepath" "path/filepath"
"reflect"
"strconv" "strconv"
"strings" "strings"
"testing" "testing"
@@ -120,6 +121,68 @@ func TestStandardValidatorJSONSchemaSuccess(t *testing.T) {
} }
} }
func TestStandardValidatorPreparedSchemaSurvivesSourceRemoval(t *testing.T) {
tmp := t.TempDir()
rootSchema := []byte(`{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"title": "original root",
"type": "object",
"required": ["value"],
"properties": {
"value": {"$ref": "value.json"}
}
}`)
rootPath := filepath.Join(tmp, "schema.json")
referencePath := filepath.Join(tmp, "value.json")
if err := os.WriteFile(rootPath, rootSchema, 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(referencePath, []byte(`{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"type": "integer",
"minimum": 2
}`), 0o644); err != nil {
t.Fatal(err)
}
validator := NewStandardValidator(tmp)
preparer, ok := validator.(ValidationPreparer)
if !ok {
t.Fatal("standard validator does not support validation preparation")
}
prepared, err := preparer.PrepareValidation(context.Background(), domain.OutputContract{
ValidationMode: domain.ValidationJSONSchema,
SchemaPath: "schema.json",
})
if err != nil {
t.Fatalf("prepare validation: %v", err)
}
assertSchemaDocument(t, prepared.SchemaDocument(), rootSchema)
if err := os.Remove(rootPath); err != nil {
t.Fatal(err)
}
if err := os.Remove(referencePath); err != nil {
t.Fatal(err)
}
valid, err := prepared.Validate(context.Background(), &domain.Artifact{Body: []byte(`{"value":3}`)})
if err != nil {
t.Fatalf("validate prepared artifact: %v", err)
}
if valid.Status != domain.ValidationPassed || !valid.IsValid {
t.Fatalf("expected passed/valid, got status=%q valid=%v errors=%v", valid.Status, valid.IsValid, valid.Errors)
}
invalid, err := prepared.Validate(context.Background(), &domain.Artifact{Body: []byte(`{"value":"changed"}`)})
if err != nil {
t.Fatalf("validate prepared artifact: %v", err)
}
if invalid.Status != domain.ValidationFailed || invalid.IsValid {
t.Fatalf("expected failed/invalid, got status=%q valid=%v", invalid.Status, invalid.IsValid)
}
}
func TestStandardValidatorJSONSchemaNestedSchemaPathSuccess(t *testing.T) { func TestStandardValidatorJSONSchemaNestedSchemaPathSuccess(t *testing.T) {
tmp := t.TempDir() tmp := t.TempDir()
nestedDir := filepath.Join(tmp, "dnd") nestedDir := filepath.Join(tmp, "dnd")
@@ -295,6 +358,64 @@ func TestFSValidatorJSONSchemaSuccess(t *testing.T) {
} }
} }
func TestFSValidatorPreparedSchemaSurvivesSourceMutation(t *testing.T) {
rootSchema := []byte(`{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"title": "original root",
"type": "object",
"required": ["value"],
"properties": {
"value": {"$ref": "value.json"}
}
}`)
fsys := fstest.MapFS{
"schema.json": &fstest.MapFile{Data: rootSchema},
"value.json": &fstest.MapFile{Data: []byte(`{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"type": "integer",
"minimum": 2
}`)},
}
validator := NewFSValidator(fsys, ".")
preparer, ok := validator.(ValidationPreparer)
if !ok {
t.Fatal("filesystem validator does not support validation preparation")
}
prepared, err := preparer.PrepareValidation(context.Background(), domain.OutputContract{
ValidationMode: domain.ValidationJSONSchema,
SchemaPath: "schema.json",
})
if err != nil {
t.Fatalf("prepare validation: %v", err)
}
assertSchemaDocument(t, prepared.SchemaDocument(), rootSchema)
fsys["schema.json"] = &fstest.MapFile{Data: []byte(`{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"type": "string"
}`)}
fsys["value.json"] = &fstest.MapFile{Data: []byte(`{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"type": "string"
}`)}
valid, err := prepared.Validate(context.Background(), &domain.Artifact{Body: []byte(`{"value":3}`)})
if err != nil {
t.Fatalf("validate prepared artifact: %v", err)
}
if valid.Status != domain.ValidationPassed || !valid.IsValid {
t.Fatalf("expected passed/valid, got status=%q valid=%v errors=%v", valid.Status, valid.IsValid, valid.Errors)
}
invalid, err := prepared.Validate(context.Background(), &domain.Artifact{Body: []byte(`{"value":"changed"}`)})
if err != nil {
t.Fatalf("validate prepared artifact: %v", err)
}
if invalid.Status != domain.ValidationFailed || invalid.IsValid {
t.Fatalf("expected failed/invalid, got status=%q valid=%v", invalid.Status, invalid.IsValid)
}
}
func TestFSValidatorJSONSchemaRegistrationError(t *testing.T) { func TestFSValidatorJSONSchemaRegistrationError(t *testing.T) {
v := NewFSValidator(fstest.MapFS{ v := NewFSValidator(fstest.MapFS{
"schemas/%zz.json": &fstest.MapFile{Data: []byte(`{"type":"object"}`)}, "schemas/%zz.json": &fstest.MapFile{Data: []byte(`{"type":"object"}`)},
@@ -548,3 +669,15 @@ func TestJSONSchemaDialectIsDraft2020(t *testing.T) {
}) })
} }
} }
func assertSchemaDocument(t *testing.T, got any, expectedJSON []byte) {
t.Helper()
var expected any
if err := json.Unmarshal(expectedJSON, &expected); err != nil {
t.Fatalf("decode expected schema document: %v", err)
}
if !reflect.DeepEqual(got, expected) {
t.Fatalf("schema document mismatch:\n got: %#v\nwant: %#v", got, expected)
}
}

View File

@@ -2,6 +2,7 @@ package validate
import ( import (
"context" "context"
"gitea.maximumdirect.net/eric/promptkit/internal/domain" "gitea.maximumdirect.net/eric/promptkit/internal/domain"
) )
@@ -10,6 +11,20 @@ type Validator interface {
Validate(ctx context.Context, artifact *domain.Artifact, contract domain.OutputContract) (domain.ValidationResult, error) Validate(ctx context.Context, artifact *domain.Artifact, contract domain.OutputContract) (domain.ValidationResult, error)
} }
// PreparedValidation validates artifacts against one frozen output contract.
type PreparedValidation interface {
Validate(ctx context.Context, artifact *domain.Artifact) (domain.ValidationResult, error)
// SchemaDocument returns the root JSON Schema document used for provider
// structured output, or nil for non-schema modes. Returned internal
// immutable state must not be mutated.
SchemaDocument() any
}
// ValidationPreparer freezes validation resources for one output contract.
type ValidationPreparer interface {
PrepareValidation(ctx context.Context, contract domain.OutputContract) (PreparedValidation, error)
}
// SchemaDocumentLoader loads JSON schema documents using validator path semantics. // SchemaDocumentLoader loads JSON schema documents using validator path semantics.
type SchemaDocumentLoader interface { type SchemaDocumentLoader interface {
LoadSchemaDocument(ctx context.Context, schemaPath string) (any, error) LoadSchemaDocument(ctx context.Context, schemaPath string) (any, error)

55
prepared_execution.go Normal file
View File

@@ -0,0 +1,55 @@
package promptkit
import "gitea.maximumdirect.net/eric/promptkit/internal/usecase"
const preparedExecutionString = "promptkit.PreparedExecution{opaque}"
// PreparedExecution is an opaque, in-process handle for one completely
// prepared execution. A handle is bound to the [Engine] that created it and
// permits one [Engine.RunPrepared] invocation.
//
// PreparedExecution contains no supported serializable state and cannot be
// used as a restartable job. Copying the value preserves the same shared
// lifecycle; it does not create another execution attempt.
type PreparedExecution struct {
internal *usecase.PreparedExecution
}
// Details returns a fresh caller-owned, credential-redacted copy of the
// prepared request details. Mutating the result cannot affect execution or a
// later Details call. Details remains available after execution or discard.
//
// A nil receiver or zero-value PreparedExecution returns a zero [PreparedRun].
func (p *PreparedExecution) Details() PreparedRun {
if p == nil || p.internal == nil {
return PreparedRun{}
}
details := fromDomainPreparedRun(p.internal.Details())
if details == nil {
return PreparedRun{}
}
return *details
}
// Discard invalidates an unclaimed handle and drops Promptkit's references to
// its execution-only state. Discard is nil-safe and idempotent. It does not
// cancel an execution that has already claimed the handle; use the
// [Engine.RunPrepared] context for cancellation.
func (p *PreparedExecution) Discard() {
if p == nil || p.internal == nil {
return
}
p.internal.Discard()
}
// String returns a constant representation that exposes no retained request,
// rendered content, or credential data.
func (p *PreparedExecution) String() string {
return preparedExecutionString
}
// GoString returns a constant Go-syntax representation that exposes no
// retained request, rendered content, or credential data.
func (p *PreparedExecution) GoString() string {
return preparedExecutionString
}

View File

@@ -0,0 +1,773 @@
package promptkit_test
import (
"context"
"encoding/json"
"errors"
"fmt"
"os"
"reflect"
"strings"
"sync"
"testing"
"testing/fstest"
"time"
"gitea.maximumdirect.net/eric/promptkit"
)
func TestPreparedExecutionFreezesSourcesAndReturnsIndependentDetails(t *testing.T) {
promptSource := preparedPromptSource("original")
profileSource := preparedProfileSource("original-model")
schemaSource := preparedSchemaSource()
reader := &mutablePreparedArtifactReader{
body: "original artifact",
hash: "original-input-hash",
}
client := &preparedRecordingClient{
response: &promptkit.GenerateResponse{Content: `{"value":3}`},
}
engine, err := promptkit.NewEngine(
promptkit.Config{},
promptkit.WithPromptFS(promptSource, "."),
promptkit.WithProfileFS(profileSource, "."),
promptkit.WithSchemaFS(schemaSource, "."),
promptkit.WithArtifactReader(reader),
promptkit.WithLLMClient(client),
)
if err != nil {
t.Fatalf("construct engine: %v", err)
}
temperature := 0.25
extraParams := map[string]any{
"nested": map[string]any{"source": "original"},
}
request := promptkit.RunRequest{
PromptID: "prepared",
Inputs: map[string]promptkit.ArtifactRef{
"input": promptkit.Inline("original request input"),
},
Vars: map[string]string{"label": "original variable"},
Execution: &promptkit.ExecutionTargetOverride{
Temperature: &temperature,
ExtraParams: extraParams,
},
}
preparationContext, cancelPreparation := context.WithCancel(context.Background())
prepared, err := engine.PrepareExecution(preparationContext, request)
if err != nil {
t.Fatalf("prepare execution: %v", err)
}
cancelPreparation()
request.PromptID = "changed"
request.Inputs["input"] = promptkit.Inline("changed request input")
request.Vars["label"] = "changed variable"
temperature = 1.5
extraParams["nested"].(map[string]any)["source"] = "changed"
promptSource["prompt.yaml"] = &fstest.MapFile{Data: []byte(`id: changed`)}
profileSource["profile.yaml"] = &fstest.MapFile{Data: []byte(`id: changed`)}
schemaSource["schema.json"] = &fstest.MapFile{Data: []byte(`{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"title": "changed root",
"type": "string"
}`)}
schemaSource["value.json"] = &fstest.MapFile{Data: []byte(`{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"type": "string"
}`)}
reader.set("changed artifact", "changed-input-hash")
first := prepared.Details()
first.Messages[0].Content = "changed details"
first.InputHashes["input"] = "changed-details-hash"
first.EffectiveModelParams.ExtraParams["nested"].(map[string]any)["source"] = "changed details"
first.StructuredOutput.JSONSchema.Schema.(map[string]any)["title"] = "changed details"
second := prepared.Details()
if second.Messages[0].Content != "Input=original artifact Label=original variable" {
t.Fatalf("details message changed: %q", second.Messages[0].Content)
}
if second.InputHashes["input"] != "original-input-hash" {
t.Fatalf("details input hash changed: %q", second.InputHashes["input"])
}
if second.EffectiveModelParams.Model != "original-model" ||
second.EffectiveModelParams.Temperature != 0.25 ||
second.EffectiveModelParams.ExtraParams["nested"].(map[string]any)["source"] != "original" {
t.Fatalf("details target changed: %+v", second.EffectiveModelParams)
}
schema := second.StructuredOutput.JSONSchema.Schema.(map[string]any)
if schema["title"] != "original root" {
t.Fatalf("details schema changed: %#v", schema)
}
result, err := engine.RunPrepared(context.Background(), prepared)
if err != nil {
t.Fatalf("run prepared after preparation-context cancellation: %v", err)
}
if result.Validation.Status != promptkit.ValidationPassed || !result.Validation.IsValid {
t.Fatalf("frozen schema did not validate original output: %+v", result.Validation)
}
if reader.callCount() != 1 {
t.Fatalf("execution reopened artifact source: calls=%d", reader.callCount())
}
requests := client.snapshot()
if len(requests) != 1 {
t.Fatalf("generation calls=%d, want 1", len(requests))
}
generated := requests[0]
if generated.Prompt.Messages[0].Content != second.Messages[0].Content ||
generated.Target.Model != second.EffectiveModelParams.Model ||
!reflect.DeepEqual(generated.Target.ExtraParams, second.EffectiveModelParams.ExtraParams) ||
!reflect.DeepEqual(generated.StructuredOutput, second.StructuredOutput) {
t.Fatalf("generation did not use frozen details:\nrequest=%+v\ndetails=%+v", generated, second)
}
if result.PromptID != second.PromptID ||
result.PromptVersion != second.PromptVersion ||
result.PromptHash != second.PromptHash ||
result.SessionID != second.SessionID ||
result.RenderedPromptHash != second.RenderedPromptHash ||
result.SelectedProfileID != second.SelectedProfileID ||
result.SelectedBackendID != second.SelectedBackendID ||
!reflect.DeepEqual(result.EffectiveModelParams, second.EffectiveModelParams) ||
!reflect.DeepEqual(result.InputHashes, second.InputHashes) {
t.Fatalf("result provenance does not match details:\nresult=%+v\ndetails=%+v", result, second)
}
}
func TestPreparedExecutionLifecycleAndEngineBinding(t *testing.T) {
ownerClient := &preparedRecordingClient{
response: &promptkit.GenerateResponse{Content: "ok"},
}
owner := newPreparedContractEngine(t, ownerClient, "owner content")
foreign := newPreparedContractEngine(t, &preparedRecordingClient{
response: &promptkit.GenerateResponse{Content: "unexpected"},
}, "foreign content")
prepared, err := owner.PrepareExecution(context.Background(), promptkit.RunRequest{PromptID: "prepared"})
if err != nil {
t.Fatalf("prepare execution: %v", err)
}
copied := *prepared
var nilEngine *promptkit.Engine
if result, err := nilEngine.RunPrepared(context.Background(), prepared); result != nil ||
!errors.Is(err, promptkit.ErrInvalidConfig) {
t.Fatalf("nil engine result=(%+v, %v), want ErrInvalidConfig", result, err)
}
if result, err := foreign.RunPrepared(context.Background(), prepared); result != nil ||
!errors.Is(err, promptkit.ErrInvalidRequest) {
t.Fatalf("foreign engine result=(%+v, %v), want ErrInvalidRequest", result, err)
}
if result, err := owner.RunPrepared(context.Background(), nil); result != nil ||
!errors.Is(err, promptkit.ErrInvalidRequest) {
t.Fatalf("nil handle result=(%+v, %v), want ErrInvalidRequest", result, err)
}
if result, err := owner.RunPrepared(context.Background(), &promptkit.PreparedExecution{}); result != nil ||
!errors.Is(err, promptkit.ErrInvalidRequest) {
t.Fatalf("zero handle result=(%+v, %v), want ErrInvalidRequest", result, err)
}
result, err := owner.RunPrepared(context.Background(), &copied)
if err != nil || result == nil {
t.Fatalf("owner run prepared=(%+v, %v), want success", result, err)
}
for name, handle := range map[string]*promptkit.PreparedExecution{
"original": prepared,
"copy": &copied,
} {
if result, err := owner.RunPrepared(context.Background(), handle); result != nil ||
!errors.Is(err, promptkit.ErrInvalidRequest) {
t.Fatalf("%s reused handle result=(%+v, %v), want ErrInvalidRequest", name, result, err)
}
if handle.Details().PromptID != "prepared" {
t.Fatalf("%s details unavailable after execution", name)
}
}
if len(ownerClient.snapshot()) != 1 {
t.Fatalf("owner generation calls=%d, want 1", len(ownerClient.snapshot()))
}
collaboratorFailure := errors.New("prepared collaborator failure")
failingClient := &preparedRecordingClient{err: collaboratorFailure}
failingEngine := newPreparedContractEngine(t, failingClient, "failure content")
failing, err := failingEngine.PrepareExecution(context.Background(), promptkit.RunRequest{PromptID: "prepared"})
if err != nil {
t.Fatalf("prepare failing execution: %v", err)
}
if result, err := failingEngine.RunPrepared(context.Background(), failing); result != nil ||
!errors.Is(err, promptkit.ErrLLMGenerate) ||
!errors.Is(err, collaboratorFailure) {
t.Fatalf("generation failure result=(%+v, %v), want public and collaborator identities", result, err)
}
if result, err := failingEngine.RunPrepared(context.Background(), failing); result != nil ||
!errors.Is(err, promptkit.ErrInvalidRequest) {
t.Fatalf("failed execution was reusable: result=(%+v, %v)", result, err)
}
cancellationRelease := make(chan struct{})
cancellationStarted := make(chan struct{}, 1)
cancelingEngine := newPreparedContractEngine(t, &preparedRecordingClient{
response: &promptkit.GenerateResponse{Content: "unexpected"},
started: cancellationStarted,
release: cancellationRelease,
}, "cancellation content")
canceling, err := cancelingEngine.PrepareExecution(
context.Background(),
promptkit.RunRequest{PromptID: "prepared"},
)
if err != nil {
t.Fatalf("prepare canceled execution: %v", err)
}
executionContext, cancelExecution := context.WithCancel(context.Background())
type canceledOutcome struct {
result *promptkit.RunResult
err error
}
canceledResult := make(chan canceledOutcome, 1)
go func() {
result, runErr := cancelingEngine.RunPrepared(executionContext, canceling)
canceledResult <- canceledOutcome{result: result, err: runErr}
}()
select {
case <-cancellationStarted:
case <-time.After(5 * time.Second):
t.Fatal("timed out waiting for cancelable generation")
}
cancelExecution()
select {
case outcome := <-canceledResult:
if outcome.result != nil ||
!errors.Is(outcome.err, promptkit.ErrLLMGenerate) ||
!errors.Is(outcome.err, context.Canceled) {
t.Fatalf(
"canceled execution=(%+v, %v), want generation and context identities",
outcome.result,
outcome.err,
)
}
case <-time.After(5 * time.Second):
t.Fatal("timed out waiting for canceled execution")
}
if result, err := cancelingEngine.RunPrepared(context.Background(), canceling); result != nil ||
!errors.Is(err, promptkit.ErrInvalidRequest) {
t.Fatalf("canceled execution was reusable: result=(%+v, %v)", result, err)
}
}
func TestPreparedExecutionConcurrentClaimAllowsOneGeneration(t *testing.T) {
release := make(chan struct{})
client := &preparedRecordingClient{
response: &promptkit.GenerateResponse{Content: "ok"},
started: make(chan struct{}, 1),
release: release,
}
engine := newPreparedContractEngine(t, client, "concurrent content")
prepared, err := engine.PrepareExecution(context.Background(), promptkit.RunRequest{PromptID: "prepared"})
if err != nil {
t.Fatalf("prepare execution: %v", err)
}
type outcome struct {
result *promptkit.RunResult
err error
}
outcomes := make(chan outcome, 2)
for i := 0; i < 2; i++ {
go func() {
result, runErr := engine.RunPrepared(context.Background(), prepared)
outcomes <- outcome{result: result, err: runErr}
}()
}
select {
case <-client.started:
case <-time.After(5 * time.Second):
t.Fatal("timed out waiting for generation")
}
select {
case loser := <-outcomes:
if loser.result != nil || !errors.Is(loser.err, promptkit.ErrInvalidRequest) {
t.Fatalf("concurrent loser=(%+v, %v), want ErrInvalidRequest", loser.result, loser.err)
}
case <-time.After(5 * time.Second):
t.Fatal("timed out waiting for rejected concurrent claim")
}
close(release)
select {
case winner := <-outcomes:
if winner.err != nil || winner.result == nil {
t.Fatalf("concurrent winner=(%+v, %v), want success", winner.result, winner.err)
}
case <-time.After(5 * time.Second):
t.Fatal("timed out waiting for successful concurrent claim")
}
if len(client.snapshot()) != 1 {
t.Fatalf("generation calls=%d, want 1", len(client.snapshot()))
}
}
func TestPreparedExecutionRunAndDiscardRaceHasOneWinner(t *testing.T) {
const attempts = 32
for i := 0; i < attempts; i++ {
client := &preparedRecordingClient{
response: &promptkit.GenerateResponse{Content: "ok"},
}
engine := newPreparedContractEngine(t, client, "race content")
prepared, err := engine.PrepareExecution(
context.Background(),
promptkit.RunRequest{PromptID: "prepared"},
)
if err != nil {
t.Fatalf("attempt %d prepare execution: %v", i, err)
}
start := make(chan struct{})
type outcome struct {
result *promptkit.RunResult
err error
}
runOutcome := make(chan outcome, 1)
discardDone := make(chan struct{})
go func() {
<-start
result, runErr := engine.RunPrepared(context.Background(), prepared)
runOutcome <- outcome{result: result, err: runErr}
}()
go func() {
<-start
prepared.Discard()
close(discardDone)
}()
close(start)
runResult := <-runOutcome
<-discardDone
calls := len(client.snapshot())
switch {
case runResult.err == nil:
if runResult.result == nil || calls != 1 {
t.Fatalf(
"attempt %d run won with outcome=(%+v, %v), generation calls=%d",
i,
runResult.result,
runResult.err,
calls,
)
}
case errors.Is(runResult.err, promptkit.ErrInvalidRequest):
if runResult.result != nil || calls != 0 {
t.Fatalf(
"attempt %d discard won with outcome=(%+v, %v), generation calls=%d",
i,
runResult.result,
runResult.err,
calls,
)
}
default:
t.Fatalf("attempt %d unexpected run outcome=(%+v, %v)", i, runResult.result, runResult.err)
}
if prepared.Details().PromptID != "prepared" {
t.Fatalf("attempt %d details unavailable after race", i)
}
}
}
func TestPreparedExecutionDiscardAndFormattingDoNotExposePrivateState(t *testing.T) {
const (
directCredential = "pk-test-direct-credential-41f7"
renderedContent = "rendered-content-sentinel-98d2"
)
client := &preparedRecordingClient{
response: &promptkit.GenerateResponse{Content: "generated output"},
}
engine, err := promptkit.NewEngine(
promptkit.Config{},
promptkit.WithPromptFS(contractPromptFS("prepared", "profile", renderedContent), "."),
promptkit.WithProfiles(promptkit.Profile{
ID: "profile",
Endpoint: "http://example.test/v1",
Model: "model",
APIKeyRequired: true,
}),
promptkit.WithLLMClient(client),
)
if err != nil {
t.Fatalf("construct engine: %v", err)
}
prepared, err := engine.PrepareExecution(context.Background(), promptkit.RunRequest{
PromptID: "prepared",
APIKey: directCredential,
})
if err != nil {
t.Fatalf("prepare execution: %v", err)
}
formattedValues := []string{
fmt.Sprint(prepared),
fmt.Sprintf("%+v", prepared),
fmt.Sprintf("%#v", prepared),
}
for _, formatted := range formattedValues {
if formatted != "promptkit.PreparedExecution{opaque}" {
t.Fatalf("unexpected opaque formatting: %q", formatted)
}
assertPreparedPrivateValuesAbsent(t, formatted, directCredential, renderedContent)
}
payload, err := json.Marshal(prepared)
if err != nil {
t.Fatalf("marshal opaque handle: %v", err)
}
assertPreparedPrivateValuesAbsent(t, string(payload), directCredential, renderedContent)
detailsBefore := prepared.Details()
detailsJSON, err := json.Marshal(detailsBefore)
if err != nil {
t.Fatalf("marshal prepared details: %v", err)
}
assertPreparedPrivateValuesAbsent(t, string(detailsJSON), directCredential)
prepared.Discard()
prepared.Discard()
result, lifecycleErr := engine.RunPrepared(context.Background(), prepared)
if result != nil || !errors.Is(lifecycleErr, promptkit.ErrInvalidRequest) {
t.Fatalf("discarded execution result=(%+v, %v), want ErrInvalidRequest", result, lifecycleErr)
}
assertPreparedPrivateValuesAbsent(t, lifecycleErr.Error(), directCredential, renderedContent)
if !reflect.DeepEqual(prepared.Details(), detailsBefore) {
t.Fatal("details changed after discard")
}
executed, err := engine.PrepareExecution(context.Background(), promptkit.RunRequest{
PromptID: "prepared",
APIKey: directCredential,
})
if err != nil {
t.Fatalf("prepare execution for request inspection: %v", err)
}
executionResult, err := engine.RunPrepared(context.Background(), executed)
if err != nil {
t.Fatalf("run execution for request inspection: %v", err)
}
requests := client.snapshot()
if len(requests) != 1 || requests[0].APIKey != directCredential {
t.Fatalf("direct credential did not reach only the client credential field: %#v", requests)
}
requestJSON, err := json.Marshal(requests[0])
if err != nil {
t.Fatalf("marshal captured generate request: %v", err)
}
for _, value := range []string{
fmt.Sprint(requests[0]),
fmt.Sprintf("%+v", requests[0]),
fmt.Sprintf("%#v", requests[0]),
string(requestJSON),
fmt.Sprint(executionResult),
} {
assertPreparedPrivateValuesAbsent(t, value, directCredential)
}
resultJSON, err := json.Marshal(executionResult)
if err != nil {
t.Fatalf("marshal execution result: %v", err)
}
assertPreparedPrivateValuesAbsent(t, string(resultJSON), directCredential)
var nilHandle *promptkit.PreparedExecution
nilHandle.Discard()
if !reflect.DeepEqual(nilHandle.Details(), promptkit.PreparedRun{}) {
t.Fatalf("nil handle details=%+v, want zero value", nilHandle.Details())
}
zeroHandle := &promptkit.PreparedExecution{}
zeroHandle.Discard()
if !reflect.DeepEqual(zeroHandle.Details(), promptkit.PreparedRun{}) {
t.Fatalf("zero handle details=%+v, want zero value", zeroHandle.Details())
}
}
func TestPreparedExecutionCredentialCapacityAndTimingBoundaries(t *testing.T) {
t.Run("credential is rechecked before generation", func(t *testing.T) {
const (
environmentName = "PROMPTKIT_PREPARED_CONTRACT_KEY"
environmentKey = "environment-credential-sentinel"
)
t.Setenv(environmentName, environmentKey)
client := &preparedRecordingClient{
response: &promptkit.GenerateResponse{Content: "unexpected"},
}
engine, err := promptkit.NewEngine(
promptkit.Config{},
promptkit.WithPromptFS(contractPromptFS("prepared", "profile", "content"), "."),
promptkit.WithProfileFS(preparedCredentialProfileSource(environmentName), "."),
promptkit.WithLLMClient(client),
)
if err != nil {
t.Fatalf("construct credential engine: %v", err)
}
prepared, err := engine.PrepareExecution(context.Background(), promptkit.RunRequest{PromptID: "prepared"})
if err != nil {
t.Fatalf("prepare credential execution: %v", err)
}
if err := os.Unsetenv(environmentName); err != nil {
t.Fatalf("unset credential environment: %v", err)
}
result, err := engine.RunPrepared(context.Background(), prepared)
if result != nil ||
!errors.Is(err, promptkit.ErrInvalidRequest) ||
!errors.Is(err, promptkit.ErrAPIKeyEnvMissing) {
t.Fatalf("credential execution=(%+v, %v), want credential identities", result, err)
}
if len(client.snapshot()) != 0 {
t.Fatalf("credential failure reached generation: %d calls", len(client.snapshot()))
}
assertPreparedPrivateValuesAbsent(t, err.Error(), environmentKey)
if result, err := engine.RunPrepared(context.Background(), prepared); result != nil ||
!errors.Is(err, promptkit.ErrInvalidRequest) {
t.Fatalf("credential failure did not consume handle: result=(%+v, %v)", result, err)
}
})
t.Run("preparation does not admit and execution timing starts after retention", func(t *testing.T) {
release := make(chan struct{})
client := newCapacityGateClient(release, 4)
engine := newBackendCapacityEngine(t, client, 1, capacityInt(0), nil)
activeRun := make(chan capacityRunResult, 1)
go runCapacityRequest(
engine,
context.Background(),
promptkit.RunRequest{PromptID: "prompt"},
activeRun,
)
awaitCapacityRequest(t, client.started)
prepared, err := engine.PrepareExecution(
context.Background(),
promptkit.RunRequest{PromptID: "prompt"},
)
if err != nil {
t.Fatalf("prepare while capacity is full: %v", err)
}
if _, _, calls := client.snapshot(); calls != 1 {
t.Fatalf("preparation invoked generation: calls=%d", calls)
}
if result, err := engine.RunPrepared(context.Background(), prepared); result != nil ||
!errors.Is(err, promptkit.ErrCapacityExceeded) {
t.Fatalf("capacity execution=(%+v, %v), want ErrCapacityExceeded", result, err)
} else {
var capacityErr *promptkit.CapacityError
if !errors.As(err, &capacityErr) || capacityErr == nil || capacityErr.BackendID != "limited" {
t.Fatalf("capacity execution=%v, want limited CapacityError", err)
}
}
if result, err := engine.RunPrepared(context.Background(), prepared); result != nil ||
!errors.Is(err, promptkit.ErrInvalidRequest) {
t.Fatalf("capacity rejection did not consume handle: result=(%+v, %v)", result, err)
}
if prepared.Details().PromptID != "prompt" {
t.Fatal("details unavailable after capacity rejection")
}
close(release)
activeOutcome := awaitCapacityRun(t, activeRun)
if activeOutcome.err != nil || activeOutcome.result == nil {
t.Fatalf("active run outcome=(%+v, %v), want success", activeOutcome.result, activeOutcome.err)
}
timed, err := engine.PrepareExecution(
context.Background(),
promptkit.RunRequest{PromptID: "prompt"},
)
if err != nil {
t.Fatalf("prepare timed execution: %v", err)
}
details := timed.Details()
time.Sleep(25 * time.Millisecond)
executionFloor := time.Now().UTC()
result, err := engine.RunPrepared(context.Background(), timed)
if err != nil {
t.Fatalf("run timed execution: %v", err)
}
if result.StartTime.Before(executionFloor) ||
!result.StartTime.After(details.EndTime) ||
result.EndTime.Before(result.StartTime) ||
result.Duration != result.EndTime.Sub(result.StartTime) {
t.Fatalf(
"execution timing includes preparation or retention: details_end=%s floor=%s result=%+v",
details.EndTime,
executionFloor,
result,
)
}
if _, _, calls := client.snapshot(); calls != 2 {
t.Fatalf("generation calls=%d, want active and timed executions only", calls)
}
})
}
type mutablePreparedArtifactReader struct {
mu sync.Mutex
body string
hash string
calls int
}
func (r *mutablePreparedArtifactReader) Read(
_ context.Context,
_ promptkit.ArtifactRef,
) (*promptkit.Artifact, error) {
r.mu.Lock()
defer r.mu.Unlock()
r.calls++
return &promptkit.Artifact{
Body: []byte(r.body),
Hash: r.hash,
}, nil
}
func (r *mutablePreparedArtifactReader) set(body, hash string) {
r.mu.Lock()
defer r.mu.Unlock()
r.body = body
r.hash = hash
}
func (r *mutablePreparedArtifactReader) callCount() int {
r.mu.Lock()
defer r.mu.Unlock()
return r.calls
}
type preparedRecordingClient struct {
mu sync.Mutex
response *promptkit.GenerateResponse
err error
requests []promptkit.GenerateRequest
started chan struct{}
release <-chan struct{}
}
func (c *preparedRecordingClient) Generate(
ctx context.Context,
request promptkit.GenerateRequest,
) (*promptkit.GenerateResponse, error) {
c.mu.Lock()
c.requests = append(c.requests, request)
c.mu.Unlock()
if c.started != nil {
c.started <- struct{}{}
}
if c.release != nil {
select {
case <-c.release:
case <-ctx.Done():
return nil, ctx.Err()
}
}
if c.err != nil {
return nil, c.err
}
return c.response, nil
}
func (c *preparedRecordingClient) snapshot() []promptkit.GenerateRequest {
c.mu.Lock()
defer c.mu.Unlock()
return append([]promptkit.GenerateRequest(nil), c.requests...)
}
func newPreparedContractEngine(
t *testing.T,
client promptkit.LLMClient,
message string,
) *promptkit.Engine {
t.Helper()
engine, err := promptkit.NewEngine(
promptkit.Config{},
promptkit.WithPromptFS(contractPromptFS("prepared", "profile", message), "."),
promptkit.WithProfiles(promptkit.Profile{
ID: "profile",
Endpoint: "http://example.test/v1",
Model: "model",
}),
promptkit.WithLLMClient(client),
)
if err != nil {
t.Fatalf("construct prepared execution engine: %v", err)
}
return engine
}
func preparedPromptSource(label string) fstest.MapFS {
return fstest.MapFS{
"prompt.yaml": &fstest.MapFile{Data: []byte(`id: prepared
version: "1"
default_profile: profile
inputs:
- name: input
required: true
messages:
- role: user
content: 'Input={{input "input"}} Label={{.label}}'
description: ` + label + `
output:
format: json
validation_mode: json_schema
schema_path: schema.json
`)},
}
}
func preparedProfileSource(model string) fstest.MapFS {
return fstest.MapFS{
"profile.yaml": &fstest.MapFile{Data: []byte(`id: profile
endpoint: http://example.test/v1
model: ` + model + `
`)},
}
}
func preparedCredentialProfileSource(environmentName string) fstest.MapFS {
return fstest.MapFS{
"profile.yaml": &fstest.MapFile{Data: []byte(`id: profile
endpoint: http://example.test/v1
model: model
api_key_env: ` + environmentName + `
`)},
}
}
func preparedSchemaSource() fstest.MapFS {
return fstest.MapFS{
"schema.json": &fstest.MapFile{Data: []byte(`{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"title": "original root",
"type": "object",
"required": ["value"],
"properties": {
"value": {"$ref": "value.json"}
}
}`)},
"value.json": &fstest.MapFile{Data: []byte(`{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"type": "integer",
"minimum": 2
}`)},
}
}
func assertPreparedPrivateValuesAbsent(t *testing.T, value string, privateValues ...string) {
t.Helper()
for _, privateValue := range privateValues {
if strings.Contains(value, privateValue) {
t.Fatalf("value exposed private data %q: %s", privateValue, value)
}
}
}

View File

@@ -5,6 +5,7 @@ import (
"encoding/json" "encoding/json"
"errors" "errors"
"fmt" "fmt"
"io/fs"
"reflect" "reflect"
"strings" "strings"
"sync" "sync"
@@ -16,6 +17,380 @@ import (
"gitea.maximumdirect.net/eric/promptkit" "gitea.maximumdirect.net/eric/promptkit"
) )
type inspectionCountingFS struct {
opens atomic.Int64
}
func (f *inspectionCountingFS) Open(string) (fs.File, error) {
f.opens.Add(1)
return nil, fs.ErrNotExist
}
func TestInspectProfileResolvesCredentialStatesWithoutPromptOrGeneration(t *testing.T) {
const environmentName = "PROMPTKIT_INSPECTION_ABSENT_KEY"
t.Setenv(environmentName, "")
client := &fakeLLMClient{response: &promptkit.GenerateResponse{Content: "unexpected"}}
engine, err := promptkit.NewEngine(promptkit.Config{},
promptkit.WithPromptFS(fstest.MapFS{}, "."),
promptkit.WithProfileFS(fstest.MapFS{
"environment.yaml": &fstest.MapFile{Data: []byte(`id: environment
endpoint: http://environment.example/v1
model: environment-model
api_key_env: PROMPTKIT_INSPECTION_ABSENT_KEY
`)},
}, "."),
promptkit.WithProfiles(
promptkit.Profile{ID: "direct", Endpoint: "http://direct.example/v1", Model: "direct-model", APIKeyRequired: true},
promptkit.Profile{ID: "none", Endpoint: "http://none.example/v1", Model: "none-model"},
),
promptkit.WithLLMClient(client),
)
if err != nil {
t.Fatalf("construct inspection engine: %v", err)
}
for _, tc := range []struct {
profileID string
wantEnv string
wantDirectKey bool
wantEndpoint string
}{
{profileID: " environment ", wantEnv: environmentName, wantEndpoint: "http://environment.example/v1"},
{profileID: "direct", wantDirectKey: true, wantEndpoint: "http://direct.example/v1"},
{profileID: "none", wantEndpoint: "http://none.example/v1"},
} {
t.Run(tc.profileID, func(t *testing.T) {
inspection, err := engine.InspectProfile(context.Background(), tc.profileID)
if err != nil {
t.Fatalf("inspect profile: %v", err)
}
if inspection.ProfileID != strings.TrimSpace(tc.profileID) ||
inspection.EffectiveModelParams.Endpoint != tc.wantEndpoint ||
inspection.EffectiveModelParams.BackendID != "" ||
inspection.EffectiveModelParams.APIKeyEnv != tc.wantEnv ||
inspection.APIKeyRequired != tc.wantDirectKey {
t.Fatalf("unexpected inspection: %#v", inspection)
}
})
}
if len(client.requests) != 0 {
t.Fatalf("inspection invoked the model client %d times", len(client.requests))
}
}
func TestInspectProfilePreservesPublicErrorIdentities(t *testing.T) {
newEngine := func(t *testing.T, options ...promptkit.Option) *promptkit.Engine {
t.Helper()
engine, err := promptkit.NewEngine(
promptkit.Config{},
append([]promptkit.Option{promptkit.WithPromptFS(fstest.MapFS{}, ".")}, options...)...,
)
if err != nil {
t.Fatalf("construct inspection engine: %v", err)
}
return engine
}
var nilEngine *promptkit.Engine
if result, err := nilEngine.InspectProfile(context.Background(), "profile"); result != nil ||
!errors.Is(err, promptkit.ErrInvalidConfig) {
t.Fatalf("nil engine result=(%#v, %v), want ErrInvalidConfig", result, err)
}
valid := newEngine(t, promptkit.WithProfiles(promptkit.Profile{
ID: "profile", Endpoint: "http://profile.example/v1", Model: "model",
}))
if result, err := valid.InspectProfile(context.Background(), " \t "); result != nil ||
!errors.Is(err, promptkit.ErrInvalidRequest) {
t.Fatalf("blank profile result=(%#v, %v), want ErrInvalidRequest", result, err)
}
if result, err := valid.InspectProfile(context.Background(), "missing"); result != nil ||
!errors.Is(err, promptkit.ErrProfileNotFound) || errors.Is(err, promptkit.ErrProfileLoad) {
t.Fatalf("missing profile result=(%#v, %v), want only ErrProfileNotFound", result, err)
}
malformed := newEngine(t, promptkit.WithProfileFS(fstest.MapFS{
"broken.yaml": &fstest.MapFile{Data: []byte("id: broken\nendpoint: http://broken.example/v1\nmodel: model\nunknown: value\n")},
}, "."))
if result, err := malformed.InspectProfile(context.Background(), "broken"); result != nil ||
!errors.Is(err, promptkit.ErrProfileLoad) {
t.Fatalf("malformed profile result=(%#v, %v), want ErrProfileLoad", result, err)
}
unknownBackend := newEngine(t, promptkit.WithProfiles(promptkit.Profile{
ID: "unknown-backend", BackendID: "unknown", Model: "model",
}))
if result, err := unknownBackend.InspectProfile(context.Background(), "unknown-backend"); result != nil ||
!errors.Is(err, promptkit.ErrProfileLoad) {
t.Fatalf("unknown backend result=(%#v, %v), want ErrProfileLoad", result, err)
}
countingFS := &inspectionCountingFS{}
canceled := newEngine(t, promptkit.WithProfileFS(countingFS, "."))
ctx, cancel := context.WithCancel(context.Background())
cancel()
if result, err := canceled.InspectProfile(ctx, "profile"); result != nil ||
!errors.Is(err, promptkit.ErrProfileLoad) || !errors.Is(err, context.Canceled) || countingFS.opens.Load() != 0 {
t.Fatalf("canceled inspection result=(%#v, %v), opens=%d", result, err, countingFS.opens.Load())
}
}
func TestInspectProfileReturnsIndependentTargetMatchingPreparation(t *testing.T) {
extraParams := map[string]any{
"nested": map[string]any{"value": "original"},
}
engine, err := promptkit.NewEngine(promptkit.Config{},
promptkit.WithPromptFS(contractPromptFS("prompt", "profile", "message"), "."),
promptkit.WithProfiles(promptkit.Profile{
ID: "profile", Endpoint: "http://profile.example/v1", Model: "model", ExtraParams: extraParams,
}),
)
if err != nil {
t.Fatalf("construct inspection engine: %v", err)
}
first, err := engine.InspectProfile(context.Background(), "profile")
if err != nil {
t.Fatalf("first inspection: %v", err)
}
first.EffectiveModelParams.ExtraParams["nested"].(map[string]any)["value"] = "changed"
first.EffectiveModelParams.ExtraParams["later"] = true
second, err := engine.InspectProfile(context.Background(), "profile")
if err != nil {
t.Fatalf("second inspection: %v", err)
}
prepared, err := engine.Prepare(context.Background(), promptkit.RunRequest{PromptID: "prompt"})
if err != nil {
t.Fatalf("prepare after inspection mutation: %v", err)
}
for _, target := range []promptkit.ExecutionTarget{second.EffectiveModelParams, prepared.EffectiveModelParams} {
if target.ExtraParams["nested"].(map[string]any)["value"] != "original" || target.ExtraParams["later"] != nil {
t.Fatalf("inspection mutation reached engine-owned target: %#v", target.ExtraParams)
}
}
if !reflect.DeepEqual(second.EffectiveModelParams, prepared.EffectiveModelParams) {
t.Fatalf("inspection target=%#v, preparation target=%#v", second.EffectiveModelParams, prepared.EffectiveModelParams)
}
}
func TestInspectPromptReturnsDeclaredMetadataWithoutExecutionWork(t *testing.T) {
client := &fakeLLMClient{response: &promptkit.GenerateResponse{Content: "unexpected"}}
engine, err := promptkit.NewEngine(promptkit.Config{},
promptkit.WithPromptFS(fstest.MapFS{
"report-v1.yaml": &fstest.MapFile{Data: []byte(`id: report
version: "1.0.0"
messages:
- role: user
content: old report
output:
format: text
validation_mode: none
`)},
"report-v2.yaml": &fstest.MapFile{Data: []byte(`id: report
version: "2.0.0"
default_profile: missing-profile
inputs:
- name: location
required: true
content_type: text/plain
description: Forecast location.
- name: units
content_type: text/plain
description: Unit preference.
messages:
- role: user
content_file: messages/report.md
output:
format: json
validation_mode: json_schema
schema_path: schemas/report.json
`)},
"messages/report.md": &fstest.MapFile{Data: []byte("rendered report body is not returned")},
}, "."),
promptkit.WithLLMClient(client),
)
if err != nil {
t.Fatalf("construct prompt-inspection engine: %v", err)
}
inspection, err := engine.InspectPrompt(context.Background(), "report", "2.0.0")
if err != nil {
t.Fatalf("inspect prompt: %v", err)
}
if inspection.PromptID != "report" ||
inspection.PromptVersion != "2.0.0" ||
inspection.PromptHash == "" ||
inspection.DefaultProfileID != "missing-profile" ||
inspection.OutputContract != (promptkit.OutputContract{
Format: promptkit.FormatJSON,
ValidationMode: promptkit.ValidationJSONSchema,
SchemaPath: "schemas/report.json",
}) {
t.Fatalf("unexpected inspection metadata: %#v", inspection)
}
wantInputs := []promptkit.PromptInputDefinition{
{Name: "location", Required: true, ContentType: "text/plain", Description: "Forecast location."},
{Name: "units", ContentType: "text/plain", Description: "Unit preference."},
}
if !reflect.DeepEqual(inspection.Inputs, wantInputs) {
t.Fatalf("inspection inputs=%#v, want %#v", inspection.Inputs, wantInputs)
}
if len(client.requests) != 0 {
t.Fatalf("inspection invoked the model client %d times", len(client.requests))
}
}
func TestInspectPromptPreservesPublicErrorIdentities(t *testing.T) {
newEngine := func(t *testing.T, source fstest.MapFS) *promptkit.Engine {
t.Helper()
engine, err := promptkit.NewEngine(promptkit.Config{}, promptkit.WithPromptFS(source, "."))
if err != nil {
t.Fatalf("construct prompt-inspection engine: %v", err)
}
return engine
}
validSource := fstest.MapFS{
"prompt.yaml": &fstest.MapFile{Data: []byte(`id: prompt
version: "1"
messages:
- role: user
content: body
output:
format: text
validation_mode: none
`)},
}
var nilEngine *promptkit.Engine
if result, err := nilEngine.InspectPrompt(context.Background(), "prompt", "1"); result != nil ||
!errors.Is(err, promptkit.ErrInvalidConfig) {
t.Fatalf("nil engine result=(%#v, %v), want ErrInvalidConfig", result, err)
}
valid := newEngine(t, validSource)
if result, err := valid.InspectPrompt(context.Background(), " \t ", "1"); result != nil ||
!errors.Is(err, promptkit.ErrInvalidRequest) {
t.Fatalf("blank prompt result=(%#v, %v), want ErrInvalidRequest", result, err)
}
if result, err := valid.InspectPrompt(context.Background(), "missing", "1"); result != nil ||
!errors.Is(err, promptkit.ErrPromptNotFound) || errors.Is(err, promptkit.ErrPromptLoad) {
t.Fatalf("missing prompt result=(%#v, %v), want only ErrPromptNotFound", result, err)
}
if result, err := valid.InspectPrompt(context.Background(), "prompt", "missing"); result != nil ||
!errors.Is(err, promptkit.ErrPromptNotFound) || errors.Is(err, promptkit.ErrPromptLoad) {
t.Fatalf("missing version result=(%#v, %v), want only ErrPromptNotFound", result, err)
}
ambiguous := newEngine(t, fstest.MapFS{
"one.yaml": &fstest.MapFile{Data: []byte(`id: prompt
version: "1"
messages:
- role: user
content: first
output:
format: text
validation_mode: none
`)},
"two.yaml": &fstest.MapFile{Data: []byte(`id: prompt
version: "2"
messages:
- role: user
content: second
output:
format: text
validation_mode: none
`)},
})
if result, err := ambiguous.InspectPrompt(context.Background(), "prompt", ""); result != nil ||
!errors.Is(err, promptkit.ErrPromptLoad) {
t.Fatalf("ambiguous prompt result=(%#v, %v), want ErrPromptLoad", result, err)
}
for name, source := range map[string]fstest.MapFS{
"malformed definition": {
"broken.yaml": &fstest.MapFile{Data: []byte("id: broken\nversion: \"1\"\nunknown: value\n")},
},
"missing content file": {
"broken.yaml": &fstest.MapFile{Data: []byte(`id: broken
version: "1"
messages:
- role: user
content_file: missing.md
output:
format: text
validation_mode: none
`)},
},
} {
t.Run(name, func(t *testing.T) {
if result, err := newEngine(t, source).InspectPrompt(context.Background(), "broken", "1"); result != nil ||
!errors.Is(err, promptkit.ErrPromptLoad) {
t.Fatalf("broken prompt result=(%#v, %v), want ErrPromptLoad", result, err)
}
})
}
countingFS := &inspectionCountingFS{}
canceled, err := promptkit.NewEngine(promptkit.Config{}, promptkit.WithPromptFS(countingFS, "."))
if err != nil {
t.Fatalf("construct canceled prompt-inspection engine: %v", err)
}
ctx, cancel := context.WithCancel(context.Background())
cancel()
if result, err := canceled.InspectPrompt(ctx, "prompt", "1"); result != nil ||
!errors.Is(err, promptkit.ErrPromptLoad) || !errors.Is(err, context.Canceled) || countingFS.opens.Load() != 0 {
t.Fatalf("canceled inspection result=(%#v, %v), opens=%d", result, err, countingFS.opens.Load())
}
}
func TestInspectPromptReturnsIndependentMetadataMatchingPreparation(t *testing.T) {
engine, err := promptkit.NewEngine(promptkit.Config{},
promptkit.WithPromptFS(fstest.MapFS{
"prompt.yaml": &fstest.MapFile{Data: []byte(`id: prompt
version: "1"
default_profile: profile
inputs:
- name: subject
content_type: text/plain
description: Summary subject.
messages:
- role: user
content: summarize
output:
format: markdown
validation_mode: basic
`)},
}, "."),
promptkit.WithProfiles(promptkit.Profile{
ID: "profile", Endpoint: "http://profile.example/v1", Model: "model",
}),
)
if err != nil {
t.Fatalf("construct prompt-inspection engine: %v", err)
}
first, err := engine.InspectPrompt(context.Background(), "prompt", "1")
if err != nil {
t.Fatalf("first inspection: %v", err)
}
first.Inputs[0].Name = "changed"
first.OutputContract.SchemaPath = "changed.json"
second, err := engine.InspectPrompt(context.Background(), "prompt", "1")
if err != nil {
t.Fatalf("second inspection: %v", err)
}
prepared, err := engine.Prepare(context.Background(), promptkit.RunRequest{PromptID: "prompt", PromptVersion: "1"})
if err != nil {
t.Fatalf("prepare after inspection mutation: %v", err)
}
if second.Inputs[0].Name != "subject" || second.OutputContract.SchemaPath != "" ||
prepared.OutputContract.SchemaPath != "" || second.PromptHash != prepared.PromptHash {
t.Fatalf("inspection mutation reached engine-owned prompt metadata: inspection=%#v prepared=%#v", second, prepared)
}
}
func TestPreparedRunJSONOmitsZeroTimingValues(t *testing.T) { func TestPreparedRunJSONOmitsZeroTimingValues(t *testing.T) {
payload, err := json.Marshal(promptkit.PreparedRun{}) payload, err := json.Marshal(promptkit.PreparedRun{})
if err != nil { if err != nil {

118
types.go
View File

@@ -51,7 +51,8 @@ const (
// ValidationPassed means the generated output satisfied its contract. // ValidationPassed means the generated output satisfied its contract.
ValidationPassed ValidationStatus = "passed" ValidationPassed ValidationStatus = "passed"
// ValidationFailed means validation completed and rejected the generated // ValidationFailed means validation completed and rejected the generated
// output. Engine.Run returns this status in a result, not as an error. // output. Engine.Run and Engine.RunPrepared return this status in a result,
// not as an error.
ValidationFailed ValidationStatus = "failed" ValidationFailed ValidationStatus = "failed"
// ValidationSkipped means ValidationNone selected no content check. // ValidationSkipped means ValidationNone selected no content check.
ValidationSkipped ValidationStatus = "skipped" ValidationSkipped ValidationStatus = "skipped"
@@ -78,9 +79,10 @@ const (
// RunRequest selects one prompt execution. It has no stable JSON // RunRequest selects one prompt execution. It has no stable JSON
// representation. // representation.
// //
// Prepare and Run copy the request's maps, pointers, and nested // Prepare, PrepareExecution, and Run copy the request's maps, pointers, and
// JSON-compatible values before using them. The caller may mutate the request // nested JSON-compatible values before using them. The caller may mutate the
// after either method returns. // request after any method returns. A successful PrepareExecution retains its
// own private execution snapshot for RunPrepared.
type RunRequest struct { type RunRequest struct {
// PromptID is the required non-empty prompt identifier. // PromptID is the required non-empty prompt identifier.
PromptID string PromptID string
@@ -98,12 +100,14 @@ type RunRequest struct {
// opaque consumer metadata, not a credential, and may be exposed in // opaque consumer metadata, not a credential, and may be exposed in
// prepared values, results, collaborator requests, provider requests, and // prepared values, results, collaborator requests, provider requests, and
// provider observability. Callers should use stable, non-sensitive // provider observability. Callers should use stable, non-sensitive
// identifiers. An overlong direct value makes Prepare or Run return an // identifiers. An overlong direct value makes Prepare, PrepareExecution, or
// error matching ErrInvalidRequest. // Run return an error matching ErrInvalidRequest.
SessionID string SessionID string
// APIKey is a request-scoped direct credential. It takes precedence over // APIKey is a request-scoped direct credential. It takes precedence over
// APIKeyEnv, is passed to the selected LLMClient, and is never included in // APIKeyEnv, is passed to the selected LLMClient, and is never included in
// prepared values, results, hashes, JSON, String, or GoString output. // prepared values, results, hashes, JSON, String, or GoString output. A
// successful PrepareExecution retains it only in the opaque handle until
// RunPrepared claims the handle or Discard invalidates it.
APIKey string `json:"-"` APIKey string `json:"-"`
// Inputs maps prompt input names to references. A nil or empty map is valid // Inputs maps prompt input names to references. A nil or empty map is valid
// only when the selected prompt and its templates require no inputs. // only when the selected prompt and its templates require no inputs.
@@ -119,9 +123,10 @@ type RunRequest struct {
Validation *OutputContract Validation *OutputContract
} }
// PreparedRun contains prepared prompt execution state. It does not include // PreparedRun contains prepared prompt execution state returned by
// resolved API key values, model output, validation results, or internal target // [Engine.Prepare] or [PreparedExecution.Details]. It does not include resolved
// presence metadata. PreparedRun has a stable JSON representation. // API key values, model output, validation results, or internal target presence
// metadata. PreparedRun has a stable JSON representation.
// //
// All maps, slices, pointers, and schema values are caller-owned copies. JSON // All maps, slices, pointers, and schema values are caller-owned copies. JSON
// timestamps use RFC 3339 and zero timing values are omitted. Hash formats are // timestamps use RFC 3339 and zero timing values are omitted. Hash formats are
@@ -154,7 +159,8 @@ type PreparedRun struct {
SessionID string `json:"session_id,omitempty"` SessionID string `json:"session_id,omitempty"`
// RenderedPromptHash is an opaque equality value for SessionID and Messages. // RenderedPromptHash is an opaque equality value for SessionID and Messages.
RenderedPromptHash string `json:"rendered_prompt_hash"` RenderedPromptHash string `json:"rendered_prompt_hash"`
// Messages are the rendered messages that Run passes to the LLM client. // Messages are the rendered messages that Run or RunPrepared passes to the
// LLM client.
Messages []RenderedMessage `json:"messages"` Messages []RenderedMessage `json:"messages"`
// StartTime is the UTC time at which preparation began. // StartTime is the UTC time at which preparation began.
StartTime time.Time `json:"start_time,omitempty"` StartTime time.Time `json:"start_time,omitempty"`
@@ -214,12 +220,15 @@ type RunResult struct {
InputHashes map[string]string `json:"input_hashes,omitempty"` InputHashes map[string]string `json:"input_hashes,omitempty"`
// Usage is the token accounting reported by the LLM client. // Usage is the token accounting reported by the LLM client.
Usage TokenUsage `json:"usage"` Usage TokenUsage `json:"usage"`
// StartTime is the UTC time immediately before preparation begins. // StartTime is the UTC time immediately before ordinary Run preparation or
// after RunPrepared claims its handle.
StartTime time.Time `json:"start_time,omitempty"` StartTime time.Time `json:"start_time,omitempty"`
// EndTime is the UTC time after generation and validation complete. // EndTime is the UTC time after generation and validation complete.
EndTime time.Time `json:"end_time,omitempty"` EndTime time.Time `json:"end_time,omitempty"`
// Duration covers preparation, generation, and validation. JSON represents // Duration covers preparation, generation, and validation for Run. For
// it as integer milliseconds in duration_ms and omits a zero value. // RunPrepared it covers only the execution attempt after claim and excludes
// preparation and consumer-held delay. JSON represents it as integer
// milliseconds in duration_ms and omits a zero value.
Duration time.Duration `json:"-"` Duration time.Duration `json:"-"`
} }
@@ -259,10 +268,11 @@ type Artifact struct {
// ArtifactReader resolves a prompt input reference into its content. // ArtifactReader resolves a prompt input reference into its content.
// //
// Read may be called concurrently. It must honor ctx cancellation to make // Read may be called concurrently. It must honor ctx cancellation to make
// Prepare and Run responsive to cancellation. The engine passes a copied ref // Prepare, PrepareExecution, and Run responsive to cancellation. The engine
// and immediately copies the returned Artifact.Body; it does not retain either // passes a copied ref and immediately copies the returned Artifact.Body; it
// value. Readers supply artifact metadata, and the engine assigns an input-map // does not retain either value. Readers supply artifact metadata, and the
// name only when the returned artifact name is empty. // engine assigns an input-map name only when the returned artifact name is
// empty.
// //
// An injected reader owns any application-specific path containment, // An injected reader owns any application-specific path containment,
// authorization, content-size, and content-type policy. It must protect // authorization, content-size, and content-type policy. It must protect
@@ -310,6 +320,61 @@ type ExecutionTarget struct {
ExtraParams map[string]any `json:"extra_params"` ExtraParams map[string]any `json:"extra_params"`
} }
// ProfileInspection is the caller-owned result of [Engine.InspectProfile].
// It has no stable JSON representation.
//
// EffectiveModelParams contains a copied effective target. APIKeyRequired is
// separate from that target to preserve ExecutionTarget's general execution
// and stable JSON contracts.
type ProfileInspection struct {
// ProfileID is the trimmed, exact profile ID inspected by the engine.
ProfileID string
// EffectiveModelParams contains framework defaults overlaid by the selected
// backend and then the profile, without a request override. APIKeyEnv is an
// environment-variable name, never its credential value.
EffectiveModelParams ExecutionTarget
// APIKeyRequired reports that a later execution must supply a direct API
// key or an explicit request environment override. It is mutually exclusive
// with a nonblank EffectiveModelParams.APIKeyEnv.
APIKeyRequired bool
}
// PromptInputDefinition describes one declared prompt input.
// It has no stable JSON representation.
type PromptInputDefinition struct {
// Name is the normalized prompt input name.
Name string
// Required is the prompt definition's declared required flag. When true,
// preparation fails if the input is omitted. A false value does not account
// for input references in message or session-ID templates.
Required bool
// ContentType is the declared input media-type metadata.
ContentType string
// Description is the declared human-readable input description.
Description string
}
// PromptInspection is the caller-owned result of [Engine.InspectPrompt].
// It has no stable JSON representation.
//
// Inputs contains copied declared input metadata in definition order.
// OutputContract is the normalized contract declared by the prompt definition,
// rather than a request-level effective override. PromptHash is opaque.
type PromptInspection struct {
// PromptID is the normalized ID of the selected prompt definition.
PromptID string
// PromptVersion is the normalized version of the selected prompt definition.
PromptVersion string
// PromptHash is the opaque equality value for the selected definition.
PromptHash string
// DefaultProfileID is declared metadata and is not resolved by inspection.
DefaultProfileID string
// Inputs contains caller-owned declared input metadata in definition order.
Inputs []PromptInputDefinition
// OutputContract is the normalized contract declared by the definition.
OutputContract OutputContract
}
// ExecutionTargetOverride represents per-request runtime setting overrides and // ExecutionTargetOverride represents per-request runtime setting overrides and
// has no stable JSON representation. // has no stable JSON representation.
// //
@@ -559,25 +624,26 @@ type StructuredOutputJSONSpec struct {
Schema any `json:"schema"` Schema any `json:"schema"`
} }
// LLMClient executes rendered prompts for [Engine.Run]. // LLMClient executes rendered prompts for [Engine.Run] and
// [Engine.RunPrepared].
// //
// Generate is scheduled according to the resolved backend's capacity policy. // Generate is scheduled according to the resolved backend's capacity policy.
// It may still be called concurrently for different backend pools or unlimited // It may still be called concurrently for different backend pools or unlimited
// backends. Cancellation while waiting for capacity can prevent Generate from // backends. Cancellation while waiting for capacity can prevent Generate from
// being called. Once invoked, it must honor context cancellation to make Run // being called. Once invoked, it must honor context cancellation to make Run
// responsive to cancellation. The request and all nested maps, slices, and // and RunPrepared responsive to cancellation. The request and all nested maps,
// pointers are client-owned copies and may be mutated or retained without // slices, and pointers are client-owned copies and may be mutated or retained
// affecting engine state. // without affecting engine state.
// //
// Generate receives rendered messages and may receive a direct API key. A // Generate receives rendered messages and may receive a direct API key. A
// client must protect those values and any raw output in its logging, storage, // client must protect those values and any raw output in its logging, storage,
// and retained copies. It is responsible for the cancellation behavior of any // and retained copies. It is responsible for the cancellation behavior of any
// work it starts and for synchronizing access to retained or shared data. // work it starts and for synchronizing access to retained or shared data.
// //
// A returned error makes Run return ErrLLMGenerate while preserving the client // A returned error makes Run or RunPrepared return ErrLLMGenerate while
// error through errors.Is. A nil response with a nil error also produces // preserving the client error through errors.Is. A nil response with a nil
// ErrLLMGenerate. Promptkit copies the non-nil response before returning from // error also produces ErrLLMGenerate. Promptkit copies the non-nil response
// Run. // before returning from either method.
type LLMClient interface { type LLMClient interface {
Generate(context.Context, GenerateRequest) (*GenerateResponse, error) Generate(context.Context, GenerateRequest) (*GenerateResponse, error)
} }