Compare commits
11 Commits
369ab5392d
...
v0.4.0
| Author | SHA1 | Date | |
|---|---|---|---|
| bd6cffc9d0 | |||
| e40c4f182b | |||
| 7428e50c2c | |||
| 63c67a4520 | |||
| 25a7052a3d | |||
| fc3255967e | |||
| e920168b30 | |||
| 272b6a4bc1 | |||
| dde48a31fc | |||
| 242eace4a7 | |||
| 0bf5f88136 |
@@ -33,6 +33,9 @@ boundary and constraints that framework work must preserve.
|
|||||||
|
|
||||||
## Release Guidance
|
## Release Guidance
|
||||||
|
|
||||||
|
Consumers upgrading from `v0.3.0` to `v0.4.0` should read the
|
||||||
|
[v0.4.0 changelog and adoption guide](docs/releases/v0.4.0.md).
|
||||||
|
|
||||||
Consumers upgrading from `v0.2.0` to `v0.3.0` should read the
|
Consumers upgrading from `v0.2.0` to `v0.3.0` should read the
|
||||||
[v0.3.0 changelog](docs/releases/v0.3.0.md).
|
[v0.3.0 changelog](docs/releases/v0.3.0.md).
|
||||||
|
|
||||||
|
|||||||
@@ -60,7 +60,18 @@ func TestEngineRejectsRunBeforeCompletionWhenAdmissionIsFull(t *testing.T) {
|
|||||||
|
|
||||||
awaitCapacitySignal(t, reader.entered, "first artifact read")
|
awaitCapacitySignal(t, reader.entered, "first artifact read")
|
||||||
|
|
||||||
result, err := engine.Run(context.Background(), capacityInputRequest("http://second.example/v1"))
|
canceledContext, cancel := context.WithCancel(context.Background())
|
||||||
|
cancel()
|
||||||
|
result, err := engine.Run(canceledContext, capacityInputRequest("http://canceled.example/v1"))
|
||||||
|
if result != nil || !errors.Is(err, context.Canceled) {
|
||||||
|
t.Fatalf("canceled capacity admission=(%+v, %v), want context cancellation", result, err)
|
||||||
|
}
|
||||||
|
var canceledCapacityErr *promptkit.CapacityError
|
||||||
|
if errors.Is(err, promptkit.ErrCapacityExceeded) || errors.As(err, &canceledCapacityErr) {
|
||||||
|
t.Fatalf("canceled admission exposed capacity rejection: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
result, err = engine.Run(context.Background(), capacityInputRequest("http://second.example/v1"))
|
||||||
if result != nil {
|
if result != nil {
|
||||||
t.Fatalf("capacity rejection returned partial result: %+v", result)
|
t.Fatalf("capacity rejection returned partial result: %+v", result)
|
||||||
}
|
}
|
||||||
@@ -70,6 +81,21 @@ func TestEngineRejectsRunBeforeCompletionWhenAdmissionIsFull(t *testing.T) {
|
|||||||
if errors.Is(err, promptkit.ErrInvalidRequest) || errors.Is(err, promptkit.ErrLLMGenerate) {
|
if errors.Is(err, promptkit.ErrInvalidRequest) || errors.Is(err, promptkit.ErrLLMGenerate) {
|
||||||
t.Fatalf("capacity rejection had an unrelated category: %v", err)
|
t.Fatalf("capacity rejection had an unrelated category: %v", err)
|
||||||
}
|
}
|
||||||
|
var capacityErr *promptkit.CapacityError
|
||||||
|
if !errors.As(err, &capacityErr) || capacityErr == nil {
|
||||||
|
t.Fatalf("capacity rejection=%v, want CapacityError", err)
|
||||||
|
}
|
||||||
|
if capacityErr.BackendID != "limited" {
|
||||||
|
t.Fatalf("capacity backend ID=%q, want limited", capacityErr.BackendID)
|
||||||
|
}
|
||||||
|
capacityErr.BackendID = "changed"
|
||||||
|
|
||||||
|
result, err = engine.Run(context.Background(), capacityInputRequest("http://third.example/v1"))
|
||||||
|
var subsequentCapacityErr *promptkit.CapacityError
|
||||||
|
if result != nil || !errors.As(err, &subsequentCapacityErr) ||
|
||||||
|
subsequentCapacityErr == nil || subsequentCapacityErr.BackendID != "limited" {
|
||||||
|
t.Fatalf("subsequent capacity rejection=(%+v, %v), want independent limited CapacityError", result, err)
|
||||||
|
}
|
||||||
if calls := reader.callCount(); calls != 1 {
|
if calls := reader.callCount(); calls != 1 {
|
||||||
t.Fatalf("artifact calls=%d, want only the admitted run", calls)
|
t.Fatalf("artifact calls=%d, want only the admitted run", calls)
|
||||||
}
|
}
|
||||||
@@ -174,6 +200,19 @@ func TestCapacityExceededSentinelContract(t *testing.T) {
|
|||||||
if promptkit.ErrCapacityExceeded == nil {
|
if promptkit.ErrCapacityExceeded == nil {
|
||||||
t.Fatal("ErrCapacityExceeded is nil")
|
t.Fatal("ErrCapacityExceeded is nil")
|
||||||
}
|
}
|
||||||
|
var nilCapacityErr *promptkit.CapacityError
|
||||||
|
zeroCapacityErr := &promptkit.CapacityError{}
|
||||||
|
populatedCapacityErr := &promptkit.CapacityError{BackendID: "limited"}
|
||||||
|
for _, capacityErr := range []error{nilCapacityErr, zeroCapacityErr} {
|
||||||
|
if !errors.Is(capacityErr, promptkit.ErrCapacityExceeded) {
|
||||||
|
t.Fatalf("capacity error=%v, want ErrCapacityExceeded", capacityErr)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
var discoveredCapacityErr *promptkit.CapacityError
|
||||||
|
if !errors.As(populatedCapacityErr, &discoveredCapacityErr) || discoveredCapacityErr != populatedCapacityErr {
|
||||||
|
t.Fatalf("populated capacity error is not discoverable: %v", populatedCapacityErr)
|
||||||
|
}
|
||||||
|
|
||||||
for _, unrelated := range []error{
|
for _, unrelated := range []error{
|
||||||
promptkit.ErrInvalidConfig,
|
promptkit.ErrInvalidConfig,
|
||||||
promptkit.ErrInvalidRequest,
|
promptkit.ErrInvalidRequest,
|
||||||
@@ -181,7 +220,8 @@ func TestCapacityExceededSentinelContract(t *testing.T) {
|
|||||||
promptkit.ErrValidation,
|
promptkit.ErrValidation,
|
||||||
} {
|
} {
|
||||||
if errors.Is(promptkit.ErrCapacityExceeded, unrelated) ||
|
if errors.Is(promptkit.ErrCapacityExceeded, unrelated) ||
|
||||||
errors.Is(unrelated, promptkit.ErrCapacityExceeded) {
|
errors.Is(unrelated, promptkit.ErrCapacityExceeded) ||
|
||||||
|
errors.Is(populatedCapacityErr, unrelated) {
|
||||||
t.Fatalf("ErrCapacityExceeded aliases unrelated sentinel %v", unrelated)
|
t.Fatalf("ErrCapacityExceeded aliases unrelated sentinel %v", unrelated)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
40
capacity_error.go
Normal file
40
capacity_error.go
Normal file
@@ -0,0 +1,40 @@
|
|||||||
|
package promptkit
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
"strings"
|
||||||
|
)
|
||||||
|
|
||||||
|
// CapacityError reports bounded admission rejected for a selected backend.
|
||||||
|
//
|
||||||
|
// Engine-produced values identify only rejection at Promptkit's bounded
|
||||||
|
// [Engine.Run] or [Engine.RunPrepared] admission boundary. BackendID is the
|
||||||
|
// normalized registered backend ID used for routing and capacity; endpoint
|
||||||
|
// overrides do not change it. Every engine-produced value is nonnil and has a
|
||||||
|
// nonblank BackendID. Provider errors, active-generation waiting, and caller
|
||||||
|
// cancellation are not represented by this type.
|
||||||
|
//
|
||||||
|
// Callers own returned values and may mutate BackendID without affecting engine
|
||||||
|
// state or another error. CapacityError and its default Go encoding have no
|
||||||
|
// stable JSON contract. Consumer-constructed values do not establish that an
|
||||||
|
// engine rejected work.
|
||||||
|
type CapacityError struct {
|
||||||
|
// BackendID is the normalized registered backend ID whose admission was
|
||||||
|
// rejected.
|
||||||
|
BackendID string
|
||||||
|
}
|
||||||
|
|
||||||
|
// Error returns diagnostic wording that is not a parsing contract. It is safe
|
||||||
|
// to call on a nil receiver or a value with a blank BackendID.
|
||||||
|
func (e *CapacityError) Error() string {
|
||||||
|
if e == nil || strings.TrimSpace(e.BackendID) == "" {
|
||||||
|
return ErrCapacityExceeded.Error()
|
||||||
|
}
|
||||||
|
return fmt.Sprintf("backend %q admission: %v", e.BackendID, ErrCapacityExceeded)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Unwrap returns ErrCapacityExceeded so errors.Is and errors.As can be used
|
||||||
|
// together. It is safe to call on a nil receiver or a zero value.
|
||||||
|
func (e *CapacityError) Unwrap() error {
|
||||||
|
return ErrCapacityExceeded
|
||||||
|
}
|
||||||
34
convert.go
34
convert.go
@@ -170,6 +170,40 @@ func fromDomainExecutionTarget(target domain.ExecutionTarget) ExecutionTarget {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func fromDomainProfileInspection(inspection *domain.ProfileInspection) *ProfileInspection {
|
||||||
|
if inspection == nil {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
return &ProfileInspection{
|
||||||
|
ProfileID: inspection.ProfileID,
|
||||||
|
EffectiveModelParams: fromDomainExecutionTarget(inspection.EffectiveModelParams),
|
||||||
|
APIKeyRequired: inspection.APIKeyRequired,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func fromDomainPromptInspection(inspection *domain.PromptInspection) *PromptInspection {
|
||||||
|
if inspection == nil {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
inputs := make([]PromptInputDefinition, len(inspection.Inputs))
|
||||||
|
for i, input := range inspection.Inputs {
|
||||||
|
inputs[i] = PromptInputDefinition{
|
||||||
|
Name: input.Name,
|
||||||
|
Required: input.Required,
|
||||||
|
ContentType: input.ContentType,
|
||||||
|
Description: input.Description,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return &PromptInspection{
|
||||||
|
PromptID: inspection.PromptID,
|
||||||
|
PromptVersion: inspection.PromptVersion,
|
||||||
|
PromptHash: inspection.PromptHash,
|
||||||
|
DefaultProfileID: inspection.DefaultProfileID,
|
||||||
|
Inputs: inputs,
|
||||||
|
OutputContract: fromDomainOutputContract(inspection.OutputContract),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func fromDomainExecutionTargetPresence(presence domain.ExecutionTargetPresence) ExecutionTargetPresence {
|
func fromDomainExecutionTargetPresence(presence domain.ExecutionTargetPresence) ExecutionTargetPresence {
|
||||||
return ExecutionTargetPresence{
|
return ExecutionTargetPresence{
|
||||||
Temperature: presence.Temperature,
|
Temperature: presence.Temperature,
|
||||||
|
|||||||
40
doc.go
40
doc.go
@@ -3,25 +3,28 @@
|
|||||||
//
|
//
|
||||||
// Applications construct an [Engine] with [NewEngine], select filesystem or
|
// Applications construct an [Engine] with [NewEngine], select filesystem or
|
||||||
// in-memory sources and optional engine-scoped [Backend] registrations, and
|
// in-memory sources and optional engine-scoped [Backend] registrations, and
|
||||||
// call [Engine.Prepare], [Engine.PrepareExecution], [Engine.Run], or
|
// call [Engine.InspectPrompt], [Engine.InspectProfile], [Engine.Prepare],
|
||||||
// [Engine.RunPrepared]. Concrete registries, repositories, validators, and the
|
// [Engine.PrepareExecution], [Engine.Run], or [Engine.RunPrepared]. Concrete
|
||||||
// built-in OpenAI-compatible client remain internal implementation details.
|
// registries, repositories, validators, and the built-in OpenAI-compatible
|
||||||
|
// client remain internal implementation details.
|
||||||
//
|
//
|
||||||
// # Concurrency and ownership
|
// # Concurrency and ownership
|
||||||
//
|
//
|
||||||
// An Engine supports concurrent Prepare, PrepareExecution, Run, and RunPrepared
|
// An Engine supports concurrent InspectPrompt, InspectProfile, Prepare,
|
||||||
// calls. Engine-local backend policies bound admitted Run and RunPrepared calls
|
// PrepareExecution, Run, and RunPrepared calls. Engine-local backend policies
|
||||||
// and model generations where configured, while different backend pools and
|
// bound admitted Run and RunPrepared calls and model generations where
|
||||||
// unlimited backends continue independently. An injected [LLMClient] or
|
// configured, while different backend pools and unlimited backends continue
|
||||||
// [ArtifactReader] can therefore still receive concurrent calls and must be
|
// independently. An injected [LLMClient] or [ArtifactReader] can therefore
|
||||||
// safe for that use.
|
// still receive concurrent calls and must be safe for that use.
|
||||||
//
|
//
|
||||||
// NewEngine copies in-memory profiles and backend definitions. Prepare,
|
// NewEngine copies in-memory profiles and backend definitions. Prepare,
|
||||||
// PrepareExecution, and Run copy request maps, slices, pointer values, and
|
// PrepareExecution, and Run copy request maps, slices, pointer values, and
|
||||||
// JSON-compatible extra parameters before using them. Returned values and
|
// JSON-compatible extra parameters before using them. InspectPrompt and
|
||||||
// values passed to extension interfaces are likewise isolated from engine
|
// InspectProfile return copied inspection values. Returned values and values
|
||||||
// state. Callers own those copies and may mutate them after the call that
|
// passed to extension interfaces are likewise isolated from engine state.
|
||||||
// supplied or returned them.
|
// Callers own those copies and may mutate them after the call that supplied or
|
||||||
|
// returned them. Returned structured errors are likewise caller-owned and may
|
||||||
|
// be mutated without affecting engine state or another error.
|
||||||
//
|
//
|
||||||
// # Security and sensitive data
|
// # Security and sensitive data
|
||||||
//
|
//
|
||||||
@@ -47,11 +50,12 @@
|
|||||||
// [GenerateResponse], [ExecutionTargetPresence], and the string value types
|
// [GenerateResponse], [ExecutionTargetPresence], and the string value types
|
||||||
// used by those values.
|
// used by those values.
|
||||||
//
|
//
|
||||||
// Construction and handle values, including [Config], [Backend], [RunRequest],
|
// Construction, inspection, handle, and error values, including [Config],
|
||||||
// [ArtifactRef], [ExecutionTargetOverride], [Profile],
|
// [Backend], [RunRequest], [ArtifactRef], [ExecutionTargetOverride], [Profile],
|
||||||
// [OpenAICompatibleProfileConfig], and [PreparedExecution], do not have stable
|
// [OpenAICompatibleProfileConfig], [ProfileInspection],
|
||||||
// JSON representations. Direct API keys are nevertheless excluded from JSON
|
// [PromptInputDefinition], [PromptInspection], [PreparedExecution], and
|
||||||
// for every public value.
|
// [CapacityError], do not have stable JSON representations. Direct API keys
|
||||||
|
// are nevertheless excluded from JSON for every public value.
|
||||||
//
|
//
|
||||||
// JSON timestamps use time.Time's RFC 3339 encoding and are omitted when zero.
|
// JSON timestamps use time.Time's RFC 3339 encoding and are omitted when zero.
|
||||||
// PreparedRun and RunResult durations are encoded as integer milliseconds in
|
// PreparedRun and RunResult durations are encoded as integer milliseconds in
|
||||||
|
|||||||
@@ -40,6 +40,30 @@ validation, and default transport behavior. Source discovery, format
|
|||||||
validation, and profile precedence are defined by the
|
validation, and profile precedence are defined by the
|
||||||
[framework format reference](../formats.md).
|
[framework format reference](../formats.md).
|
||||||
|
|
||||||
|
## Inspect A Prompt Before Preparation
|
||||||
|
|
||||||
|
Use [`Engine.InspectPrompt`](../../engine.go) to check one configured prompt's
|
||||||
|
declared inputs and output workflow without creating placeholder inputs or
|
||||||
|
resolving a profile:
|
||||||
|
|
||||||
|
```go
|
||||||
|
inspection, err := engine.InspectPrompt(ctx, "meeting.summary", "")
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, input := range inspection.Inputs {
|
||||||
|
// Compare the declared input with application configuration.
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
Use this configuration-time boundary when the application needs only the
|
||||||
|
declared prompt interface. Use `InspectProfile` separately when it must also
|
||||||
|
check a configured profile. Use `Prepare` when it needs inputs, schemas, or
|
||||||
|
rendered messages, and use prepared execution when that work must remain tied
|
||||||
|
to later execution. The method's [GoDoc](../../engine.go) owns exact fields,
|
||||||
|
hash, ownership, and error semantics.
|
||||||
|
|
||||||
## Prepare Without Model Execution
|
## Prepare Without Model Execution
|
||||||
|
|
||||||
[`Engine.Prepare`](../../engine.go) resolves the selected prompt and profile,
|
[`Engine.Prepare`](../../engine.go) resolves the selected prompt and profile,
|
||||||
@@ -135,6 +159,34 @@ For programmatic profiles,
|
|||||||
[`OpenAICompatibleProfile`](../../profiles.go) converts ordinary
|
[`OpenAICompatibleProfile`](../../profiles.go) converts ordinary
|
||||||
OpenAI-compatible settings into a value accepted by `WithProfiles`.
|
OpenAI-compatible settings into a value accepted by `WithProfiles`.
|
||||||
|
|
||||||
|
### Inspect A Profile Before Prompt Work
|
||||||
|
|
||||||
|
Use [`Engine.InspectProfile`](../../engine.go) to validate one configured
|
||||||
|
profile without constructing a synthetic prompt or placeholder inputs. It
|
||||||
|
resolves the profile's effective target but does not prepare or execute a
|
||||||
|
prompt:
|
||||||
|
|
||||||
|
```go
|
||||||
|
inspection, err := engine.InspectProfile(ctx, profileID)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
|
||||||
|
target := inspection.EffectiveModelParams
|
||||||
|
if target.APIKeyEnv != "" {
|
||||||
|
// Apply application policy for the named environment variable.
|
||||||
|
} else if inspection.APIKeyRequired {
|
||||||
|
// Arrange a direct credential before later execution.
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
Use this configuration-time boundary when only the profile and its target need
|
||||||
|
checking. Use `Prepare` when the application also needs prompt, input, schema,
|
||||||
|
or rendering work; use prepared execution when that work must remain tied to a
|
||||||
|
later execution. Inspection reports credential requirements but leaves the
|
||||||
|
timing of credential enforcement to the application. The method's
|
||||||
|
[GoDoc](../../engine.go) owns its exact result and error contract.
|
||||||
|
|
||||||
### Set A Per-Run Session And Reasoning
|
### Set A Per-Run Session And Reasoning
|
||||||
|
|
||||||
Supply a direct session ID when one prompt should be correlated with a
|
Supply a direct session ID when one prompt should be correlated with a
|
||||||
@@ -325,6 +377,11 @@ When a limited backend has admitted all active and waiting calls, handle
|
|||||||
```go
|
```go
|
||||||
result, err := engine.Run(ctx, request)
|
result, err := engine.Run(ctx, request)
|
||||||
if errors.Is(err, promptkit.ErrCapacityExceeded) {
|
if errors.Is(err, promptkit.ErrCapacityExceeded) {
|
||||||
|
var capacityErr *promptkit.CapacityError
|
||||||
|
if errors.As(err, &capacityErr) {
|
||||||
|
// Record capacityErr.BackendID using application-owned diagnostics.
|
||||||
|
}
|
||||||
|
|
||||||
// Apply application policy: shed work, report overload, or retry later.
|
// Apply application policy: shed work, report overload, or retry later.
|
||||||
}
|
}
|
||||||
```
|
```
|
||||||
@@ -332,8 +389,9 @@ if errors.Is(err, promptkit.ErrCapacityExceeded) {
|
|||||||
A rejected call returns no partial result and does not invoke the model
|
A rejected call returns no partial result and does not invoke the model
|
||||||
client. Promptkit does not prescribe retries or map this error to an HTTP
|
client. Promptkit does not prescribe retries or map this error to an HTTP
|
||||||
status; those choices remain with the consuming application. The
|
status; those choices remain with the consuming application. The
|
||||||
[`Engine.Run` and error GoDoc](../../engine.go) owns exact error and
|
[`CapacityError` GoDoc](../../capacity_error.go) owns the exact typed-error
|
||||||
cancellation identities.
|
contract, while the [`Engine.Run` and error GoDoc](../../engine.go) owns broad
|
||||||
|
error and cancellation identities.
|
||||||
|
|
||||||
## Application Boundary
|
## Application Boundary
|
||||||
|
|
||||||
|
|||||||
@@ -58,6 +58,11 @@ When a request omits a version, the selected prompt ID must identify exactly
|
|||||||
one definition. When it supplies a version, the ID and version pair must be
|
one definition. When it supplies a version, the ID and version pair must be
|
||||||
unique.
|
unique.
|
||||||
|
|
||||||
|
Exact prompt inspection uses this same configured source, strict decoding,
|
||||||
|
referenced content-file resolution, and ID/version selection. It reports the
|
||||||
|
selected definition's declared metadata without changing the prompt format or
|
||||||
|
executing the definition.
|
||||||
|
|
||||||
### Inputs
|
### Inputs
|
||||||
|
|
||||||
Each `inputs` item has these fields:
|
Each `inputs` item has these fields:
|
||||||
@@ -152,7 +157,7 @@ extra_params:
|
|||||||
| Field | Required | Meaning |
|
| Field | Required | Meaning |
|
||||||
| --- | --- | --- |
|
| --- | --- | --- |
|
||||||
| `id` | yes | Non-empty profile identifier. IDs must be unique within one source. |
|
| `id` | yes | Non-empty profile identifier. IDs must be unique within one source. |
|
||||||
| `backend` | unless `endpoint` is present | Backend registry ID. It is trimmed and registry membership is checked when the profile is prepared. |
|
| `backend` | unless `endpoint` is present | Backend registry ID. It is trimmed and registry membership is checked when the profile is prepared or inspected. |
|
||||||
| `endpoint` | unless `backend` is present | Non-empty OpenAI-compatible base URL, including an API version path when required. When both connection fields are present, this overrides the backend endpoint without changing backend identity. |
|
| `endpoint` | unless `backend` is present | Non-empty OpenAI-compatible base URL, including an API version path when required. When both connection fields are present, this overrides the backend endpoint without changing backend identity. |
|
||||||
| `model` | yes | Non-empty provider model name. |
|
| `model` | yes | Non-empty provider model name. |
|
||||||
| `temperature` | no | Number from 0 through 2. |
|
| `temperature` | no | Number from 0 through 2. |
|
||||||
@@ -219,7 +224,9 @@ defines how the effective settings are serialized.
|
|||||||
### Source And Profile Precedence
|
### Source And Profile Precedence
|
||||||
|
|
||||||
An explicit request profile ID takes precedence over the prompt's
|
An explicit request profile ID takes precedence over the prompt's
|
||||||
`default_profile`. If neither is present, preparation fails.
|
`default_profile`. If neither is present, preparation fails. Exact profile
|
||||||
|
inspection instead takes one explicit profile ID and does not use a prompt
|
||||||
|
default.
|
||||||
|
|
||||||
Profile sources resolve matching IDs in this order:
|
Profile sources resolve matching IDs in this order:
|
||||||
|
|
||||||
@@ -231,6 +238,7 @@ A higher-precedence source falls back only when the profile is absent. An
|
|||||||
invalid matching profile is an error and does not fall back. In-memory
|
invalid matching profile is an error and does not fall back. In-memory
|
||||||
`Profile` values follow the same ranges as YAML profiles. They use
|
`Profile` values follow the same ranges as YAML profiles. They use
|
||||||
`APIKeyRequired` for request-scoped credentials instead of `api_key_env`.
|
`APIKeyRequired` for request-scoped credentials instead of `api_key_env`.
|
||||||
|
Preparation and exact profile inspection use this same source precedence.
|
||||||
|
|
||||||
## Built-In Profile Catalog
|
## Built-In Profile Catalog
|
||||||
|
|
||||||
|
|||||||
@@ -38,10 +38,10 @@ output contract, but before schema loading, artifact loading, or rendering.
|
|||||||
rechecks credential availability, and then asks the manager to admit the
|
rechecks credential availability, and then asks the manager to admit the
|
||||||
frozen backend before generation.
|
frozen backend before generation.
|
||||||
|
|
||||||
Admission is immediate: a limited pool either reserves a slot or returns the
|
Admission is immediate: a limited pool either reserves a slot or returns only
|
||||||
internal `ErrCapacityExceeded` identity. The root facade maps that identity to
|
the internal `ErrCapacityExceeded` identity. The runner attaches the selected
|
||||||
the public error without treating it as an invalid request or generation
|
backend identity at its use-case boundary, and the root facade translates that
|
||||||
failure.
|
typed value without treating it as an invalid request or generation failure.
|
||||||
|
|
||||||
The total admitted bound is the active-generation limit plus its configured
|
The total admitted bound is the active-generation limit plus its configured
|
||||||
waiting capacity. The returned release function is idempotent. The runner
|
waiting capacity. The returned release function is idempotent. The runner
|
||||||
|
|||||||
@@ -11,7 +11,7 @@ contributor workflow and validation.
|
|||||||
|
|
||||||
| Component | Implemented responsibility | References |
|
| Component | Implemented responsibility | References |
|
||||||
| --- | --- | --- |
|
| --- | --- | --- |
|
||||||
| Root `promptkit` package | Provides the supported engine facade, source, backend-registration, and injection options, public request and result values, opaque prepared-execution handles, profile construction, extension interfaces, value conversion, redacted formatting, public error mapping, and engine-local assembly. | [Package GoDoc](../../doc.go), [prepared execution](../../prepared_execution.go), [backend API](../../backends.go), [engine assembly](../../engine.go) |
|
| Root `promptkit` package | Provides the supported engine facade, source, backend-registration, and injection options, public request, result, prompt-inspection, and profile-inspection values, opaque prepared-execution handles, profile construction, extension interfaces, value conversion, redacted formatting, typed capacity errors, public error mapping, and engine-local assembly. | [Package GoDoc](../../doc.go), [prepared execution](../../prepared_execution.go), [backend API](../../backends.go), [engine assembly](../../engine.go) |
|
||||||
| `examples/go-library/prepare` | Demonstrates an offline downstream consumer using a prompt file, in-memory profile, inline input, and `Prepare`. It is not a public library package. | [Example program](../../examples/go-library/prepare/main.go) |
|
| `examples/go-library/prepare` | Demonstrates an offline downstream consumer using a prompt file, in-memory profile, inline input, and `Prepare`. It is not a public library package. | [Example program](../../examples/go-library/prepare/main.go) |
|
||||||
| `examples/go-library/run` | Demonstrates an offline downstream consumer using a prompt file, in-memory profile, inline input, an injected deterministic model client, and `Run`. It is not a public library package. | [Example program](../../examples/go-library/run/main.go) |
|
| `examples/go-library/run` | Demonstrates an offline downstream consumer using a prompt file, in-memory profile, inline input, an injected deterministic model client, and `Run`. It is not a public library package. | [Example program](../../examples/go-library/run/main.go) |
|
||||||
| `internal/backend` | Constructs each engine's immutable registry from the built-in OpenRouter definition and consumer additions, validates and defensively copies definitions through the shared JSON-value package, and consumes the LLM-owned OpenAI-compatible reserved request-field rule. | [Backend registry](../../internal/backend/registry.go) |
|
| `internal/backend` | Constructs each engine's immutable registry from the built-in OpenRouter definition and consumer additions, validates and defensively copies definitions through the shared JSON-value package, and consumes the LLM-owned OpenAI-compatible reserved request-field rule. | [Backend registry](../../internal/backend/registry.go) |
|
||||||
@@ -27,7 +27,7 @@ contributor workflow and validation.
|
|||||||
| `internal/artifact` | Resolves ordinary inline and unrestricted caller-selected file references into copied artifacts with metadata and hashes. | [Internal sources and validation](sources.md) |
|
| `internal/artifact` | Resolves ordinary inline and unrestricted caller-selected file references into copied artifacts with metadata and hashes. | [Internal sources and validation](sources.md) |
|
||||||
| `internal/validate` | Validates basic, JSON, and JSON Schema output using operating-system filesystem or `fs.FS` schema sources and creates frozen validation plans for prepared execution. | [Framework formats](../formats.md#schemas), [internal sources and validation](sources.md) |
|
| `internal/validate` | Validates basic, JSON, and JSON Schema output using operating-system filesystem or `fs.FS` schema sources and creates frozen validation plans for prepared execution. | [Framework formats](../formats.md#schemas), [internal sources and validation](sources.md) |
|
||||||
| `internal/llm` | Defines the internal generation boundary and implements outbound OpenAI-compatible chat requests from resolved execution targets, including response decoding, authentication, deadline handling, and ownership of the OpenAI-compatible reserved request-field policy. | [Internal model client](llm.md) |
|
| `internal/llm` | Defines the internal generation boundary and implements outbound OpenAI-compatible chat requests from resolved execution targets, including response decoding, authentication, deadline handling, and ownership of the OpenAI-compatible reserved request-field policy. | [Internal model client](llm.md) |
|
||||||
| `internal/usecase` | Resolves backend, profile, and request settings and coordinates preparation, ordinary execution, and one-attempt prepared execution across internal sources, rendering, artifact loading, generation, validation, capacity, and optional repair. | [Internal runner](runner.md), [prepared-execution implementation](../../internal/usecase/prepared_execution.go) |
|
| `internal/usecase` | Resolves prompt definitions and hashes, profiles, backends, and targets for exact inspection and request settings for preparation, and coordinates ordinary execution and one-attempt prepared execution across internal sources, rendering, artifact loading, generation, validation, capacity, and optional repair. | [Internal runner](runner.md), [prepared-execution implementation](../../internal/usecase/prepared_execution.go) |
|
||||||
|
|
||||||
The root package assembles these internal components without exposing their
|
The root package assembles these internal components without exposing their
|
||||||
representations. Consumers depend only on the root facade.
|
representations. Consumers depend only on the root facade.
|
||||||
|
|||||||
@@ -28,6 +28,34 @@ constructor does not enable one.
|
|||||||
Each invocation carries its state in request, prepared-run, and result values.
|
Each invocation carries its state in request, prepared-run, and result values.
|
||||||
The runner has no durable run or session store.
|
The runner has no durable run or session store.
|
||||||
|
|
||||||
|
## Shared Prompt Selection
|
||||||
|
|
||||||
|
The runner uses one prompt-selection and hashing boundary for ordinary
|
||||||
|
preparation and exact prompt inspection. Preparation retains its early
|
||||||
|
request-ID check before direct-session normalization; both operations then use
|
||||||
|
the configured prompt repository to select one definition, load referenced
|
||||||
|
message content, and calculate the same prompt hash.
|
||||||
|
|
||||||
|
Inspection stops after that structural lookup. It does not parse templates or
|
||||||
|
touch profile, artifact, schema, renderer, validator, admission, or model
|
||||||
|
collaborators. The root [`Engine.InspectPrompt`](../../engine.go) GoDoc owns
|
||||||
|
the public operation's exact contract.
|
||||||
|
|
||||||
|
## Shared Profile Selection
|
||||||
|
|
||||||
|
The runner uses one profile-selection and target-resolution boundary for
|
||||||
|
ordinary preparation and exact profile inspection. Preparation first selects a
|
||||||
|
request profile or a prompt default; inspection begins with its required
|
||||||
|
explicit profile ID. Both then apply the ordinary source precedence, resolve a
|
||||||
|
named backend, and construct the effective target from framework, backend, and
|
||||||
|
profile values.
|
||||||
|
|
||||||
|
Inspection stops after the resulting endpoint and model are structurally
|
||||||
|
validated. It does not check credential availability or perform prompt,
|
||||||
|
artifact, schema, rendering, admission, or model-client work. The root
|
||||||
|
[`Engine.InspectProfile`](../../engine.go) GoDoc owns the public operation's
|
||||||
|
exact contract.
|
||||||
|
|
||||||
## Shared Preparation Pipeline
|
## Shared Preparation Pipeline
|
||||||
|
|
||||||
`Prepare` and `Run` share one private preparation pipeline split at the point
|
`Prepare` and `Run` share one private preparation pipeline split at the point
|
||||||
@@ -124,15 +152,17 @@ validation failures. Wrapping preserves the package identities mapped by the
|
|||||||
public facade and retains collaborator identities where they are part of the
|
public facade and retains collaborator identities where they are part of the
|
||||||
internal contract.
|
internal contract.
|
||||||
|
|
||||||
Admission capacity exhaustion retains the internal capacity identity and adds
|
Admission capacity exhaustion retains the internal capacity identity. At the
|
||||||
the selected backend ID as context. It is not recategorized as an invalid
|
use-case boundary, the runner attaches the selected backend ID in an internal
|
||||||
request or generation failure, and no partial result is returned. A context
|
typed error, and the root facade copies that value into the public
|
||||||
already done at admission retains its context identity directly. Cancellation
|
[`CapacityError`](../../capacity_error.go) without parsing diagnostic text. It
|
||||||
while waiting for an active generation permit prevents client invocation when
|
is not recategorized as an invalid request or generation failure, and no
|
||||||
it wins the grant race; the model-client boundary then preserves the context
|
partial result is returned. A context already done at admission retains its
|
||||||
error through the generation-failure category. Deferred release restores the
|
context identity directly. Cancellation while waiting for an active generation
|
||||||
admission lease on preparation, generation, validation, repair, and
|
permit prevents client invocation when it wins the grant race; the model-client
|
||||||
cancellation failures.
|
boundary then preserves the context error through the generation-failure
|
||||||
|
category. Deferred release restores the admission lease on preparation,
|
||||||
|
generation, validation, repair, and cancellation failures.
|
||||||
|
|
||||||
Other context cancellation propagates through the invoked collaborator and is
|
Other context cancellation propagates through the invoked collaborator and is
|
||||||
classified by the owning operation.
|
classified by the owning operation.
|
||||||
|
|||||||
@@ -16,6 +16,11 @@ validation modes, built-in catalog, and source precedence.
|
|||||||
definitions, selects an ID and optional version, and resolves file-backed
|
definitions, selects an ID and optional version, and resolves file-backed
|
||||||
message content within the selected operating-system or `fs.FS` source.
|
message content within the selected operating-system or `fs.FS` source.
|
||||||
|
|
||||||
|
Exact prompt inspection performs one point-in-time lookup through that same
|
||||||
|
repository and validates referenced message content before returning declared
|
||||||
|
metadata. It does not parse templates or read profile, input, or schema
|
||||||
|
sources, and it does not retain the definition for a later execution.
|
||||||
|
|
||||||
Its package tests own prompt selection, strict decoding, definition validation,
|
Its package tests own prompt selection, strict decoding, definition validation,
|
||||||
duplicate detection, and source containment:
|
duplicate detection, and source containment:
|
||||||
[prompt-definition repository tests](../../internal/promptdef/repository_test.go).
|
[prompt-definition repository tests](../../internal/promptdef/repository_test.go).
|
||||||
@@ -28,7 +33,12 @@ with fallback only when the primary reports that a profile is absent. Strict
|
|||||||
YAML decoding recognizes the optional `backend` field, trims its value, and
|
YAML decoding recognizes the optional `backend` field, trims its value, and
|
||||||
requires a model plus at least one non-blank backend or endpoint. Loading does
|
requires a model plus at least one non-blank backend or endpoint. Loading does
|
||||||
not check registry membership because the available registry belongs to the
|
not check registry membership because the available registry belongs to the
|
||||||
assembled engine; the runner checks membership during preparation.
|
assembled engine; the runner checks membership during preparation and exact
|
||||||
|
profile inspection.
|
||||||
|
|
||||||
|
Exact profile inspection performs one point-in-time lookup through those
|
||||||
|
profile sources and checks the resolved target without reading prompt, input,
|
||||||
|
or schema sources. It does not retain that lookup for a later execution.
|
||||||
|
|
||||||
`internal/profile/builtin` embeds the maintained built-in profile catalog and
|
`internal/profile/builtin` embeds the maintained built-in profile catalog and
|
||||||
can place a caller-selected repository ahead of that catalog. Every embedded
|
can place a caller-selected repository ahead of that catalog. Every embedded
|
||||||
@@ -79,7 +89,7 @@ captured root document.
|
|||||||
loading and hashing, session and message rendering, and target resolution.
|
loading and hashing, session and message rendering, and target resolution.
|
||||||
`RunPrepared` uses the retained source-derived state and validation plan; it
|
`RunPrepared` uses the retained source-derived state and validation plan; it
|
||||||
does not reopen prompt, profile, input, or schema sources and does not rerender
|
does not reopen prompt, profile, input, or schema sources and does not rerender
|
||||||
the request. By contrast, ordinary `Prepare` produces an inspection value only:
|
the request. By contrast, ordinary `Prepare` produces a preparation value only:
|
||||||
a later `Run` performs its own source resolution and preparation.
|
a later `Run` performs its own source resolution and preparation.
|
||||||
|
|
||||||
The [validator tests](../../internal/validate/standard_validator_test.go) own
|
The [validator tests](../../internal/validate/standard_validator_test.go) own
|
||||||
|
|||||||
189
docs/releases/v0.4.0.md
Normal file
189
docs/releases/v0.4.0.md
Normal file
@@ -0,0 +1,189 @@
|
|||||||
|
# Promptkit v0.4.0
|
||||||
|
|
||||||
|
This supplemental changelog and adoption guide summarizes the consumer-facing
|
||||||
|
changes from `v0.3.0` to `v0.4.0`. The annotated `v0.4.0` tag is the
|
||||||
|
authoritative release record. Exact current contracts belong to the linked
|
||||||
|
GoDoc and durable documentation.
|
||||||
|
|
||||||
|
## Summary
|
||||||
|
|
||||||
|
`v0.4.0` adds four complementary capabilities:
|
||||||
|
|
||||||
|
- opaque prepared-execution handles for preparing once, inspecting safe
|
||||||
|
details, and executing the same frozen snapshot;
|
||||||
|
- exact prompt-definition inspection without profile resolution or execution;
|
||||||
|
- exact profile inspection without selecting a prompt or checking credential
|
||||||
|
availability; and
|
||||||
|
- structured backend identity on engine admission-capacity rejection.
|
||||||
|
|
||||||
|
These APIs let consumers perform more precise preflight work and retain useful
|
||||||
|
operational context without reproducing Promptkit's internal resolution logic.
|
||||||
|
|
||||||
|
## Compatibility
|
||||||
|
|
||||||
|
The release is additive for `v0.3.0` consumers. Existing uses of `Prepare`,
|
||||||
|
`Run`, backend registration, endpoint-only profiles, local-backend helpers,
|
||||||
|
runtime overrides, public JSON values, and error sentinels continue to work
|
||||||
|
without migration.
|
||||||
|
|
||||||
|
Capacity rejection now returns a structured error while continuing to match
|
||||||
|
`ErrCapacityExceeded` through `errors.Is`. Error-string wording and direct
|
||||||
|
sentinel equality were not public contracts.
|
||||||
|
|
||||||
|
The new inspection values, capacity error, and prepared-execution handle do not
|
||||||
|
have stable JSON representations. `PreparedExecution.Details` returns the
|
||||||
|
existing stable `PreparedRun` value.
|
||||||
|
|
||||||
|
## Upgrade
|
||||||
|
|
||||||
|
Update the module dependency with:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
go get gitea.maximumdirect.net/eric/promptkit@v0.4.0
|
||||||
|
go mod tidy
|
||||||
|
```
|
||||||
|
|
||||||
|
Run the consuming project's ordinary and race-enabled tests after upgrading.
|
||||||
|
No source migration is required.
|
||||||
|
|
||||||
|
## Prepare Once And Execute The Same Snapshot
|
||||||
|
|
||||||
|
Consumers that need to persist preparation details before generation can now
|
||||||
|
prepare an opaque, engine-bound execution:
|
||||||
|
|
||||||
|
```go
|
||||||
|
prepared, err := engine.PrepareExecution(ctx, request)
|
||||||
|
if err != nil {
|
||||||
|
// Handle preparation failure.
|
||||||
|
}
|
||||||
|
defer prepared.Discard()
|
||||||
|
|
||||||
|
details := prepared.Details()
|
||||||
|
// Persist a consumer-selected, appropriately protected preparation record.
|
||||||
|
|
||||||
|
result, err := engine.RunPrepared(ctx, prepared)
|
||||||
|
```
|
||||||
|
|
||||||
|
Preparation freezes the selected sources, rendered messages, effective
|
||||||
|
settings, input content, structured-output metadata, and validation resources
|
||||||
|
needed by execution. `Details` returns a fresh, caller-owned,
|
||||||
|
credential-redacted `PreparedRun`.
|
||||||
|
|
||||||
|
A handle belongs to its creating engine and permits one execution attempt.
|
||||||
|
`RunPrepared` consumes that attempt on success and on operational failure.
|
||||||
|
`Discard` is idempotent and releases an unclaimed handle's execution-only
|
||||||
|
state. Consumers should discard handles they will not execute, particularly
|
||||||
|
when a direct request API key may be retained privately until claim or
|
||||||
|
discard.
|
||||||
|
|
||||||
|
Prepared execution does not reserve backend admission during preparation.
|
||||||
|
Credential availability and backend admission are checked when execution
|
||||||
|
begins. The execution context is independent of the preparation context.
|
||||||
|
|
||||||
|
See the
|
||||||
|
[prepared-execution consumer guide](../consumers/pkg-promptkit.md#prepare-now-and-execute-the-same-snapshot-later),
|
||||||
|
the [`PreparedExecution` GoDoc](../../prepared_execution.go), and the
|
||||||
|
[`Engine.PrepareExecution` and `Engine.RunPrepared` GoDoc](../../engine.go)
|
||||||
|
for the exact lifecycle, ownership, cancellation, capacity, timing, and
|
||||||
|
failure contracts.
|
||||||
|
|
||||||
|
## Inspect A Prompt
|
||||||
|
|
||||||
|
`Engine.InspectPrompt` resolves one prompt ID and optional version through the
|
||||||
|
engine's configured prompt source:
|
||||||
|
|
||||||
|
```go
|
||||||
|
inspection, err := engine.InspectPrompt(ctx, "report.summary", "")
|
||||||
|
```
|
||||||
|
|
||||||
|
The result includes prompt identity, the opaque prompt hash, declared default
|
||||||
|
profile ID, declared input metadata, and normalized output contract. It
|
||||||
|
structurally loads the selected definition and referenced message content but
|
||||||
|
does not resolve a profile, load schemas or artifacts, render templates,
|
||||||
|
reserve capacity, or contact a model.
|
||||||
|
|
||||||
|
Use inspection for exact configuration checks and metadata discovery. Use
|
||||||
|
`PrepareExecution` rather than relying on a prior inspection when later
|
||||||
|
execution must freeze one exact source state, because filesystem-backed
|
||||||
|
inspection is only a point-in-time lookup.
|
||||||
|
|
||||||
|
See the
|
||||||
|
[prompt-inspection consumer guide](../consumers/pkg-promptkit.md#inspect-a-prompt-before-preparation)
|
||||||
|
and [`Engine.InspectPrompt` GoDoc](../../engine.go) for exact selection,
|
||||||
|
ownership, and error behavior.
|
||||||
|
|
||||||
|
## Inspect A Profile
|
||||||
|
|
||||||
|
`Engine.InspectProfile` resolves one explicit profile independently of a
|
||||||
|
prompt:
|
||||||
|
|
||||||
|
```go
|
||||||
|
inspection, err := engine.InspectProfile(ctx, "report-production")
|
||||||
|
```
|
||||||
|
|
||||||
|
The result includes the resolved effective execution target and whether a
|
||||||
|
later request must provide a direct credential. Environment-variable names may
|
||||||
|
be reported, but inspection does not read credential values or require the
|
||||||
|
named variable to be populated.
|
||||||
|
|
||||||
|
Inspection applies the ordinary configured and built-in profile precedence and
|
||||||
|
resolves any selected backend. It does not load a prompt, render content,
|
||||||
|
reserve capacity, or contact a model.
|
||||||
|
|
||||||
|
See the
|
||||||
|
[profile-inspection consumer guide](../consumers/pkg-promptkit.md#inspect-a-profile-before-prompt-work)
|
||||||
|
and [`Engine.InspectProfile` GoDoc](../../engine.go) for the exact resolution,
|
||||||
|
credential, ownership, and error contracts.
|
||||||
|
|
||||||
|
## Identify Capacity-Rejected Backends
|
||||||
|
|
||||||
|
Calls rejected at Promptkit's bounded engine admission boundary continue to
|
||||||
|
match `ErrCapacityExceeded`. Consumers can additionally obtain the selected
|
||||||
|
registered backend ID without parsing diagnostic text:
|
||||||
|
|
||||||
|
```go
|
||||||
|
result, err := engine.Run(ctx, request)
|
||||||
|
if errors.Is(err, promptkit.ErrCapacityExceeded) {
|
||||||
|
var capacityErr *promptkit.CapacityError
|
||||||
|
if errors.As(err, &capacityErr) {
|
||||||
|
// Record capacityErr.BackendID using application-owned diagnostics.
|
||||||
|
}
|
||||||
|
|
||||||
|
// Apply application-owned overload or retry policy.
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
The structured error applies to `Run` and `RunPrepared` admission rejection.
|
||||||
|
It does not represent provider throttling, quota exhaustion, cancellation
|
||||||
|
while waiting for generation capacity, or another model-client failure.
|
||||||
|
Promptkit does not prescribe retry timing or transport status mapping.
|
||||||
|
|
||||||
|
See the
|
||||||
|
[error-handling consumer guide](../consumers/pkg-promptkit.md#handle-errors),
|
||||||
|
the [`CapacityError` GoDoc](../../capacity_error.go), and the
|
||||||
|
[`ErrCapacityExceeded` GoDoc](../../engine.go) for the canonical contracts.
|
||||||
|
|
||||||
|
## Public API Additions
|
||||||
|
|
||||||
|
The release adds:
|
||||||
|
|
||||||
|
- `Engine.PrepareExecution`;
|
||||||
|
- `Engine.RunPrepared`;
|
||||||
|
- `PreparedExecution`, including `Details`, `Discard`, `String`, and
|
||||||
|
`GoString`;
|
||||||
|
- `Engine.InspectPrompt`;
|
||||||
|
- `PromptInspection`;
|
||||||
|
- `PromptInputDefinition`;
|
||||||
|
- `Engine.InspectProfile`;
|
||||||
|
- `ProfileInspection`; and
|
||||||
|
- `CapacityError`.
|
||||||
|
|
||||||
|
No public API was removed.
|
||||||
|
|
||||||
|
## Consumer Action
|
||||||
|
|
||||||
|
None. Existing `v0.3.0` workflows may upgrade without adopting the new APIs.
|
||||||
|
|
||||||
|
Consumers that adopt prepared execution should discard unused handles.
|
||||||
|
Consumers that need backend-specific capacity diagnostics may add an
|
||||||
|
`errors.As` check while retaining their existing `errors.Is` classification.
|
||||||
@@ -1,237 +0,0 @@
|
|||||||
# Backend-Specific Concurrency Management
|
|
||||||
|
|
||||||
**Status:** Complete.
|
|
||||||
|
|
||||||
## Purpose
|
|
||||||
|
|
||||||
This roadmap defines the scope and target end state for engine-local,
|
|
||||||
backend-specific concurrency management. It records the intended capability,
|
|
||||||
consumer value, and important policy choices.
|
|
||||||
|
|
||||||
This document is planning material, not a description of current behavior.
|
|
||||||
Current exported contracts remain owned by Go declarations and GoDoc, backend
|
|
||||||
registration guidance by the
|
|
||||||
[consumer guide](../consumers/pkg-promptkit.md#configure-a-local-openai-compatible-endpoint),
|
|
||||||
and
|
|
||||||
implemented orchestration by the
|
|
||||||
[internal runner document](../internal/runner.md).
|
|
||||||
|
|
||||||
## Motivation
|
|
||||||
|
|
||||||
Different model backends can sustain very different request loads. A local
|
|
||||||
network endpoint may need a small concurrency limit, while OpenRouter can
|
|
||||||
usually accept substantially more simultaneous work. Requiring every consumer
|
|
||||||
to build its own semaphores and queues would duplicate routing knowledge,
|
|
||||||
create inconsistent cancellation behavior, and make it easy for one caller to
|
|
||||||
bypass the intended backend limit.
|
|
||||||
|
|
||||||
Promptkit should own this coordination because it already resolves each run to
|
|
||||||
an engine-scoped backend identity and owns every model-generation call made by
|
|
||||||
the runner. Consumers should continue submitting ready-to-run requests through
|
|
||||||
the synchronous API, including concurrently from multiple goroutines, without
|
|
||||||
implementing their own backend scheduler.
|
|
||||||
|
|
||||||
The buffered queue is a safety boundary, not an ordinary throughput
|
|
||||||
restriction. Its primary purpose is to prevent a bug or unintended submission
|
|
||||||
loop from creating an unbounded in-memory backlog.
|
|
||||||
|
|
||||||
## Scope
|
|
||||||
|
|
||||||
The feature will add optional concurrency policy to registered backends and
|
|
||||||
coordinate `Run` calls against independent per-backend capacity pools.
|
|
||||||
|
|
||||||
Each policy has two distinct controls:
|
|
||||||
|
|
||||||
- an active-generation limit, which protects the backend from too many
|
|
||||||
simultaneous model requests; and
|
|
||||||
- a bounded waiting capacity, which protects the process from admitting an
|
|
||||||
unbounded backlog.
|
|
||||||
|
|
||||||
Concurrency policy belongs to a backend registration. It is not a profile
|
|
||||||
model parameter and cannot be overridden per run. Profiles select the policy
|
|
||||||
through their backend ID, while a profile or request endpoint override remains
|
|
||||||
in the selected backend's pool.
|
|
||||||
|
|
||||||
`Prepare` does not call a model and will remain outside concurrency admission.
|
|
||||||
|
|
||||||
## Defaults And Configuration
|
|
||||||
|
|
||||||
The built-in OpenRouter backend will use:
|
|
||||||
|
|
||||||
- an active-generation limit of 16; and
|
|
||||||
- a waiting capacity of 1024.
|
|
||||||
|
|
||||||
The waiting default is intentionally generous. Reaching it should indicate
|
|
||||||
abnormal submission pressure rather than normal application behavior.
|
|
||||||
|
|
||||||
Consumer-registered backends will remain unlimited unless the consumer
|
|
||||||
configures an active-generation limit. When a consumer enables a limit and
|
|
||||||
does not specify waiting capacity, the waiting capacity will default to 1024.
|
|
||||||
Consumers may configure a different bounded capacity, including zero when
|
|
||||||
they want no admitted backlog beyond the active-limit-sized run set.
|
|
||||||
|
|
||||||
The public representation must distinguish an omitted waiting capacity from
|
|
||||||
an explicit zero.
|
|
||||||
|
|
||||||
Endpoint-only profiles have no backend registration from which to obtain
|
|
||||||
policy and will remain unlimited. A future engine-wide or endpoint-keyed
|
|
||||||
policy can be considered separately if consumers demonstrate that need.
|
|
||||||
|
|
||||||
Invalid limits or capacities will fail engine construction as invalid
|
|
||||||
configuration. Policy values will be copied into engine-owned immutable state
|
|
||||||
along with the rest of the backend registration.
|
|
||||||
|
|
||||||
## Admission And Execution Behavior
|
|
||||||
|
|
||||||
`Run` remains a synchronous, wait-for-result operation. Concurrent callers may
|
|
||||||
block inside `Run` while waiting for their selected backend, then receive the
|
|
||||||
ordinary result or error from that invocation.
|
|
||||||
|
|
||||||
For a configured pool, the active-generation limit plus the waiting capacity
|
|
||||||
defines the maximum number of concurrent `Run` invocations that Promptkit will
|
|
||||||
accept for that backend. A waiting capacity of zero therefore accepts no more
|
|
||||||
runs than the active limit. Admission is immediate: a call either reserves one
|
|
||||||
of those bounded slots or receives the capacity error. An accepted run may
|
|
||||||
then wait internally for active-generation capacity.
|
|
||||||
|
|
||||||
For a limited backend, Promptkit will bound the number of accepted runs before
|
|
||||||
expensive artifact loading, prompt rendering, and large defensive copies where
|
|
||||||
practical. Lightweight prompt, profile, and backend resolution may occur first
|
|
||||||
when it is required to identify the correct capacity pool. This pre-admission
|
|
||||||
resolution must not become a second execution-precedence path with behavior
|
|
||||||
that can drift from `Prepare`.
|
|
||||||
|
|
||||||
An accepted run retains its admission until it completes or fails. Every
|
|
||||||
actual model-generation call for that run must separately observe the
|
|
||||||
backend's active-generation limit. This includes:
|
|
||||||
|
|
||||||
- the initial generation;
|
|
||||||
- every output-repair generation; and
|
|
||||||
- calls made through either the built-in or an injected model client.
|
|
||||||
|
|
||||||
Preparation and output validation should not hold an active-generation permit.
|
|
||||||
A repair remains part of its already-admitted run, but reacquires active
|
|
||||||
generation capacity so repairs cannot exceed the backend limit. It must not be
|
|
||||||
rejected merely because new runs filled the waiting queue after its initial
|
|
||||||
generation.
|
|
||||||
|
|
||||||
Within one backend pool, waiting generation calls should be served in FIFO
|
|
||||||
order, subject to canceled calls being removed. Different backend pools make
|
|
||||||
progress independently; a saturated local backend must not consume
|
|
||||||
OpenRouter's active or waiting capacity.
|
|
||||||
|
|
||||||
The feature will not promise ordering across backend pools or completion order
|
|
||||||
among admitted runs.
|
|
||||||
|
|
||||||
## Capacity Failure And Cancellation
|
|
||||||
|
|
||||||
When a backend's bounded waiting capacity is full, a new `Run` call will fail
|
|
||||||
promptly rather than waiting outside the bounded admission system. The public
|
|
||||||
API will expose a recognizable capacity-exhaustion error identity distinct
|
|
||||||
from invalid configuration, invalid requests, and model-client failures.
|
|
||||||
Rejected calls return no partial result and do not invoke the model client.
|
|
||||||
|
|
||||||
Waiting within the admitted backlog or for active-generation capacity must
|
|
||||||
honor the caller's context. Cancellation or deadline expiry will:
|
|
||||||
|
|
||||||
- stop waiting promptly;
|
|
||||||
- release any admission or generation capacity held by that invocation;
|
|
||||||
- preserve the applicable context error identity; and
|
|
||||||
- avoid invoking the model client if cancellation wins before generation
|
|
||||||
starts.
|
|
||||||
|
|
||||||
Capacity must also be released after preparation, generation, validation,
|
|
||||||
repair, or collaborator failure. One failed or canceled run must not reduce
|
|
||||||
the backend's future usable capacity.
|
|
||||||
|
|
||||||
Elapsed `Run` timing will include time spent waiting after the call is
|
|
||||||
accepted. `PreparedRun` timing will continue to describe preparation rather
|
|
||||||
than queue waiting.
|
|
||||||
|
|
||||||
## Engine And Client Boundaries
|
|
||||||
|
|
||||||
All pools and queued state belong to one `Engine`. Separate engines do not
|
|
||||||
share capacity, even when they register the same backend ID or endpoint. The
|
|
||||||
feature introduces no process-global scheduler.
|
|
||||||
|
|
||||||
The engine will apply policy consistently to the built-in model client and an
|
|
||||||
injected `LLMClient`. Consumers calling their own client outside Promptkit are
|
|
||||||
outside this boundary. Injected clients remain responsible for their internal
|
|
||||||
thread safety and cancellation behavior.
|
|
||||||
|
|
||||||
Backend policy is keyed by the resolved backend ID rather than endpoint text.
|
|
||||||
This preserves stable routing when a selected backend's endpoint is overridden
|
|
||||||
and avoids accidentally combining unrelated registrations that happen to use
|
|
||||||
the same URL.
|
|
||||||
|
|
||||||
## Queue Lifetime And Observability
|
|
||||||
|
|
||||||
Admission state is buffered, ephemeral, and in-process. It is not persisted
|
|
||||||
and has no survival guarantee across engine disposal or process termination.
|
|
||||||
Promptkit will not introduce background job ownership or require consumers to
|
|
||||||
start or stop workers.
|
|
||||||
|
|
||||||
The initial feature does not require public queue-depth metrics, callbacks, or
|
|
||||||
inspection APIs. Capacity errors and ordinary call timing provide the
|
|
||||||
consumer-visible behavior. Operational observability can be added later
|
|
||||||
without coupling the scheduling mechanism to an application logging or
|
|
||||||
metrics system.
|
|
||||||
|
|
||||||
## Compatibility
|
|
||||||
|
|
||||||
Consumer-registered backends and endpoint-only profiles remain unlimited
|
|
||||||
unless concurrency is explicitly configured, preserving their existing
|
|
||||||
behavior.
|
|
||||||
|
|
||||||
The built-in OpenRouter backend will change from unlimited concurrency to a
|
|
||||||
limit of 16 with a bounded waiting capacity of 1024. Ordinary synchronous
|
|
||||||
calls remain unchanged, while unusually high concurrent use may now wait or
|
|
||||||
return the capacity error. This behavioral change must be identified in the
|
|
||||||
release notes for the version that publishes it.
|
|
||||||
|
|
||||||
Adding backend policy fields and a public capacity error is otherwise
|
|
||||||
additive. The change will use a pre-`v1` minor release under Promptkit's
|
|
||||||
[release policy](../release.md#release-model).
|
|
||||||
|
|
||||||
## Non-Goals
|
|
||||||
|
|
||||||
This scope does not include:
|
|
||||||
|
|
||||||
- asynchronous job handles, polling, or detached result delivery;
|
|
||||||
- durable or cross-process queues;
|
|
||||||
- persistence or recovery across engine or process shutdown;
|
|
||||||
- priorities, scheduling weights, or consumer-defined fairness classes;
|
|
||||||
- automatic retries, backoff, rate-limit interpretation, or provider quota
|
|
||||||
discovery;
|
|
||||||
- token-per-minute or request-per-minute rate limiting;
|
|
||||||
- dynamic reconfiguration after engine construction;
|
|
||||||
- per-profile or per-run concurrency overrides;
|
|
||||||
- endpoint-keyed pooling for profiles without a backend ID;
|
|
||||||
- process-global coordination across engines;
|
|
||||||
- application worker lifecycle, logging, tracing, or metrics policy; or
|
|
||||||
- changes to prompt, profile, schema, or model-provider wire formats.
|
|
||||||
|
|
||||||
## Target End State
|
|
||||||
|
|
||||||
This roadmap reaches its target end state when:
|
|
||||||
|
|
||||||
- each engine independently coordinates configured backend capacity;
|
|
||||||
- the built-in OpenRouter backend allows 16 active generations and up to 1024
|
|
||||||
waiting runs;
|
|
||||||
- consumer backends can opt into their own active and waiting limits while
|
|
||||||
remaining unlimited by default;
|
|
||||||
- endpoint overrides retain the selected backend's capacity pool and
|
|
||||||
endpoint-only profiles remain unlimited;
|
|
||||||
- synchronous `Run` callers wait for and receive their ordinary result;
|
|
||||||
- admission is bounded before expensive preparation work where practical;
|
|
||||||
- every initial and repair generation observes the backend's active limit
|
|
||||||
without serializing preparation or validation;
|
|
||||||
- a full waiting queue returns a recognizable capacity error without invoking
|
|
||||||
the model client;
|
|
||||||
- cancellation and all failure paths promptly release capacity and preserve
|
|
||||||
context error identity;
|
|
||||||
- built-in and injected model clients receive the same scheduling behavior;
|
|
||||||
- pools remain ephemeral, engine-scoped, and independent across backend IDs;
|
|
||||||
and
|
|
||||||
- current-state GoDoc, consumer, internal, and release documentation describe
|
|
||||||
the implemented behavior once it lands.
|
|
||||||
@@ -33,36 +33,7 @@ consumers.
|
|||||||
|
|
||||||
## Ideas
|
## Ideas
|
||||||
|
|
||||||
Prompt-independent profile inspection has been selected for active planning in
|
No ideas currently await selection.
|
||||||
the [focused feature roadmap](profile-inspection.md). The remaining ideas are
|
|
||||||
still available for future selection.
|
|
||||||
|
|
||||||
### Prompt-definition inspection
|
|
||||||
|
|
||||||
Provide exact prompt-definition lookup without rendering, placeholder inputs,
|
|
||||||
profile resolution, or model generation, as requested by
|
|
||||||
[Weatherreporter](weatherreporter-promptkit-wishlist.md#priority-2-prompt-definition-inspection).
|
|
||||||
|
|
||||||
- Return caller-owned identity, version, input definitions, default-profile,
|
|
||||||
output-contract, and opaque definition-equality information.
|
|
||||||
- Apply ordinary prompt-source precedence and exact ID/version selection.
|
|
||||||
- Validate the selected definition and referenced prompt content
|
|
||||||
structurally, without returning source bodies or rendered messages.
|
|
||||||
- Leave complete cross-source corpus validation and enumeration outside the
|
|
||||||
initial inspection contract.
|
|
||||||
|
|
||||||
### Structured capacity errors
|
|
||||||
|
|
||||||
Add safe structured context to backend admission rejection, as requested by
|
|
||||||
[Notarius](notarius-promptkit-wishlist.md#priority-4-structured-capacity-errors)
|
|
||||||
and
|
|
||||||
[Weatherreporter](weatherreporter-promptkit-wishlist.md#structured-capacity-errors).
|
|
||||||
|
|
||||||
- Preserve compatibility with `errors.Is(err, ErrCapacityExceeded)`.
|
|
||||||
- Support `errors.As` to obtain the stable backend ID.
|
|
||||||
- Do not expose endpoints, credential configuration or values, request
|
|
||||||
content, or speculative retry timing.
|
|
||||||
- Keep retry and backoff policy with downstream consumers.
|
|
||||||
|
|
||||||
## Entry Format
|
## Entry Format
|
||||||
|
|
||||||
|
|||||||
@@ -1,670 +0,0 @@
|
|||||||
# Prompt-Independent Profile Inspection Implementation Plan
|
|
||||||
|
|
||||||
**Status:** Ready for implementation.
|
|
||||||
|
|
||||||
## Purpose
|
|
||||||
|
|
||||||
This document is the decision-complete implementation plan for
|
|
||||||
[prompt-independent profile inspection](profile-inspection.md). It is written
|
|
||||||
for a coding agent that will implement each stage in order.
|
|
||||||
|
|
||||||
The feature roadmap owns the motivation, consumer workflow, policy choices,
|
|
||||||
compatibility requirements, non-goals, and target end state. This document
|
|
||||||
owns the concrete design, file-level changes, implementation sequence, test
|
|
||||||
ownership, documentation work, validation commands, and completion gates.
|
|
||||||
|
|
||||||
## Implementation Rules
|
|
||||||
|
|
||||||
- Complete the stages in order. Keep the repository compiling and the focused
|
|
||||||
tests passing at every stage boundary.
|
|
||||||
- Preserve unrelated working-tree changes. In particular, retain the accepted
|
|
||||||
feature roadmap and the corresponding future-roadmap and downstream-wishlist
|
|
||||||
edits that may already be uncommitted.
|
|
||||||
- Follow every policy under `docs/policy/`, the task-specific reading guide in
|
|
||||||
`docs/development.md`, and the accepted behavior in
|
|
||||||
`profile-inspection.md`.
|
|
||||||
- Keep the supported API in the root `promptkit` package. Profile repositories,
|
|
||||||
backend resolution, and target assembly remain below Go's `internal/`
|
|
||||||
boundary.
|
|
||||||
- Reuse the exact profile-source, backend-membership, and execution-target
|
|
||||||
precedence used by `Prepare`, `PrepareExecution`, and `Run`. Do not create a
|
|
||||||
second profile loader, backend registry, or target-merging implementation.
|
|
||||||
- Preserve the observable behavior and error ordering of existing preparation
|
|
||||||
and execution methods. The new inspection operation must not require a
|
|
||||||
prompt, prompt default profile, request override, credential value, capacity
|
|
||||||
admission, model client, renderer, artifact reader, schema source, or
|
|
||||||
validator.
|
|
||||||
- Treat structural validation as the existing profile, backend, and effective
|
|
||||||
target invariants. Do not add new endpoint URL policy, provider
|
|
||||||
connectivity checks, reserved-extra-parameter policy, credential syntax, or
|
|
||||||
file-format validation rules as part of this feature.
|
|
||||||
- Never read, retain, return, format, or log an environment credential value.
|
|
||||||
The result may contain only the effective environment-variable name and the
|
|
||||||
direct-key-required boolean.
|
|
||||||
- Return caller-owned public values. In particular, nested
|
|
||||||
`ExecutionTarget.ExtraParams` maps, slices, and objects must not alias
|
|
||||||
engine-owned state or another inspection result.
|
|
||||||
- Keep tests lean and behavior-focused. Reuse existing profile repository,
|
|
||||||
backend registry, target precedence, and preparation tests rather than
|
|
||||||
duplicating their complete matrices.
|
|
||||||
- Update exact contracts in GoDoc with the exported declarations. Update
|
|
||||||
current-state consumer, format, and internal documentation only after the
|
|
||||||
corresponding code exists.
|
|
||||||
- Do not add release notes, change a module version, create a release, or tag a
|
|
||||||
commit as part of this work.
|
|
||||||
|
|
||||||
## Fixed Design
|
|
||||||
|
|
||||||
### Public API
|
|
||||||
|
|
||||||
Add this root-package value immediately after `ExecutionTarget` in `types.go`:
|
|
||||||
|
|
||||||
```go
|
|
||||||
// ProfileInspection is the caller-owned result of Engine.InspectProfile.
|
|
||||||
// The exact GoDoc is specified below.
|
|
||||||
type ProfileInspection struct {
|
|
||||||
ProfileID string
|
|
||||||
EffectiveModelParams ExecutionTarget
|
|
||||||
APIKeyRequired bool
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
Add this method to `Engine` in `engine.go`, immediately before `Prepare`:
|
|
||||||
|
|
||||||
```go
|
|
||||||
func (e *Engine) InspectProfile(
|
|
||||||
ctx context.Context,
|
|
||||||
profileID string,
|
|
||||||
) (*ProfileInspection, error)
|
|
||||||
```
|
|
||||||
|
|
||||||
Do not accept `RunRequest`, `ExecutionTargetOverride`, an API key, an
|
|
||||||
environment override, or options on this method. Exact inspection of one
|
|
||||||
explicit profile ID is the complete operation.
|
|
||||||
|
|
||||||
`ProfileInspection` has no stable JSON contract. Do not add JSON tags,
|
|
||||||
`MarshalJSON`, `UnmarshalJSON`, or a custom string representation. Ordinary Go
|
|
||||||
encoding of the exported fields is not prohibited, but consumers must not be
|
|
||||||
promised compatibility for that encoding.
|
|
||||||
|
|
||||||
The exact type and method GoDoc must establish:
|
|
||||||
|
|
||||||
- surrounding whitespace is trimmed from `profileID`; the resulting nonblank
|
|
||||||
ID is looked up exactly and case-sensitively;
|
|
||||||
- the result applies the engine's ordinary in-memory, configured-source, and
|
|
||||||
built-in profile precedence;
|
|
||||||
- `EffectiveModelParams` contains framework defaults overlaid by the selected
|
|
||||||
backend and then the selected profile, with no request override;
|
|
||||||
- `EffectiveModelParams.BackendID` is empty for endpoint-only profiles;
|
|
||||||
- `EffectiveModelParams.APIKeyEnv` is an environment-variable name and never
|
|
||||||
its value;
|
|
||||||
- `APIKeyRequired` means a direct request credential is required and is
|
|
||||||
mutually exclusive with a nonblank effective `APIKeyEnv`;
|
|
||||||
- the method never derives an ID from a prompt's `default_profile`;
|
|
||||||
- the method performs no prompt lookup, rendering, artifact or schema work,
|
|
||||||
backend admission, provider connectivity check, or model generation;
|
|
||||||
- credential availability is not checked, so an absent or blank named
|
|
||||||
environment variable is not an error;
|
|
||||||
- the returned value and all nested mutable values are caller-owned;
|
|
||||||
- filesystem-backed inspection is a point-in-time lookup and does not freeze
|
|
||||||
a profile for a later execution;
|
|
||||||
- a nil engine matches `ErrInvalidConfig`;
|
|
||||||
- a blank ID matches `ErrInvalidRequest`;
|
|
||||||
- an absent exact ID matches `ErrProfileNotFound` and not `ErrProfileLoad`;
|
|
||||||
- malformed or unreadable profile data, an unknown backend, or an invalid
|
|
||||||
resolved target matches `ErrProfileLoad`;
|
|
||||||
- cancellation observed during profile loading matches `ErrProfileLoad` while
|
|
||||||
preserving the context error through `errors.Is`; and
|
|
||||||
- the method returns no partial result on error.
|
|
||||||
|
|
||||||
Update the `Engine` type GoDoc to include `InspectProfile` among the operations
|
|
||||||
safe for concurrent calls. Do not imply that injected collaborators become
|
|
||||||
safe when their existing contracts do not provide that guarantee.
|
|
||||||
|
|
||||||
Update `doc.go` in the same stage:
|
|
||||||
|
|
||||||
- include `Engine.InspectProfile` in the package's operation list;
|
|
||||||
- include inspection in the concurrency and caller-ownership summary; and
|
|
||||||
- list `ProfileInspection` among construction and inspection values without a
|
|
||||||
stable JSON representation.
|
|
||||||
|
|
||||||
Do not add `ProfileInspection` to the stable JSON list.
|
|
||||||
|
|
||||||
### Internal Result
|
|
||||||
|
|
||||||
Add this internal value to `internal/domain/domain.go` near
|
|
||||||
`ExecutionProfile` and `ExecutionTarget`:
|
|
||||||
|
|
||||||
```go
|
|
||||||
// ProfileInspection is the resolved result of exact profile inspection.
|
|
||||||
type ProfileInspection struct {
|
|
||||||
ProfileID string
|
|
||||||
EffectiveModelParams ExecutionTarget
|
|
||||||
APIKeyRequired bool
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
Do not add JSON or YAML tags. The internal value is a use-case result, not a
|
|
||||||
file format or persistence contract.
|
|
||||||
|
|
||||||
Add a root conversion in `convert.go`:
|
|
||||||
|
|
||||||
```go
|
|
||||||
func fromDomainProfileInspection(
|
|
||||||
inspection *domain.ProfileInspection,
|
|
||||||
) *ProfileInspection
|
|
||||||
```
|
|
||||||
|
|
||||||
The conversion must:
|
|
||||||
|
|
||||||
- return `nil` for a nil internal value;
|
|
||||||
- copy the scalar fields;
|
|
||||||
- convert the target through the existing `fromDomainExecutionTarget`; and
|
|
||||||
- therefore use the existing recursive `copyAnyMap` boundary for nested extra
|
|
||||||
parameters.
|
|
||||||
|
|
||||||
Do not add `APIKeyRequired` to the public `ExecutionTarget`. Keeping the
|
|
||||||
requirement on `ProfileInspection` preserves the existing stable JSON and
|
|
||||||
general execution-target contract.
|
|
||||||
|
|
||||||
### Shared Profile Selection
|
|
||||||
|
|
||||||
Add `internal/usecase/profile_inspection.go`. Define one private selection
|
|
||||||
value:
|
|
||||||
|
|
||||||
```go
|
|
||||||
type resolvedProfileSelection struct {
|
|
||||||
id string
|
|
||||||
profile *domain.ExecutionProfile
|
|
||||||
backend *domain.Backend
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
Add a private runner helper:
|
|
||||||
|
|
||||||
```go
|
|
||||||
func (r *Runner) resolveProfileSelection(
|
|
||||||
ctx context.Context,
|
|
||||||
profileID string,
|
|
||||||
) (*resolvedProfileSelection, error)
|
|
||||||
```
|
|
||||||
|
|
||||||
The helper must perform these operations in order:
|
|
||||||
|
|
||||||
1. trim surrounding whitespace from `profileID`;
|
|
||||||
2. reject a blank result with `ErrInvalidRequest`;
|
|
||||||
3. reject a nil runner profile repository with `ErrProfileLoad` rather than
|
|
||||||
panicking;
|
|
||||||
4. call the existing `profile.Repository.GetProfile` exactly once;
|
|
||||||
5. wrap every repository error with `ErrProfileLoad` while preserving the
|
|
||||||
underlying error with `%w`;
|
|
||||||
6. reject a nil profile returned without an error as `ErrProfileLoad`;
|
|
||||||
7. make a value copy of the selected profile before normalization so the
|
|
||||||
runner does not mutate repository-owned state;
|
|
||||||
8. trim the copied profile's backend ID;
|
|
||||||
9. when that ID is nonblank, resolve it exactly once through the existing
|
|
||||||
`BackendResolver`, preserving the current unknown-backend
|
|
||||||
`ErrProfileLoad` wrapping and backend ID context; and
|
|
||||||
10. return the normalized exact ID, copied profile, and optional defensive
|
|
||||||
backend value.
|
|
||||||
|
|
||||||
Do not inspect all sources, enumerate profiles, expose a source path, fall back
|
|
||||||
after a malformed higher-precedence match, or infer a backend from an endpoint
|
|
||||||
or model.
|
|
||||||
|
|
||||||
Refactor the profile-loading and backend-resolution block in
|
|
||||||
`Runner.resolvePreparation` to call `resolveProfileSelection`. Leave prompt
|
|
||||||
selection, prompt default-profile selection, prompt hashing, target overrides,
|
|
||||||
request credentials, output-contract resolution, artifact work, and all later
|
|
||||||
ordering where they currently occur.
|
|
||||||
|
|
||||||
This refactor must preserve:
|
|
||||||
|
|
||||||
- explicit request profile selection over prompt `default_profile`;
|
|
||||||
- the existing `ErrProfileRequired` result when neither exists;
|
|
||||||
- exact repository and backend lookup behavior;
|
|
||||||
- existing public not-found versus profile-load classification;
|
|
||||||
- request override precedence and target-presence tracking;
|
|
||||||
- credential checking during ordinary preparation; and
|
|
||||||
- current `Prepare`, `PrepareExecution`, `Run`, and `RunPrepared` behavior.
|
|
||||||
|
|
||||||
### Shared Target Resolution And Structural Validation
|
|
||||||
|
|
||||||
Continue using the existing `resolveExecutionTarget` for framework, backend,
|
|
||||||
profile, and optional request precedence. Do not duplicate or move the
|
|
||||||
individual merge rules into the inspection code.
|
|
||||||
|
|
||||||
Extract the two existing effective-target requiredness checks from
|
|
||||||
`resolvePreparation` into a private pure helper:
|
|
||||||
|
|
||||||
```go
|
|
||||||
func validateResolvedExecutionTarget(
|
|
||||||
target domain.ExecutionTarget,
|
|
||||||
) error
|
|
||||||
```
|
|
||||||
|
|
||||||
It checks only that the trimmed endpoint and model are nonblank and returns a
|
|
||||||
plain descriptive error. It does not validate URL syntax, contact the
|
|
||||||
endpoint, validate credentials, or introduce new parameter rules.
|
|
||||||
|
|
||||||
`resolvePreparation` calls this helper at the same current point: after
|
|
||||||
request override resolution and direct API-key assignment, and before
|
|
||||||
`validateAPIKey`. It wraps a failure with `ErrInvalidRequest`, preserving
|
|
||||||
ordinary preparation behavior.
|
|
||||||
|
|
||||||
Add the internal inspection method:
|
|
||||||
|
|
||||||
```go
|
|
||||||
func (r *Runner) InspectProfile(
|
|
||||||
ctx context.Context,
|
|
||||||
profileID string,
|
|
||||||
) (*domain.ProfileInspection, error)
|
|
||||||
```
|
|
||||||
|
|
||||||
It performs these operations:
|
|
||||||
|
|
||||||
1. trim and reject a blank ID as `ErrInvalidRequest`;
|
|
||||||
2. if the supplied context is already canceled, return an error wrapping both
|
|
||||||
`ErrProfileLoad` and `ctx.Err()` without touching the repository;
|
|
||||||
3. call `resolveProfileSelection`;
|
|
||||||
4. call `resolveExecutionTarget` with the selected backend, selected profile,
|
|
||||||
and a nil request override;
|
|
||||||
5. wrap any target-resolution error with `ErrProfileLoad`;
|
|
||||||
6. call `validateResolvedExecutionTarget` and wrap a failure with
|
|
||||||
`ErrProfileLoad`;
|
|
||||||
7. defensively clear `target.APIKey`;
|
|
||||||
8. copy `target.APIKeyRequired` into the result's `APIKeyRequired`; and
|
|
||||||
9. return the normalized profile ID and effective target.
|
|
||||||
|
|
||||||
Do not call `validateAPIKey`, `os.Getenv`, prompt repositories, artifact
|
|
||||||
readers, renderers, validators, capacity admission, output repair, or the
|
|
||||||
model client. Do not populate request target-presence state.
|
|
||||||
|
|
||||||
The internal `ExecutionTarget` may retain its private `APIKeyRequired` field;
|
|
||||||
the root conversion intentionally omits that private field from the public
|
|
||||||
target and publishes the separate inspection boolean.
|
|
||||||
|
|
||||||
### Root Facade And Error Mapping
|
|
||||||
|
|
||||||
`Engine.InspectProfile` follows the existing facade pattern:
|
|
||||||
|
|
||||||
1. reject a nil engine or nil runner with `ErrInvalidConfig`;
|
|
||||||
2. pass the context and string ID directly to `Runner.InspectProfile`;
|
|
||||||
3. map internal failures through the existing `mapPublicError`; and
|
|
||||||
4. convert a successful result with `fromDomainProfileInspection`.
|
|
||||||
|
|
||||||
No request conversion is needed. Do not add a public sentinel or typed error.
|
|
||||||
|
|
||||||
The existing error mapping already has the required ordering:
|
|
||||||
|
|
||||||
- an underlying `profile.ErrProfileNotFound` maps to
|
|
||||||
`ErrProfileNotFound` before the enclosing use-case `ErrProfileLoad` is
|
|
||||||
considered;
|
|
||||||
- other use-case `ErrProfileLoad` failures map to `ErrProfileLoad`; and
|
|
||||||
- `usecase.ErrInvalidRequest` maps to `ErrInvalidRequest`.
|
|
||||||
|
|
||||||
Do not reorder or otherwise change `errors.go` unless a focused test proves
|
|
||||||
that the existing mapping does not meet this plan. Preserve underlying
|
|
||||||
repository, backend, and context errors through `errors.Is`.
|
|
||||||
|
|
||||||
### Credentials, Ownership, And Concurrency
|
|
||||||
|
|
||||||
Inspection has no input through which a direct credential value can enter.
|
|
||||||
Backend and file-profile `APIKeyEnv` values remain names only. An in-memory
|
|
||||||
profile with `APIKeyRequired` clears an inherited backend environment name
|
|
||||||
through the existing target merge behavior.
|
|
||||||
|
|
||||||
The implementation must succeed for:
|
|
||||||
|
|
||||||
- a backend or profile naming an unset environment variable;
|
|
||||||
- a backend or profile naming an environment variable whose value is blank;
|
|
||||||
- an in-memory profile requiring a direct key when no key is supplied; and
|
|
||||||
- a profile requiring no credential.
|
|
||||||
|
|
||||||
The result expresses these effective states:
|
|
||||||
|
|
||||||
| `EffectiveModelParams.APIKeyEnv` | `APIKeyRequired` | Meaning |
|
|
||||||
| --- | --- | --- |
|
|
||||||
| nonblank | `false` | The profile resolves to the named environment source. |
|
|
||||||
| blank | `true` | A later execution must supply a direct key or explicit request environment override. |
|
|
||||||
| blank | `false` | The resolved target declares no credential requirement. |
|
|
||||||
|
|
||||||
The implementation must never produce a successful public result with both a
|
|
||||||
nonblank `APIKeyEnv` and `APIKeyRequired == true`.
|
|
||||||
|
|
||||||
Public ownership is enforced at the root conversion boundary. A consumer may
|
|
||||||
mutate the returned target and arbitrarily nested JSON-compatible extra
|
|
||||||
parameters without affecting:
|
|
||||||
|
|
||||||
- the engine registry or profile repository;
|
|
||||||
- a later `InspectProfile` call;
|
|
||||||
- `Prepare`, prepared execution, or `Run`; or
|
|
||||||
- another result already returned to a caller.
|
|
||||||
|
|
||||||
No new mutable engine state, cache, global registry, lock, or goroutine is
|
|
||||||
needed. Concurrency safety follows from the existing immutable registry and
|
|
||||||
repository contracts plus per-call values.
|
|
||||||
|
|
||||||
### Test Ownership
|
|
||||||
|
|
||||||
Add focused internal tests in
|
|
||||||
`internal/usecase/profile_inspection_test.go`. Use small repository and backend
|
|
||||||
fakes already present in the package where practical; do not build a parallel
|
|
||||||
fixture framework.
|
|
||||||
|
|
||||||
The internal tests own:
|
|
||||||
|
|
||||||
- blank-ID rejection;
|
|
||||||
- pre-canceled context classification without repository access;
|
|
||||||
- one exact profile and backend lookup;
|
|
||||||
- framework-default, backend, and profile target precedence through the
|
|
||||||
existing resolver;
|
|
||||||
- endpoint-only behavior;
|
|
||||||
- effective environment-name reporting without availability checks;
|
|
||||||
- direct-key-required behavior clearing an inherited backend environment
|
|
||||||
name;
|
|
||||||
- missing-profile and unknown-backend wrapping; and
|
|
||||||
- nil repository, nil returned profile, and invalid resolved target defenses
|
|
||||||
only if these cases are not already cheaply covered through existing runner
|
|
||||||
fakes.
|
|
||||||
|
|
||||||
Keep the internal matrix compact. Profile parser tests continue to own YAML,
|
|
||||||
duplicates, raw-key rejection, ranges, and source discovery. Backend tests
|
|
||||||
continue to own registration validation. Existing target tests continue to
|
|
||||||
own every merge field and numeric override boundary.
|
|
||||||
|
|
||||||
Add external-package public contract tests in `public_contract_test.go`. They
|
|
||||||
own:
|
|
||||||
|
|
||||||
- the exported method and result shape through normal Go use;
|
|
||||||
- a nil engine returning `ErrInvalidConfig`;
|
|
||||||
- blank, missing, malformed, and unknown-backend public error identities,
|
|
||||||
including that not-found does not also match `ErrProfileLoad`;
|
|
||||||
- a pre-canceled inspection preserving both `ErrProfileLoad` and the context
|
|
||||||
error without consulting the profile repository;
|
|
||||||
- successful inspection with an absent credential environment value;
|
|
||||||
- no prompt dependency, demonstrated with an empty configured prompt
|
|
||||||
`fs.FS`;
|
|
||||||
- no model invocation, using the existing deterministic fake client;
|
|
||||||
- the three credential-requirement result states;
|
|
||||||
- an endpoint-only profile's empty backend ID;
|
|
||||||
- deep caller ownership of nested extra parameters across repeated
|
|
||||||
inspections; and
|
|
||||||
- representative equivalence between `InspectProfile.EffectiveModelParams`
|
|
||||||
and `Prepare.EffectiveModelParams` for the same engine, profile, and source
|
|
||||||
state with no request execution override.
|
|
||||||
|
|
||||||
Combine closely related assertions into a few readable behavioral tests.
|
|
||||||
Do not add a JSON golden test because `ProfileInspection` deliberately has no
|
|
||||||
stable JSON contract. Do not duplicate the complete repository-precedence,
|
|
||||||
target-field, parser-error, or credential-execution suites at the root layer.
|
|
||||||
|
|
||||||
Existing tests that must continue passing without semantic edits include:
|
|
||||||
|
|
||||||
- profile repository and built-in fallback tests;
|
|
||||||
- backend registry defensive-copy and lookup tests;
|
|
||||||
- execution-target precedence tests;
|
|
||||||
- `Prepare` and `Run` profile/backend equivalence tests;
|
|
||||||
- prepared-execution snapshot and credential tests; and
|
|
||||||
- public error mapping tests.
|
|
||||||
|
|
||||||
### Documentation Ownership
|
|
||||||
|
|
||||||
After the implementation and public tests pass, update current-state
|
|
||||||
documentation:
|
|
||||||
|
|
||||||
- `docs/consumers/pkg-promptkit.md`: add a concise task-oriented section near
|
|
||||||
profile selection showing `Engine.InspectProfile`, explaining when to use it
|
|
||||||
instead of a synthetic `Prepare`, how to interpret `APIKeyEnv` and
|
|
||||||
`APIKeyRequired`, and that credential enforcement timing remains
|
|
||||||
application policy. Link to the exact GoDoc rather than restating every
|
|
||||||
error and field contract.
|
|
||||||
- `docs/formats.md`: update profile selection and backend-membership wording
|
|
||||||
so exact inspection is recognized as another consumer of the existing
|
|
||||||
source and profile precedence. Do not redefine the exported method here.
|
|
||||||
- `docs/internal/runner.md`: describe the shared profile-selection boundary
|
|
||||||
and explain that inspection stops after structural target resolution,
|
|
||||||
before credential availability and all prompt-dependent work.
|
|
||||||
- `docs/internal/sources.md`: record that exact profile inspection performs
|
|
||||||
one point-in-time profile-source lookup without reading prompt, input, or
|
|
||||||
schema sources, and replace any ambiguous use of “inspection value” for
|
|
||||||
`PreparedRun` with “preparation value.”
|
|
||||||
- `docs/internal/overview.md`: add profile inspection to the existing root
|
|
||||||
facade and `internal/usecase` responsibility descriptions. Do not add a new
|
|
||||||
component or package row.
|
|
||||||
- `docs/roadmap/future.md`: remove the statement that profile inspection is in
|
|
||||||
active planning and leave the remaining unselected ideas intact.
|
|
||||||
- `docs/roadmap/notarius-promptkit-wishlist.md` and
|
|
||||||
`docs/roadmap/weatherreporter-promptkit-wishlist.md`: change the profile
|
|
||||||
inspection disposition from accepted planning to implemented behavior and
|
|
||||||
link to the durable consumer guidance or GoDoc rather than treating the
|
|
||||||
roadmap as current-state documentation.
|
|
||||||
- `docs/roadmap/profile-inspection.md`: change its status to `Complete` only
|
|
||||||
after the code, tests, current-state documentation, and full validation are
|
|
||||||
complete.
|
|
||||||
- `docs/roadmap/implementation.md`: change its status to `Complete` only after
|
|
||||||
every completion gate in this plan is satisfied.
|
|
||||||
|
|
||||||
Do not update release guidance in this feature implementation. A later release
|
|
||||||
pass decides whether the change warrants a supplemental release document.
|
|
||||||
|
|
||||||
## Stage 1: Implement Shared Internal Profile Resolution
|
|
||||||
|
|
||||||
### Objective
|
|
||||||
|
|
||||||
Add the internal inspection result and operation, share profile/backend
|
|
||||||
selection and structural target validation with ordinary preparation, and
|
|
||||||
prove the internal behavior without publishing the root API yet.
|
|
||||||
|
|
||||||
### Implementation Prompt
|
|
||||||
|
|
||||||
Implement only Stage 1 of
|
|
||||||
`docs/roadmap/implementation.md`. Read the complete feature roadmap and the
|
|
||||||
implementation rules and fixed design above before editing.
|
|
||||||
|
|
||||||
1. Add `domain.ProfileInspection` to `internal/domain/domain.go` without
|
|
||||||
serialization tags.
|
|
||||||
2. Add `internal/usecase/profile_inspection.go` with
|
|
||||||
`resolvedProfileSelection`, `resolveProfileSelection`,
|
|
||||||
`validateResolvedExecutionTarget`, and `Runner.InspectProfile` exactly as
|
|
||||||
specified.
|
|
||||||
3. Refactor `Runner.resolvePreparation` in `internal/usecase/runner.go` to use
|
|
||||||
the shared selection and target-validation helpers while preserving its
|
|
||||||
current ordering, error classification, target overrides, credentials, and
|
|
||||||
results.
|
|
||||||
4. Add lean behavioral tests in
|
|
||||||
`internal/usecase/profile_inspection_test.go`.
|
|
||||||
5. Run the focused validation below and repair regressions before ending the
|
|
||||||
stage.
|
|
||||||
|
|
||||||
Do not add the root public type or method, update current-state documentation,
|
|
||||||
or alter profile formats, backend registration, credential availability,
|
|
||||||
capacity, model-client, or provider behavior in this stage.
|
|
||||||
|
|
||||||
### Focused Validation
|
|
||||||
|
|
||||||
Run:
|
|
||||||
|
|
||||||
```sh
|
|
||||||
gofmt -w internal/domain/domain.go \
|
|
||||||
internal/usecase/profile_inspection.go \
|
|
||||||
internal/usecase/profile_inspection_test.go \
|
|
||||||
internal/usecase/runner.go
|
|
||||||
go test ./internal/usecase ./internal/profile ./internal/profile/builtin \
|
|
||||||
./internal/backend
|
|
||||||
go test ./internal/usecase -run \
|
|
||||||
'TestRunner(InspectProfile|Prepare|Run|PrepareExecution|RunPrepared)'
|
|
||||||
go vet ./internal/usecase ./internal/profile ./internal/backend
|
|
||||||
```
|
|
||||||
|
|
||||||
### Completion Gate
|
|
||||||
|
|
||||||
Stage 1 is complete only when:
|
|
||||||
|
|
||||||
- inspection reaches profile and backend resolution without any prompt or
|
|
||||||
execution collaborator;
|
|
||||||
- profile/backend selection is shared with ordinary preparation;
|
|
||||||
- target merging and requiredness checks are shared rather than duplicated;
|
|
||||||
- inspection does not check credential values or capacity;
|
|
||||||
- internal errors preserve the required identities;
|
|
||||||
- the ordinary preparation and execution tests still pass without weakened
|
|
||||||
assertions; and
|
|
||||||
- no root public API or current-state documentation claims the feature yet.
|
|
||||||
|
|
||||||
## Stage 2: Publish The Root Facade And Public Contract
|
|
||||||
|
|
||||||
### Objective
|
|
||||||
|
|
||||||
Expose the minimal caller-owned inspection API through `Engine`, preserve
|
|
||||||
public error and JSON compatibility, and protect the consumer-visible
|
|
||||||
contract.
|
|
||||||
|
|
||||||
### Implementation Prompt
|
|
||||||
|
|
||||||
Implement only Stage 2 of
|
|
||||||
`docs/roadmap/implementation.md` after Stage 1 satisfies its completion gate.
|
|
||||||
Re-read the fixed public API, error, credential, ownership, and test sections
|
|
||||||
before editing.
|
|
||||||
|
|
||||||
1. Add `ProfileInspection` and its exact GoDoc to `types.go` immediately after
|
|
||||||
`ExecutionTarget`. Do not add JSON tags or change `ExecutionTarget`.
|
|
||||||
2. Add `fromDomainProfileInspection` to `convert.go` using
|
|
||||||
`fromDomainExecutionTarget`.
|
|
||||||
3. Add `Engine.InspectProfile` and its exact GoDoc to `engine.go` immediately
|
|
||||||
before `Prepare`.
|
|
||||||
4. Update `Engine` GoDoc to include concurrent inspection.
|
|
||||||
5. Update the package GoDoc in `doc.go` for operation discovery, concurrency,
|
|
||||||
ownership, and the non-stable JSON classification.
|
|
||||||
6. Add compact external-package contract coverage in
|
|
||||||
`public_contract_test.go`, using existing fixtures and fakes where they
|
|
||||||
remain clear.
|
|
||||||
7. Confirm that the existing `errors.go` mapping meets the plan; do not change
|
|
||||||
it unless a required public identity test fails for a genuine mapping
|
|
||||||
reason.
|
|
||||||
8. Run the focused validation below and repair regressions before ending the
|
|
||||||
stage.
|
|
||||||
|
|
||||||
Do not add enumeration, request overrides, profile fingerprints, JSON
|
|
||||||
stability, prompt inspection, environment credential checks, caching, or
|
|
||||||
current-state prose documentation in this stage.
|
|
||||||
|
|
||||||
### Focused Validation
|
|
||||||
|
|
||||||
Run:
|
|
||||||
|
|
||||||
```sh
|
|
||||||
gofmt -w doc.go types.go convert.go engine.go public_contract_test.go
|
|
||||||
go test .
|
|
||||||
go test ./internal/usecase
|
|
||||||
go test . -run 'Test(InspectProfile|.*Profile.*PublicError|.*Profile.*Contract)'
|
|
||||||
go vet .
|
|
||||||
go build .
|
|
||||||
```
|
|
||||||
|
|
||||||
If the repository's actual focused test names differ, use the implemented test
|
|
||||||
names rather than weakening or skipping the intended assertions.
|
|
||||||
|
|
||||||
### Completion Gate
|
|
||||||
|
|
||||||
Stage 2 is complete only when:
|
|
||||||
|
|
||||||
- a consumer can call `Engine.InspectProfile` through the root package;
|
|
||||||
- the result contains only the normalized ID, caller-owned effective target,
|
|
||||||
and direct-key-required signal;
|
|
||||||
- no credential value can enter the result;
|
|
||||||
- environment, direct, and no-credential states are unambiguous;
|
|
||||||
- nil, blank, missing, malformed, unknown-backend, and cancellation errors
|
|
||||||
have the required public identities;
|
|
||||||
- not-found remains distinct from profile-load failure;
|
|
||||||
- the effective target matches ordinary preparation absent request overrides;
|
|
||||||
- repeated calls and caller mutation cannot alter engine-owned state;
|
|
||||||
- no stable JSON or new error contract was introduced; and
|
|
||||||
- existing public preparation, execution, and stable JSON tests remain
|
|
||||||
unchanged and passing.
|
|
||||||
|
|
||||||
## Stage 3: Update Documentation And Validate The Repository
|
|
||||||
|
|
||||||
### Objective
|
|
||||||
|
|
||||||
Make implemented profile inspection discoverable in its canonical
|
|
||||||
documentation, reconcile temporary roadmap state, and complete full repository
|
|
||||||
validation.
|
|
||||||
|
|
||||||
### Implementation Prompt
|
|
||||||
|
|
||||||
Implement only Stage 3 of
|
|
||||||
`docs/roadmap/implementation.md` after Stages 1 and 2 satisfy their completion
|
|
||||||
gates.
|
|
||||||
|
|
||||||
1. Update `docs/consumers/pkg-promptkit.md`, `docs/formats.md`,
|
|
||||||
`docs/internal/runner.md`, `docs/internal/sources.md`, and
|
|
||||||
`docs/internal/overview.md` according to the documentation ownership
|
|
||||||
section above.
|
|
||||||
2. Update the future catalog and both downstream wishlist dispositions so they
|
|
||||||
no longer describe profile inspection as merely accepted work.
|
|
||||||
3. Check every changed Markdown link and confirm its file and heading target.
|
|
||||||
4. Run the full validation sequence below.
|
|
||||||
5. Only after every check passes, set the feature roadmap and this
|
|
||||||
implementation plan to `**Status:** Complete.`
|
|
||||||
6. Re-run `git diff --check` after the status edits.
|
|
||||||
|
|
||||||
Do not add release notes, an example program, a new public package, or a
|
|
||||||
duplicate API reference. Keep detailed contracts in GoDoc and task-oriented
|
|
||||||
usage in the consumer guide.
|
|
||||||
|
|
||||||
### Full Validation
|
|
||||||
|
|
||||||
Run from the repository root:
|
|
||||||
|
|
||||||
```sh
|
|
||||||
gofmt -w internal/domain/domain.go \
|
|
||||||
internal/usecase/profile_inspection.go \
|
|
||||||
internal/usecase/profile_inspection_test.go \
|
|
||||||
internal/usecase/runner.go \
|
|
||||||
doc.go types.go convert.go engine.go public_contract_test.go
|
|
||||||
gofmt -l $(git ls-files '*.go')
|
|
||||||
go test ./...
|
|
||||||
go test -race ./...
|
|
||||||
go vet ./...
|
|
||||||
go build ./...
|
|
||||||
go run ./examples/go-library/prepare
|
|
||||||
git diff --check
|
|
||||||
git status --short
|
|
||||||
```
|
|
||||||
|
|
||||||
The `gofmt -l` command must print no paths. The maintained example must remain
|
|
||||||
offline and must not require a real credential or provider.
|
|
||||||
|
|
||||||
Inspect the final diff and confirm:
|
|
||||||
|
|
||||||
- only files required by this feature and pre-existing user changes are
|
|
||||||
present;
|
|
||||||
- no credential values, private infrastructure details, workspace files,
|
|
||||||
local replacements, generated binaries, or unrelated formatting changes
|
|
||||||
were added;
|
|
||||||
- the public declarations and GoDoc own exact API behavior;
|
|
||||||
- current-state documentation describes only implemented behavior and links
|
|
||||||
to canonical owners;
|
|
||||||
- roadmap documents contain future scope or completion status rather than a
|
|
||||||
duplicate current API reference; and
|
|
||||||
- no release, commit, or tag was created.
|
|
||||||
|
|
||||||
### Completion Gate
|
|
||||||
|
|
||||||
The implementation is complete only when:
|
|
||||||
|
|
||||||
- every Stage 1 and Stage 2 gate remains satisfied;
|
|
||||||
- the complete ordinary and race-enabled suites pass;
|
|
||||||
- vet, build, formatting, the maintained offline example, Markdown links, and
|
|
||||||
whitespace checks pass;
|
|
||||||
- consumer, format, internal, future, and wishlist documentation are
|
|
||||||
consistent with the implemented boundary;
|
|
||||||
- both roadmap statuses are `Complete`;
|
|
||||||
- the working tree contains no unintended files or changes; and
|
|
||||||
- the repository is ready for maintainer review without a commit or release
|
|
||||||
having been created by this plan.
|
|
||||||
|
|
||||||
## Open Questions
|
|
||||||
|
|
||||||
None. The accepted feature roadmap and the fixed decisions above fully specify
|
|
||||||
the implementation boundary.
|
|
||||||
@@ -1,163 +0,0 @@
|
|||||||
# Local Backend Convenience
|
|
||||||
|
|
||||||
**Status:** Complete.
|
|
||||||
|
|
||||||
## Purpose
|
|
||||||
|
|
||||||
Make the common case of using a local OpenAI-compatible endpoint concise and
|
|
||||||
easy to discover without introducing implicit configuration or a separate
|
|
||||||
backend abstraction.
|
|
||||||
|
|
||||||
The existing `Backend` type and registry remain the canonical, fully
|
|
||||||
configurable interface. A small convenience constructor will cover the usual
|
|
||||||
local-network case, while improved consumer documentation will make it clear
|
|
||||||
when an endpoint-only profile, the convenience constructor, or a complete
|
|
||||||
`Backend` value is appropriate.
|
|
||||||
|
|
||||||
## Motivation
|
|
||||||
|
|
||||||
Consumers can already use a local endpoint by setting `Profile.Endpoint`, or
|
|
||||||
register one as a backend with `WithBackend`. The first option is concise but
|
|
||||||
does not provide shared backend-level concurrency control. The second supports
|
|
||||||
the complete backend feature set but requires consumers to understand and
|
|
||||||
populate several fields for a common configuration.
|
|
||||||
|
|
||||||
Most consumers adding a local backend need only:
|
|
||||||
|
|
||||||
- a stable backend ID;
|
|
||||||
- an OpenAI-compatible endpoint; and
|
|
||||||
- a concurrency limit appropriate for the local server.
|
|
||||||
|
|
||||||
Promptkit should provide a direct path for that case while keeping all
|
|
||||||
configuration explicit and preserving the full registry interface for
|
|
||||||
advanced needs.
|
|
||||||
|
|
||||||
## Consumer Paths
|
|
||||||
|
|
||||||
Documentation should present three progressively more configurable paths:
|
|
||||||
|
|
||||||
1. Set `Profile.Endpoint` when a profile only needs to target a local endpoint
|
|
||||||
and does not need shared backend policy.
|
|
||||||
2. Use the local-backend convenience constructor when profiles should share a
|
|
||||||
named local endpoint and its concurrency limit.
|
|
||||||
3. Construct a complete `Backend` value when the consumer needs a custom
|
|
||||||
backend ID, authentication, extra request parameters, an explicit queue
|
|
||||||
capacity, or multiple local backends.
|
|
||||||
|
|
||||||
These are complementary interfaces. The convenience constructor must return an
|
|
||||||
ordinary `Backend`, so it does not create a second configuration model.
|
|
||||||
|
|
||||||
## Public Convenience API
|
|
||||||
|
|
||||||
The public package should expose:
|
|
||||||
|
|
||||||
```go
|
|
||||||
const BackendLocal = "local"
|
|
||||||
|
|
||||||
func LocalBackend(endpoint string, concurrencyLimit int) Backend
|
|
||||||
```
|
|
||||||
|
|
||||||
`LocalBackend` should return a `Backend` with:
|
|
||||||
|
|
||||||
- `ID` set to `BackendLocal`;
|
|
||||||
- `Endpoint` set to the supplied endpoint;
|
|
||||||
- `ConcurrencyLimit` set to the supplied limit; and
|
|
||||||
- all other fields left at their zero values.
|
|
||||||
|
|
||||||
The returned value is passed to `WithBackend` and follows the same copying,
|
|
||||||
normalization, validation, and registration rules as any consumer-constructed
|
|
||||||
`Backend`.
|
|
||||||
|
|
||||||
The constructor should be a transparent value constructor. It should not read
|
|
||||||
environment variables, mutate global state, register the backend, validate
|
|
||||||
arguments independently, or create profiles. Consumers may inspect or modify
|
|
||||||
the returned value before registration, although documentation should direct
|
|
||||||
substantially customized configurations to the full `Backend` form.
|
|
||||||
|
|
||||||
## Identity and Registration
|
|
||||||
|
|
||||||
`BackendLocal` is a conventional ID used by the convenience constructor. It is
|
|
||||||
not pre-registered and should not become a specially reserved registry ID.
|
|
||||||
Consumers remain responsible for registering the returned backend with
|
|
||||||
`WithBackend` and naming it from profiles through `BackendID`.
|
|
||||||
|
|
||||||
This distinction preserves compatibility with consumers that may already
|
|
||||||
register their own backend using the ID `"local"`. Normal duplicate-ID rules
|
|
||||||
still apply if a consumer attempts to register more than one backend with that
|
|
||||||
ID.
|
|
||||||
|
|
||||||
Consumers that need multiple local endpoints should choose distinct IDs and
|
|
||||||
use complete `Backend` values rather than the single conventional helper ID.
|
|
||||||
|
|
||||||
## Concurrency and Queue Semantics
|
|
||||||
|
|
||||||
The constructor must preserve the existing backend concurrency contract:
|
|
||||||
|
|
||||||
- a positive concurrency limit bounds simultaneous requests and uses the
|
|
||||||
existing default queue capacity because `QueueCapacity` remains `nil`;
|
|
||||||
- a zero concurrency limit leaves the backend unconstrained; and
|
|
||||||
- a negative concurrency limit is rejected through the existing engine
|
|
||||||
configuration validation path.
|
|
||||||
|
|
||||||
The constructor should not select a hidden default concurrency limit. Local
|
|
||||||
servers vary substantially in capacity, so the consumer should make this
|
|
||||||
choice explicitly.
|
|
||||||
|
|
||||||
## Documentation
|
|
||||||
|
|
||||||
The final documentation state has two canonical surfaces:
|
|
||||||
|
|
||||||
- Public Go documentation describes the exact contract of
|
|
||||||
`BackendLocal` and `LocalBackend`, including their conventional,
|
|
||||||
non-pre-registered nature.
|
|
||||||
- The [promptkit consumer guide](../consumers/pkg-promptkit.md) includes
|
|
||||||
a task-oriented local-endpoint section that shows the three consumer paths,
|
|
||||||
explains the decision between them, and provides concise examples of the
|
|
||||||
endpoint-only and convenience-constructor forms.
|
|
||||||
|
|
||||||
The consumer guide continues to document the full `Backend` interface as
|
|
||||||
the advanced path rather than attempting to reproduce every configuration
|
|
||||||
variation through convenience APIs.
|
|
||||||
|
|
||||||
## Compatibility
|
|
||||||
|
|
||||||
This feature is additive:
|
|
||||||
|
|
||||||
- existing endpoint-only profiles continue to work unchanged;
|
|
||||||
- existing `Backend` values and `WithBackend` registrations remain the
|
|
||||||
canonical general-purpose interface;
|
|
||||||
- existing registrations using the literal ID `"local"` remain valid; and
|
|
||||||
- OpenRouter defaults and all other backend behavior remain unchanged.
|
|
||||||
|
|
||||||
No consumer is required to adopt the convenience constructor.
|
|
||||||
|
|
||||||
## Non-Goals
|
|
||||||
|
|
||||||
This work does not include:
|
|
||||||
|
|
||||||
- pre-registering or implicitly enabling a local backend;
|
|
||||||
- discovering a local endpoint, API key, or concurrency limit from environment
|
|
||||||
variables;
|
|
||||||
- adding local-backend fields to `Config`;
|
|
||||||
- selecting a default local model or creating a profile automatically;
|
|
||||||
- adding a combined backend-and-profile constructor;
|
|
||||||
- adding convenience parameters for API keys, extra request parameters, or
|
|
||||||
queue capacity;
|
|
||||||
- replacing or redesigning the backend registry;
|
|
||||||
- adding support for non-OpenAI-compatible local APIs; or
|
|
||||||
- changing backend routing, scheduling, or queue behavior.
|
|
||||||
|
|
||||||
## Target End State
|
|
||||||
|
|
||||||
After this work:
|
|
||||||
|
|
||||||
- consumers with a simple one-profile local endpoint can continue to configure
|
|
||||||
it directly on the profile;
|
|
||||||
- consumers needing a shared local endpoint and concurrency policy can express
|
|
||||||
it with one `LocalBackend` call and register the returned value normally;
|
|
||||||
- consumers with advanced or multiple-local-backend requirements have a clear
|
|
||||||
path to the complete `Backend` interface;
|
|
||||||
- all local configuration remains explicit, inspectable, and compatible with
|
|
||||||
dependency injection; and
|
|
||||||
- canonical documentation makes the simplest suitable interface easy to find
|
|
||||||
without obscuring the underlying registry model.
|
|
||||||
@@ -6,21 +6,19 @@ This document records features and interface changes that would be useful
|
|||||||
additions to PromptKit from the perspective of the maintainers of Notarius, a
|
additions to PromptKit from the perspective of the maintainers of Notarius, a
|
||||||
downstream application that consumes PromptKit.
|
downstream application that consumes PromptKit.
|
||||||
|
|
||||||
PromptKit v0.3.0 provides the capabilities Notarius currently needs. None of
|
PromptKit now provides the capabilities Notarius currently needs. The
|
||||||
the ideas below blocks current Notarius development. They are opportunities to
|
remaining deferred ideas are optional opportunities to improve checkpointing
|
||||||
reduce downstream workarounds, improve integration correctness, and make
|
and operational observability.
|
||||||
PromptKit more ergonomic for applications with configuration validation,
|
|
||||||
debugging, checkpointing, and operational-observability requirements.
|
|
||||||
|
|
||||||
The examples are API sketches intended to communicate the desired capability,
|
The examples are API sketches intended to communicate the desired capability,
|
||||||
not prescriptive names or finalized Go contracts.
|
not prescriptive names or finalized Go contracts.
|
||||||
|
|
||||||
## Priority 1: Atomic Execution With Prepared Details
|
## Priority 1: Atomic Execution With Prepared Details
|
||||||
|
|
||||||
**Disposition:** Covered by the accepted
|
**Disposition:** Implemented through [`Engine.PrepareExecution` and
|
||||||
[executable preparation handles](prepared-execution.md) roadmap. The shared
|
`Engine.RunPrepared`](../../engine.go). See the
|
||||||
two-phase capability should provide the required single-preparation
|
[consumer guidance](../consumers/pkg-promptkit.md#prepare-now-and-execute-the-same-snapshot-later).
|
||||||
consistency; a separate `RunDetailed` method is not cataloged initially.
|
A separate `RunDetailed` method is not cataloged.
|
||||||
|
|
||||||
### Downstream need
|
### Downstream need
|
||||||
|
|
||||||
@@ -33,63 +31,24 @@ Notarius needs both:
|
|||||||
Notarius uses the prepared details to construct redaction-aware debug bundles
|
Notarius uses the prepared details to construct redaction-aware debug bundles
|
||||||
and retain enough information to diagnose model behavior.
|
and retain enough information to diagnose model behavior.
|
||||||
|
|
||||||
### Current integration
|
### Implemented behavior
|
||||||
|
|
||||||
Notarius currently calls `Engine.Prepare` and then `Engine.Run` with the same
|
Notarius can prepare one frozen execution snapshot, retain a caller-owned and
|
||||||
request. Because `Run` performs preparation internally, a successful request
|
credential-redacted `Details` value for its debug bundle, and execute the same
|
||||||
resolves and prepares the same work twice.
|
snapshot through `RunPrepared`. The opaque handle is engine-bound and
|
||||||
|
single-use; an unused handle can be released with `Discard`. The consumer
|
||||||
This duplicates profile resolution, input hashing, schema loading, and prompt
|
guide and exported GoDoc own the exact lifecycle and failure contracts.
|
||||||
rendering. It also creates a theoretical consistency window in which a
|
|
||||||
filesystem-backed prompt, profile, schema, or input could change between the
|
|
||||||
explicit preparation and the preparation performed by `Run`.
|
|
||||||
|
|
||||||
### Requested capability
|
|
||||||
|
|
||||||
Add an opt-in execution method that prepares exactly once and returns both the
|
|
||||||
prepared details and completed result:
|
|
||||||
|
|
||||||
```go
|
|
||||||
type RunReport struct {
|
|
||||||
Prepared PreparedRun
|
|
||||||
Result RunResult
|
|
||||||
}
|
|
||||||
|
|
||||||
func (e *Engine) RunDetailed(
|
|
||||||
ctx context.Context,
|
|
||||||
req RunRequest,
|
|
||||||
) (*RunReport, error)
|
|
||||||
```
|
|
||||||
|
|
||||||
The exact names are flexible. The important contract is that preparation
|
|
||||||
occurs once and that the returned prepared state describes the execution that
|
|
||||||
produced the returned result.
|
|
||||||
|
|
||||||
Existing `Prepare` and `Run` behavior should remain available for consumers
|
|
||||||
that need only one side of the operation.
|
|
||||||
|
|
||||||
### Design considerations
|
|
||||||
|
|
||||||
- Keep this API additive and preserve the existing simple `Run` workflow.
|
|
||||||
- Return caller-owned copies under PromptKit's existing ownership rules.
|
|
||||||
- Define whether any prepared details are available after an operational
|
|
||||||
generation or validation error. Notarius does not require partial results
|
|
||||||
for the initial use case, but an explicit contract would be valuable.
|
|
||||||
- Preserve cancellation and backend-admission semantics.
|
|
||||||
- Do not add all prepared content directly to `RunResult`. Rendered prompt
|
|
||||||
content can be large and sensitive, and consumers should opt in to receiving
|
|
||||||
it.
|
|
||||||
|
|
||||||
### Value to Notarius
|
### Value to Notarius
|
||||||
|
|
||||||
This is the highest-value wishlist item. It would remove duplicate work from
|
This removes duplicate work from PromptKit-backed calls and ensures that
|
||||||
every successful PromptKit-backed call and ensure that retained debug material
|
retained debug material corresponds atomically to the actual execution.
|
||||||
corresponds atomically to the actual execution.
|
|
||||||
|
|
||||||
## Priority 2: Prompt-Independent Profile Inspection
|
## Priority 2: Prompt-Independent Profile Inspection
|
||||||
|
|
||||||
**Disposition:** Covered by the accepted
|
**Disposition:** Implemented as
|
||||||
[prompt-independent profile inspection](profile-inspection.md) roadmap.
|
[`Engine.InspectProfile`](../../engine.go). See the
|
||||||
|
[consumer guidance](../consumers/pkg-promptkit.md#inspect-a-profile-before-prompt-work).
|
||||||
|
|
||||||
### Downstream need
|
### Downstream need
|
||||||
|
|
||||||
@@ -104,62 +63,22 @@ needs to determine whether:
|
|||||||
|
|
||||||
This validation should not require model generation.
|
This validation should not require model generation.
|
||||||
|
|
||||||
### Current integration
|
### Previous integration
|
||||||
|
|
||||||
Notarius constructs a synthetic prompt using `testing/fstest.MapFS`, supplies a
|
Before profile inspection was available, Notarius constructed a synthetic
|
||||||
dummy transcript, and calls `Engine.Prepare` solely to exercise profile and
|
prompt using `testing/fstest.MapFS`, supplied a dummy transcript, and called
|
||||||
backend resolution. This works, but prompt preparation is serving as a
|
`Engine.Prepare` solely to exercise profile and backend resolution.
|
||||||
substitute for a profile-inspection interface.
|
|
||||||
|
|
||||||
### Requested capability
|
|
||||||
|
|
||||||
Add a prompt-independent profile-resolution API, for example:
|
|
||||||
|
|
||||||
```go
|
|
||||||
type ResolvedProfile struct {
|
|
||||||
ProfileID string
|
|
||||||
BackendID string
|
|
||||||
EffectiveTarget ExecutionTarget
|
|
||||||
APIKeyEnv string
|
|
||||||
}
|
|
||||||
|
|
||||||
func (e *Engine) ResolveProfile(
|
|
||||||
ctx context.Context,
|
|
||||||
profileID string,
|
|
||||||
) (ResolvedProfile, error)
|
|
||||||
```
|
|
||||||
|
|
||||||
The returned shape may differ, but it should provide enough information for a
|
|
||||||
consumer to validate an explicit profile selection without inventing a prompt
|
|
||||||
or supplying placeholder inputs.
|
|
||||||
|
|
||||||
### Design considerations
|
|
||||||
|
|
||||||
- Resolve built-in, file-backed, and programmatic profiles using normal
|
|
||||||
PromptKit precedence.
|
|
||||||
- Validate that a referenced backend registration exists.
|
|
||||||
- Do not resolve, retain, or expose credential values.
|
|
||||||
- Report credential requirements, such as an environment-variable name, so
|
|
||||||
the consuming application can decide whether availability is required at
|
|
||||||
configuration-validation time or only at execution time.
|
|
||||||
- Return caller-owned values.
|
|
||||||
- Preserve typed or sentinel error classification for missing and invalid
|
|
||||||
profiles.
|
|
||||||
- Consider accepting an `ExecutionTargetOverride` if consumers need to inspect
|
|
||||||
the same effective target that a run-level override would produce.
|
|
||||||
- Enumeration of all profiles is not required for the Notarius use case; exact
|
|
||||||
lookup by ID is sufficient.
|
|
||||||
|
|
||||||
### Value to Notarius
|
### Value to Notarius
|
||||||
|
|
||||||
This would eliminate a synthetic production-only prompt fixture and establish
|
The implemented interface eliminates a synthetic production-only prompt
|
||||||
a direct, supported contract for configuration-time profile and backend
|
fixture and establishes a direct, supported contract for configuration-time
|
||||||
validation.
|
profile and backend validation.
|
||||||
|
|
||||||
## Priority 3: Semantic Execution-Target Fingerprints
|
## Priority 3: Semantic Execution-Target Fingerprints
|
||||||
|
|
||||||
**Disposition:** Deferred until prompt-independent profile inspection defines
|
**Disposition:** Deferred pending a separate semantic-equality design for
|
||||||
the resolved target whose configuration identity would be fingerprinted.
|
resolved execution targets.
|
||||||
|
|
||||||
### Downstream need
|
### Downstream need
|
||||||
|
|
||||||
@@ -244,8 +163,8 @@ resolution.
|
|||||||
|
|
||||||
## Priority 4: Structured Capacity Errors
|
## Priority 4: Structured Capacity Errors
|
||||||
|
|
||||||
**Disposition:** Accepted into the
|
**Disposition:** Implemented behavior. See the consumer guide's
|
||||||
[future catalog](future.md#structured-capacity-errors).
|
[Handle Errors](../consumers/pkg-promptkit.md#handle-errors) section.
|
||||||
|
|
||||||
### Downstream need
|
### Downstream need
|
||||||
|
|
||||||
@@ -254,46 +173,16 @@ provider-neutral application error. When multiple backends are active,
|
|||||||
operators would benefit from knowing which backend rejected admission without
|
operators would benefit from knowing which backend rejected admission without
|
||||||
parsing an error string or exposing endpoint details.
|
parsing an error string or exposing endpoint details.
|
||||||
|
|
||||||
### Current integration
|
### Implemented behavior
|
||||||
|
|
||||||
PromptKit provides the useful `ErrCapacityExceeded` sentinel. Notarius can
|
PromptKit now retains the broad capacity classification while allowing
|
||||||
classify the failure reliably, but it retains only a sanitized diagnostic
|
Notarius to obtain the selected backend ID without parsing diagnostic text.
|
||||||
string as additional context.
|
The consumer guide owns the application workflow, including retry and backoff
|
||||||
|
policy.
|
||||||
### Requested capability
|
|
||||||
|
|
||||||
Add a typed error that continues to match `ErrCapacityExceeded`:
|
|
||||||
|
|
||||||
```go
|
|
||||||
type CapacityError struct {
|
|
||||||
BackendID string
|
|
||||||
}
|
|
||||||
|
|
||||||
func (e *CapacityError) Is(target error) bool {
|
|
||||||
return target == ErrCapacityExceeded
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
The exact implementation may use `Unwrap` or another idiomatic mechanism. The
|
|
||||||
important properties are compatibility with `errors.Is` and discoverability
|
|
||||||
through `errors.As`.
|
|
||||||
|
|
||||||
### Design considerations
|
|
||||||
|
|
||||||
- Include the stable backend ID.
|
|
||||||
- Do not expose the backend endpoint, credential environment, credential
|
|
||||||
value, request content, or other sensitive configuration.
|
|
||||||
- Consider including the configured concurrency and queue limits if they are
|
|
||||||
useful and safe, but backend identity alone provides most of the downstream
|
|
||||||
value.
|
|
||||||
- Add a retry delay only if PromptKit can provide a meaningful value. A full
|
|
||||||
queue does not necessarily imply a reliable `Retry-After` duration.
|
|
||||||
- Keep retry and backoff policy with the consuming application. PromptKit
|
|
||||||
should classify the admission failure rather than silently retry it.
|
|
||||||
|
|
||||||
### Value to Notarius
|
### Value to Notarius
|
||||||
|
|
||||||
This would improve operational diagnostics and future metrics while preserving
|
This improves operational diagnostics and future metrics while preserving
|
||||||
the provider-neutral error boundary used by Notarius.
|
the provider-neutral error boundary used by Notarius.
|
||||||
|
|
||||||
## Capabilities PromptKit Already Provides Well
|
## Capabilities PromptKit Already Provides Well
|
||||||
@@ -344,15 +233,13 @@ application-policy decision, not an upstream PromptKit gap.
|
|||||||
|
|
||||||
## Suggested Upstream Sequence
|
## Suggested Upstream Sequence
|
||||||
|
|
||||||
If the PromptKit team chooses to pursue these ideas, the most useful order for
|
For downstream adoption and any remaining upstream work, the useful order is:
|
||||||
Notarius would be:
|
|
||||||
|
|
||||||
1. Add atomic execution that returns prepared details and the completed result.
|
1. Adopt prepared execution for atomic details and results.
|
||||||
2. Add prompt-independent profile inspection.
|
2. Add a semantic execution-target digest, preferably alongside profile
|
||||||
3. Add a semantic execution-target digest, preferably as part of profile
|
|
||||||
inspection.
|
inspection.
|
||||||
4. Add a typed capacity error carrying backend identity.
|
3. Use the implemented typed capacity error where backend admission diagnostics
|
||||||
|
are needed.
|
||||||
|
|
||||||
The first two address concrete workarounds in current Notarius code. The third
|
The first removes the concrete execution workaround. The second would improve
|
||||||
would improve checkpoint correctness and reduce coupling. The fourth is
|
checkpoint correctness and reduce coupling. The third is operational polish.
|
||||||
operational polish.
|
|
||||||
|
|||||||
@@ -1,280 +0,0 @@
|
|||||||
# Executable Preparation Handles
|
|
||||||
|
|
||||||
**Status:** Complete.
|
|
||||||
|
|
||||||
## Purpose
|
|
||||||
|
|
||||||
Allow a consumer to prepare one exact Promptkit execution, inspect and retain
|
|
||||||
its credential-redacted public preparation details, and later execute that
|
|
||||||
already-prepared work without resolving or rendering the request again.
|
|
||||||
|
|
||||||
This provides a supported preflight-before-generation boundary for
|
|
||||||
[Weatherreporter](weatherreporter-promptkit-wishlist.md#priority-1-executable-preparation-handles)
|
|
||||||
and removes the duplicate `Prepare`-then-`Run` workaround described by
|
|
||||||
[Notarius](notarius-promptkit-wishlist.md#priority-1-atomic-execution-with-prepared-details).
|
|
||||||
|
|
||||||
## Motivation
|
|
||||||
|
|
||||||
`Engine.Prepare` currently returns the provenance and rendered details that
|
|
||||||
consumers need for debugging, persistence, and preflight checks. `Engine.Run`
|
|
||||||
then performs its own preparation before generation. A consumer that needs
|
|
||||||
both values must therefore prepare the same logical request twice.
|
|
||||||
|
|
||||||
That workaround duplicates source loading, input hashing, schema work, and
|
|
||||||
template rendering. It also permits filesystem-backed prompts, profiles,
|
|
||||||
schemas, or inputs to change between the public preparation and the
|
|
||||||
preparation that actually produces the result.
|
|
||||||
|
|
||||||
Promptkit already owns the complete preparation and execution pipeline. An
|
|
||||||
opt-in prepared-execution handle should expose the missing boundary without
|
|
||||||
putting rendered content into every `RunResult` or moving persistence policy
|
|
||||||
into the library.
|
|
||||||
|
|
||||||
## Consumer Workflow
|
|
||||||
|
|
||||||
The target public workflow is:
|
|
||||||
|
|
||||||
```go
|
|
||||||
prepared, err := engine.PrepareExecution(ctx, request)
|
|
||||||
if err != nil {
|
|
||||||
// Handle preparation failure.
|
|
||||||
return
|
|
||||||
}
|
|
||||||
defer prepared.Discard()
|
|
||||||
|
|
||||||
details := prepared.Details()
|
|
||||||
// Persist or inspect a consumer-selected safe subset of details.
|
|
||||||
|
|
||||||
result, err := engine.RunPrepared(ctx, prepared)
|
|
||||||
```
|
|
||||||
|
|
||||||
The target public surface is:
|
|
||||||
|
|
||||||
```go
|
|
||||||
type PreparedExecution struct {
|
|
||||||
// Opaque Promptkit-owned state.
|
|
||||||
}
|
|
||||||
|
|
||||||
func (e *Engine) PrepareExecution(
|
|
||||||
ctx context.Context,
|
|
||||||
req RunRequest,
|
|
||||||
) (*PreparedExecution, error)
|
|
||||||
|
|
||||||
func (p *PreparedExecution) Details() PreparedRun
|
|
||||||
|
|
||||||
func (p *PreparedExecution) Discard()
|
|
||||||
|
|
||||||
func (e *Engine) RunPrepared(
|
|
||||||
ctx context.Context,
|
|
||||||
prepared *PreparedExecution,
|
|
||||||
) (*RunResult, error)
|
|
||||||
```
|
|
||||||
|
|
||||||
The declarations and GoDoc will own the exact implemented contract. The
|
|
||||||
important public shape is an opaque handle, caller-owned `PreparedRun`
|
|
||||||
details, an explicit discard operation, and execution through the engine that
|
|
||||||
created the handle.
|
|
||||||
|
|
||||||
## Prepared Snapshot
|
|
||||||
|
|
||||||
`PrepareExecution` performs complete preparation without model generation or
|
|
||||||
backend-capacity admission. It applies the same request validation, source
|
|
||||||
precedence, backend and profile resolution, credential requirement checks,
|
|
||||||
output-contract resolution, artifact loading, hashing, schema loading,
|
|
||||||
session resolution, and rendering behavior as `Prepare`.
|
|
||||||
|
|
||||||
A successful handle freezes all source-derived state required for later
|
|
||||||
execution, including:
|
|
||||||
|
|
||||||
- the selected prompt definition, profile, and backend identity;
|
|
||||||
- the complete effective execution target and request-field presence;
|
|
||||||
- rendered messages and effective session ID;
|
|
||||||
- prompt, rendered-prompt, and input hashes;
|
|
||||||
- the effective output contract and provider-facing structured-output
|
|
||||||
constraint; and
|
|
||||||
- private validation state sufficient to validate generated output without
|
|
||||||
reopening schema files or `fs.FS` resources, including required schema
|
|
||||||
references.
|
|
||||||
|
|
||||||
After `PrepareExecution` succeeds, changes to prompt, profile, schema, input,
|
|
||||||
or request-owned data cannot change what `RunPrepared` sends to the model or
|
|
||||||
how it validates the generated output.
|
|
||||||
|
|
||||||
The handle retains an internal snapshot independent from values returned by
|
|
||||||
`Details`. Mutating a returned `PreparedRun`, its maps, slices, messages, or
|
|
||||||
schema values does not affect later execution. Each `Details` call returns a
|
|
||||||
fresh caller-owned copy under the existing `PreparedRun` ownership and stable
|
|
||||||
JSON rules.
|
|
||||||
|
|
||||||
## Handle Lifecycle
|
|
||||||
|
|
||||||
A `PreparedExecution` is:
|
|
||||||
|
|
||||||
- created only by a successful `PrepareExecution` call;
|
|
||||||
- bound to the exact `Engine` that created it;
|
|
||||||
- valid for one `RunPrepared` invocation;
|
|
||||||
- safe for repeated `Details` calls;
|
|
||||||
- intentionally opaque and without a supported JSON representation; and
|
|
||||||
- in-process state rather than a durable or restartable job.
|
|
||||||
|
|
||||||
`RunPrepared` atomically claims a valid handle before beginning the execution
|
|
||||||
attempt. A second or concurrent invocation fails without starting another
|
|
||||||
execution, including when the first invocation ended in cancellation, capacity
|
|
||||||
rejection, generation failure, or operational validation failure. Copies of
|
|
||||||
the public handle share the same one-attempt state and cannot bypass this rule.
|
|
||||||
|
|
||||||
A nil, zero-value, foreign-engine, discarded, already-claimed, or already-used
|
|
||||||
handle is invalid. `RunPrepared` reports these lifecycle errors through the
|
|
||||||
ordinary public invalid-request category. A failed foreign-engine invocation
|
|
||||||
does not consume a handle that remains valid for its owning engine.
|
|
||||||
|
|
||||||
`Discard` idempotently makes an unclaimed handle unavailable for execution and
|
|
||||||
drops Promptkit's references to secret-bearing or execution-only state.
|
|
||||||
`RunPrepared` performs the same cleanup automatically after claiming a handle.
|
|
||||||
Credential-redacted public preparation details remain available after discard,
|
|
||||||
success, or failure so consumers can retain diagnostics. Promptkit does not
|
|
||||||
promise secure erasure of Go string memory.
|
|
||||||
|
|
||||||
Lifecycle transitions are concurrency-safe. When `RunPrepared` and `Discard`
|
|
||||||
race, exactly one claims the ready handle. `Discard` is not an execution
|
|
||||||
cancellation mechanism and does not interrupt an attempt that has already
|
|
||||||
claimed the handle; consumers cancel that attempt through its context.
|
|
||||||
|
|
||||||
## Credentials And Sensitive Data
|
|
||||||
|
|
||||||
`Details` has the same security contract as `PreparedRun`: it can contain
|
|
||||||
rendered messages, schemas, identifiers, and hashes, but never a resolved API
|
|
||||||
key value. Consumers remain responsible for selecting, redacting, storing, and
|
|
||||||
retaining any persisted preparation material.
|
|
||||||
|
|
||||||
A direct `RunRequest.APIKey` is retained only in opaque execution state until
|
|
||||||
the handle is run or discarded. It is never added to details, hashes, JSON,
|
|
||||||
`String`, or `GoString` output.
|
|
||||||
|
|
||||||
An environment-variable name is frozen as part of the effective target, but
|
|
||||||
its credential value is not captured for the lifetime of the handle.
|
|
||||||
`PrepareExecution` applies the existing preparation-time availability check.
|
|
||||||
`RunPrepared` rechecks availability before admission, and the selected model
|
|
||||||
client uses the environment value visible during execution. This preserves
|
|
||||||
current secret ownership and avoids retaining an environment credential while
|
|
||||||
a consumer persists preflight material.
|
|
||||||
|
|
||||||
The opaque handle must not expose retained request data or credentials through
|
|
||||||
default formatting, JSON, or error messages.
|
|
||||||
|
|
||||||
## Admission, Cancellation, And Execution
|
|
||||||
|
|
||||||
`PrepareExecution` never reserves backend admission or an active-generation
|
|
||||||
permit. Its context governs preparation only; cancellation after it returns
|
|
||||||
does not invalidate the handle.
|
|
||||||
|
|
||||||
`RunPrepared` uses its own context for credential revalidation, backend
|
|
||||||
admission, active-generation waiting, model generation, output validation, and
|
|
||||||
any internal repair. Admission occurs when `RunPrepared` begins so a consumer
|
|
||||||
cannot occupy bounded capacity while inspecting or persisting preparation
|
|
||||||
details.
|
|
||||||
|
|
||||||
For a limited backend, the admission lease covers the complete prepared
|
|
||||||
execution attempt after admission: generation, validation, internal repair,
|
|
||||||
and every success or failure exit. Actual generation continues to use the
|
|
||||||
backend's FIFO active-generation permit. Existing capacity error identity,
|
|
||||||
cancellation behavior, and release guarantees remain in force.
|
|
||||||
|
|
||||||
Because a prepared handle is one-attempt, cancellation or capacity rejection
|
|
||||||
does not make it reusable. Retry and backoff policy remains with the consumer,
|
|
||||||
which may create a new prepared handle when another attempt is appropriate.
|
|
||||||
|
|
||||||
## Results And Failures
|
|
||||||
|
|
||||||
On success, `RunPrepared` returns the existing caller-owned `RunResult`. Its
|
|
||||||
source-derived provenance must match `Details`, including prompt identity and
|
|
||||||
hash, rendered-prompt hash, session ID, selected profile and backend,
|
|
||||||
effective target, and input hashes.
|
|
||||||
|
|
||||||
`RunResult.StartTime`, `EndTime`, and `Duration` describe the
|
|
||||||
`RunPrepared` execution attempt. They exclude preparation time and any delay
|
|
||||||
while the consumer retained the handle. Preparation timing remains in
|
|
||||||
`PreparedRun`.
|
|
||||||
|
|
||||||
Preparation failure returns no handle. After successful preparation,
|
|
||||||
`RunPrepared` retains the existing rule that an operational failure returns no
|
|
||||||
partial `RunResult`; the consumer already has independent preparation details.
|
|
||||||
A completed content-validation failure remains a successful result with
|
|
||||||
`ValidationFailed`.
|
|
||||||
|
|
||||||
`PrepareExecution` preserves the public error categories of `Prepare`.
|
|
||||||
`RunPrepared` preserves applicable invalid-request, credential, capacity,
|
|
||||||
generation, validation, collaborator, and cancellation identities without
|
|
||||||
reintroducing source-loading or rendering failures from frozen state.
|
|
||||||
|
|
||||||
## Compatibility And Existing Workflows
|
|
||||||
|
|
||||||
This feature is additive:
|
|
||||||
|
|
||||||
- `Prepare` remains the simple preparation-only operation;
|
|
||||||
- `Run` remains the simple prepare-and-execute operation with its current
|
|
||||||
early-admission and error-ordering behavior;
|
|
||||||
- `PreparedRun` and `RunResult` retain their existing stable JSON
|
|
||||||
representations;
|
|
||||||
- model-client and artifact-reader extension interfaces remain unchanged; and
|
|
||||||
- backend routing, concurrency limits, queue capacities, and provider wire
|
|
||||||
behavior remain unchanged.
|
|
||||||
|
|
||||||
The new workflow may share internal machinery with `Prepare` and `Run`, but it
|
|
||||||
must not change their observable behavior merely to simplify implementation.
|
|
||||||
|
|
||||||
## Documentation
|
|
||||||
|
|
||||||
The completed documentation set has these ownership boundaries:
|
|
||||||
|
|
||||||
- exported declarations and GoDoc own the exact handle, method, lifecycle,
|
|
||||||
ownership, concurrency, credential, error, and cancellation contracts;
|
|
||||||
- the promptkit consumer guide explains when to use `Prepare`, `Run`, or the
|
|
||||||
two-phase prepared-execution workflow; and
|
|
||||||
- internal runner, source-validation, capacity, and model-client documentation
|
|
||||||
describe the implemented collaborator boundaries without duplicating public
|
|
||||||
contracts.
|
|
||||||
|
|
||||||
No release document is part of the feature implementation itself. Release
|
|
||||||
guidance is prepared only when the resulting public API is selected for
|
|
||||||
publication.
|
|
||||||
|
|
||||||
## Non-Goals
|
|
||||||
|
|
||||||
This work does not include:
|
|
||||||
|
|
||||||
- serializable, durable, resumable, or cross-process execution handles;
|
|
||||||
- reuse for multiple consumer-initiated executions;
|
|
||||||
- concurrent execution of one handle;
|
|
||||||
- capacity reservation during preparation;
|
|
||||||
- a background task queue, priorities, worker lifecycle, or job status;
|
|
||||||
- retry, backoff, or provider failover policy;
|
|
||||||
- a separate `RunDetailed` convenience method;
|
|
||||||
- adding preparation details or rendered messages to every `RunResult`;
|
|
||||||
- prompt or profile inspection APIs;
|
|
||||||
- structured capacity or generation errors;
|
|
||||||
- freezing environment-variable credential values for the handle lifetime;
|
|
||||||
- snapshotting provider state or mutable behavior inside an injected
|
|
||||||
`LLMClient`;
|
|
||||||
- allowing target, validation, session, variable, or input overrides after
|
|
||||||
preparation; or
|
|
||||||
- changing existing `Prepare`, `Run`, file-format, provider-request, or stable
|
|
||||||
JSON contracts.
|
|
||||||
|
|
||||||
## Target End State
|
|
||||||
|
|
||||||
After this work:
|
|
||||||
|
|
||||||
- consumers can perform and persist preflight before starting provider work;
|
|
||||||
- one prepared handle executes exactly the source-derived prompt, target,
|
|
||||||
schema, inputs, session, and messages described by its public details;
|
|
||||||
- execution never reloads or rerenders consumer sources;
|
|
||||||
- direct credentials remain confined to opaque, explicitly discardable state;
|
|
||||||
- environment credentials are not retained across the preflight boundary;
|
|
||||||
- backend capacity is reserved only when execution begins;
|
|
||||||
- one handle can start at most one execution attempt, including any internal
|
|
||||||
repair calls owned by that attempt;
|
|
||||||
- preparation details remain available after execution success or failure;
|
|
||||||
- existing simple `Prepare` and `Run` consumers remain unaffected; and
|
|
||||||
- Promptkit continues to own reusable execution mechanics without taking on
|
|
||||||
downstream persistence, redaction, retry, or job-management policy.
|
|
||||||
@@ -1,246 +0,0 @@
|
|||||||
# Prompt-Independent Profile Inspection
|
|
||||||
|
|
||||||
**Status:** Accepted.
|
|
||||||
|
|
||||||
## Purpose
|
|
||||||
|
|
||||||
Allow consumers to look up one execution profile by ID and inspect its
|
|
||||||
structurally resolved model target without selecting a prompt, supplying
|
|
||||||
placeholder inputs, checking credential availability, or invoking a model.
|
|
||||||
|
|
||||||
This provides a direct configuration-validation boundary for
|
|
||||||
[Notarius](notarius-promptkit-wishlist.md#priority-2-prompt-independent-profile-inspection)
|
|
||||||
and
|
|
||||||
[Weatherreporter](weatherreporter-promptkit-wishlist.md#priority-3-prompt-independent-profile-inspection).
|
|
||||||
|
|
||||||
## Motivation
|
|
||||||
|
|
||||||
Both downstream consumers need to reject invalid configured profile IDs before
|
|
||||||
starting application work. They currently have to construct a synthetic prompt
|
|
||||||
and call `Engine.Prepare` merely to exercise profile loading, backend lookup,
|
|
||||||
and execution-target resolution.
|
|
||||||
|
|
||||||
That workaround couples profile validation to unrelated prompt definitions,
|
|
||||||
fixture inputs, rendering, schema behavior, and current credential
|
|
||||||
availability. Promptkit already owns profile precedence, backend membership,
|
|
||||||
and target resolution, so it should expose that cohesive capability directly.
|
|
||||||
|
|
||||||
## Consumer Workflow
|
|
||||||
|
|
||||||
The target public workflow is:
|
|
||||||
|
|
||||||
```go
|
|
||||||
inspection, err := engine.InspectProfile(ctx, profileID)
|
|
||||||
if err != nil {
|
|
||||||
// Reject or report the configured profile.
|
|
||||||
return
|
|
||||||
}
|
|
||||||
|
|
||||||
target := inspection.EffectiveModelParams
|
|
||||||
if target.APIKeyEnv != "" {
|
|
||||||
// Apply application policy for the named environment variable.
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
The target public surface is:
|
|
||||||
|
|
||||||
```go
|
|
||||||
type ProfileInspection struct {
|
|
||||||
ProfileID string
|
|
||||||
EffectiveModelParams ExecutionTarget
|
|
||||||
APIKeyRequired bool
|
|
||||||
}
|
|
||||||
|
|
||||||
func (e *Engine) InspectProfile(
|
|
||||||
ctx context.Context,
|
|
||||||
profileID string,
|
|
||||||
) (*ProfileInspection, error)
|
|
||||||
```
|
|
||||||
|
|
||||||
The declarations and GoDoc will own the exact implemented contract. The
|
|
||||||
important public shape is exact lookup through the existing engine, one
|
|
||||||
caller-owned inspection value, the resolved `ExecutionTarget`, and an explicit
|
|
||||||
signal for a direct API-key requirement.
|
|
||||||
|
|
||||||
`EffectiveModelParams.BackendID` identifies the selected registered backend
|
|
||||||
and remains empty for endpoint-only profiles.
|
|
||||||
`EffectiveModelParams.APIKeyEnv` reports the effective credential environment
|
|
||||||
variable name. `APIKeyRequired` reports that the profile requires a direct
|
|
||||||
request credential instead. These states are mutually exclusive after normal
|
|
||||||
profile and backend precedence is applied.
|
|
||||||
|
|
||||||
`ProfileInspection` does not need a stable JSON representation. Consumers that
|
|
||||||
persist application configuration or diagnostics can select the fields their
|
|
||||||
own format requires.
|
|
||||||
|
|
||||||
## Lookup And Precedence
|
|
||||||
|
|
||||||
`InspectProfile` requires a non-blank explicit profile ID. It trims surrounding
|
|
||||||
whitespace and otherwise performs the same case-sensitive exact lookup used by
|
|
||||||
ordinary execution.
|
|
||||||
|
|
||||||
Lookup applies the engine's normal profile-source precedence:
|
|
||||||
|
|
||||||
- programmatic profiles supplied through `WithProfiles`;
|
|
||||||
- the configured file, directory, or `fs.FS` profile source; and
|
|
||||||
- the built-in profile catalog.
|
|
||||||
|
|
||||||
A valid higher-precedence match shadows a lower-precedence profile with the
|
|
||||||
same ID. A malformed or unreadable higher-precedence match fails rather than
|
|
||||||
silently falling back. The
|
|
||||||
[profile format reference](../formats.md#profile-definitions) remains the
|
|
||||||
canonical owner of profile-source and file-format behavior.
|
|
||||||
|
|
||||||
Inspection never derives a profile ID from a prompt's `default_profile`; the
|
|
||||||
caller is inspecting one explicitly named profile.
|
|
||||||
|
|
||||||
## Structural Resolution
|
|
||||||
|
|
||||||
Inspection loads and validates the selected profile, verifies that a named
|
|
||||||
backend exists in the engine's immutable backend registry, and applies normal
|
|
||||||
framework-default, backend, and profile precedence to produce the effective
|
|
||||||
target.
|
|
||||||
|
|
||||||
For the same engine state and profile ID, with no per-run execution override,
|
|
||||||
the inspected target must match the target that ordinary preparation would
|
|
||||||
resolve before applying request credentials and checking their availability.
|
|
||||||
This equivalence must use one shared resolution path rather than a second set
|
|
||||||
of precedence rules.
|
|
||||||
|
|
||||||
Structural resolution includes:
|
|
||||||
|
|
||||||
- backend routing identity;
|
|
||||||
- endpoint and model;
|
|
||||||
- sampling, token, timeout, service-tier, and reasoning settings;
|
|
||||||
- the effective credential environment-variable name or direct-key
|
|
||||||
requirement; and
|
|
||||||
- deeply copied provider-specific extra parameters.
|
|
||||||
|
|
||||||
Inspection returns the effective target rather than a raw profile definition.
|
|
||||||
This keeps framework and backend defaults visible to consumers without
|
|
||||||
creating a second public profile-loading interface.
|
|
||||||
|
|
||||||
The result does not include request-override presence because no
|
|
||||||
`ExecutionTargetOverride` participates in inspection.
|
|
||||||
|
|
||||||
## Credentials And Sensitive Data
|
|
||||||
|
|
||||||
Inspection reports credential requirements but never resolves, retains, or
|
|
||||||
returns a credential value.
|
|
||||||
|
|
||||||
The operation does not read the named environment variable and succeeds when
|
|
||||||
that variable is absent or blank. It accepts neither a direct API key nor an
|
|
||||||
API-key environment override. Consumers decide whether credential availability
|
|
||||||
must be enforced during application configuration, while `Prepare`,
|
|
||||||
`PrepareExecution`, `Run`, and `RunPrepared` retain their execution-time
|
|
||||||
credential contracts.
|
|
||||||
|
|
||||||
Error messages, formatting, and returned values must not expose environment
|
|
||||||
values or other resolved secrets. Existing restrictions against raw API keys
|
|
||||||
in profile sources remain unchanged.
|
|
||||||
|
|
||||||
## Ownership, Consistency, And Concurrency
|
|
||||||
|
|
||||||
Each successful call returns a caller-owned snapshot. Mutating the returned
|
|
||||||
target or any nested extra-parameter map or slice cannot affect the engine,
|
|
||||||
later inspection, or later execution.
|
|
||||||
|
|
||||||
Inspection is safe to call concurrently under the engine's existing immutable
|
|
||||||
registry and repository contracts. It does not mutate profile sources or
|
|
||||||
cache a result globally.
|
|
||||||
|
|
||||||
For filesystem-backed sources, an inspection describes the state observed by
|
|
||||||
that call. It does not freeze the profile for a later `Run`; a source may
|
|
||||||
change between operations. Consumers requiring an exact preflight-to-execution
|
|
||||||
snapshot should use the existing prepared-execution workflow.
|
|
||||||
|
|
||||||
## Errors And Cancellation
|
|
||||||
|
|
||||||
The operation uses existing public error categories:
|
|
||||||
|
|
||||||
- a blank profile ID matches `ErrInvalidRequest`;
|
|
||||||
- an absent exact ID matches `ErrProfileNotFound` and not `ErrProfileLoad`;
|
|
||||||
- read, decode, validation, and source-selection failures match
|
|
||||||
`ErrProfileLoad`; and
|
|
||||||
- an unknown referenced backend or an invalid structurally resolved target
|
|
||||||
matches `ErrProfileLoad`.
|
|
||||||
|
|
||||||
Errors should preserve useful underlying collaborator and context identities
|
|
||||||
through `errors.Is` where the existing facade does so, without exposing
|
|
||||||
internal package types. Context cancellation governs inspection and no partial
|
|
||||||
inspection result is returned on failure.
|
|
||||||
|
|
||||||
Missing credential values are not inspection errors. The operation cannot
|
|
||||||
return capacity or model-generation failures because it performs neither
|
|
||||||
backend admission nor generation.
|
|
||||||
|
|
||||||
## Compatibility And Boundaries
|
|
||||||
|
|
||||||
This feature is additive. Existing profile formats, source precedence,
|
|
||||||
backend registration, `Prepare`, prepared execution, and `Run` behavior remain
|
|
||||||
unchanged.
|
|
||||||
|
|
||||||
The method belongs on the root `Engine` facade. Profile repositories and the
|
|
||||||
backend registry remain internal implementation details, and no new public
|
|
||||||
repository interface is introduced.
|
|
||||||
|
|
||||||
Inspection does not require prompt lookup, rendering, artifact loading, schema
|
|
||||||
loading, validation, backend-capacity admission, or model-client access.
|
|
||||||
Engine construction retains its ordinary configuration requirements; this
|
|
||||||
feature does not introduce a separate profile-only engine.
|
|
||||||
|
|
||||||
## Documentation
|
|
||||||
|
|
||||||
The completed documentation set has these ownership boundaries:
|
|
||||||
|
|
||||||
- exported declarations and GoDoc own the exact method, result, ownership,
|
|
||||||
credential, error, and cancellation contracts;
|
|
||||||
- the promptkit consumer guide explains configuration-time profile inspection
|
|
||||||
and distinguishes it from `Prepare` and prepared execution;
|
|
||||||
- the profile format reference continues to own profile fields and source
|
|
||||||
precedence; and
|
|
||||||
- internal documentation describes shared profile and target resolution
|
|
||||||
without duplicating public contracts.
|
|
||||||
|
|
||||||
## Non-Goals
|
|
||||||
|
|
||||||
This work does not include:
|
|
||||||
|
|
||||||
- enumerating or searching profiles;
|
|
||||||
- returning raw profile definitions or profile source paths;
|
|
||||||
- accepting per-run execution overrides, direct API keys, or API-key
|
|
||||||
environment overrides;
|
|
||||||
- checking environment-variable contents or other credential availability;
|
|
||||||
- semantic execution-target fingerprints or profile hashes;
|
|
||||||
- prompt-definition inspection or prompt default-profile resolution;
|
|
||||||
- full-corpus validation across every profile source;
|
|
||||||
- freezing a filesystem-backed profile for later execution;
|
|
||||||
- exposing backend concurrency limits, queue state, or capacity policy;
|
|
||||||
- model generation, provider health checks, or endpoint connectivity tests;
|
|
||||||
- dynamic backend or profile registration after engine construction; or
|
|
||||||
- changing current profile, backend, prepared-execution, or stable JSON
|
|
||||||
contracts.
|
|
||||||
|
|
||||||
## Target End State
|
|
||||||
|
|
||||||
After this work:
|
|
||||||
|
|
||||||
- consumers can validate one configured profile without inventing a prompt or
|
|
||||||
placeholder inputs;
|
|
||||||
- lookup observes ordinary programmatic, configured-source, and built-in
|
|
||||||
precedence;
|
|
||||||
- a successful result proves that the profile exists, is valid, references a
|
|
||||||
registered backend when applicable, and resolves to a structurally valid
|
|
||||||
effective target;
|
|
||||||
- the inspected target matches ordinary preparation for the same profile and
|
|
||||||
engine state before per-run overrides and credential availability checks;
|
|
||||||
- credential requirements are visible without reading or exposing credential
|
|
||||||
values;
|
|
||||||
- returned targets and nested data are caller-owned;
|
|
||||||
- inspection performs no rendering, source loading unrelated to the profile,
|
|
||||||
capacity admission, or model work;
|
|
||||||
- existing execution workflows and compatibility contracts remain unchanged;
|
|
||||||
and
|
|
||||||
- Promptkit owns reusable profile validation while downstream applications
|
|
||||||
retain configuration policy, persistence, logging, and credential-timing
|
|
||||||
decisions.
|
|
||||||
@@ -7,10 +7,9 @@ additions to PromptKit from the perspective of the maintainers of
|
|||||||
Weatherreporter, a downstream application planning to replace its Scriptorium
|
Weatherreporter, a downstream application planning to replace its Scriptorium
|
||||||
CLI integration with PromptKit.
|
CLI integration with PromptKit.
|
||||||
|
|
||||||
PromptKit v0.3.0 provides the capabilities Weatherreporter needs for the
|
PromptKit now provides the capabilities Weatherreporter needs for the
|
||||||
migration. None of the ideas below is a hard adoption requirement. They are
|
migration. The remaining deferred ideas are optional opportunities to validate
|
||||||
opportunities to avoid duplicate preparation, validate configuration earlier,
|
configuration earlier and improve durable failure diagnostics.
|
||||||
improve durable failure diagnostics, and make the integration more direct.
|
|
||||||
|
|
||||||
The examples are API sketches intended to communicate the desired capability,
|
The examples are API sketches intended to communicate the desired capability,
|
||||||
not prescriptive names or finalized Go contracts. The related
|
not prescriptive names or finalized Go contracts. The related
|
||||||
@@ -19,8 +18,9 @@ overlapping features from another downstream consumer's perspective.
|
|||||||
|
|
||||||
## Priority 1: Executable Preparation Handles
|
## Priority 1: Executable Preparation Handles
|
||||||
|
|
||||||
**Disposition:** Accepted into the
|
**Disposition:** Implemented as [`Engine.PrepareExecution` and
|
||||||
[executable preparation handles](prepared-execution.md) feature roadmap.
|
`Engine.RunPrepared`](../../engine.go). See the
|
||||||
|
[consumer guidance](../consumers/pkg-promptkit.md#prepare-now-and-execute-the-same-snapshot-later).
|
||||||
|
|
||||||
### Downstream need
|
### Downstream need
|
||||||
|
|
||||||
@@ -35,79 +35,26 @@ needs to:
|
|||||||
Persisting preflight before generation leaves useful evidence when a provider
|
Persisting preflight before generation leaves useful evidence when a provider
|
||||||
call fails or the process is interrupted during generation.
|
call fails or the process is interrupted during generation.
|
||||||
|
|
||||||
### Current integration option
|
### Implemented behavior
|
||||||
|
|
||||||
With PromptKit v0.3.0, Weatherreporter can call `Engine.Prepare`, save selected
|
Weatherreporter can prepare one frozen execution snapshot, persist a
|
||||||
fields from the returned `PreparedRun`, and then call `Engine.Run` with the
|
caller-owned and credential-redacted `Details` value, and execute that same
|
||||||
same request. Because `Run` performs preparation internally, the work is
|
snapshot through `RunPrepared`. The opaque handle is engine-bound and
|
||||||
repeated.
|
single-use; an unused handle can be released with `Discard`. The consumer
|
||||||
|
guide and exported GoDoc own the exact lifecycle, credential, cancellation,
|
||||||
Weatherreporter plans to use embedded prompt and schema files plus immutable
|
and capacity contracts.
|
||||||
inline input bytes, which removes most of the consistency risk. An external
|
|
||||||
profile file or directory can still change between the two calls, and the
|
|
||||||
second preparation remains unnecessary work.
|
|
||||||
|
|
||||||
The atomic `RunDetailed` operation proposed by the
|
|
||||||
[Notarius wishlist](notarius-promptkit-wishlist.md#priority-1-atomic-execution-with-prepared-details)
|
|
||||||
would guarantee that returned preparation details describe the completed
|
|
||||||
execution. However, returning those details only after generation would not
|
|
||||||
preserve Weatherreporter's preflight-before-generation persistence boundary.
|
|
||||||
|
|
||||||
### Requested capability
|
|
||||||
|
|
||||||
Add an opt-in two-phase API that returns a prepared execution handle:
|
|
||||||
|
|
||||||
```go
|
|
||||||
prepared, err := engine.PrepareExecution(ctx, request)
|
|
||||||
if err != nil {
|
|
||||||
// Handle preparation failure.
|
|
||||||
}
|
|
||||||
|
|
||||||
details := prepared.Details()
|
|
||||||
// Persist a consumer-selected safe preparation record.
|
|
||||||
|
|
||||||
result, err := engine.RunPrepared(ctx, prepared)
|
|
||||||
```
|
|
||||||
|
|
||||||
The exact names and shapes are flexible. The important contract is that
|
|
||||||
`RunPrepared` executes the already prepared prompt and does not reload or
|
|
||||||
rerender its prompt, profile, schema, or input sources.
|
|
||||||
|
|
||||||
`Details` should return the same caller-owned public preparation information
|
|
||||||
currently represented by `PreparedRun`. The execution handle may retain opaque
|
|
||||||
engine-owned state needed to invoke the model and validate the response.
|
|
||||||
|
|
||||||
### Design considerations
|
|
||||||
|
|
||||||
- Keep `Prepare` and `Run` available for consumers that do not need a
|
|
||||||
two-phase execution boundary.
|
|
||||||
- Bind a prepared handle to the engine that constructed it.
|
|
||||||
- Define whether a handle is one-shot, reusable, or safe for concurrent use.
|
|
||||||
A one-shot contract may be the safest initial design.
|
|
||||||
- Do not give the opaque handle a stable JSON representation.
|
|
||||||
- Do not expose or serialize resolved credential values through `Details`.
|
|
||||||
- Define how a direct request API key is retained and released when an opaque
|
|
||||||
handle must carry it until execution.
|
|
||||||
- Preserve caller-owned copies for all public details.
|
|
||||||
- Make context cancellation and backend admission timing explicit.
|
|
||||||
- Document whether profile credential environment values are resolved during
|
|
||||||
preparation or execution.
|
|
||||||
- Ensure an execution error does not invalidate the public details already
|
|
||||||
returned to the consumer.
|
|
||||||
- Consider whether an atomic `RunDetailed` can share the same internal
|
|
||||||
prepared-execution implementation.
|
|
||||||
|
|
||||||
### Value to Weatherreporter
|
### Value to Weatherreporter
|
||||||
|
|
||||||
This is the highest-value upstream addition. It would preserve
|
This preserves Weatherreporter's durable preflight behavior, removes duplicate
|
||||||
Weatherreporter's durable preflight behavior, remove duplicate work, eliminate
|
work, eliminates the source-consistency window, and ensures that persisted
|
||||||
the remaining source-consistency window, and ensure that persisted provenance
|
provenance describes the actual execution.
|
||||||
describes the actual execution.
|
|
||||||
|
|
||||||
## Priority 2: Prompt-Definition Inspection
|
## Priority 2: Prompt-Definition Inspection
|
||||||
|
|
||||||
**Disposition:** Accepted into the
|
**Disposition:** Implemented as
|
||||||
[future catalog](future.md#prompt-definition-inspection).
|
[`Engine.InspectPrompt`](../../engine.go). See the
|
||||||
|
[consumer guidance](../consumers/pkg-promptkit.md#inspect-a-prompt-before-preparation).
|
||||||
|
|
||||||
### Downstream need
|
### Downstream need
|
||||||
|
|
||||||
@@ -123,113 +70,42 @@ response schemas that implement those reports. It needs to validate that the
|
|||||||
report registry and embedded prompt corpus agree before weather collection or
|
report registry and embedded prompt corpus agree before weather collection or
|
||||||
provider execution.
|
provider execution.
|
||||||
|
|
||||||
### Current integration option
|
### Previous integration option
|
||||||
|
|
||||||
Weatherreporter can maintain synthetic data-package fixtures and call
|
Before prompt inspection was available, Weatherreporter could maintain
|
||||||
`Engine.Prepare` for every report prompt during tests. Runtime validation can
|
synthetic data-package fixtures and call `Engine.Prepare` for every report
|
||||||
also occur through the ordinary per-report preparation stage.
|
prompt during tests. Runtime validation could also occur through the ordinary
|
||||||
|
per-report preparation stage.
|
||||||
|
|
||||||
This works, but it requires complete placeholder inputs and profile resolution
|
This required complete placeholder inputs and profile resolution when the
|
||||||
when the application primarily wants to inspect prompt identity and declared
|
application primarily wanted to inspect prompt identity and declared contracts.
|
||||||
contracts.
|
|
||||||
|
|
||||||
### Requested capability
|
|
||||||
|
|
||||||
Add exact prompt-definition lookup without rendering or generation:
|
|
||||||
|
|
||||||
```go
|
|
||||||
type PromptInfo struct {
|
|
||||||
PromptID string
|
|
||||||
PromptVersion string
|
|
||||||
PromptHash string
|
|
||||||
DefaultProfileID string
|
|
||||||
Inputs []InputDefinition
|
|
||||||
OutputContract OutputContract
|
|
||||||
}
|
|
||||||
|
|
||||||
func (e *Engine) ResolvePrompt(
|
|
||||||
ctx context.Context,
|
|
||||||
promptID string,
|
|
||||||
promptVersion string,
|
|
||||||
) (PromptInfo, error)
|
|
||||||
```
|
|
||||||
|
|
||||||
The exact returned shape may differ. Weatherreporter needs enough information
|
|
||||||
to verify prompt existence, version selection, declared inputs, default
|
|
||||||
profile identity, output format, validation mode, and schema selection without
|
|
||||||
supplying synthetic prompt input.
|
|
||||||
|
|
||||||
### Design considerations
|
|
||||||
|
|
||||||
- Use ordinary PromptKit prompt-source precedence and exact ID/version
|
|
||||||
selection.
|
|
||||||
- Fully load and structurally validate the selected prompt definition.
|
|
||||||
- Validate referenced prompt content files without rendering their templates.
|
|
||||||
- Resolve and validate the selected output contract and schema reference where
|
|
||||||
practical.
|
|
||||||
- Return an opaque prompt-definition equality value rather than raw source
|
|
||||||
bytes.
|
|
||||||
- Do not return rendered messages, schema bodies, profile credentials, or
|
|
||||||
another source of sensitive content.
|
|
||||||
- Preserve typed or sentinel errors for missing and invalid prompts.
|
|
||||||
- Return caller-owned values.
|
|
||||||
- Enumeration of all known prompts is not required for Weatherreporter; exact
|
|
||||||
lookup is sufficient.
|
|
||||||
|
|
||||||
### Value to Weatherreporter
|
### Value to Weatherreporter
|
||||||
|
|
||||||
This would let Weatherreporter directly verify that every report prompt
|
The implemented interface lets Weatherreporter directly verify that every
|
||||||
exists, requires the curated `data_package` input, and declares the expected
|
report prompt exists, requires the curated `data_package` input, and declares
|
||||||
Markdown or JSON Schema output contract. It would reduce synthetic test setup
|
the expected Markdown or JSON Schema output contract. It reduces synthetic
|
||||||
and move failures ahead of weather collection.
|
test setup and moves failures ahead of weather collection.
|
||||||
|
|
||||||
## Priority 3: Prompt-Independent Profile Inspection
|
## Priority 3: Prompt-Independent Profile Inspection
|
||||||
|
|
||||||
**Disposition:** Covered by the accepted
|
**Disposition:** Implemented as
|
||||||
[prompt-independent profile inspection](profile-inspection.md) roadmap.
|
[`Engine.InspectProfile`](../../engine.go). See the
|
||||||
|
[consumer guidance](../consumers/pkg-promptkit.md#inspect-a-profile-before-prompt-work).
|
||||||
|
|
||||||
### Downstream need
|
### Downstream need
|
||||||
|
|
||||||
Weatherreporter will allow operators to select an external PromptKit profile
|
Weatherreporter will allow operators to select an external PromptKit profile
|
||||||
source and may allow an explicit profile override. It should reject a missing
|
source and may allow an explicit profile override. It needs to reject a missing
|
||||||
profile, unknown backend, malformed execution target, or unsatisfied credential
|
profile, unknown backend, or malformed execution target before collecting
|
||||||
requirement before collecting weather data or writing report artifacts.
|
weather data or writing report artifacts, and to apply its own policy to a
|
||||||
|
reported credential requirement.
|
||||||
### Current integration option
|
|
||||||
|
|
||||||
Weatherreporter can validate an explicit profile by preparing one embedded
|
|
||||||
prompt with fixture input. Prompts that use their own default profiles can be
|
|
||||||
validated during their normal preparation stage.
|
|
||||||
|
|
||||||
This couples configuration validation to one prompt and requires placeholder
|
|
||||||
input even when only profile and backend resolution are relevant.
|
|
||||||
|
|
||||||
### Requested capability
|
|
||||||
|
|
||||||
The prompt-independent `ResolveProfile` API proposed by the
|
|
||||||
[Notarius wishlist](notarius-promptkit-wishlist.md#priority-2-prompt-independent-profile-inspection)
|
|
||||||
would satisfy this need. It should resolve built-in, file-backed, and
|
|
||||||
programmatic profiles, validate backend membership, report credential
|
|
||||||
requirements without resolving credential values, and preserve typed error
|
|
||||||
classification.
|
|
||||||
|
|
||||||
### Additional Weatherreporter considerations
|
|
||||||
|
|
||||||
- An explicit application profile override should be inspectable without
|
|
||||||
selecting a report prompt.
|
|
||||||
- A prompt-definition inspection result may expose its default profile ID so
|
|
||||||
Weatherreporter can inspect that profile separately.
|
|
||||||
- Inspection should distinguish structural profile validity from current
|
|
||||||
credential availability so configuration validation can apply explicit
|
|
||||||
application policy.
|
|
||||||
- An optional execution-target override should be considered only if it
|
|
||||||
describes the same target that a later run will use.
|
|
||||||
|
|
||||||
### Value to Weatherreporter
|
### Value to Weatherreporter
|
||||||
|
|
||||||
This would improve fail-fast configuration validation and give operator-facing
|
The implemented interface improves fail-fast configuration validation and gives
|
||||||
errors direct profile and backend context. It is valuable but not required for
|
operator-facing errors direct profile and backend context. It remains optional
|
||||||
the initial migration.
|
for the initial migration.
|
||||||
|
|
||||||
## Priority 4: Eager Source Validation
|
## Priority 4: Eager Source Validation
|
||||||
|
|
||||||
@@ -352,24 +228,22 @@ the existing sentinel remain available.
|
|||||||
|
|
||||||
### Structured Capacity Errors
|
### Structured Capacity Errors
|
||||||
|
|
||||||
**Disposition:** Accepted into the
|
**Disposition:** Implemented behavior. See the consumer guide's
|
||||||
[future catalog](future.md#structured-capacity-errors).
|
[Handle Errors](../consumers/pkg-promptkit.md#handle-errors) section.
|
||||||
|
|
||||||
The typed capacity error proposed by the
|
PromptKit now exposes the stable backend ID for capacity rejection without
|
||||||
[Notarius wishlist](notarius-promptkit-wishlist.md#priority-4-structured-capacity-errors)
|
requiring Weatherreporter to parse error text.
|
||||||
would improve Weatherreporter diagnostics by exposing the stable backend ID
|
|
||||||
without parsing error text.
|
|
||||||
|
|
||||||
Weatherreporter currently generates batch reports sequentially and constructs
|
Weatherreporter currently generates batch reports sequentially and constructs
|
||||||
one engine per invocation, so engine-local capacity exhaustion is unlikely in
|
one engine per invocation, so engine-local capacity exhaustion is unlikely in
|
||||||
the initial design. The feature would become more valuable if report
|
the initial design. The typed error becomes more valuable if report generation
|
||||||
generation later becomes concurrent or PromptKit engines become longer-lived.
|
later becomes concurrent or PromptKit engines become longer-lived.
|
||||||
It should not block adoption.
|
It should not block adoption.
|
||||||
|
|
||||||
### Semantic Execution-Target Fingerprints
|
### Semantic Execution-Target Fingerprints
|
||||||
|
|
||||||
**Disposition:** Deferred until prompt-independent profile inspection defines
|
**Disposition:** Deferred pending a separate semantic-equality design for
|
||||||
the resolved target whose configuration identity would be fingerprinted.
|
resolved execution targets.
|
||||||
|
|
||||||
The semantic target digest proposed by the
|
The semantic target digest proposed by the
|
||||||
[Notarius wishlist](notarius-promptkit-wishlist.md#priority-3-semantic-execution-target-fingerprints)
|
[Notarius wishlist](notarius-promptkit-wishlist.md#priority-3-semantic-execution-target-fingerprints)
|
||||||
@@ -384,12 +258,13 @@ migration priority.
|
|||||||
|
|
||||||
## Capabilities PromptKit Already Provides Well
|
## Capabilities PromptKit Already Provides Well
|
||||||
|
|
||||||
PromptKit v0.3.0 already provides the essential Weatherreporter integration
|
PromptKit already provides the essential Weatherreporter integration surface:
|
||||||
surface:
|
|
||||||
|
|
||||||
- importable in-process engine construction;
|
- importable in-process engine construction;
|
||||||
- filesystem, `fs.FS`, and programmatic prompt, profile, and schema sources;
|
- filesystem, `fs.FS`, and programmatic prompt, profile, and schema sources;
|
||||||
- offline preparation without model execution;
|
- offline preparation without model execution;
|
||||||
|
- prepared execution handles for a durable preflight boundary;
|
||||||
|
- exact prompt and profile inspection;
|
||||||
- versioned prompt selection;
|
- versioned prompt selection;
|
||||||
- text, Markdown, JSON, and JSON Schema output contracts;
|
- text, Markdown, JSON, and JSON Schema output contracts;
|
||||||
- single-pass output validation with raw output retained after completed
|
- single-pass output validation with raw output retained after completed
|
||||||
@@ -401,7 +276,8 @@ surface:
|
|||||||
- injected model-client and artifact-reader interfaces;
|
- injected model-client and artifact-reader interfaces;
|
||||||
- caller cancellation, generation timeout, and transport timeout behavior; and
|
- caller cancellation, generation timeout, and transport timeout behavior; and
|
||||||
- public error sentinels for configuration, prompt, profile, artifact,
|
- public error sentinels for configuration, prompt, profile, artifact,
|
||||||
validation, capacity, and generation failures.
|
validation, capacity, and generation failures; and
|
||||||
|
- structured backend identity for capacity rejection.
|
||||||
|
|
||||||
These capabilities are sufficient for Weatherreporter to adopt PromptKit
|
These capabilities are sufficient for Weatherreporter to adopt PromptKit
|
||||||
without waiting for new upstream work.
|
without waiting for new upstream work.
|
||||||
@@ -426,43 +302,30 @@ PromptKit:
|
|||||||
|
|
||||||
## Suggested Upstream Sequence
|
## Suggested Upstream Sequence
|
||||||
|
|
||||||
If the PromptKit team chooses to pursue these ideas, the most useful order for
|
For downstream adoption and any remaining upstream work, the useful order is:
|
||||||
Weatherreporter would be:
|
|
||||||
|
|
||||||
1. Add executable preparation handles, ideally sharing implementation with an
|
1. Adopt the implemented executable preparation handles.
|
||||||
atomic detailed-run API.
|
2. Consider eager source validation after evaluating whether the two exact
|
||||||
2. Add prompt-definition inspection.
|
|
||||||
3. Add prompt-independent profile inspection.
|
|
||||||
4. Consider eager source validation after evaluating whether the two exact
|
|
||||||
inspection APIs are sufficient.
|
inspection APIs are sufficient.
|
||||||
5. Add structured generation errors.
|
3. Add structured generation errors.
|
||||||
6. Add structured capacity errors and semantic execution-target fingerprints
|
4. Use structured capacity errors and consider semantic execution-target
|
||||||
as lower-priority operational improvements.
|
fingerprints as lower-priority operational improvements.
|
||||||
|
|
||||||
The first item removes the only material integration workaround. Prompt and
|
The first item removes the material integration workaround. Prompt and profile
|
||||||
profile inspection improve fail-fast validation. The remaining items improve
|
inspection improve fail-fast validation. The remaining items are optional
|
||||||
ergonomics and diagnostics.
|
ergonomic and diagnostic improvements.
|
||||||
|
|
||||||
## Adoption Sequencing
|
## Adoption Sequencing
|
||||||
|
|
||||||
Weatherreporter should not wait for the complete wishlist. PromptKit v0.3.0 is
|
Weatherreporter should not wait for the deferred wishlist items. The current
|
||||||
already sufficient when Weatherreporter:
|
PromptKit interface is sufficient when Weatherreporter:
|
||||||
|
|
||||||
- embeds immutable prompt and schema assets;
|
- embeds immutable prompt and schema assets;
|
||||||
- supplies immutable inline data-package bytes;
|
- supplies immutable inline data-package bytes;
|
||||||
- constructs one engine per CLI invocation;
|
- constructs one engine per CLI invocation;
|
||||||
- calls `Prepare` and `Run` with the same request; and
|
- prepares an execution, persists selected `Details`, and calls
|
||||||
|
`RunPrepared`; and
|
||||||
- keeps PromptKit behind a weatherreporter-owned adapter contract.
|
- keeps PromptKit behind a weatherreporter-owned adapter contract.
|
||||||
|
|
||||||
If executable preparation handles are scheduled for a near-term PromptKit
|
|
||||||
release, Weatherreporter may defer only its final adapter implementation to
|
|
||||||
avoid implementing and then removing duplicate preparation. Prompt corpus
|
|
||||||
retrieval, application-contract design, configuration work, embedded assets,
|
|
||||||
state contracts, and offline fixtures can proceed independently.
|
|
||||||
|
|
||||||
If the feature is not scheduled, Weatherreporter can adopt v0.3.0 and keep the
|
|
||||||
duplicate `Prepare` and `Run` sequence inside its adapter. A later PromptKit
|
|
||||||
upgrade would remain localized behind that neutral boundary.
|
|
||||||
|
|
||||||
Prompt inspection, profile inspection, source validation, structured errors,
|
Prompt inspection, profile inspection, source validation, structured errors,
|
||||||
capacity details, and semantic fingerprints should not gate adoption.
|
capacity details, and semantic fingerprints should not gate adoption.
|
||||||
|
|||||||
121
engine.go
121
engine.go
@@ -65,8 +65,8 @@ var (
|
|||||||
ErrPromptRender = errors.New("failed to render prompt")
|
ErrPromptRender = errors.New("failed to render prompt")
|
||||||
// ErrCapacityExceeded identifies a Run or RunPrepared rejected because the
|
// ErrCapacityExceeded identifies a Run or RunPrepared rejected because the
|
||||||
// selected backend already admitted ConcurrencyLimit + QueueCapacity calls.
|
// selected backend already admitted ConcurrencyLimit + QueueCapacity calls.
|
||||||
// It is not an invalid request, an LLM or provider rate-limit response, or
|
// A [CapacityError] reports the selected backend ID. It is not an invalid
|
||||||
// ErrLLMGenerate.
|
// request, an LLM or provider rate-limit response, or ErrLLMGenerate.
|
||||||
ErrCapacityExceeded = errors.New("backend capacity exceeded")
|
ErrCapacityExceeded = errors.New("backend capacity exceeded")
|
||||||
// ErrLLMGenerate identifies a model-client failure or a nil successful
|
// ErrLLMGenerate identifies a model-client failure or a nil successful
|
||||||
// response. Errors returned by an injected LLMClient remain available
|
// response. Errors returned by an injected LLMClient remain available
|
||||||
@@ -78,14 +78,15 @@ var (
|
|||||||
ErrValidation = errors.New("failed to validate output")
|
ErrValidation = errors.New("failed to validate output")
|
||||||
)
|
)
|
||||||
|
|
||||||
// Engine prepares and runs Promptkit prompt requests.
|
// Engine inspects prompts and profiles and prepares and runs Promptkit prompt
|
||||||
|
// requests.
|
||||||
//
|
//
|
||||||
// An Engine is safe for concurrent calls to [Engine.Prepare],
|
// An Engine is safe for concurrent calls to [Engine.InspectPrompt],
|
||||||
// [Engine.PrepareExecution], [Engine.Run], and [Engine.RunPrepared]. Each
|
// [Engine.InspectProfile], [Engine.Prepare], [Engine.PrepareExecution],
|
||||||
// Engine owns independent backend-capacity pools that coordinate Run and
|
// [Engine.Run], and [Engine.RunPrepared]. Each Engine owns independent
|
||||||
// RunPrepared admission and model generation. Injected collaborators may still
|
// backend-capacity pools that coordinate Run and RunPrepared admission and
|
||||||
// be invoked concurrently across different backend pools or for unlimited
|
// model generation. Injected collaborators may still be invoked concurrently
|
||||||
// backends.
|
// across different backend pools or for unlimited backends.
|
||||||
type Engine struct {
|
type Engine struct {
|
||||||
runner *usecase.Runner
|
runner *usecase.Runner
|
||||||
}
|
}
|
||||||
@@ -428,6 +429,93 @@ func fileSource(name string) (fs.FS, string, error) {
|
|||||||
return os.DirFS(dir), filepath.ToSlash(base), nil
|
return os.DirFS(dir), filepath.ToSlash(base), nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// InspectPrompt resolves one explicit prompt definition without selecting a
|
||||||
|
// profile or starting execution work.
|
||||||
|
//
|
||||||
|
// InspectPrompt requires a nonblank promptID. It passes nonblank promptID and
|
||||||
|
// promptVersion values unchanged to the engine's ordinary, case-sensitive
|
||||||
|
// prompt selection. An empty version succeeds only when that source has one
|
||||||
|
// selected ID; a nonempty version selects one exact ID/version pair. The
|
||||||
|
// configured prompt source is used without merging, fallback, or enumeration.
|
||||||
|
//
|
||||||
|
// A successful result proves that the selected definition and any referenced
|
||||||
|
// message content files were structurally loaded. Inputs are returned in
|
||||||
|
// definition order. DefaultProfileID is declared metadata only and is not
|
||||||
|
// resolved. OutputContract is the normalized declared contract, with a JSON
|
||||||
|
// Schema path when declared but without loading or compiling that schema.
|
||||||
|
// PromptHash is the same opaque equality value as PreparedRun.PromptHash for
|
||||||
|
// the selected definition and observed source state; its spelling, length,
|
||||||
|
// encoding, algorithm, and security properties are not contracts.
|
||||||
|
//
|
||||||
|
// This method does not return prompt bodies, templates, source paths, schemas,
|
||||||
|
// rendered messages, or execution settings. It does not resolve a profile or
|
||||||
|
// credential, read artifacts or schemas, render, validate, admit capacity,
|
||||||
|
// contact a provider, or generate model output. The returned PromptInspection
|
||||||
|
// and its input slice are caller-owned. Filesystem-backed inspection is a
|
||||||
|
// point-in-time lookup and does not freeze a definition for later execution.
|
||||||
|
//
|
||||||
|
// A nil Engine returns an error matching ErrInvalidConfig. A blank prompt ID
|
||||||
|
// matches ErrInvalidRequest. An absent exact ID or version matches
|
||||||
|
// ErrPromptNotFound and not ErrPromptLoad. Malformed, unreadable, duplicate,
|
||||||
|
// ambiguous, referenced-content, or hashing failures match ErrPromptLoad.
|
||||||
|
// Cancellation during lookup matches ErrPromptLoad while preserving the
|
||||||
|
// context error. InspectPrompt returns no partial result on error.
|
||||||
|
func (e *Engine) InspectPrompt(
|
||||||
|
ctx context.Context,
|
||||||
|
promptID string,
|
||||||
|
promptVersion string,
|
||||||
|
) (*PromptInspection, error) {
|
||||||
|
if e == nil || e.runner == nil {
|
||||||
|
return nil, fmt.Errorf("%w: engine is nil", ErrInvalidConfig)
|
||||||
|
}
|
||||||
|
|
||||||
|
inspection, err := e.runner.InspectPrompt(ctx, promptID, promptVersion)
|
||||||
|
if err != nil {
|
||||||
|
return nil, mapPublicError(err)
|
||||||
|
}
|
||||||
|
return fromDomainPromptInspection(inspection), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// InspectProfile resolves one explicit profile without selecting a prompt or
|
||||||
|
// starting execution work.
|
||||||
|
//
|
||||||
|
// InspectProfile trims surrounding whitespace from profileID and looks up the
|
||||||
|
// resulting nonblank ID exactly and case-sensitively through the engine's
|
||||||
|
// ordinary in-memory, configured-source, and built-in profile precedence. It
|
||||||
|
// applies framework defaults, the selected backend, and then the selected
|
||||||
|
// profile to EffectiveModelParams without a request override. BackendID is
|
||||||
|
// empty for an endpoint-only profile.
|
||||||
|
//
|
||||||
|
// APIKeyEnv in the returned target is an environment-variable name, never its
|
||||||
|
// value. APIKeyRequired instead reports a direct credential requirement and is
|
||||||
|
// mutually exclusive with a nonblank APIKeyEnv. InspectProfile neither derives
|
||||||
|
// an ID from a prompt default_profile nor checks credential availability, so an
|
||||||
|
// absent or blank named environment variable is not an error.
|
||||||
|
//
|
||||||
|
// The returned ProfileInspection and all nested mutable values are
|
||||||
|
// caller-owned. Filesystem-backed inspection is a point-in-time lookup and
|
||||||
|
// does not freeze the profile for a later execution. This method does not load
|
||||||
|
// a prompt, render, read artifacts or schemas, admit backend capacity, contact
|
||||||
|
// a provider, or generate model output.
|
||||||
|
//
|
||||||
|
// A nil Engine returns an error matching ErrInvalidConfig. A blank profile ID
|
||||||
|
// matches ErrInvalidRequest. An absent exact ID matches ErrProfileNotFound and
|
||||||
|
// not ErrProfileLoad. Malformed or unreadable profile data, an unknown backend,
|
||||||
|
// or an invalid resolved target matches ErrProfileLoad. Cancellation during
|
||||||
|
// profile loading matches ErrProfileLoad while preserving the context error.
|
||||||
|
// InspectProfile returns no partial result on error.
|
||||||
|
func (e *Engine) InspectProfile(ctx context.Context, profileID string) (*ProfileInspection, error) {
|
||||||
|
if e == nil || e.runner == nil {
|
||||||
|
return nil, fmt.Errorf("%w: engine is nil", ErrInvalidConfig)
|
||||||
|
}
|
||||||
|
|
||||||
|
inspection, err := e.runner.InspectProfile(ctx, profileID)
|
||||||
|
if err != nil {
|
||||||
|
return nil, mapPublicError(err)
|
||||||
|
}
|
||||||
|
return fromDomainProfileInspection(inspection), nil
|
||||||
|
}
|
||||||
|
|
||||||
// Prepare resolves and renders a prompt request without calling an LLM.
|
// Prepare resolves and renders a prompt request without calling an LLM.
|
||||||
//
|
//
|
||||||
// Prepare selects the prompt and profile, resolves any selected backend and
|
// Prepare selects the prompt and profile, resolves any selected backend and
|
||||||
@@ -504,9 +592,10 @@ func (e *Engine) PrepareExecution(ctx context.Context, req RunRequest) (*Prepare
|
|||||||
// single-pass even when OutputContract.RepairAttempts is positive.
|
// single-pass even when OutputContract.RepairAttempts is positive.
|
||||||
//
|
//
|
||||||
// Run can return every error category documented by [Engine.Prepare], plus
|
// Run can return every error category documented by [Engine.Prepare], plus
|
||||||
// ErrCapacityExceeded and ErrLLMGenerate. ErrCapacityExceeded identifies
|
// ErrCapacityExceeded and ErrLLMGenerate. An engine admission rejection is
|
||||||
// rejection before artifacts, schemas, rendering, or model generation because
|
// discoverable as [CapacityError] and still matches ErrCapacityExceeded. It
|
||||||
// the selected backend's admission capacity is full; it does not match
|
// occurs before artifacts, schemas, rendering, or model generation because the
|
||||||
|
// selected backend's admission capacity is full; it does not match
|
||||||
// ErrInvalidRequest or ErrLLMGenerate. Errors from injected clients remain
|
// ErrInvalidRequest or ErrLLMGenerate. Errors from injected clients remain
|
||||||
// available through errors.Is. Cancellation while waiting for model-generation
|
// available through errors.Is. Cancellation while waiting for model-generation
|
||||||
// capacity matches both ErrLLMGenerate and the context error. Cancellation
|
// capacity matches both ErrLLMGenerate and the context error. Cancellation
|
||||||
@@ -547,9 +636,11 @@ func (e *Engine) Run(ctx context.Context, req RunRequest) (*RunResult, error) {
|
|||||||
//
|
//
|
||||||
// RunPrepared can return ErrInvalidRequest, ErrAPIKeyEnvMissing,
|
// RunPrepared can return ErrInvalidRequest, ErrAPIKeyEnvMissing,
|
||||||
// ErrCapacityExceeded, ErrLLMGenerate, or ErrValidation as applicable while
|
// ErrCapacityExceeded, ErrLLMGenerate, or ErrValidation as applicable while
|
||||||
// preserving documented collaborator and context identities. A completed
|
// preserving documented collaborator and context identities. An engine
|
||||||
// content-validation rejection is returned in RunResult, not as an
|
// admission rejection is discoverable as [CapacityError] and still matches
|
||||||
// operational error. An operational error returns no partial RunResult.
|
// ErrCapacityExceeded. A completed content-validation rejection is returned
|
||||||
|
// in RunResult, not as an operational error. An operational error returns no
|
||||||
|
// partial RunResult.
|
||||||
func (e *Engine) RunPrepared(ctx context.Context, prepared *PreparedExecution) (*RunResult, error) {
|
func (e *Engine) RunPrepared(ctx context.Context, prepared *PreparedExecution) (*RunResult, error) {
|
||||||
if e == nil || e.runner == nil {
|
if e == nil || e.runner == nil {
|
||||||
return nil, fmt.Errorf("%w: engine is nil", ErrInvalidConfig)
|
return nil, fmt.Errorf("%w: engine is nil", ErrInvalidConfig)
|
||||||
|
|||||||
@@ -3,6 +3,7 @@ package promptkit
|
|||||||
import (
|
import (
|
||||||
"errors"
|
"errors"
|
||||||
"fmt"
|
"fmt"
|
||||||
|
"strings"
|
||||||
|
|
||||||
"gitea.maximumdirect.net/eric/promptkit/internal/capacity"
|
"gitea.maximumdirect.net/eric/promptkit/internal/capacity"
|
||||||
"gitea.maximumdirect.net/eric/promptkit/internal/profile"
|
"gitea.maximumdirect.net/eric/promptkit/internal/profile"
|
||||||
@@ -14,6 +15,11 @@ func mapPublicError(err error) error {
|
|||||||
if err == nil {
|
if err == nil {
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
var internalCapacityError *usecase.CapacityError
|
||||||
|
if errors.As(err, &internalCapacityError) && internalCapacityError != nil &&
|
||||||
|
strings.TrimSpace(internalCapacityError.BackendID) != "" {
|
||||||
|
return &CapacityError{BackendID: internalCapacityError.BackendID}
|
||||||
|
}
|
||||||
publicErr := publicErrorFor(err)
|
publicErr := publicErrorFor(err)
|
||||||
if publicErr == nil {
|
if publicErr == nil {
|
||||||
return err
|
return err
|
||||||
|
|||||||
@@ -20,3 +20,31 @@ func TestMapPublicErrorPreservesGenerationCancellation(t *testing.T) {
|
|||||||
t.Fatalf("mapped error=%v, want context.Canceled", err)
|
t.Fatalf("mapped error=%v, want context.Canceled", err)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func TestMapPublicErrorTranslatesCapacityError(t *testing.T) {
|
||||||
|
internalErr := &usecase.CapacityError{BackendID: "limited"}
|
||||||
|
|
||||||
|
err := mapPublicError(internalErr)
|
||||||
|
var publicErr *CapacityError
|
||||||
|
if !errors.As(err, &publicErr) || publicErr == nil {
|
||||||
|
t.Fatalf("mapped error=%v, want public CapacityError", err)
|
||||||
|
}
|
||||||
|
if publicErr.BackendID != "limited" {
|
||||||
|
t.Fatalf("mapped backend ID=%q, want limited", publicErr.BackendID)
|
||||||
|
}
|
||||||
|
if !errors.Is(err, ErrCapacityExceeded) {
|
||||||
|
t.Fatalf("mapped error=%v, want ErrCapacityExceeded", err)
|
||||||
|
}
|
||||||
|
if errors.Is(err, ErrInvalidRequest) || errors.Is(err, ErrLLMGenerate) {
|
||||||
|
t.Fatalf("mapped capacity error has an unrelated category: %v", err)
|
||||||
|
}
|
||||||
|
var leakedInternalErr *usecase.CapacityError
|
||||||
|
if errors.As(err, &leakedInternalErr) {
|
||||||
|
t.Fatalf("mapped error exposes internal CapacityError: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
internalErr.BackendID = "changed"
|
||||||
|
if publicErr.BackendID != "limited" {
|
||||||
|
t.Fatalf("mapped backend ID changed with source error: %q", publicErr.BackendID)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -124,6 +124,11 @@ func TestManagerAdmissionHonorsContextAndUnlimitedBackends(t *testing.T) {
|
|||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("construct manager: %v", err)
|
t.Fatalf("construct manager: %v", err)
|
||||||
}
|
}
|
||||||
|
release, err := manager.Admit(context.Background(), "limited")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("fill limited pool: %v", err)
|
||||||
|
}
|
||||||
|
defer release()
|
||||||
|
|
||||||
ctx, cancel := context.WithCancel(context.Background())
|
ctx, cancel := context.WithCancel(context.Background())
|
||||||
cancel()
|
cancel()
|
||||||
|
|||||||
@@ -145,6 +145,16 @@ type PromptDefinition struct {
|
|||||||
Validation OutputContract `yaml:"validation"`
|
Validation OutputContract `yaml:"validation"`
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// PromptInspection is the resolved result of exact prompt inspection.
|
||||||
|
type PromptInspection struct {
|
||||||
|
PromptID string
|
||||||
|
PromptVersion string
|
||||||
|
PromptHash string
|
||||||
|
DefaultProfileID string
|
||||||
|
Inputs []PromptInput
|
||||||
|
OutputContract OutputContract
|
||||||
|
}
|
||||||
|
|
||||||
// PromptInput describes one named input expected by a prompt definition.
|
// PromptInput describes one named input expected by a prompt definition.
|
||||||
type PromptInput struct {
|
type PromptInput struct {
|
||||||
Name string `yaml:"name"`
|
Name string `yaml:"name"`
|
||||||
@@ -236,6 +246,13 @@ type ExecutionTarget struct {
|
|||||||
ExtraParams map[string]any `yaml:"extra_params" json:"extra_params"`
|
ExtraParams map[string]any `yaml:"extra_params" json:"extra_params"`
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ProfileInspection is the resolved result of exact profile inspection.
|
||||||
|
type ProfileInspection struct {
|
||||||
|
ProfileID string
|
||||||
|
EffectiveModelParams ExecutionTarget
|
||||||
|
APIKeyRequired bool
|
||||||
|
}
|
||||||
|
|
||||||
// OutputContract defines the requirements for the output artifact.
|
// OutputContract defines the requirements for the output artifact.
|
||||||
type OutputContract struct {
|
type OutputContract struct {
|
||||||
Format OutputFormat `yaml:"format"`
|
Format OutputFormat `yaml:"format"`
|
||||||
|
|||||||
24
internal/usecase/capacity_error.go
Normal file
24
internal/usecase/capacity_error.go
Normal file
@@ -0,0 +1,24 @@
|
|||||||
|
package usecase
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
"strings"
|
||||||
|
|
||||||
|
"gitea.maximumdirect.net/eric/promptkit/internal/capacity"
|
||||||
|
)
|
||||||
|
|
||||||
|
// CapacityError identifies bounded admission rejected for one selected backend.
|
||||||
|
type CapacityError struct {
|
||||||
|
BackendID string
|
||||||
|
}
|
||||||
|
|
||||||
|
func (e *CapacityError) Error() string {
|
||||||
|
if e == nil || strings.TrimSpace(e.BackendID) == "" {
|
||||||
|
return capacity.ErrCapacityExceeded.Error()
|
||||||
|
}
|
||||||
|
return fmt.Sprintf("backend %q admission: %v", e.BackendID, capacity.ErrCapacityExceeded)
|
||||||
|
}
|
||||||
|
|
||||||
|
func (e *CapacityError) Unwrap() error {
|
||||||
|
return capacity.ErrCapacityExceeded
|
||||||
|
}
|
||||||
103
internal/usecase/profile_inspection.go
Normal file
103
internal/usecase/profile_inspection.go
Normal file
@@ -0,0 +1,103 @@
|
|||||||
|
package usecase
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"errors"
|
||||||
|
"fmt"
|
||||||
|
"strings"
|
||||||
|
|
||||||
|
"gitea.maximumdirect.net/eric/promptkit/internal/domain"
|
||||||
|
)
|
||||||
|
|
||||||
|
type resolvedProfileSelection struct {
|
||||||
|
id string
|
||||||
|
profile *domain.ExecutionProfile
|
||||||
|
backend *domain.Backend
|
||||||
|
}
|
||||||
|
|
||||||
|
func (r *Runner) resolveProfileSelection(
|
||||||
|
ctx context.Context,
|
||||||
|
profileID string,
|
||||||
|
) (*resolvedProfileSelection, error) {
|
||||||
|
normalizedID := strings.TrimSpace(profileID)
|
||||||
|
if normalizedID == "" {
|
||||||
|
return nil, fmt.Errorf("%w: profile id is required", ErrInvalidRequest)
|
||||||
|
}
|
||||||
|
if r == nil || r.profiles == nil {
|
||||||
|
return nil, fmt.Errorf("%w: profile repository is not configured", ErrProfileLoad)
|
||||||
|
}
|
||||||
|
|
||||||
|
selectedProfile, err := r.profiles.GetProfile(ctx, normalizedID)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("%w: %w", ErrProfileLoad, err)
|
||||||
|
}
|
||||||
|
if selectedProfile == nil {
|
||||||
|
return nil, fmt.Errorf("%w: profile repository returned nil profile", ErrProfileLoad)
|
||||||
|
}
|
||||||
|
|
||||||
|
profileValue := *selectedProfile
|
||||||
|
profileValue.BackendID = strings.TrimSpace(profileValue.BackendID)
|
||||||
|
|
||||||
|
var selectedBackend *domain.Backend
|
||||||
|
if profileValue.BackendID != "" {
|
||||||
|
if r.backends == nil {
|
||||||
|
return nil, fmt.Errorf("%w: backend %q cannot be resolved", ErrProfileLoad, profileValue.BackendID)
|
||||||
|
}
|
||||||
|
backendValue, err := r.backends.GetBackend(profileValue.BackendID)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("%w: backend %q: %w", ErrProfileLoad, profileValue.BackendID, err)
|
||||||
|
}
|
||||||
|
selectedBackend = &backendValue
|
||||||
|
}
|
||||||
|
|
||||||
|
return &resolvedProfileSelection{
|
||||||
|
id: normalizedID,
|
||||||
|
profile: &profileValue,
|
||||||
|
backend: selectedBackend,
|
||||||
|
}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func validateResolvedExecutionTarget(target domain.ExecutionTarget) error {
|
||||||
|
if strings.TrimSpace(target.Endpoint) == "" {
|
||||||
|
return errors.New("execution endpoint is required")
|
||||||
|
}
|
||||||
|
if strings.TrimSpace(target.Model) == "" {
|
||||||
|
return errors.New("execution model is required")
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// InspectProfile resolves one explicit profile without prompt or execution work.
|
||||||
|
func (r *Runner) InspectProfile(
|
||||||
|
ctx context.Context,
|
||||||
|
profileID string,
|
||||||
|
) (*domain.ProfileInspection, error) {
|
||||||
|
normalizedID := strings.TrimSpace(profileID)
|
||||||
|
if normalizedID == "" {
|
||||||
|
return nil, fmt.Errorf("%w: profile id is required", ErrInvalidRequest)
|
||||||
|
}
|
||||||
|
select {
|
||||||
|
case <-ctx.Done():
|
||||||
|
return nil, fmt.Errorf("%w: %w", ErrProfileLoad, ctx.Err())
|
||||||
|
default:
|
||||||
|
}
|
||||||
|
|
||||||
|
selection, err := r.resolveProfileSelection(ctx, normalizedID)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
target, _, err := resolveExecutionTarget(selection.backend, selection.profile, nil)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("%w: %w", ErrProfileLoad, err)
|
||||||
|
}
|
||||||
|
if err := validateResolvedExecutionTarget(target); err != nil {
|
||||||
|
return nil, fmt.Errorf("%w: %w", ErrProfileLoad, err)
|
||||||
|
}
|
||||||
|
target.APIKey = ""
|
||||||
|
|
||||||
|
return &domain.ProfileInspection{
|
||||||
|
ProfileID: selection.id,
|
||||||
|
EffectiveModelParams: target,
|
||||||
|
APIKeyRequired: target.APIKeyRequired,
|
||||||
|
}, nil
|
||||||
|
}
|
||||||
213
internal/usecase/profile_inspection_test.go
Normal file
213
internal/usecase/profile_inspection_test.go
Normal file
@@ -0,0 +1,213 @@
|
|||||||
|
package usecase
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"errors"
|
||||||
|
"reflect"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"gitea.maximumdirect.net/eric/promptkit/internal/defaults"
|
||||||
|
"gitea.maximumdirect.net/eric/promptkit/internal/domain"
|
||||||
|
"gitea.maximumdirect.net/eric/promptkit/internal/profile"
|
||||||
|
)
|
||||||
|
|
||||||
|
type inspectionProfileRepository struct {
|
||||||
|
profile *domain.ExecutionProfile
|
||||||
|
err error
|
||||||
|
calls int
|
||||||
|
id string
|
||||||
|
}
|
||||||
|
|
||||||
|
func (r *inspectionProfileRepository) GetProfile(
|
||||||
|
_ context.Context,
|
||||||
|
id string,
|
||||||
|
) (*domain.ExecutionProfile, error) {
|
||||||
|
r.calls++
|
||||||
|
r.id = id
|
||||||
|
if r.err != nil {
|
||||||
|
return nil, r.err
|
||||||
|
}
|
||||||
|
return r.profile, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
type inspectionBackendResolver struct {
|
||||||
|
backend domain.Backend
|
||||||
|
err error
|
||||||
|
calls int
|
||||||
|
id string
|
||||||
|
}
|
||||||
|
|
||||||
|
func (r *inspectionBackendResolver) GetBackend(id string) (domain.Backend, error) {
|
||||||
|
r.calls++
|
||||||
|
r.id = id
|
||||||
|
if r.err != nil {
|
||||||
|
return domain.Backend{}, r.err
|
||||||
|
}
|
||||||
|
return r.backend, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunnerInspectProfileResolvesProfileAndBackendOnce(t *testing.T) {
|
||||||
|
profiles := &inspectionProfileRepository{profile: &domain.ExecutionProfile{
|
||||||
|
ID: "profile",
|
||||||
|
BackendID: " backend ",
|
||||||
|
Model: "profile-model",
|
||||||
|
Temperature: 0.4,
|
||||||
|
MaxTokens: 32,
|
||||||
|
TimeoutSeconds: 45,
|
||||||
|
ServiceTier: "priority",
|
||||||
|
ReasoningEffort: "high",
|
||||||
|
ExtraParams: map[string]any{
|
||||||
|
"profile": "value",
|
||||||
|
},
|
||||||
|
}}
|
||||||
|
backends := &inspectionBackendResolver{backend: domain.Backend{
|
||||||
|
ID: "backend",
|
||||||
|
Endpoint: "https://backend.example/v1",
|
||||||
|
APIKeyEnv: "BACKEND_KEY",
|
||||||
|
ExtraParams: map[string]any{
|
||||||
|
"backend": "value",
|
||||||
|
},
|
||||||
|
}}
|
||||||
|
runner := &Runner{profiles: profiles, backends: backends}
|
||||||
|
|
||||||
|
inspection, err := runner.InspectProfile(context.Background(), " profile ")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("inspect profile: %v", err)
|
||||||
|
}
|
||||||
|
if profiles.calls != 1 || profiles.id != "profile" {
|
||||||
|
t.Fatalf("profile lookup=(calls=%d id=%q), want one exact lookup", profiles.calls, profiles.id)
|
||||||
|
}
|
||||||
|
if backends.calls != 1 || backends.id != "backend" {
|
||||||
|
t.Fatalf("backend lookup=(calls=%d id=%q), want one exact lookup", backends.calls, backends.id)
|
||||||
|
}
|
||||||
|
if profiles.profile.BackendID != " backend " {
|
||||||
|
t.Fatalf("inspection mutated repository profile backend: %q", profiles.profile.BackendID)
|
||||||
|
}
|
||||||
|
|
||||||
|
wantTarget := domain.ExecutionTarget{
|
||||||
|
BackendID: "backend",
|
||||||
|
Endpoint: "https://backend.example/v1",
|
||||||
|
Model: "profile-model",
|
||||||
|
Temperature: 0.4,
|
||||||
|
MaxTokens: 32,
|
||||||
|
TopP: defaults.ExecutionTargetDefault().TopP,
|
||||||
|
TimeoutSeconds: 45,
|
||||||
|
ServiceTier: "priority",
|
||||||
|
ReasoningEffort: "high",
|
||||||
|
APIKeyEnv: "BACKEND_KEY",
|
||||||
|
ExtraParams: map[string]any{
|
||||||
|
"profile": "value",
|
||||||
|
},
|
||||||
|
}
|
||||||
|
if inspection.ProfileID != "profile" || inspection.APIKeyRequired ||
|
||||||
|
!reflect.DeepEqual(inspection.EffectiveModelParams, wantTarget) {
|
||||||
|
t.Fatalf("inspection=%#v, want profile=%q target=%#v", inspection, "profile", wantTarget)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunnerInspectProfileDoesNotNeedExecutionCollaboratorsOrCredentials(t *testing.T) {
|
||||||
|
t.Setenv("PROMPTKIT_INSPECTION_TEST_KEY", "")
|
||||||
|
profiles := &inspectionProfileRepository{profile: &domain.ExecutionProfile{
|
||||||
|
ID: "endpoint-only",
|
||||||
|
Endpoint: "https://profile.example/v1",
|
||||||
|
Model: "profile-model",
|
||||||
|
APIKeyEnv: "PROMPTKIT_INSPECTION_TEST_KEY",
|
||||||
|
}}
|
||||||
|
runner := &Runner{profiles: profiles}
|
||||||
|
|
||||||
|
inspection, err := runner.InspectProfile(context.Background(), "endpoint-only")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("inspect endpoint-only profile: %v", err)
|
||||||
|
}
|
||||||
|
if inspection.EffectiveModelParams.BackendID != "" ||
|
||||||
|
inspection.EffectiveModelParams.APIKeyEnv != "PROMPTKIT_INSPECTION_TEST_KEY" ||
|
||||||
|
inspection.APIKeyRequired {
|
||||||
|
t.Fatalf("unexpected endpoint-only inspection: %#v", inspection)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunnerInspectProfileDirectCredentialRequirementClearsBackendEnvironment(t *testing.T) {
|
||||||
|
profiles := &inspectionProfileRepository{profile: &domain.ExecutionProfile{
|
||||||
|
ID: "direct-key",
|
||||||
|
BackendID: "backend",
|
||||||
|
Model: "profile-model",
|
||||||
|
APIKeyRequired: true,
|
||||||
|
}}
|
||||||
|
backends := &inspectionBackendResolver{backend: domain.Backend{
|
||||||
|
ID: "backend",
|
||||||
|
Endpoint: "https://backend.example/v1",
|
||||||
|
APIKeyEnv: "BACKEND_KEY",
|
||||||
|
}}
|
||||||
|
|
||||||
|
inspection, err := (&Runner{profiles: profiles, backends: backends}).InspectProfile(
|
||||||
|
context.Background(),
|
||||||
|
"direct-key",
|
||||||
|
)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("inspect direct-key profile: %v", err)
|
||||||
|
}
|
||||||
|
if !inspection.APIKeyRequired || inspection.EffectiveModelParams.APIKeyEnv != "" {
|
||||||
|
t.Fatalf("credential requirement was not resolved exclusively: %#v", inspection)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunnerInspectProfileClassifiesFailuresWithoutRepositoryWorkAfterCancellation(t *testing.T) {
|
||||||
|
t.Run("blank ID", func(t *testing.T) {
|
||||||
|
profiles := &inspectionProfileRepository{}
|
||||||
|
_, err := (&Runner{profiles: profiles}).InspectProfile(context.Background(), " \t ")
|
||||||
|
if !errors.Is(err, ErrInvalidRequest) || profiles.calls != 0 {
|
||||||
|
t.Fatalf("blank inspection=(%v, calls=%d), want invalid request without lookup", err, profiles.calls)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("canceled context", func(t *testing.T) {
|
||||||
|
ctx, cancel := context.WithCancel(context.Background())
|
||||||
|
cancel()
|
||||||
|
profiles := &inspectionProfileRepository{}
|
||||||
|
_, err := (&Runner{profiles: profiles}).InspectProfile(ctx, "profile")
|
||||||
|
if !errors.Is(err, ErrProfileLoad) || !errors.Is(err, context.Canceled) || profiles.calls != 0 {
|
||||||
|
t.Fatalf("canceled inspection=(%v, calls=%d), want profile load and context identities without lookup", err, profiles.calls)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("missing profile", func(t *testing.T) {
|
||||||
|
profiles := &inspectionProfileRepository{err: profile.ErrProfileNotFound}
|
||||||
|
_, err := (&Runner{profiles: profiles}).InspectProfile(context.Background(), "missing")
|
||||||
|
if !errors.Is(err, ErrProfileLoad) || !errors.Is(err, profile.ErrProfileNotFound) {
|
||||||
|
t.Fatalf("missing profile error=%v, want profile load and not-found identities", err)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("unknown backend", func(t *testing.T) {
|
||||||
|
backendErr := errors.New("unknown backend")
|
||||||
|
profiles := &inspectionProfileRepository{profile: &domain.ExecutionProfile{
|
||||||
|
ID: "profile", BackendID: "backend", Model: "profile-model",
|
||||||
|
}}
|
||||||
|
backends := &inspectionBackendResolver{err: backendErr}
|
||||||
|
_, err := (&Runner{profiles: profiles, backends: backends}).InspectProfile(context.Background(), "profile")
|
||||||
|
if !errors.Is(err, ErrProfileLoad) || !errors.Is(err, backendErr) {
|
||||||
|
t.Fatalf("unknown backend error=%v, want profile load and backend identities", err)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("defensive invalid dependencies", func(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
runner *Runner
|
||||||
|
}{
|
||||||
|
{name: "nil repository", runner: &Runner{}},
|
||||||
|
{name: "nil profile", runner: &Runner{profiles: &inspectionProfileRepository{}}},
|
||||||
|
{name: "invalid target", runner: &Runner{profiles: &inspectionProfileRepository{
|
||||||
|
profile: &domain.ExecutionProfile{ID: "profile", Endpoint: "https://profile.example/v1"},
|
||||||
|
}}},
|
||||||
|
}
|
||||||
|
for _, tc := range cases {
|
||||||
|
t.Run(tc.name, func(t *testing.T) {
|
||||||
|
_, err := tc.runner.InspectProfile(context.Background(), "profile")
|
||||||
|
if !errors.Is(err, ErrProfileLoad) {
|
||||||
|
t.Fatalf("inspection error=%v, want profile load", err)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
73
internal/usecase/prompt_inspection.go
Normal file
73
internal/usecase/prompt_inspection.go
Normal file
@@ -0,0 +1,73 @@
|
|||||||
|
package usecase
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"fmt"
|
||||||
|
"strings"
|
||||||
|
|
||||||
|
"gitea.maximumdirect.net/eric/promptkit/internal/domain"
|
||||||
|
)
|
||||||
|
|
||||||
|
type resolvedPromptDefinition struct {
|
||||||
|
definition *domain.PromptDefinition
|
||||||
|
hash string
|
||||||
|
}
|
||||||
|
|
||||||
|
func (r *Runner) resolvePromptDefinition(
|
||||||
|
ctx context.Context,
|
||||||
|
promptID string,
|
||||||
|
promptVersion string,
|
||||||
|
) (*resolvedPromptDefinition, error) {
|
||||||
|
if strings.TrimSpace(promptID) == "" {
|
||||||
|
return nil, fmt.Errorf("%w: prompt id is required", ErrInvalidRequest)
|
||||||
|
}
|
||||||
|
if r == nil || r.promptDefs == nil {
|
||||||
|
return nil, fmt.Errorf("%w: prompt repository is not configured", ErrPromptLoad)
|
||||||
|
}
|
||||||
|
|
||||||
|
definition, err := r.promptDefs.GetPromptDefinition(ctx, promptID, promptVersion)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("%w: %w", ErrPromptLoad, err)
|
||||||
|
}
|
||||||
|
if definition == nil {
|
||||||
|
return nil, fmt.Errorf("%w: prompt repository returned nil definition", ErrPromptLoad)
|
||||||
|
}
|
||||||
|
|
||||||
|
hash, err := hashPromptDefinition(definition)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("%w: failed to hash prompt definition: %v", ErrPromptLoad, err)
|
||||||
|
}
|
||||||
|
return &resolvedPromptDefinition{definition: definition, hash: hash}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// InspectPrompt resolves one explicit prompt without execution work.
|
||||||
|
func (r *Runner) InspectPrompt(
|
||||||
|
ctx context.Context,
|
||||||
|
promptID string,
|
||||||
|
promptVersion string,
|
||||||
|
) (*domain.PromptInspection, error) {
|
||||||
|
if strings.TrimSpace(promptID) == "" {
|
||||||
|
return nil, fmt.Errorf("%w: prompt id is required", ErrInvalidRequest)
|
||||||
|
}
|
||||||
|
select {
|
||||||
|
case <-ctx.Done():
|
||||||
|
return nil, fmt.Errorf("%w: %w", ErrPromptLoad, ctx.Err())
|
||||||
|
default:
|
||||||
|
}
|
||||||
|
|
||||||
|
selection, err := r.resolvePromptDefinition(ctx, promptID, promptVersion)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
inputs := make([]domain.PromptInput, len(selection.definition.Inputs))
|
||||||
|
copy(inputs, selection.definition.Inputs)
|
||||||
|
|
||||||
|
return &domain.PromptInspection{
|
||||||
|
PromptID: selection.definition.ID,
|
||||||
|
PromptVersion: selection.definition.Version,
|
||||||
|
PromptHash: selection.hash,
|
||||||
|
DefaultProfileID: selection.definition.DefaultProfile,
|
||||||
|
Inputs: inputs,
|
||||||
|
OutputContract: selection.definition.Validation,
|
||||||
|
}, nil
|
||||||
|
}
|
||||||
157
internal/usecase/prompt_inspection_test.go
Normal file
157
internal/usecase/prompt_inspection_test.go
Normal file
@@ -0,0 +1,157 @@
|
|||||||
|
package usecase
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"errors"
|
||||||
|
"reflect"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"gitea.maximumdirect.net/eric/promptkit/internal/domain"
|
||||||
|
"gitea.maximumdirect.net/eric/promptkit/internal/promptdef"
|
||||||
|
)
|
||||||
|
|
||||||
|
type inspectionPromptRepository struct {
|
||||||
|
definition *domain.PromptDefinition
|
||||||
|
err error
|
||||||
|
calls int
|
||||||
|
id string
|
||||||
|
version string
|
||||||
|
}
|
||||||
|
|
||||||
|
func (r *inspectionPromptRepository) GetPromptDefinition(
|
||||||
|
_ context.Context,
|
||||||
|
id string,
|
||||||
|
version string,
|
||||||
|
) (*domain.PromptDefinition, error) {
|
||||||
|
r.calls++
|
||||||
|
r.id = id
|
||||||
|
r.version = version
|
||||||
|
if r.err != nil {
|
||||||
|
return nil, r.err
|
||||||
|
}
|
||||||
|
return r.definition, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunnerInspectPromptResolvesOneDefinitionWithoutExecutionCollaborators(t *testing.T) {
|
||||||
|
definition := &domain.PromptDefinition{
|
||||||
|
ID: "normalized.prompt",
|
||||||
|
Version: "1.2.3",
|
||||||
|
DefaultProfile: "not-resolved",
|
||||||
|
Inputs: []domain.PromptInput{
|
||||||
|
{Name: "document", Required: true, ContentType: "text/plain", Description: "Source document."},
|
||||||
|
{Name: "audience", ContentType: "text/plain", Description: "Intended reader."},
|
||||||
|
},
|
||||||
|
Validation: domain.OutputContract{
|
||||||
|
Format: domain.FormatJSON,
|
||||||
|
ValidationMode: domain.ValidationJSONSchema,
|
||||||
|
SchemaPath: "schemas/result.json",
|
||||||
|
RepairAttempts: 2,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
repository := &inspectionPromptRepository{definition: definition}
|
||||||
|
runner := &Runner{promptDefs: repository}
|
||||||
|
|
||||||
|
inspection, err := runner.InspectPrompt(context.Background(), " prompt-id ", " version ")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("inspect prompt: %v", err)
|
||||||
|
}
|
||||||
|
wantHash, err := hashPromptDefinition(definition)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("hash prompt definition: %v", err)
|
||||||
|
}
|
||||||
|
if repository.calls != 1 || repository.id != " prompt-id " || repository.version != " version " {
|
||||||
|
t.Fatalf("prompt lookup=(calls=%d id=%q version=%q), want one unchanged lookup", repository.calls, repository.id, repository.version)
|
||||||
|
}
|
||||||
|
if inspection.PromptID != definition.ID ||
|
||||||
|
inspection.PromptVersion != definition.Version ||
|
||||||
|
inspection.PromptHash != wantHash ||
|
||||||
|
inspection.DefaultProfileID != definition.DefaultProfile ||
|
||||||
|
!reflect.DeepEqual(inspection.Inputs, definition.Inputs) ||
|
||||||
|
inspection.OutputContract != definition.Validation {
|
||||||
|
t.Fatalf("inspection=%#v, want definition metadata", inspection)
|
||||||
|
}
|
||||||
|
|
||||||
|
inspection.Inputs[0].Name = "changed"
|
||||||
|
second, err := runner.InspectPrompt(context.Background(), " prompt-id ", " version ")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("inspect prompt again: %v", err)
|
||||||
|
}
|
||||||
|
if definition.Inputs[0].Name != "document" || second.Inputs[0].Name != "document" {
|
||||||
|
t.Fatalf("inspection input mutation escaped caller result: definition=%#v next=%#v", definition.Inputs, second.Inputs)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunnerInspectPromptClassifiesFailuresWithoutRepositoryWorkAfterCancellation(t *testing.T) {
|
||||||
|
t.Run("blank ID", func(t *testing.T) {
|
||||||
|
repository := &inspectionPromptRepository{}
|
||||||
|
_, err := (&Runner{promptDefs: repository}).InspectPrompt(context.Background(), " \t ", "version")
|
||||||
|
if !errors.Is(err, ErrInvalidRequest) || repository.calls != 0 {
|
||||||
|
t.Fatalf("blank inspection=(%v, calls=%d), want invalid request without lookup", err, repository.calls)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("canceled context", func(t *testing.T) {
|
||||||
|
ctx, cancel := context.WithCancel(context.Background())
|
||||||
|
cancel()
|
||||||
|
repository := &inspectionPromptRepository{}
|
||||||
|
_, err := (&Runner{promptDefs: repository}).InspectPrompt(ctx, "prompt", "version")
|
||||||
|
if !errors.Is(err, ErrPromptLoad) || !errors.Is(err, context.Canceled) || repository.calls != 0 {
|
||||||
|
t.Fatalf("canceled inspection=(%v, calls=%d), want prompt load and context identities without lookup", err, repository.calls)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("missing prompt", func(t *testing.T) {
|
||||||
|
repository := &inspectionPromptRepository{err: promptdef.ErrPromptDefinitionNotFound}
|
||||||
|
_, err := (&Runner{promptDefs: repository}).InspectPrompt(context.Background(), "missing", "version")
|
||||||
|
if !errors.Is(err, ErrPromptLoad) || !errors.Is(err, promptdef.ErrPromptDefinitionNotFound) {
|
||||||
|
t.Fatalf("missing prompt error=%v, want prompt load and not-found identities", err)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("defensive prompt dependencies", func(t *testing.T) {
|
||||||
|
var nilRunner *Runner
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
runner *Runner
|
||||||
|
}{
|
||||||
|
{name: "nil runner", runner: nilRunner},
|
||||||
|
{name: "nil repository", runner: &Runner{}},
|
||||||
|
{name: "nil definition", runner: &Runner{promptDefs: &inspectionPromptRepository{}}},
|
||||||
|
}
|
||||||
|
for _, tc := range cases {
|
||||||
|
t.Run(tc.name, func(t *testing.T) {
|
||||||
|
_, err := tc.runner.InspectPrompt(context.Background(), "prompt", "version")
|
||||||
|
if !errors.Is(err, ErrPromptLoad) {
|
||||||
|
t.Fatalf("inspection error=%v, want prompt load", err)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunnerPrepareUsesThePromptInspectionSelectionAndHash(t *testing.T) {
|
||||||
|
definition := promptDef(domain.FormatMarkdown, domain.ValidationBasic, 0)
|
||||||
|
repository := &fakePromptRepo{def: definition}
|
||||||
|
runner := NewRunner(
|
||||||
|
repository,
|
||||||
|
&fakeExecutionProfileRepo{profiles: map[string]*domain.ExecutionProfile{"exec": defaultExecutionProfile()}},
|
||||||
|
nil,
|
||||||
|
&fakeArtifactReader{},
|
||||||
|
&fakeRenderer{rendered: &domain.RenderedPrompt{Messages: []domain.RenderedMessage{{Role: "user", Content: "hello"}}}},
|
||||||
|
nil,
|
||||||
|
nil,
|
||||||
|
nil,
|
||||||
|
)
|
||||||
|
|
||||||
|
inspection, err := runner.InspectPrompt(context.Background(), definition.ID, definition.Version)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("inspect prompt: %v", err)
|
||||||
|
}
|
||||||
|
prepared, err := runner.Prepare(context.Background(), domain.RunRequest{PromptID: definition.ID, PromptVersion: definition.Version, ProfileID: "exec"})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("prepare prompt: %v", err)
|
||||||
|
}
|
||||||
|
if inspection.PromptHash != prepared.PromptHash {
|
||||||
|
t.Fatalf("inspection hash=%q, preparation hash=%q", inspection.PromptHash, prepared.PromptHash)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -266,14 +266,12 @@ func (r *Runner) resolvePreparation(
|
|||||||
return nil, fmt.Errorf("%w: session_id: %v", ErrInvalidRequest, err)
|
return nil, fmt.Errorf("%w: session_id: %v", ErrInvalidRequest, err)
|
||||||
}
|
}
|
||||||
|
|
||||||
def, err := r.promptDefs.GetPromptDefinition(ctx, req.PromptID, req.PromptVersion)
|
promptSelection, err := r.resolvePromptDefinition(ctx, req.PromptID, req.PromptVersion)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, fmt.Errorf("%w: %w", ErrPromptLoad, err)
|
return nil, err
|
||||||
}
|
|
||||||
promptDefinitionHash, err := hashPromptDefinition(def)
|
|
||||||
if err != nil {
|
|
||||||
return nil, fmt.Errorf("%w: failed to hash prompt definition: %v", ErrPromptLoad, err)
|
|
||||||
}
|
}
|
||||||
|
def := promptSelection.definition
|
||||||
|
promptDefinitionHash := promptSelection.hash
|
||||||
|
|
||||||
selectedProfileID := strings.TrimSpace(req.ProfileID)
|
selectedProfileID := strings.TrimSpace(req.ProfileID)
|
||||||
if selectedProfileID == "" {
|
if selectedProfileID == "" {
|
||||||
@@ -283,34 +281,18 @@ func (r *Runner) resolvePreparation(
|
|||||||
return nil, fmt.Errorf("%w: %w: profile id is required either in request or prompt default_profile", ErrInvalidRequest, ErrProfileRequired)
|
return nil, fmt.Errorf("%w: %w: profile id is required either in request or prompt default_profile", ErrInvalidRequest, ErrProfileRequired)
|
||||||
}
|
}
|
||||||
|
|
||||||
execProfile, err := r.profiles.GetProfile(ctx, selectedProfileID)
|
selection, err := r.resolveProfileSelection(ctx, selectedProfileID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, fmt.Errorf("%w: %w", ErrProfileLoad, err)
|
return nil, err
|
||||||
}
|
}
|
||||||
|
|
||||||
var selectedBackend *domain.Backend
|
effectiveModel, targetPresence, err := resolveExecutionTarget(selection.backend, selection.profile, req.Execution)
|
||||||
if backendID := strings.TrimSpace(execProfile.BackendID); backendID != "" {
|
|
||||||
execProfile.BackendID = backendID
|
|
||||||
if r.backends == nil {
|
|
||||||
return nil, fmt.Errorf("%w: backend %q cannot be resolved", ErrProfileLoad, backendID)
|
|
||||||
}
|
|
||||||
resolvedBackend, resolveErr := r.backends.GetBackend(backendID)
|
|
||||||
if resolveErr != nil {
|
|
||||||
return nil, fmt.Errorf("%w: backend %q: %w", ErrProfileLoad, backendID, resolveErr)
|
|
||||||
}
|
|
||||||
selectedBackend = &resolvedBackend
|
|
||||||
}
|
|
||||||
|
|
||||||
effectiveModel, targetPresence, err := resolveExecutionTarget(selectedBackend, execProfile, req.Execution)
|
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, fmt.Errorf("%w: %w", ErrInvalidRequest, err)
|
return nil, fmt.Errorf("%w: %w", ErrInvalidRequest, err)
|
||||||
}
|
}
|
||||||
effectiveModel.APIKey = req.APIKey
|
effectiveModel.APIKey = req.APIKey
|
||||||
if strings.TrimSpace(effectiveModel.Endpoint) == "" {
|
if err := validateResolvedExecutionTarget(effectiveModel); err != nil {
|
||||||
return nil, fmt.Errorf("%w: execution endpoint is required", ErrInvalidRequest)
|
return nil, fmt.Errorf("%w: %w", ErrInvalidRequest, err)
|
||||||
}
|
|
||||||
if strings.TrimSpace(effectiveModel.Model) == "" {
|
|
||||||
return nil, fmt.Errorf("%w: execution model is required", ErrInvalidRequest)
|
|
||||||
}
|
}
|
||||||
if err := validateAPIKey(effectiveModel.APIKeyEnv, effectiveModel.APIKey, effectiveModel.APIKeyRequired); err != nil {
|
if err := validateAPIKey(effectiveModel.APIKeyEnv, effectiveModel.APIKey, effectiveModel.APIKeyRequired); err != nil {
|
||||||
return nil, fmt.Errorf("%w: %w", ErrInvalidRequest, err)
|
return nil, fmt.Errorf("%w: %w", ErrInvalidRequest, err)
|
||||||
@@ -321,7 +303,7 @@ func (r *Runner) resolvePreparation(
|
|||||||
definition: def,
|
definition: def,
|
||||||
directSessionID: directSessionID,
|
directSessionID: directSessionID,
|
||||||
promptDefinitionHash: promptDefinitionHash,
|
promptDefinitionHash: promptDefinitionHash,
|
||||||
selectedProfileID: selectedProfileID,
|
selectedProfileID: selection.id,
|
||||||
effectiveModel: effectiveModel,
|
effectiveModel: effectiveModel,
|
||||||
targetPresence: targetPresence,
|
targetPresence: targetPresence,
|
||||||
effectiveContract: effectiveContract,
|
effectiveContract: effectiveContract,
|
||||||
@@ -408,8 +390,8 @@ func (r *Runner) admitRun(ctx context.Context, backendID string) (func(), error)
|
|||||||
}
|
}
|
||||||
release, err := r.admitter.Admit(ctx, backendID)
|
release, err := r.admitter.Admit(ctx, backendID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
if errors.Is(err, capacity.ErrCapacityExceeded) {
|
if errors.Is(err, capacity.ErrCapacityExceeded) && strings.TrimSpace(backendID) != "" {
|
||||||
return nil, fmt.Errorf("backend %q admission: %w", backendID, err)
|
return nil, &CapacityError{BackendID: backendID}
|
||||||
}
|
}
|
||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1357,14 +1357,14 @@ func TestRunnerAdmissionUsesResolvedBackendIdentity(t *testing.T) {
|
|||||||
|
|
||||||
func TestRunnerAdmissionFailureSkipsCompletionCollaborators(t *testing.T) {
|
func TestRunnerAdmissionFailureSkipsCompletionCollaborators(t *testing.T) {
|
||||||
tests := []struct {
|
tests := []struct {
|
||||||
name string
|
name string
|
||||||
admissionError error
|
admissionError error
|
||||||
wantBackendContext bool
|
wantCapacityType bool
|
||||||
}{
|
}{
|
||||||
{
|
{
|
||||||
name: "capacity exhausted",
|
name: "capacity exhausted",
|
||||||
admissionError: capacity.ErrCapacityExceeded,
|
admissionError: capacity.ErrCapacityExceeded,
|
||||||
wantBackendContext: true,
|
wantCapacityType: true,
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
name: "context canceled",
|
name: "context canceled",
|
||||||
@@ -1414,8 +1414,16 @@ func TestRunnerAdmissionFailureSkipsCompletionCollaborators(t *testing.T) {
|
|||||||
if errors.Is(err, ErrInvalidRequest) || errors.Is(err, ErrLLMGenerate) {
|
if errors.Is(err, ErrInvalidRequest) || errors.Is(err, ErrLLMGenerate) {
|
||||||
t.Fatalf("admission error was recategorized: %v", err)
|
t.Fatalf("admission error was recategorized: %v", err)
|
||||||
}
|
}
|
||||||
if tc.wantBackendContext && !strings.Contains(err.Error(), "custom") {
|
var capacityErr *CapacityError
|
||||||
t.Fatalf("capacity error lacks backend context: %v", err)
|
if tc.wantCapacityType {
|
||||||
|
if !errors.As(err, &capacityErr) {
|
||||||
|
t.Fatalf("capacity error=%v, want internal typed identity", err)
|
||||||
|
}
|
||||||
|
if capacityErr.BackendID != "custom" {
|
||||||
|
t.Fatalf("capacity backend ID=%q, want custom", capacityErr.BackendID)
|
||||||
|
}
|
||||||
|
} else if errors.As(err, &capacityErr) {
|
||||||
|
t.Fatalf("non-capacity admission error exposed typed capacity identity: %v", err)
|
||||||
}
|
}
|
||||||
if !reflect.DeepEqual(admitter.backendIDs, []string{"custom"}) {
|
if !reflect.DeepEqual(admitter.backendIDs, []string{"custom"}) {
|
||||||
t.Fatalf("admitted backend IDs=%#v, want custom", admitter.backendIDs)
|
t.Fatalf("admitted backend IDs=%#v, want custom", admitter.backendIDs)
|
||||||
|
|||||||
@@ -560,6 +560,11 @@ func TestPreparedExecutionCredentialCapacityAndTimingBoundaries(t *testing.T) {
|
|||||||
if result, err := engine.RunPrepared(context.Background(), prepared); result != nil ||
|
if result, err := engine.RunPrepared(context.Background(), prepared); result != nil ||
|
||||||
!errors.Is(err, promptkit.ErrCapacityExceeded) {
|
!errors.Is(err, promptkit.ErrCapacityExceeded) {
|
||||||
t.Fatalf("capacity execution=(%+v, %v), want ErrCapacityExceeded", result, err)
|
t.Fatalf("capacity execution=(%+v, %v), want ErrCapacityExceeded", result, err)
|
||||||
|
} else {
|
||||||
|
var capacityErr *promptkit.CapacityError
|
||||||
|
if !errors.As(err, &capacityErr) || capacityErr == nil || capacityErr.BackendID != "limited" {
|
||||||
|
t.Fatalf("capacity execution=%v, want limited CapacityError", err)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
if result, err := engine.RunPrepared(context.Background(), prepared); result != nil ||
|
if result, err := engine.RunPrepared(context.Background(), prepared); result != nil ||
|
||||||
!errors.Is(err, promptkit.ErrInvalidRequest) {
|
!errors.Is(err, promptkit.ErrInvalidRequest) {
|
||||||
|
|||||||
@@ -5,6 +5,7 @@ import (
|
|||||||
"encoding/json"
|
"encoding/json"
|
||||||
"errors"
|
"errors"
|
||||||
"fmt"
|
"fmt"
|
||||||
|
"io/fs"
|
||||||
"reflect"
|
"reflect"
|
||||||
"strings"
|
"strings"
|
||||||
"sync"
|
"sync"
|
||||||
@@ -16,6 +17,380 @@ import (
|
|||||||
"gitea.maximumdirect.net/eric/promptkit"
|
"gitea.maximumdirect.net/eric/promptkit"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
type inspectionCountingFS struct {
|
||||||
|
opens atomic.Int64
|
||||||
|
}
|
||||||
|
|
||||||
|
func (f *inspectionCountingFS) Open(string) (fs.File, error) {
|
||||||
|
f.opens.Add(1)
|
||||||
|
return nil, fs.ErrNotExist
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestInspectProfileResolvesCredentialStatesWithoutPromptOrGeneration(t *testing.T) {
|
||||||
|
const environmentName = "PROMPTKIT_INSPECTION_ABSENT_KEY"
|
||||||
|
t.Setenv(environmentName, "")
|
||||||
|
client := &fakeLLMClient{response: &promptkit.GenerateResponse{Content: "unexpected"}}
|
||||||
|
engine, err := promptkit.NewEngine(promptkit.Config{},
|
||||||
|
promptkit.WithPromptFS(fstest.MapFS{}, "."),
|
||||||
|
promptkit.WithProfileFS(fstest.MapFS{
|
||||||
|
"environment.yaml": &fstest.MapFile{Data: []byte(`id: environment
|
||||||
|
endpoint: http://environment.example/v1
|
||||||
|
model: environment-model
|
||||||
|
api_key_env: PROMPTKIT_INSPECTION_ABSENT_KEY
|
||||||
|
`)},
|
||||||
|
}, "."),
|
||||||
|
promptkit.WithProfiles(
|
||||||
|
promptkit.Profile{ID: "direct", Endpoint: "http://direct.example/v1", Model: "direct-model", APIKeyRequired: true},
|
||||||
|
promptkit.Profile{ID: "none", Endpoint: "http://none.example/v1", Model: "none-model"},
|
||||||
|
),
|
||||||
|
promptkit.WithLLMClient(client),
|
||||||
|
)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("construct inspection engine: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, tc := range []struct {
|
||||||
|
profileID string
|
||||||
|
wantEnv string
|
||||||
|
wantDirectKey bool
|
||||||
|
wantEndpoint string
|
||||||
|
}{
|
||||||
|
{profileID: " environment ", wantEnv: environmentName, wantEndpoint: "http://environment.example/v1"},
|
||||||
|
{profileID: "direct", wantDirectKey: true, wantEndpoint: "http://direct.example/v1"},
|
||||||
|
{profileID: "none", wantEndpoint: "http://none.example/v1"},
|
||||||
|
} {
|
||||||
|
t.Run(tc.profileID, func(t *testing.T) {
|
||||||
|
inspection, err := engine.InspectProfile(context.Background(), tc.profileID)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("inspect profile: %v", err)
|
||||||
|
}
|
||||||
|
if inspection.ProfileID != strings.TrimSpace(tc.profileID) ||
|
||||||
|
inspection.EffectiveModelParams.Endpoint != tc.wantEndpoint ||
|
||||||
|
inspection.EffectiveModelParams.BackendID != "" ||
|
||||||
|
inspection.EffectiveModelParams.APIKeyEnv != tc.wantEnv ||
|
||||||
|
inspection.APIKeyRequired != tc.wantDirectKey {
|
||||||
|
t.Fatalf("unexpected inspection: %#v", inspection)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
if len(client.requests) != 0 {
|
||||||
|
t.Fatalf("inspection invoked the model client %d times", len(client.requests))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestInspectProfilePreservesPublicErrorIdentities(t *testing.T) {
|
||||||
|
newEngine := func(t *testing.T, options ...promptkit.Option) *promptkit.Engine {
|
||||||
|
t.Helper()
|
||||||
|
engine, err := promptkit.NewEngine(
|
||||||
|
promptkit.Config{},
|
||||||
|
append([]promptkit.Option{promptkit.WithPromptFS(fstest.MapFS{}, ".")}, options...)...,
|
||||||
|
)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("construct inspection engine: %v", err)
|
||||||
|
}
|
||||||
|
return engine
|
||||||
|
}
|
||||||
|
|
||||||
|
var nilEngine *promptkit.Engine
|
||||||
|
if result, err := nilEngine.InspectProfile(context.Background(), "profile"); result != nil ||
|
||||||
|
!errors.Is(err, promptkit.ErrInvalidConfig) {
|
||||||
|
t.Fatalf("nil engine result=(%#v, %v), want ErrInvalidConfig", result, err)
|
||||||
|
}
|
||||||
|
|
||||||
|
valid := newEngine(t, promptkit.WithProfiles(promptkit.Profile{
|
||||||
|
ID: "profile", Endpoint: "http://profile.example/v1", Model: "model",
|
||||||
|
}))
|
||||||
|
if result, err := valid.InspectProfile(context.Background(), " \t "); result != nil ||
|
||||||
|
!errors.Is(err, promptkit.ErrInvalidRequest) {
|
||||||
|
t.Fatalf("blank profile result=(%#v, %v), want ErrInvalidRequest", result, err)
|
||||||
|
}
|
||||||
|
if result, err := valid.InspectProfile(context.Background(), "missing"); result != nil ||
|
||||||
|
!errors.Is(err, promptkit.ErrProfileNotFound) || errors.Is(err, promptkit.ErrProfileLoad) {
|
||||||
|
t.Fatalf("missing profile result=(%#v, %v), want only ErrProfileNotFound", result, err)
|
||||||
|
}
|
||||||
|
|
||||||
|
malformed := newEngine(t, promptkit.WithProfileFS(fstest.MapFS{
|
||||||
|
"broken.yaml": &fstest.MapFile{Data: []byte("id: broken\nendpoint: http://broken.example/v1\nmodel: model\nunknown: value\n")},
|
||||||
|
}, "."))
|
||||||
|
if result, err := malformed.InspectProfile(context.Background(), "broken"); result != nil ||
|
||||||
|
!errors.Is(err, promptkit.ErrProfileLoad) {
|
||||||
|
t.Fatalf("malformed profile result=(%#v, %v), want ErrProfileLoad", result, err)
|
||||||
|
}
|
||||||
|
|
||||||
|
unknownBackend := newEngine(t, promptkit.WithProfiles(promptkit.Profile{
|
||||||
|
ID: "unknown-backend", BackendID: "unknown", Model: "model",
|
||||||
|
}))
|
||||||
|
if result, err := unknownBackend.InspectProfile(context.Background(), "unknown-backend"); result != nil ||
|
||||||
|
!errors.Is(err, promptkit.ErrProfileLoad) {
|
||||||
|
t.Fatalf("unknown backend result=(%#v, %v), want ErrProfileLoad", result, err)
|
||||||
|
}
|
||||||
|
|
||||||
|
countingFS := &inspectionCountingFS{}
|
||||||
|
canceled := newEngine(t, promptkit.WithProfileFS(countingFS, "."))
|
||||||
|
ctx, cancel := context.WithCancel(context.Background())
|
||||||
|
cancel()
|
||||||
|
if result, err := canceled.InspectProfile(ctx, "profile"); result != nil ||
|
||||||
|
!errors.Is(err, promptkit.ErrProfileLoad) || !errors.Is(err, context.Canceled) || countingFS.opens.Load() != 0 {
|
||||||
|
t.Fatalf("canceled inspection result=(%#v, %v), opens=%d", result, err, countingFS.opens.Load())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestInspectProfileReturnsIndependentTargetMatchingPreparation(t *testing.T) {
|
||||||
|
extraParams := map[string]any{
|
||||||
|
"nested": map[string]any{"value": "original"},
|
||||||
|
}
|
||||||
|
engine, err := promptkit.NewEngine(promptkit.Config{},
|
||||||
|
promptkit.WithPromptFS(contractPromptFS("prompt", "profile", "message"), "."),
|
||||||
|
promptkit.WithProfiles(promptkit.Profile{
|
||||||
|
ID: "profile", Endpoint: "http://profile.example/v1", Model: "model", ExtraParams: extraParams,
|
||||||
|
}),
|
||||||
|
)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("construct inspection engine: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
first, err := engine.InspectProfile(context.Background(), "profile")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("first inspection: %v", err)
|
||||||
|
}
|
||||||
|
first.EffectiveModelParams.ExtraParams["nested"].(map[string]any)["value"] = "changed"
|
||||||
|
first.EffectiveModelParams.ExtraParams["later"] = true
|
||||||
|
|
||||||
|
second, err := engine.InspectProfile(context.Background(), "profile")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("second inspection: %v", err)
|
||||||
|
}
|
||||||
|
prepared, err := engine.Prepare(context.Background(), promptkit.RunRequest{PromptID: "prompt"})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("prepare after inspection mutation: %v", err)
|
||||||
|
}
|
||||||
|
for _, target := range []promptkit.ExecutionTarget{second.EffectiveModelParams, prepared.EffectiveModelParams} {
|
||||||
|
if target.ExtraParams["nested"].(map[string]any)["value"] != "original" || target.ExtraParams["later"] != nil {
|
||||||
|
t.Fatalf("inspection mutation reached engine-owned target: %#v", target.ExtraParams)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if !reflect.DeepEqual(second.EffectiveModelParams, prepared.EffectiveModelParams) {
|
||||||
|
t.Fatalf("inspection target=%#v, preparation target=%#v", second.EffectiveModelParams, prepared.EffectiveModelParams)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestInspectPromptReturnsDeclaredMetadataWithoutExecutionWork(t *testing.T) {
|
||||||
|
client := &fakeLLMClient{response: &promptkit.GenerateResponse{Content: "unexpected"}}
|
||||||
|
engine, err := promptkit.NewEngine(promptkit.Config{},
|
||||||
|
promptkit.WithPromptFS(fstest.MapFS{
|
||||||
|
"report-v1.yaml": &fstest.MapFile{Data: []byte(`id: report
|
||||||
|
version: "1.0.0"
|
||||||
|
messages:
|
||||||
|
- role: user
|
||||||
|
content: old report
|
||||||
|
output:
|
||||||
|
format: text
|
||||||
|
validation_mode: none
|
||||||
|
`)},
|
||||||
|
"report-v2.yaml": &fstest.MapFile{Data: []byte(`id: report
|
||||||
|
version: "2.0.0"
|
||||||
|
default_profile: missing-profile
|
||||||
|
inputs:
|
||||||
|
- name: location
|
||||||
|
required: true
|
||||||
|
content_type: text/plain
|
||||||
|
description: Forecast location.
|
||||||
|
- name: units
|
||||||
|
content_type: text/plain
|
||||||
|
description: Unit preference.
|
||||||
|
messages:
|
||||||
|
- role: user
|
||||||
|
content_file: messages/report.md
|
||||||
|
output:
|
||||||
|
format: json
|
||||||
|
validation_mode: json_schema
|
||||||
|
schema_path: schemas/report.json
|
||||||
|
`)},
|
||||||
|
"messages/report.md": &fstest.MapFile{Data: []byte("rendered report body is not returned")},
|
||||||
|
}, "."),
|
||||||
|
promptkit.WithLLMClient(client),
|
||||||
|
)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("construct prompt-inspection engine: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
inspection, err := engine.InspectPrompt(context.Background(), "report", "2.0.0")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("inspect prompt: %v", err)
|
||||||
|
}
|
||||||
|
if inspection.PromptID != "report" ||
|
||||||
|
inspection.PromptVersion != "2.0.0" ||
|
||||||
|
inspection.PromptHash == "" ||
|
||||||
|
inspection.DefaultProfileID != "missing-profile" ||
|
||||||
|
inspection.OutputContract != (promptkit.OutputContract{
|
||||||
|
Format: promptkit.FormatJSON,
|
||||||
|
ValidationMode: promptkit.ValidationJSONSchema,
|
||||||
|
SchemaPath: "schemas/report.json",
|
||||||
|
}) {
|
||||||
|
t.Fatalf("unexpected inspection metadata: %#v", inspection)
|
||||||
|
}
|
||||||
|
wantInputs := []promptkit.PromptInputDefinition{
|
||||||
|
{Name: "location", Required: true, ContentType: "text/plain", Description: "Forecast location."},
|
||||||
|
{Name: "units", ContentType: "text/plain", Description: "Unit preference."},
|
||||||
|
}
|
||||||
|
if !reflect.DeepEqual(inspection.Inputs, wantInputs) {
|
||||||
|
t.Fatalf("inspection inputs=%#v, want %#v", inspection.Inputs, wantInputs)
|
||||||
|
}
|
||||||
|
if len(client.requests) != 0 {
|
||||||
|
t.Fatalf("inspection invoked the model client %d times", len(client.requests))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestInspectPromptPreservesPublicErrorIdentities(t *testing.T) {
|
||||||
|
newEngine := func(t *testing.T, source fstest.MapFS) *promptkit.Engine {
|
||||||
|
t.Helper()
|
||||||
|
engine, err := promptkit.NewEngine(promptkit.Config{}, promptkit.WithPromptFS(source, "."))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("construct prompt-inspection engine: %v", err)
|
||||||
|
}
|
||||||
|
return engine
|
||||||
|
}
|
||||||
|
validSource := fstest.MapFS{
|
||||||
|
"prompt.yaml": &fstest.MapFile{Data: []byte(`id: prompt
|
||||||
|
version: "1"
|
||||||
|
messages:
|
||||||
|
- role: user
|
||||||
|
content: body
|
||||||
|
output:
|
||||||
|
format: text
|
||||||
|
validation_mode: none
|
||||||
|
`)},
|
||||||
|
}
|
||||||
|
|
||||||
|
var nilEngine *promptkit.Engine
|
||||||
|
if result, err := nilEngine.InspectPrompt(context.Background(), "prompt", "1"); result != nil ||
|
||||||
|
!errors.Is(err, promptkit.ErrInvalidConfig) {
|
||||||
|
t.Fatalf("nil engine result=(%#v, %v), want ErrInvalidConfig", result, err)
|
||||||
|
}
|
||||||
|
|
||||||
|
valid := newEngine(t, validSource)
|
||||||
|
if result, err := valid.InspectPrompt(context.Background(), " \t ", "1"); result != nil ||
|
||||||
|
!errors.Is(err, promptkit.ErrInvalidRequest) {
|
||||||
|
t.Fatalf("blank prompt result=(%#v, %v), want ErrInvalidRequest", result, err)
|
||||||
|
}
|
||||||
|
if result, err := valid.InspectPrompt(context.Background(), "missing", "1"); result != nil ||
|
||||||
|
!errors.Is(err, promptkit.ErrPromptNotFound) || errors.Is(err, promptkit.ErrPromptLoad) {
|
||||||
|
t.Fatalf("missing prompt result=(%#v, %v), want only ErrPromptNotFound", result, err)
|
||||||
|
}
|
||||||
|
if result, err := valid.InspectPrompt(context.Background(), "prompt", "missing"); result != nil ||
|
||||||
|
!errors.Is(err, promptkit.ErrPromptNotFound) || errors.Is(err, promptkit.ErrPromptLoad) {
|
||||||
|
t.Fatalf("missing version result=(%#v, %v), want only ErrPromptNotFound", result, err)
|
||||||
|
}
|
||||||
|
|
||||||
|
ambiguous := newEngine(t, fstest.MapFS{
|
||||||
|
"one.yaml": &fstest.MapFile{Data: []byte(`id: prompt
|
||||||
|
version: "1"
|
||||||
|
messages:
|
||||||
|
- role: user
|
||||||
|
content: first
|
||||||
|
output:
|
||||||
|
format: text
|
||||||
|
validation_mode: none
|
||||||
|
`)},
|
||||||
|
"two.yaml": &fstest.MapFile{Data: []byte(`id: prompt
|
||||||
|
version: "2"
|
||||||
|
messages:
|
||||||
|
- role: user
|
||||||
|
content: second
|
||||||
|
output:
|
||||||
|
format: text
|
||||||
|
validation_mode: none
|
||||||
|
`)},
|
||||||
|
})
|
||||||
|
if result, err := ambiguous.InspectPrompt(context.Background(), "prompt", ""); result != nil ||
|
||||||
|
!errors.Is(err, promptkit.ErrPromptLoad) {
|
||||||
|
t.Fatalf("ambiguous prompt result=(%#v, %v), want ErrPromptLoad", result, err)
|
||||||
|
}
|
||||||
|
|
||||||
|
for name, source := range map[string]fstest.MapFS{
|
||||||
|
"malformed definition": {
|
||||||
|
"broken.yaml": &fstest.MapFile{Data: []byte("id: broken\nversion: \"1\"\nunknown: value\n")},
|
||||||
|
},
|
||||||
|
"missing content file": {
|
||||||
|
"broken.yaml": &fstest.MapFile{Data: []byte(`id: broken
|
||||||
|
version: "1"
|
||||||
|
messages:
|
||||||
|
- role: user
|
||||||
|
content_file: missing.md
|
||||||
|
output:
|
||||||
|
format: text
|
||||||
|
validation_mode: none
|
||||||
|
`)},
|
||||||
|
},
|
||||||
|
} {
|
||||||
|
t.Run(name, func(t *testing.T) {
|
||||||
|
if result, err := newEngine(t, source).InspectPrompt(context.Background(), "broken", "1"); result != nil ||
|
||||||
|
!errors.Is(err, promptkit.ErrPromptLoad) {
|
||||||
|
t.Fatalf("broken prompt result=(%#v, %v), want ErrPromptLoad", result, err)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
countingFS := &inspectionCountingFS{}
|
||||||
|
canceled, err := promptkit.NewEngine(promptkit.Config{}, promptkit.WithPromptFS(countingFS, "."))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("construct canceled prompt-inspection engine: %v", err)
|
||||||
|
}
|
||||||
|
ctx, cancel := context.WithCancel(context.Background())
|
||||||
|
cancel()
|
||||||
|
if result, err := canceled.InspectPrompt(ctx, "prompt", "1"); result != nil ||
|
||||||
|
!errors.Is(err, promptkit.ErrPromptLoad) || !errors.Is(err, context.Canceled) || countingFS.opens.Load() != 0 {
|
||||||
|
t.Fatalf("canceled inspection result=(%#v, %v), opens=%d", result, err, countingFS.opens.Load())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestInspectPromptReturnsIndependentMetadataMatchingPreparation(t *testing.T) {
|
||||||
|
engine, err := promptkit.NewEngine(promptkit.Config{},
|
||||||
|
promptkit.WithPromptFS(fstest.MapFS{
|
||||||
|
"prompt.yaml": &fstest.MapFile{Data: []byte(`id: prompt
|
||||||
|
version: "1"
|
||||||
|
default_profile: profile
|
||||||
|
inputs:
|
||||||
|
- name: subject
|
||||||
|
content_type: text/plain
|
||||||
|
description: Summary subject.
|
||||||
|
messages:
|
||||||
|
- role: user
|
||||||
|
content: summarize
|
||||||
|
output:
|
||||||
|
format: markdown
|
||||||
|
validation_mode: basic
|
||||||
|
`)},
|
||||||
|
}, "."),
|
||||||
|
promptkit.WithProfiles(promptkit.Profile{
|
||||||
|
ID: "profile", Endpoint: "http://profile.example/v1", Model: "model",
|
||||||
|
}),
|
||||||
|
)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("construct prompt-inspection engine: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
first, err := engine.InspectPrompt(context.Background(), "prompt", "1")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("first inspection: %v", err)
|
||||||
|
}
|
||||||
|
first.Inputs[0].Name = "changed"
|
||||||
|
first.OutputContract.SchemaPath = "changed.json"
|
||||||
|
|
||||||
|
second, err := engine.InspectPrompt(context.Background(), "prompt", "1")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("second inspection: %v", err)
|
||||||
|
}
|
||||||
|
prepared, err := engine.Prepare(context.Background(), promptkit.RunRequest{PromptID: "prompt", PromptVersion: "1"})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("prepare after inspection mutation: %v", err)
|
||||||
|
}
|
||||||
|
if second.Inputs[0].Name != "subject" || second.OutputContract.SchemaPath != "" ||
|
||||||
|
prepared.OutputContract.SchemaPath != "" || second.PromptHash != prepared.PromptHash {
|
||||||
|
t.Fatalf("inspection mutation reached engine-owned prompt metadata: inspection=%#v prepared=%#v", second, prepared)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func TestPreparedRunJSONOmitsZeroTimingValues(t *testing.T) {
|
func TestPreparedRunJSONOmitsZeroTimingValues(t *testing.T) {
|
||||||
payload, err := json.Marshal(promptkit.PreparedRun{})
|
payload, err := json.Marshal(promptkit.PreparedRun{})
|
||||||
if err != nil {
|
if err != nil {
|
||||||
|
|||||||
55
types.go
55
types.go
@@ -320,6 +320,61 @@ type ExecutionTarget struct {
|
|||||||
ExtraParams map[string]any `json:"extra_params"`
|
ExtraParams map[string]any `json:"extra_params"`
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ProfileInspection is the caller-owned result of [Engine.InspectProfile].
|
||||||
|
// It has no stable JSON representation.
|
||||||
|
//
|
||||||
|
// EffectiveModelParams contains a copied effective target. APIKeyRequired is
|
||||||
|
// separate from that target to preserve ExecutionTarget's general execution
|
||||||
|
// and stable JSON contracts.
|
||||||
|
type ProfileInspection struct {
|
||||||
|
// ProfileID is the trimmed, exact profile ID inspected by the engine.
|
||||||
|
ProfileID string
|
||||||
|
// EffectiveModelParams contains framework defaults overlaid by the selected
|
||||||
|
// backend and then the profile, without a request override. APIKeyEnv is an
|
||||||
|
// environment-variable name, never its credential value.
|
||||||
|
EffectiveModelParams ExecutionTarget
|
||||||
|
// APIKeyRequired reports that a later execution must supply a direct API
|
||||||
|
// key or an explicit request environment override. It is mutually exclusive
|
||||||
|
// with a nonblank EffectiveModelParams.APIKeyEnv.
|
||||||
|
APIKeyRequired bool
|
||||||
|
}
|
||||||
|
|
||||||
|
// PromptInputDefinition describes one declared prompt input.
|
||||||
|
// It has no stable JSON representation.
|
||||||
|
type PromptInputDefinition struct {
|
||||||
|
// Name is the normalized prompt input name.
|
||||||
|
Name string
|
||||||
|
// Required is the prompt definition's declared required flag. When true,
|
||||||
|
// preparation fails if the input is omitted. A false value does not account
|
||||||
|
// for input references in message or session-ID templates.
|
||||||
|
Required bool
|
||||||
|
// ContentType is the declared input media-type metadata.
|
||||||
|
ContentType string
|
||||||
|
// Description is the declared human-readable input description.
|
||||||
|
Description string
|
||||||
|
}
|
||||||
|
|
||||||
|
// PromptInspection is the caller-owned result of [Engine.InspectPrompt].
|
||||||
|
// It has no stable JSON representation.
|
||||||
|
//
|
||||||
|
// Inputs contains copied declared input metadata in definition order.
|
||||||
|
// OutputContract is the normalized contract declared by the prompt definition,
|
||||||
|
// rather than a request-level effective override. PromptHash is opaque.
|
||||||
|
type PromptInspection struct {
|
||||||
|
// PromptID is the normalized ID of the selected prompt definition.
|
||||||
|
PromptID string
|
||||||
|
// PromptVersion is the normalized version of the selected prompt definition.
|
||||||
|
PromptVersion string
|
||||||
|
// PromptHash is the opaque equality value for the selected definition.
|
||||||
|
PromptHash string
|
||||||
|
// DefaultProfileID is declared metadata and is not resolved by inspection.
|
||||||
|
DefaultProfileID string
|
||||||
|
// Inputs contains caller-owned declared input metadata in definition order.
|
||||||
|
Inputs []PromptInputDefinition
|
||||||
|
// OutputContract is the normalized contract declared by the definition.
|
||||||
|
OutputContract OutputContract
|
||||||
|
}
|
||||||
|
|
||||||
// ExecutionTargetOverride represents per-request runtime setting overrides and
|
// ExecutionTargetOverride represents per-request runtime setting overrides and
|
||||||
// has no stable JSON representation.
|
// has no stable JSON representation.
|
||||||
//
|
//
|
||||||
|
|||||||
Reference in New Issue
Block a user