Compare commits
13 Commits
v0.6.0
...
2d44305a8a
| Author | SHA1 | Date | |
|---|---|---|---|
| 2d44305a8a | |||
| a11c80291e | |||
| 3239567297 | |||
| c239304c2a | |||
| 159b02116f | |||
| e5b7adfb49 | |||
| fa2384e696 | |||
| af0bd3f31a | |||
| 5fff8cd623 | |||
| b2a6c47778 | |||
| a1805fe550 | |||
| 93af155254 | |||
| d783b687a5 |
25
backends.go
25
backends.go
@@ -9,6 +9,10 @@ import (
|
||||
// backend.
|
||||
const BackendOpenRouter = backend.OpenRouterID
|
||||
|
||||
// BackendRakestrawHome is the reserved ID of Promptkit's built-in
|
||||
// Rakestrawhome backend.
|
||||
const BackendRakestrawHome = backend.RakestrawHomeID
|
||||
|
||||
// BackendLocal is the case-sensitive conventional ID used by [LocalBackend].
|
||||
// It is not a built-in or reserved backend and must be registered with
|
||||
// [WithBackend].
|
||||
@@ -20,15 +24,18 @@ const BackendLocal = "local"
|
||||
// to this configuration value do not break source compatibility.
|
||||
type Backend struct {
|
||||
// ID is the stable, case-sensitive registry key. NewEngine trims it and
|
||||
// requires a non-blank value. BackendOpenRouter is reserved.
|
||||
// requires a non-blank value. Built-in backend IDs are reserved.
|
||||
ID string
|
||||
// Endpoint is the OpenAI-compatible base endpoint. NewEngine trims it and
|
||||
// requires an absolute HTTP or HTTPS URL with a host and without user
|
||||
// information, a query string, or a fragment. Paths are allowed.
|
||||
Endpoint string
|
||||
// APIKeyEnv optionally names the environment variable containing the API
|
||||
// key. NewEngine trims it and requires the portable form
|
||||
// [A-Za-z_][A-Za-z0-9_]*. Store only the name, never a credential value.
|
||||
// APIKeyEnv optionally names an environment lookup source for an API key.
|
||||
// NewEngine trims it and requires the portable form [A-Za-z_][A-Za-z0-9_]*.
|
||||
// A direct RunRequest.APIKey takes precedence. When no usable credential is
|
||||
// available, the built-in client omits Authorization; injected clients own
|
||||
// their own credential-resolution behavior. Store only the name, never a
|
||||
// credential value.
|
||||
APIKeyEnv string
|
||||
// ExtraParams contains backend-wide request defaults. Values must be
|
||||
// JSON-compatible, finite, acyclic, and keyed by non-empty strings. Keys
|
||||
@@ -74,11 +81,11 @@ func LocalBackend(endpoint string, concurrencyLimit int) Backend {
|
||||
//
|
||||
// Registrations accumulate in option order. Every normalized ID must be unique
|
||||
// across consumer registrations and built-ins; a duplicate or invalid
|
||||
// definition makes NewEngine fail with ErrInvalidConfig. In particular,
|
||||
// BackendOpenRouter cannot be replaced. The immutable registration is scoped
|
||||
// to the resulting Engine and cannot be enumerated, replaced, removed, or
|
||||
// mutated after construction. WithBackend does not install package-global
|
||||
// state.
|
||||
// definition makes NewEngine fail with ErrInvalidConfig. Built-in IDs,
|
||||
// including [BackendOpenRouter] and [BackendRakestrawHome], cannot be
|
||||
// replaced. The immutable registration is scoped to the resulting Engine and
|
||||
// cannot be enumerated, replaced, removed, or mutated after construction.
|
||||
// WithBackend does not install package-global state.
|
||||
func WithBackend(backend Backend) Option {
|
||||
queueCapacity := 0
|
||||
queueCapacitySet := backend.QueueCapacity != nil
|
||||
|
||||
16
doc.go
16
doc.go
@@ -23,8 +23,9 @@
|
||||
// InspectProfile return copied inspection values. Returned values and values
|
||||
// passed to extension interfaces are likewise isolated from engine state.
|
||||
// Callers own those copies and may mutate them after the call that supplied or
|
||||
// returned them. Returned structured errors are likewise caller-owned and may
|
||||
// be mutated without affecting engine state or another error.
|
||||
// returned them. [CapacityError] values are caller-owned and may be mutated
|
||||
// without affecting engine state or another error. Immutable [GenerationError]
|
||||
// values are also caller-owned and do not retain shared engine state.
|
||||
//
|
||||
// # Security and sensitive data
|
||||
//
|
||||
@@ -53,9 +54,14 @@
|
||||
// Construction, inspection, handle, and error values, including [Config],
|
||||
// [Backend], [RunRequest], [ArtifactRef], [ExecutionTargetOverride], [Profile],
|
||||
// [OpenAICompatibleProfileConfig], [ProfileInspection],
|
||||
// [PromptInputDefinition], [PromptInspection], [PreparedExecution], and
|
||||
// [CapacityError], do not have stable JSON representations. Direct API keys
|
||||
// are nevertheless excluded from JSON for every public value.
|
||||
// [PromptInputDefinition], [PromptInspection], [PreparedExecution],
|
||||
// [CapacityError], and [GenerationError], do not have stable JSON
|
||||
// representations. Direct API keys are nevertheless excluded from JSON for
|
||||
// every public value.
|
||||
|
||||
// Provider-derived [GenerationError] accessor values are untrusted and can
|
||||
// contain sensitive request or schema fragments. Applications must apply their
|
||||
// own disclosure policy before logging, displaying, or returning them.
|
||||
//
|
||||
// JSON timestamps use time.Time's RFC 3339 encoding and are omitted when zero.
|
||||
// PreparedRun and RunResult durations are encoded as integer milliseconds in
|
||||
|
||||
@@ -189,6 +189,33 @@ For programmatic profiles,
|
||||
[`OpenAICompatibleProfile`](../../profiles.go) converts ordinary
|
||||
OpenAI-compatible settings into a value accepted by `WithProfiles`.
|
||||
|
||||
### Use The Rakestrawhome Built-In Profile
|
||||
|
||||
Set `RAKESTRAWHOME_INFERENCE_API_KEY` in the application environment, then
|
||||
select `rakestrawhome-gemma-4-31b` as an ordinary profile ID. For example, a
|
||||
prepared result identifies the selected built-in through
|
||||
`BackendRakestrawHome`:
|
||||
|
||||
```go
|
||||
prepared, err := engine.Prepare(ctx, promptkit.RunRequest{
|
||||
PromptID: "meeting.summary",
|
||||
ProfileID: "rakestrawhome-gemma-4-31b",
|
||||
Inputs: inputs,
|
||||
})
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if prepared.SelectedBackendID != promptkit.BackendRakestrawHome {
|
||||
return fmt.Errorf("unexpected backend %q", prepared.SelectedBackendID)
|
||||
}
|
||||
```
|
||||
|
||||
Do not register `rakestrawhome` manually. When adopting this built-in, remove
|
||||
an existing `WithBackend` registration with that exact ID; retaining it causes
|
||||
the intentional duplicate-ID configuration error. Direct request credentials
|
||||
and runtime endpoint overrides remain supported under their ordinary GoDoc and
|
||||
format contracts.
|
||||
|
||||
### Inspect A Profile Before Prompt Work
|
||||
|
||||
Use [`Engine.InspectProfile`](../../engine.go) to validate one configured
|
||||
@@ -204,7 +231,7 @@ if err != nil {
|
||||
|
||||
target := inspection.EffectiveModelParams
|
||||
if target.APIKeyEnv != "" {
|
||||
// Apply application policy for the named environment variable.
|
||||
// This is a configured optional environment lookup source.
|
||||
} else if inspection.APIKeyRequired {
|
||||
// Arrange a direct credential before later execution.
|
||||
}
|
||||
@@ -213,9 +240,11 @@ if target.APIKeyEnv != "" {
|
||||
Use this configuration-time boundary when only the profile and its target need
|
||||
checking. Use `Prepare` when the application also needs prompt, input, schema,
|
||||
or rendering work; use prepared execution when that work must remain tied to a
|
||||
later execution. Inspection reports credential requirements but leaves the
|
||||
timing of credential enforcement to the application. The method's
|
||||
[GoDoc](../../engine.go) owns its exact result and error contract.
|
||||
later execution. A reported `APIKeyEnv` is a configured optional source, while
|
||||
`APIKeyRequired` is the explicit local requirement. The
|
||||
[credential format reference](../formats.md#credentials) and the method's
|
||||
[GoDoc](../../engine.go) own the exact precedence, timing, result, and error
|
||||
contracts.
|
||||
|
||||
### Set A Per-Run Session And Reasoning
|
||||
|
||||
@@ -423,6 +452,24 @@ status; those choices remain with the consuming application. The
|
||||
contract, while the [`Engine.Run` and error GoDoc](../../engine.go) owns broad
|
||||
error and cancellation identities.
|
||||
|
||||
For a non-2xx response from the built-in OpenAI-compatible client, inspect the
|
||||
status and deliberately selected provider diagnostic when useful:
|
||||
|
||||
```go
|
||||
var generationErr *promptkit.GenerationError
|
||||
if errors.As(err, &generationErr) {
|
||||
status := generationErr.StatusCode()
|
||||
message := generationErr.ProviderMessage()
|
||||
_, _ = status, message // Apply application retry and presentation policy.
|
||||
}
|
||||
```
|
||||
|
||||
All provider fields are untrusted and can contain sensitive request or schema
|
||||
fragments. Do not log, display, or return them without an application-specific
|
||||
disclosure policy. Promptkit does not assign retry or presentation behavior.
|
||||
The [`GenerationError` GoDoc](../../generation_error.go) owns the exact typed
|
||||
error contract.
|
||||
|
||||
## Application Boundary
|
||||
|
||||
Promptkit is an importable library. It does not own a command, inbound HTTP
|
||||
|
||||
@@ -144,6 +144,17 @@ A request-level `OutputContract` replaces the complete prompt output contract.
|
||||
It does not merge individual fields. If its format is empty, Promptkit uses
|
||||
`text`.
|
||||
|
||||
## Built-In Backends
|
||||
|
||||
Every engine provides these reserved OpenAI-compatible backend IDs. Consumers
|
||||
must not register either ID with `WithBackend`; exact registration and
|
||||
reservation behavior belongs to the [`Backend` GoDoc](../backends.go).
|
||||
|
||||
| ID | Base endpoint | API-key environment variable | Active generation limit | Default queue capacity |
|
||||
| --- | --- | --- | ---: | ---: |
|
||||
| `openrouter` | `https://openrouter.ai/api/v1` | `OPENROUTER_API_KEY` | 16 | 1024 |
|
||||
| `rakestrawhome` | `https://inference.ai.rakestrawhome.com/v1` | `RAKESTRAWHOME_INFERENCE_API_KEY` | 4 | 1024 |
|
||||
|
||||
## Profile Definitions
|
||||
|
||||
A profile supplies model execution settings:
|
||||
@@ -174,7 +185,7 @@ extra_params:
|
||||
| `timeout_seconds` | no | Per-generation deadline in whole seconds; integer zero or greater. |
|
||||
| `service_tier` | no | Provider-specific request tier. |
|
||||
| `reasoning_effort` | no | Provider-specific reasoning setting. |
|
||||
| `api_key_env` | no | Name of an environment variable containing the API key. |
|
||||
| `api_key_env` | no | Optional environment-variable lookup source for an API key. |
|
||||
| `extra_params` | no | JSON-compatible provider-specific outbound fields. |
|
||||
|
||||
Raw `api_key` is prohibited in profile YAML. Store only an environment
|
||||
@@ -182,8 +193,8 @@ variable name in `api_key_env`.
|
||||
|
||||
Promptkit does not infer a backend from a model or endpoint. Endpoint-only
|
||||
profiles remain supported and have no effective backend ID.
|
||||
The engine always provides the built-in `openrouter` ID. Consumers can add
|
||||
engine-scoped IDs with
|
||||
The engine always provides the built-in `openrouter` and `rakestrawhome` IDs.
|
||||
Consumers can add engine-scoped IDs with
|
||||
[`WithBackend`](../backends.go); exact registration validation belongs to its
|
||||
GoDoc.
|
||||
|
||||
@@ -257,39 +268,38 @@ precedence.
|
||||
|
||||
## Built-In Profile Catalog
|
||||
|
||||
Every built-in selects the `openrouter` backend. The engine's built-in backend
|
||||
registry supplies `https://openrouter.ai/api/v1` and the environment-variable
|
||||
name `OPENROUTER_API_KEY`, so individual profiles contain only model and
|
||||
generation settings. Built-in profile files do not repeat those connection
|
||||
values. A configured, application fallback, or in-memory profile with the same
|
||||
profile ID takes precedence.
|
||||
Every built-in profile selects one maintained built-in backend and inherits
|
||||
that backend's connection and credential metadata. Profile files do not repeat
|
||||
those values. A configured, application fallback, or in-memory profile with
|
||||
the same profile ID takes precedence.
|
||||
|
||||
| Provider | ID | Model |
|
||||
| --- | --- | --- |
|
||||
| aion-labs | `aion-2` | `aion-labs/aion-2.0` |
|
||||
| anthropic | `claude-fable-latest` | `~anthropic/claude-fable-latest` |
|
||||
| anthropic | `claude-haiku-latest` | `~anthropic/claude-haiku-latest` |
|
||||
| anthropic | `claude-opus-latest` | `~anthropic/claude-opus-latest` |
|
||||
| anthropic | `claude-sonnet-latest` | `~anthropic/claude-sonnet-latest` |
|
||||
| deepseek | `deepseek-3-2` | `deepseek/deepseek-v3.2` |
|
||||
| deepseek | `deepseek-4-flash` | `deepseek/deepseek-v4-flash` |
|
||||
| deepseek | `deepseek-4-pro` | `deepseek/deepseek-v4-pro` |
|
||||
| google | `gemini-2-flash` | `google/gemini-2.5-flash` |
|
||||
| google | `gemini-2-flash-lite` | `google/gemini-2.5-flash-lite` |
|
||||
| google | `gemini-2-pro` | `google/gemini-2.5-pro` |
|
||||
| google | `gemini-3-flash-lite` | `google/gemini-3.1-flash-lite` |
|
||||
| google | `gemini-flash-latest` | `~google/gemini-flash-latest` |
|
||||
| google | `gemini-pro-latest` | `~google/gemini-pro-latest` |
|
||||
| google | `gemma-4-31b` | `google/gemma-4-31b-it:exacto` |
|
||||
| minimax | `minimax-m2` | `minimax/minimax-m2.5` |
|
||||
| minimax | `minimax-m3` | `minimax/minimax-m3` |
|
||||
| mistral | `mistral-large-2512` | `mistralai/mistral-large-2512` |
|
||||
| mistral | `mistral-medium-3-5` | `mistralai/mistral-medium-3-5` |
|
||||
| mistral | `mistral-small-3` | `mistralai/mistral-small-3.2-24b-instruct` |
|
||||
| mistral | `mistral-small-4` | `mistralai/mistral-small-2603` |
|
||||
| nvidia | `nemotron-3-ultra` | `nvidia/nemotron-3-ultra-550b-a55b` |
|
||||
| openai | `gpt-5-mini` | `openai/gpt-5.4-mini` |
|
||||
| openai | `gpt-5-nano` | `openai/gpt-5.4-nano` |
|
||||
| Provider | ID | Backend | Model |
|
||||
| --- | --- | --- | --- |
|
||||
| aion-labs | `aion-2` | `openrouter` | `aion-labs/aion-2.0` |
|
||||
| anthropic | `claude-fable-latest` | `openrouter` | `~anthropic/claude-fable-latest` |
|
||||
| anthropic | `claude-haiku-latest` | `openrouter` | `~anthropic/claude-haiku-latest` |
|
||||
| anthropic | `claude-opus-latest` | `openrouter` | `~anthropic/claude-opus-latest` |
|
||||
| anthropic | `claude-sonnet-latest` | `openrouter` | `~anthropic/claude-sonnet-latest` |
|
||||
| deepseek | `deepseek-3-2` | `openrouter` | `deepseek/deepseek-v3.2` |
|
||||
| deepseek | `deepseek-4-flash` | `openrouter` | `deepseek/deepseek-v4-flash` |
|
||||
| deepseek | `deepseek-4-pro` | `openrouter` | `deepseek/deepseek-v4-pro` |
|
||||
| google | `gemini-2-flash` | `openrouter` | `google/gemini-2.5-flash` |
|
||||
| google | `gemini-2-flash-lite` | `openrouter` | `google/gemini-2.5-flash-lite` |
|
||||
| google | `gemini-2-pro` | `openrouter` | `google/gemini-2.5-pro` |
|
||||
| google | `gemini-3-flash-lite` | `openrouter` | `google/gemini-3.1-flash-lite` |
|
||||
| google | `gemini-flash-latest` | `openrouter` | `~google/gemini-flash-latest` |
|
||||
| google | `gemini-pro-latest` | `openrouter` | `~google/gemini-pro-latest` |
|
||||
| google | `gemma-4-31b` | `openrouter` | `google/gemma-4-31b-it:exacto` |
|
||||
| google | `rakestrawhome-gemma-4-31b` | `rakestrawhome` | `google/gemma-4-31b-it` |
|
||||
| minimax | `minimax-m2` | `openrouter` | `minimax/minimax-m2.5` |
|
||||
| minimax | `minimax-m3` | `openrouter` | `minimax/minimax-m3` |
|
||||
| mistral | `mistral-large-2512` | `openrouter` | `mistralai/mistral-large-2512` |
|
||||
| mistral | `mistral-medium-3-5` | `openrouter` | `mistralai/mistral-medium-3-5` |
|
||||
| mistral | `mistral-small-3` | `openrouter` | `mistralai/mistral-small-3.2-24b-instruct` |
|
||||
| mistral | `mistral-small-4` | `openrouter` | `mistralai/mistral-small-2603` |
|
||||
| nvidia | `nemotron-3-ultra` | `openrouter` | `nvidia/nemotron-3-ultra-550b-a55b` |
|
||||
| openai | `gpt-5-mini` | `openrouter` | `openai/gpt-5.4-mini` |
|
||||
| openai | `gpt-5-nano` | `openrouter` | `openai/gpt-5.4-nano` |
|
||||
|
||||
## Schemas
|
||||
|
||||
@@ -308,16 +318,25 @@ schema produces a failed validation result.
|
||||
Credential values belong at the request or environment boundary, never in
|
||||
prompt, profile, schema, or example files:
|
||||
|
||||
- a file profile names an environment variable with `api_key_env`;
|
||||
- an in-memory profile may set `APIKeyRequired`;
|
||||
- a request can provide a direct `APIKey` or override `APIKeyEnv`; and
|
||||
- a backend or file profile can name an optional environment lookup source
|
||||
with `APIKeyEnv` or `api_key_env`;
|
||||
- an in-memory profile may set `APIKeyRequired` as an explicit local
|
||||
requirement;
|
||||
- a request can provide a direct `APIKey` or override the optional `APIKeyEnv`
|
||||
source; and
|
||||
- a direct request key takes precedence over environment lookup.
|
||||
|
||||
After a direct request key, the credential-source precedence is request
|
||||
`APIKeyEnv`, profile `api_key_env`, then the backend default. An in-memory
|
||||
profile with `APIKeyRequired` clears an inherited backend environment name and
|
||||
requires a direct key unless the request explicitly supplies `APIKeyEnv`.
|
||||
Promptkit validates required credential availability during preparation.
|
||||
Named environment sources are optional: when the selected source is absent,
|
||||
empty, or whitespace-only, the built-in client omits the `Authorization`
|
||||
header and handles the provider response normally. `APIKeyRequired` is the
|
||||
only explicit local availability requirement. Promptkit validates required
|
||||
credential availability during preparation and rechecks it when a prepared
|
||||
execution runs. Injected clients receive resolved source metadata but define
|
||||
their own credential-resolution behavior.
|
||||
Direct keys are excluded from JSON results and redacted by public string
|
||||
formatters. Environment-variable names may appear in prepared metadata, but
|
||||
their values do not.
|
||||
|
||||
@@ -31,11 +31,13 @@ does not serialize it in the provider request.
|
||||
|
||||
## Authentication
|
||||
|
||||
A non-empty API key supplied directly on the execution target takes
|
||||
precedence. Otherwise, when an API-key environment-variable name is supplied,
|
||||
the client reads that variable and requires a non-empty value. The selected
|
||||
key is sent as `Authorization: Bearer <key>`. No authorization header is sent
|
||||
when neither mechanism is configured.
|
||||
A usable API key supplied directly on the execution target takes precedence.
|
||||
Otherwise, when an API-key environment-variable name is supplied, the client
|
||||
reads and trims that variable. A bearer header is sent only when the resolved
|
||||
direct or environment credential is non-empty. When neither source is usable,
|
||||
the client omits `Authorization` and handles the provider response normally.
|
||||
An explicitly required target with no usable source is rejected before
|
||||
transport.
|
||||
|
||||
The target contains the already resolved environment-variable name: an
|
||||
explicit request override takes precedence over profile metadata, which takes
|
||||
@@ -102,10 +104,26 @@ and cache-write token counts. Invalid or truncated JSON, trailing non-whitespace
|
||||
data, a second JSON value, absent choices, empty first-choice content, and size
|
||||
overflow are malformed responses and return no partial result.
|
||||
|
||||
For a non-2xx status, the error includes the status code but never the provider
|
||||
response body. Promptkit does not yet parse provider error envelopes; bounded
|
||||
non-success parsing belongs to the
|
||||
[structured-generation-error roadmap](../roadmap/structured-generation-errors.md).
|
||||
For a non-2xx status, Promptkit recognizes one JSON document with a top-level
|
||||
object-valued `error` member. Its optional `message` and `type` fields must be
|
||||
strings, and `code` may be a string or JSON number. Valid supported fields are
|
||||
handled independently, numeric codes retain their JSON number text, and
|
||||
unknown fields are ignored. Missing, invalid, malformed, or multiply framed
|
||||
envelopes contribute no provider detail.
|
||||
|
||||
Non-success bodies have a 65,536-byte limit. A larger declared
|
||||
`Content-Length` is not read; otherwise the client reads at most one additional
|
||||
byte to detect streamed or underreported overflow. Empty, unreadable,
|
||||
oversized, malformed, and unrecognized bodies retain only the received status.
|
||||
The body is always closed and no oversized stream is drained beyond that probe.
|
||||
|
||||
Extracted strings are made valid UTF-8, trimmed, and converted to one line by
|
||||
collapsing Unicode whitespace, control, and format-character runs. Blank
|
||||
values are omitted. Codes and types longer than 256 Unicode code points are
|
||||
omitted; messages longer than 4,096 code points are truncated at a code-point
|
||||
boundary with an ellipsis inside the limit. Promptkit never exposes raw bodies,
|
||||
headers, endpoints, credentials, request data, schemas, generated content, or
|
||||
unsupported provider metadata through this handling.
|
||||
|
||||
An outbound `http.Client.Do` failure retains both Promptkit's request-failure
|
||||
identity and the exact transport error for `errors.Is` and `errors.As` checks.
|
||||
|
||||
@@ -37,8 +37,9 @@ resolved request target may supply the endpoint. Generation then:
|
||||
4. composes `/chat/completions` through parsed URL path operations;
|
||||
5. resolves authentication;
|
||||
6. performs the outbound request under the applicable deadlines; and
|
||||
7. decodes one strictly framed, size-bounded response object and maps its first
|
||||
choice and token usage.
|
||||
7. decodes one strictly framed, size-bounded successful response object and
|
||||
maps its first choice and token usage, or decodes bounded structured
|
||||
non-success detail.
|
||||
|
||||
`internal/llm` owns the set of reserved OpenAI-compatible request fields used
|
||||
when validating extra parameters. Backend registration consumes the same rule
|
||||
@@ -54,12 +55,14 @@ the target, rendered messages, and structured-output constraint retained by
|
||||
executable preparation. Execution does not reopen or rerender consumer
|
||||
sources.
|
||||
|
||||
Before backend admission, the runner rechecks that the frozen credential
|
||||
environment-variable name is available. The handle does not retain the
|
||||
environment value; the model client resolves the value visible when generation
|
||||
begins. A direct request key remains in private execution state only until the
|
||||
claimed execution finishes or an unclaimed handle is discarded. Exact public
|
||||
ownership and redaction semantics belong to the
|
||||
Before backend admission, the runner rechecks a frozen credential
|
||||
environment-variable name only when the target explicitly requires a
|
||||
credential. The handle does not retain the environment value; the model client
|
||||
resolves the value visible when generation begins. For optional sources with no
|
||||
usable value, the built-in client omits `Authorization` and continues to the
|
||||
provider. A direct request key remains in private execution state only until
|
||||
the claimed execution finishes or an unclaimed handle is discarded. Exact
|
||||
public ownership and redaction semantics belong to the
|
||||
[`PreparedExecution` GoDoc](../../prepared_execution.go).
|
||||
|
||||
## Failure Categories
|
||||
@@ -67,21 +70,32 @@ ownership and redaction semantics belong to the
|
||||
The package preserves distinct error identities for invalid client
|
||||
configuration, invalid generation requests, request execution failures,
|
||||
non-success provider statuses, and malformed successful responses. Provider
|
||||
response bodies are not included in non-success errors.
|
||||
response bodies are never exposed in raw form through non-success errors.
|
||||
|
||||
Invalid nonempty configured endpoints are configuration failures. A missing or
|
||||
invalid final selected endpoint is an invalid generation request and is
|
||||
rejected before transport.
|
||||
|
||||
Authentication resolves a trimmed direct key before a trimmed configured
|
||||
environment value. Optional missing, empty, or whitespace-only sources do not
|
||||
block transport and produce no `Authorization` header. An explicitly required
|
||||
target with no usable source is rejected before transport with the existing
|
||||
invalid-request diagnostics.
|
||||
|
||||
Successful response bodies have a fixed 16 MiB limit enforced by declared
|
||||
length and by reading at most one byte beyond the boundary. The decoder accepts
|
||||
exactly one JSON object plus trailing whitespace and EOF. Size overflow,
|
||||
truncation, malformed JSON, trailing data, and a second value are malformed
|
||||
responses with no partial result or provider content in the error. Every body
|
||||
is closed, and an unbounded oversized stream is not drained. Non-success
|
||||
responses remain status-only; bounded provider error-envelope parsing belongs
|
||||
to the
|
||||
[structured-generation-error roadmap](../roadmap/structured-generation-errors.md).
|
||||
is closed, and an unbounded oversized stream is not drained.
|
||||
|
||||
For a non-success response, `ProviderHTTPError` retains the HTTP status and
|
||||
only normalized detail from the bounded recognized envelope. It retains
|
||||
`ErrUnexpectedStatus` through unwrapping. The client owns response closure;
|
||||
its bounded reader and parser never close or drain a body themselves. The root
|
||||
facade converts this concrete internal error into the public
|
||||
[`GenerationError`](../../generation_error.go), while arbitrary injected-client
|
||||
errors continue through the ordinary generation-error mapping unchanged.
|
||||
|
||||
An `http.Client.Do` failure is represented by a redacting multi-cause error:
|
||||
the package request-failure sentinel and the exact returned transport error are
|
||||
@@ -99,10 +113,14 @@ The
|
||||
own configuration, client cloning, deterministic deadline precedence,
|
||||
authentication, request and response mapping, malformed data, error identity,
|
||||
cancellation, endpoint selection and composition, pre-transport rejection, and
|
||||
bounded single-document response framing, closure, and response-body
|
||||
suppression. The root
|
||||
transport contract tests also verify that resolved backend settings reach this
|
||||
client without serializing backend identity and that ordinary-run cancellation
|
||||
retains its public generation and context identities. All use local test
|
||||
servers or controlled test transports; the default suite makes no live or paid
|
||||
provider requests.
|
||||
bounded single-document successful-response framing, closure, and
|
||||
response-body suppression. The focused
|
||||
[provider HTTP error tests](../../internal/llm/provider_http_error_test.go)
|
||||
own envelope parsing, normalization, and bounded-reader cases; their
|
||||
[transport tests](../../internal/llm/provider_http_error_transport_test.go)
|
||||
own non-success response closure and integration. Root transport contract tests
|
||||
own public `GenerationError` conversion, while also verifying that resolved
|
||||
backend settings reach this client without serializing backend identity and
|
||||
that ordinary-run cancellation retains its public generation and context
|
||||
identities. All use local test servers or controlled test transports; the
|
||||
default suite makes no live or paid provider requests.
|
||||
|
||||
@@ -11,10 +11,10 @@ contributor workflow and validation.
|
||||
|
||||
| Component | Implemented responsibility | References |
|
||||
| --- | --- | --- |
|
||||
| Root `promptkit` package | Provides the supported engine facade, source, backend-registration, and injection options, public request, result, prompt-inspection, and profile-inspection values, opaque prepared-execution handles, profile construction, extension interfaces, value conversion, redacted formatting, typed capacity errors, public error mapping, and engine-local profile-source assembly including application fallbacks. | [Package GoDoc](../../doc.go), [prepared execution](../../prepared_execution.go), [backend API](../../backends.go), [engine assembly](../../engine.go) |
|
||||
| Root `promptkit` package | Provides the supported engine facade, source, backend-registration, and injection options, public request, result, prompt-inspection, and profile-inspection values, opaque prepared-execution handles, profile construction, extension interfaces, value conversion, redacted formatting, typed capacity and generation error mapping, and engine-local profile-source assembly including application fallbacks. | [Package GoDoc](../../doc.go), [prepared execution](../../prepared_execution.go), [backend API](../../backends.go), [engine assembly](../../engine.go) |
|
||||
| `examples/go-library/prepare` | Demonstrates an offline downstream consumer using a prompt file, in-memory profile, inline input, and `Prepare`. It is not a public library package. | [Example program](../../examples/go-library/prepare/main.go) |
|
||||
| `examples/go-library/run` | Demonstrates an offline downstream consumer using a prompt file, in-memory profile, inline input, an injected deterministic model client, and `Run`. It is not a public library package. | [Example program](../../examples/go-library/run/main.go) |
|
||||
| `internal/backend` | Constructs each engine's immutable registry from the built-in OpenRouter definition and consumer additions, validates and defensively copies definitions through the shared JSON-value package, and consumes the LLM-owned OpenAI-compatible reserved request-field rule. | [Backend registry](../../internal/backend/registry.go) |
|
||||
| `internal/backend` | Constructs each engine's immutable registry from the maintained built-in definitions and consumer additions, validates and defensively copies definitions through the shared JSON-value package, and consumes the LLM-owned OpenAI-compatible reserved request-field rule. | [Backend registry](../../internal/backend/registry.go) |
|
||||
| `internal/capacity` | Owns engine-local bounded execution admission and FIFO model-generation permits for limited backend IDs, including cancellation-safe waiter removal and client wrapping. | [Internal capacity management](capacity.md) |
|
||||
| `internal/domain` | Defines internal framework values for requests, artifacts, prompt definitions, profiles, execution targets, rendering, generation, and validation, and owns source-neutral invariants for shared execution settings, OpenAI-compatible base endpoints, session identifiers, and output contracts. Source parsing, required fields, other source-specific normalization, defaulting, and boundary-specific error classification remain with their callers. | [Domain declarations](../../internal/domain/domain.go), [endpoint invariant](../../internal/domain/endpoint.go) |
|
||||
| `internal/defaults` | Defines application-neutral framework constants and constructs the default execution target. It contains no CLI, server, or inbound HTTP limits. | [Framework defaults](../../internal/defaults/defaults.go) |
|
||||
@@ -22,11 +22,11 @@ contributor workflow and validation.
|
||||
| `internal/jsonvalue` | Validates and deeply copies bounded JSON-compatible extra-parameter and prepared-schema trees while preserving supported concrete value types and rejecting cycles or excessive depth and work. | [JSON values](../../internal/jsonvalue/jsonvalue.go) |
|
||||
| `internal/promptdef` | Loads strictly decoded, validated prompt definitions from filesystem and `fs.FS` sources, including version selection and contained file-backed message content. | [Framework formats](../formats.md), [prompt-definition repository](../../internal/promptdef/filesystem_repository.go) |
|
||||
| `internal/profile` | Loads strictly decoded, validated execution profiles, including backend selection, from filesystem and `fs.FS` sources and composes repositories with error-preserving fallback. | [Framework formats](../formats.md), [profile repositories](../../internal/profile/filesystem_repository.go) |
|
||||
| `internal/profile/builtin` | Embeds the built-in profile catalog, whose entries select OpenRouter. | [Built-in catalog](../formats.md#built-in-profile-catalog), [repository](../../internal/profile/builtin/repository.go) |
|
||||
| `internal/profile/builtin` | Embeds the built-in profile catalog, whose entries select maintained built-in backends. | [Built-in catalog](../formats.md#built-in-profile-catalog), [repository](../../internal/profile/builtin/repository.go) |
|
||||
| `internal/prompt` | Renders prompt messages from Go templates with artifact, variable, session, and cache-control data. | [Go-template renderer](../../internal/prompt/go_renderer.go) |
|
||||
| `internal/artifact` | Resolves ordinary inline and unrestricted caller-selected file references into copied artifacts with metadata and hashes. | [Internal sources and validation](sources.md) |
|
||||
| `internal/validate` | Validates basic, JSON, and JSON Schema output using operating-system filesystem or `fs.FS` schema sources and creates operation-local validation plans with canonical contained schema resources. | [Framework formats](../formats.md#schemas), [internal sources and validation](sources.md) |
|
||||
| `internal/llm` | Defines the internal generation boundary and implements outbound OpenAI-compatible chat requests from resolved execution targets, including response decoding, authentication, deadline handling, and ownership of the OpenAI-compatible reserved request-field policy. | [Internal model client](llm.md) |
|
||||
| `internal/llm` | Defines the internal generation boundary and implements outbound OpenAI-compatible chat requests from resolved execution targets, including bounded structured non-success response decoding, successful-response decoding, authentication, deadline handling, and ownership of the OpenAI-compatible reserved request-field policy. | [Internal model client](llm.md) |
|
||||
| `internal/usecase` | Resolves prompt definitions and hashes, profiles, backends, and targets for exact inspection and request settings for preparation, and coordinates ordinary execution and one-attempt prepared execution across internal sources, rendering, artifact loading, operation-local validation plans, generation, capacity, and optional repair. | [Internal runner](runner.md), [prepared-execution implementation](../../internal/usecase/prepared_execution.go) |
|
||||
|
||||
The root package assembles these internal components without exposing their
|
||||
|
||||
@@ -69,12 +69,13 @@ profile sources and checks the resolved target without reading prompt, input,
|
||||
or schema sources. It does not retain that lookup for a later execution.
|
||||
|
||||
`internal/profile/builtin` embeds the maintained built-in profile catalog.
|
||||
Every embedded profile selects `openrouter` and inherits its endpoint and
|
||||
credential environment-variable name from the built-in backend registry rather
|
||||
than repeating those values. Profile loading and overlay behavior are owned by
|
||||
the [profile repository tests](../../internal/profile/repository_test.go),
|
||||
while catalog completeness, the backend-selection invariant, and duplicate IDs
|
||||
are owned by the
|
||||
Every embedded profile selects a maintained built-in backend and inherits that
|
||||
backend's endpoint and credential environment-variable name from the built-in
|
||||
backend registry rather than repeating those values. Profile loading and
|
||||
overlay behavior are owned by the
|
||||
[profile repository tests](../../internal/profile/repository_test.go), while
|
||||
catalog completeness, the backend-selection invariant, and duplicate IDs are
|
||||
owned by the
|
||||
[built-in repository tests](../../internal/profile/builtin/repository_test.go).
|
||||
|
||||
## Ordinary Artifacts
|
||||
|
||||
@@ -22,7 +22,7 @@ The implemented internal components consist of:
|
||||
- `internal/domain`, which owns framework data values and source-neutral
|
||||
invariants shared by later internal components;
|
||||
- `internal/backend`, which owns validated immutable OpenAI-compatible backend
|
||||
definitions and the built-in OpenRouter definition;
|
||||
definitions and the maintained built-in definitions;
|
||||
- `internal/capacity`, which owns engine-local bounded run admission and
|
||||
model-generation scheduling for limited backends;
|
||||
- `internal/defaults`, which owns application-neutral framework defaults and
|
||||
|
||||
332
docs/roadmap/implementation.md
Normal file
332
docs/roadmap/implementation.md
Normal file
@@ -0,0 +1,332 @@
|
||||
# Optional API-Key Environment Implementation Plan
|
||||
|
||||
## Purpose
|
||||
|
||||
Change Promptkit's credential handling so an effective `APIKeyEnv` names an
|
||||
optional credential source rather than implicitly requiring that environment
|
||||
variable to be populated. When neither a direct key nor a populated optional
|
||||
environment variable is available, the built-in OpenAI-compatible client must
|
||||
omit the `Authorization` header and let the upstream provider accept or reject
|
||||
the unauthenticated request.
|
||||
|
||||
This document owns implementation sequencing for that change. Follow the
|
||||
architecture, documentation, and testing policies under [`docs/policy/`](../policy/)
|
||||
and the task-specific reading guide in [`docs/development.md`](../development.md)
|
||||
throughout the work.
|
||||
|
||||
## Target Outcome
|
||||
|
||||
For ordinary and prepared execution:
|
||||
|
||||
1. a nonblank direct `RunRequest.APIKey` remains the highest-precedence
|
||||
credential and produces `Authorization: Bearer <key>`;
|
||||
2. otherwise, a nonblank effective `APIKeyEnv` is read when generation begins;
|
||||
3. a nonblank trimmed environment value produces the bearer header;
|
||||
4. an absent, empty, or whitespace-only optional environment value produces no
|
||||
`Authorization` header and does not prevent preparation or execution; and
|
||||
5. the provider response follows the normal success or error path, including
|
||||
public `GenerationError` conversion for a non-2xx response.
|
||||
|
||||
An explicitly required credential retains local validation. `APIKeyRequired`
|
||||
continues to require either a nonblank direct key or a populated environment
|
||||
variable explicitly selected by the request. Missing required credentials
|
||||
fail before provider transport.
|
||||
|
||||
## Fixed Decisions
|
||||
|
||||
- `APIKeyEnv` is a lookup location, not an assertion that authentication is
|
||||
required. This applies whether the name comes from a built-in backend, a
|
||||
consumer backend, a file profile, or a request override.
|
||||
- An environment value is usable only after `strings.TrimSpace`; unset, empty,
|
||||
and whitespace-only values are equivalent and cause header omission when
|
||||
optional.
|
||||
- Direct keys retain precedence. When a direct key is nonblank, Promptkit does
|
||||
not need the environment value and must not let a missing environment value
|
||||
block the request.
|
||||
- `APIKeyRequired` remains the only existing explicit local requirement flag.
|
||||
Do not add a backend field, YAML field, request field, or new public API.
|
||||
- With `APIKeyRequired` true:
|
||||
- a nonblank direct key satisfies the requirement;
|
||||
- an explicitly selected nonblank `APIKeyEnv` satisfies it only when that
|
||||
variable currently contains a nonblank value;
|
||||
- no selected environment name retains the existing required-key failure;
|
||||
and
|
||||
- a selected but unset environment retains `ErrAPIKeyEnvMissing` and
|
||||
`ErrInvalidRequest` at the public boundary.
|
||||
- `ErrAPIKeyEnvMissing` remains exported for compatibility but is narrowed to
|
||||
a missing environment credential in an explicitly required flow. Optional
|
||||
missing environment values do not return it.
|
||||
- Optional credential availability is not frozen during preparation. The
|
||||
built-in client reads the selected environment variable for each generation,
|
||||
including repair generation. Prepared execution therefore uses the value
|
||||
visible when it runs.
|
||||
- Required prepared execution continues to recheck credential availability
|
||||
before admission. Optional prepared execution proceeds even if the variable
|
||||
becomes unset after preparation.
|
||||
- The built-in transport sets no `Authorization` header at all when no usable
|
||||
credential exists. It must not send `Bearer ` with an empty value.
|
||||
- Injected `LLMClient` behavior is not redefined. Promptkit stops rejecting an
|
||||
optional missing environment before the injected client is called and
|
||||
continues to pass the effective target through the public adapter without
|
||||
resolving or exposing a secret on the client's behalf.
|
||||
- Upstream authentication policy remains upstream. Promptkit adds no provider-
|
||||
specific authentication rules, retry behavior, status mapping, or special
|
||||
handling beyond the existing structured non-2xx error path.
|
||||
|
||||
## Execution Rules
|
||||
|
||||
- Complete the stages in numerical order. Each stage is scoped for one coding-
|
||||
agent prompt and must finish its focused tests before the next stage begins.
|
||||
- Treat all stages as one behavior change. Intermediate stages intentionally
|
||||
leave the use-case and transport policies temporarily different; do not tag,
|
||||
release, or claim the change is complete until Stage 3 passes.
|
||||
- At the start of each stage, inspect the working tree and preserve all
|
||||
unrelated changes. In particular, merge rather than overwrite any existing
|
||||
edits in files touched by this plan.
|
||||
- Use controlled transports, local servers, and `t.Setenv`; never contact a
|
||||
live provider or depend on the developer machine's credentials.
|
||||
- Update existing tests whose asserted policy is changing instead of retaining
|
||||
contradictory tests or adding duplicate coverage under new names.
|
||||
- Do not add dependencies, commit, tag, push, or edit release documentation
|
||||
unless separately instructed.
|
||||
- Current-state prose documentation changes only in Stage 3, after the runtime
|
||||
behavior exists. GoDoc changes alongside the public contract in that stage.
|
||||
|
||||
## Stage 1: Make Use-Case Credential Validation Requirement-Aware
|
||||
|
||||
### Objective
|
||||
|
||||
Stop preparation and execution orchestration from treating every named
|
||||
environment variable as required, while preserving the explicit
|
||||
`APIKeyRequired` and prepared-execution contracts.
|
||||
|
||||
### Implementation
|
||||
|
||||
1. Update `validateAPIKey` in `internal/usecase/runner.go` with this policy:
|
||||
- return success immediately for a nonblank direct key;
|
||||
- return success when `apiKeyRequired` is false, regardless of whether
|
||||
`apiKeyEnv` is blank, populated, or missing;
|
||||
- when `apiKeyRequired` is true and the trimmed environment name is blank,
|
||||
return the existing `ErrAPIKeyRequired`;
|
||||
- when `apiKeyRequired` is true and the selected environment variable is
|
||||
unset, empty, or whitespace-only, return the existing wrapped
|
||||
`ErrAPIKeyEnvMissing` diagnostic naming the variable; and
|
||||
- otherwise return success.
|
||||
2. Keep the calls from preparation and `RunPrepared` in place. Do not move
|
||||
environment lookup into backend/profile resolution, admission, the domain
|
||||
model, or the root facade.
|
||||
3. Do not clear `APIKeyEnv` from effective or prepared metadata merely because
|
||||
its current value is absent. The name remains the configured lookup source;
|
||||
the secret value remains excluded.
|
||||
4. Preserve credential-source precedence and `APIKeyRequired` merge behavior.
|
||||
A request environment override remains capable of satisfying an in-memory
|
||||
profile's explicit requirement when its value is populated.
|
||||
|
||||
### Tests
|
||||
|
||||
Update the narrow `internal/usecase` behavior owners rather than adding tests
|
||||
for the private helper itself:
|
||||
|
||||
1. Replace `TestRunnerRunAPIKeyEnvMissingEnvironmentValueFailsClearly` with a
|
||||
test proving an unset optional profile `APIKeyEnv` reaches the injected
|
||||
model client and completes successfully. Explicitly set the named variable
|
||||
to an empty value so the test is independent of the process environment.
|
||||
2. Preserve the existing populated-environment and direct-key-precedence tests.
|
||||
3. Preserve `TestRunnerPrepareAPIKeyRequiredFailsWithoutDirectKey` and the
|
||||
existing direct-key success coverage.
|
||||
4. Revise the prepared credential test matrix to prove both distinct cases:
|
||||
- an optional environment value that disappears after
|
||||
`PrepareExecution` does not prevent admission and generation; and
|
||||
- an `APIKeyRequired` profile using an explicit request `APIKeyEnv` override
|
||||
still fails with `ErrInvalidRequest` and `ErrAPIKeyEnvMissing` if that
|
||||
value disappears before `RunPrepared`, before admission or generation.
|
||||
5. Keep prepared handles single-use in both outcomes and avoid duplicating
|
||||
unrelated preparation, validation, or admission assertions.
|
||||
|
||||
### Verification
|
||||
|
||||
```sh
|
||||
go test ./internal/usecase -run 'APIKey|Credential|Prepared'
|
||||
go test ./internal/usecase
|
||||
go test ./...
|
||||
```
|
||||
|
||||
Stage 1 is complete when optional missing environment values no longer block
|
||||
the use-case layer and explicit requirements retain their prior identities and
|
||||
timing.
|
||||
|
||||
**Status:** Complete.
|
||||
|
||||
## Stage 2: Omit Authentication in the Built-In Client When Optional Keys Are Missing
|
||||
|
||||
### Objective
|
||||
|
||||
Make the OpenAI-compatible transport implement the optional lookup contract at
|
||||
the boundary that owns the outbound `Authorization` header.
|
||||
|
||||
### Implementation
|
||||
|
||||
1. Refactor the authentication branch in
|
||||
`internal/llm/openai_compatible_client.go` so it resolves one credential in
|
||||
this order:
|
||||
- trimmed `req.Target.APIKey`; then
|
||||
- the trimmed value of the environment variable named by the trimmed
|
||||
`req.Target.APIKeyEnv`.
|
||||
2. Set `Authorization` to `Bearer <credential>` only when the resolved value is
|
||||
nonblank. If the direct key is blank and the optional environment lookup is
|
||||
absent, empty, or whitespace-only, leave the header unset and continue to
|
||||
`http.Client.Do`.
|
||||
3. Retain defensive enforcement for direct internal callers of the transport:
|
||||
- if `req.Target.APIKeyRequired` is true and no usable source exists, return
|
||||
`ErrInvalidRequest` before transport;
|
||||
- use the existing missing-environment diagnostic when a nonblank
|
||||
environment name was selected; and
|
||||
- use a concise required-key diagnostic when no environment name exists.
|
||||
Do not add an internal or public error type for this branch.
|
||||
4. Preserve request construction, endpoint validation, timeout behavior,
|
||||
response-body ownership, and direct-key redaction. Never serialize
|
||||
`APIKeyEnv`, `APIKeyRequired`, or a credential into the JSON request body.
|
||||
5. Allow an unauthenticated upstream non-2xx response to flow through the
|
||||
existing `ProviderHTTPError` and public `GenerationError` machinery without
|
||||
adding a credential-specific translation.
|
||||
|
||||
### Tests
|
||||
|
||||
Update `TestOpenAICompatibleClientAuthentication` in
|
||||
`internal/llm/openai_compatible_client_test.go` as the transport contract owner.
|
||||
Its table should cover only the meaningful credential states:
|
||||
|
||||
- a direct key takes precedence over a populated environment value;
|
||||
- a populated environment value supplies the bearer header when no direct key
|
||||
is present;
|
||||
- no configured source sends the request without `Authorization`;
|
||||
- a named but unset optional environment sends the request without
|
||||
`Authorization`;
|
||||
- a named but unset required environment fails with `ErrInvalidRequest` before
|
||||
transport; and
|
||||
- a required target with no source also fails before transport.
|
||||
|
||||
Use exact header assertions because presence or absence of `Authorization` is
|
||||
the wire contract. Do not repeat endpoint, body, timeout, or response-error
|
||||
matrices in this test.
|
||||
|
||||
### Verification
|
||||
|
||||
```sh
|
||||
go test ./internal/llm -run TestOpenAICompatibleClientAuthentication
|
||||
go test ./internal/llm
|
||||
go test ./...
|
||||
```
|
||||
|
||||
Stage 2 is complete when the built-in transport sends unauthenticated requests
|
||||
for optional missing sources, still enforces explicit requirements, and all
|
||||
existing provider-response behavior remains green.
|
||||
|
||||
**Status:** Complete.
|
||||
|
||||
## Stage 3: Align the Public Contract, Durable Documentation, and Full Validation
|
||||
|
||||
### Objective
|
||||
|
||||
Prove the assembled consumer workflow, update canonical contract owners, and
|
||||
complete repository-wide validation.
|
||||
|
||||
### Public Contract And Tests
|
||||
|
||||
1. Update the GoDoc for `ErrAPIKeyEnvMissing` in `engine.go` so it applies only
|
||||
when an explicitly required credential names an unset or empty environment
|
||||
variable. Retain the exported value and its `ErrInvalidRequest` identity.
|
||||
2. Update credential GoDoc where the exact public semantics are exposed:
|
||||
- `Backend.APIKeyEnv` in `backends.go`;
|
||||
- `ExecutionTarget.APIKeyEnv` and
|
||||
`ExecutionTargetOverride.APIKeyEnv` in `types.go`; and
|
||||
- any nearby `APIKeyRequired` wording that would otherwise imply every
|
||||
named environment source is mandatory.
|
||||
State that optional missing values cause the built-in client to omit
|
||||
`Authorization`; do not promise that injected clients resolve environment
|
||||
variables identically.
|
||||
3. Replace `TestMissingCredentialsFailClearlyWhenProfileRequiresAuth` in
|
||||
`engine_test.go`, whose old assertion is intentionally obsolete, with one
|
||||
focused external-package contract test. Assemble a real engine with a
|
||||
controlled HTTP client or local server, select a backend or file profile
|
||||
whose `APIKeyEnv` is explicitly empty in the process environment, and have
|
||||
the controlled upstream return a structured authentication failure. Assert:
|
||||
- the request reaches upstream;
|
||||
- `Authorization` is absent;
|
||||
- the result is nil;
|
||||
- the error does not match `ErrInvalidRequest` or
|
||||
`ErrAPIKeyEnvMissing`; and
|
||||
- the error matches `ErrLLMGenerate` and is discoverable as a
|
||||
`*GenerationError` with the upstream status and representative provider
|
||||
detail.
|
||||
4. Keep this root test representative. The transport authentication table owns
|
||||
the full credential matrix, and the existing generation-error tests own
|
||||
envelope parsing, bounds, and formatting.
|
||||
|
||||
### Documentation
|
||||
|
||||
Update each canonical owner only for its topic:
|
||||
|
||||
1. In `docs/formats.md`, define profile, backend, and request `api_key_env`
|
||||
values as optional lookup sources. In the credentials section, distinguish
|
||||
them from `APIKeyRequired`, document header omission for a missing optional
|
||||
value, retain direct-key precedence, and state that required availability is
|
||||
validated during preparation and rechecked for prepared execution.
|
||||
2. In `docs/integrations/openai-compatible-chat.md`, document the outbound
|
||||
authentication wire behavior: a bearer header is sent only for a usable
|
||||
direct or environment credential; otherwise the header is omitted and the
|
||||
provider response is handled normally.
|
||||
3. In `docs/internal/llm.md`, update the internal flow and failure categories to
|
||||
distinguish optional omission from explicit required-key rejection. Preserve
|
||||
any unrelated edits already present in this file.
|
||||
4. In `docs/consumers/pkg-promptkit.md`, clarify near profile inspection or
|
||||
credential guidance that a reported `APIKeyEnv` is a configured optional
|
||||
source, while `APIKeyRequired` is the explicit local requirement. Link to
|
||||
the exact public GoDoc or format reference rather than reproducing the full
|
||||
precedence contract.
|
||||
5. Do not change architecture policy, backend IDs/default names, release notes,
|
||||
the README, or examples unless implementation uncovers a concrete
|
||||
inaccurate current-state statement in one of those owners.
|
||||
|
||||
### Final Validation
|
||||
|
||||
Run the complete maintainer workflow from
|
||||
[`docs/development.md#maintainer-validation`](../development.md#maintainer-validation):
|
||||
|
||||
```sh
|
||||
go test ./...
|
||||
go test -race ./...
|
||||
go vet ./...
|
||||
go build ./...
|
||||
go run ./examples/go-library/prepare
|
||||
go run ./examples/go-library/run
|
||||
```
|
||||
|
||||
Also run the documented tracked-Go formatting check, local Markdown-link
|
||||
validation, repository-hygiene checks, ignored-file check, credential scan, and
|
||||
`git diff --check`. Review the complete diff and confirm that:
|
||||
|
||||
- optional absent, empty, and whitespace-only environment values omit
|
||||
authentication and reach transport;
|
||||
- direct and populated environment credentials still produce the correct
|
||||
bearer header;
|
||||
- explicit required flows still fail locally with the intended identities;
|
||||
- ordinary, prepared, built-in, injected-client, and repair paths follow their
|
||||
stated ownership boundaries;
|
||||
- upstream non-2xx responses remain structured generation errors rather than
|
||||
local credential errors;
|
||||
- no credential value is serialized, retained in public metadata, or exposed
|
||||
by formatting; and
|
||||
- durable documentation describes only the now-implemented behavior with one
|
||||
canonical owner per exact contract.
|
||||
|
||||
Stage 3 is complete when the public workflow, documentation, and full
|
||||
maintainer validation all match the target outcome.
|
||||
|
||||
**Status:** Complete.
|
||||
|
||||
## Open Questions
|
||||
|
||||
None. Optional environment lookup, explicit requirement behavior, precedence,
|
||||
prepared-execution timing, outbound header semantics, error compatibility,
|
||||
documentation ownership, and test boundaries are fixed by this plan.
|
||||
@@ -5,67 +5,256 @@
|
||||
Promptkit should give downstream applications actionable, machine-readable
|
||||
details when the built-in OpenAI-compatible client receives a non-success HTTP
|
||||
response. Today the client reports only the status code and discards the
|
||||
provider response body. This makes ordinary configuration failures—such as an
|
||||
unsupported strict JSON Schema keyword—unnecessarily difficult to diagnose.
|
||||
provider response body. This makes ordinary configuration failures, such as an
|
||||
unsupported strict JSON Schema keyword, unnecessarily difficult to diagnose.
|
||||
|
||||
This feature supplies bounded facts about the provider response. It does not
|
||||
make retry, presentation, or logging decisions for consumers.
|
||||
|
||||
## Target End State
|
||||
|
||||
Failures from the built-in transport are available through a public typed error
|
||||
that works with `errors.As` while continuing to match `ErrLLMGenerate` through
|
||||
`errors.Is`. The error should expose:
|
||||
Every non-2xx response received by Promptkit's built-in OpenAI-compatible
|
||||
client becomes a public typed generation error. A consumer can use
|
||||
`errors.As` to obtain the HTTP status and any safely extracted provider fields,
|
||||
and `errors.Is` continues to match `ErrLLMGenerate`.
|
||||
|
||||
- the HTTP status code;
|
||||
- a normalized provider error code or type when supplied; and
|
||||
- a bounded provider message extracted from a recognized OpenAI-compatible
|
||||
JSON error envelope.
|
||||
The typed contract is available from both `Run` and `RunPrepared`. It is not
|
||||
produced during preparation, which performs no model request. Successful
|
||||
responses, transport failures before a response is received, cancellation,
|
||||
capacity failures, validation failures, and nil responses from injected model
|
||||
clients retain their existing categories and behavior.
|
||||
|
||||
The ordinary `Error()` string should remain safe and concise: it should include
|
||||
the status and provider code or type, but not automatically include the
|
||||
provider message. Consumers that deliberately want the provider's diagnostic
|
||||
text can retrieve it from the typed error and apply their own disclosure and
|
||||
logging policy.
|
||||
An unusable response body never hides the known HTTP status. Empty, malformed,
|
||||
unrecognized, unreadable, or oversized bodies therefore produce the same typed
|
||||
error with status-only detail rather than falling back to an unstructured
|
||||
error or becoming a malformed-success response.
|
||||
|
||||
This contract should be available for both ordinary and prepared execution.
|
||||
Errors returned by injected model clients must continue to preserve their own
|
||||
identity and should not be converted into fabricated HTTP details.
|
||||
## Public Contract
|
||||
|
||||
## Safety And Compatibility Boundaries
|
||||
The root package exposes an immutable `GenerationError` type with unexported
|
||||
state and these read-only accessors:
|
||||
|
||||
- Never expose the raw response body, response headers, endpoint, credentials,
|
||||
request messages, schema document, or generated content through this API.
|
||||
- Read only a small fixed maximum response body, reject malformed or
|
||||
unrecognized envelopes, normalize invalid UTF-8 and control characters, and
|
||||
cap every retained diagnostic field independently.
|
||||
- Treat the extracted provider message as untrusted and potentially sensitive:
|
||||
its GoDoc must tell consumers not to log or display it without applying their
|
||||
own policy.
|
||||
- Preserve the existing generic behavior when a response is empty, non-JSON,
|
||||
oversized, or does not match a recognized error envelope.
|
||||
- Do not assign retryability from an HTTP status. Promptkit supplies facts;
|
||||
downstream applications retain retry and presentation policy.
|
||||
- `StatusCode() int` returns the received HTTP status code;
|
||||
- `ProviderCode() string` returns a normalized provider code, when present;
|
||||
- `ProviderType() string` returns a normalized provider error type, when
|
||||
present; and
|
||||
- `ProviderMessage() string` returns the bounded normalized diagnostic message,
|
||||
when present.
|
||||
|
||||
## Recommended API Direction
|
||||
The engine returns a `*GenerationError`, so the idiomatic inspection form is:
|
||||
|
||||
Prefer one immutable public `GenerationError` value, constructed internally and
|
||||
carrying accessors for HTTP status, provider code or type, and provider message.
|
||||
This keeps the exact representation evolvable while giving consumers an
|
||||
idiomatic `errors.As` contract. Public Go declarations and GoDoc should own the
|
||||
final exact names and semantics.
|
||||
```go
|
||||
var generationErr *promptkit.GenerationError
|
||||
if errors.As(err, &generationErr) {
|
||||
status := generationErr.StatusCode()
|
||||
message := generationErr.ProviderMessage()
|
||||
_, _ = status, message
|
||||
}
|
||||
```
|
||||
|
||||
The internal OpenAI-compatible client should parse only the conventional
|
||||
top-level `error` envelope and pass normalized details through the use-case and
|
||||
public error-mapping layers. The integration documentation should continue to
|
||||
own wire behavior; the public declarations should own the consumer contract.
|
||||
There is no public constructor or mutation API. The type implements `error`,
|
||||
unwraps to `ErrLLMGenerate`, and provides safe ordinary and Go-syntax
|
||||
formatting. `Error()` and `GoString()` include the HTTP status but no provider-
|
||||
controlled code, type, or message. Consumers must use the accessors
|
||||
deliberately when they want provider details and must not classify failures by
|
||||
matching error text.
|
||||
|
||||
The zero value and a nil `*GenerationError` receiver are safe: accessors return
|
||||
zero or empty values, formatting returns a generic redacted generation-failure
|
||||
description, and unwrapping still identifies `ErrLLMGenerate`. Engine-produced
|
||||
values always have the non-2xx status received from the provider. The type has
|
||||
no stable JSON representation.
|
||||
|
||||
All provider-derived strings remain untrusted even after normalization. GoDoc
|
||||
must warn consumers that provider fields can contain sensitive request or
|
||||
schema fragments and must not be logged, displayed, or returned to another
|
||||
caller without an application-appropriate disclosure policy.
|
||||
|
||||
## Recognized Provider Envelope
|
||||
|
||||
Promptkit recognizes only the conventional OpenAI-compatible top-level error
|
||||
object:
|
||||
|
||||
```json
|
||||
{
|
||||
"error": {
|
||||
"message": "diagnostic text",
|
||||
"type": "invalid_request_error",
|
||||
"code": "unsupported_parameter"
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
The envelope must be one JSON document followed only by JSON whitespace. The
|
||||
top-level `error` value must be an object. Unknown top-level and error-object
|
||||
fields are ignored. The optional supported fields are interpreted
|
||||
independently:
|
||||
|
||||
- `message` and `type` must be JSON strings;
|
||||
- `code` may be a JSON string or number and is exposed as normalized text;
|
||||
numeric codes retain their validated JSON number text without floating-point
|
||||
coercion; and
|
||||
- `null`, booleans, arrays, objects, or otherwise invalid values are treated as
|
||||
absent for that field.
|
||||
|
||||
An invalid optional field does not discard other valid supported fields. An
|
||||
absent `error` object, malformed or multiply framed JSON, or an object with no
|
||||
usable supported fields simply leaves all provider accessors empty while
|
||||
preserving the typed status error.
|
||||
|
||||
Promptkit does not expose `param`, metadata objects, nested causes, headers, or
|
||||
provider-specific extensions in this feature.
|
||||
|
||||
## Bounded Reading And Normalization
|
||||
|
||||
Non-success bodies have a separate fixed limit of 64 KiB (65,536 bytes). This
|
||||
is intentionally much smaller than the successful completion-body limit while
|
||||
remaining large enough for useful schema diagnostics.
|
||||
|
||||
- A declared `Content-Length` above the limit is rejected without reading the
|
||||
body for detail extraction.
|
||||
- Otherwise Promptkit reads at most one byte beyond the limit so streamed,
|
||||
chunked, and underreported bodies are bounded.
|
||||
- A body over the limit contributes no provider fields; Promptkit does not
|
||||
parse or retain a prefix as though it were a complete envelope.
|
||||
- Read failures likewise discard provider fields while preserving the status.
|
||||
- The response body is closed on every outcome and is not drained beyond the
|
||||
bounded read.
|
||||
|
||||
Extracted strings are converted to valid UTF-8, trimmed, and made single-line:
|
||||
invalid UTF-8 is replaced, and runs of Unicode whitespace, control characters,
|
||||
and formatting controls are replaced with one ASCII space. Empty normalized
|
||||
values are treated as absent.
|
||||
|
||||
Normalized provider codes and types are retained only when they contain at
|
||||
most 256 Unicode code points. Longer values are omitted rather than truncated
|
||||
so consumers never classify on a fabricated partial identifier. A provider
|
||||
message is limited to 4,096 Unicode code points; a longer normalized message is
|
||||
truncated at a code-point boundary with a visible ellipsis inside that limit.
|
||||
The raw response body and pre-normalized strings are never exposed or retained
|
||||
in the public error.
|
||||
|
||||
## Error Propagation And Compatibility
|
||||
|
||||
- Every built-in-client non-2xx response matches `ErrLLMGenerate` and supports
|
||||
`errors.As` to `*GenerationError`, including status-only cases.
|
||||
- The internal model client retains its non-success-status identity for its
|
||||
own package tests. The use-case layer remains provider-neutral and continues
|
||||
to add only its generation category.
|
||||
- The root error boundary converts only the built-in transport's structured
|
||||
status error. It does not parse arbitrary error text, inspect consumer error
|
||||
fields, or fabricate HTTP details for an injected `LLMClient`.
|
||||
- Errors returned by injected clients remain in the chain exactly as today.
|
||||
If an injected client deliberately returns an existing `*GenerationError`,
|
||||
its identity may pass through ordinary wrapping, but Promptkit does not
|
||||
construct or enrich one on that client's behalf.
|
||||
- Existing cancellation and deadline identities, capacity errors, validation
|
||||
behavior, repair behavior, and successful response decoding remain
|
||||
unchanged.
|
||||
- This is an additive public API. Existing consumers that use
|
||||
`errors.Is(err, ErrLLMGenerate)` continue to work; consumers should not rely
|
||||
on the previous rendered wording of non-success errors.
|
||||
|
||||
## Architecture And Ownership
|
||||
|
||||
The provider-envelope parser and bounded body reader belong in `internal/llm`,
|
||||
which owns the OpenAI-compatible transport. The internal transport error owns
|
||||
only normalized status facts and continues to match the package's existing
|
||||
non-success-status sentinel.
|
||||
|
||||
The use-case package does not gain HTTP DTOs, status policy, or a provider-
|
||||
specific branch. Its existing wrapping carries the internal error to the root
|
||||
facade. The root error mapper recognizes the internal structured status error
|
||||
and constructs the public `GenerationError` without exposing an internal type
|
||||
or raw cause through public fields. No transport error is added to
|
||||
`internal/domain`.
|
||||
|
||||
The public type and its exact Go semantics are owned by its declaration and
|
||||
GoDoc. The
|
||||
[OpenAI-compatible integration contract](../integrations/openai-compatible-chat.md)
|
||||
owns recognized wire shapes, limits, and observable response behavior. The
|
||||
[internal model-client document](../internal/llm.md) owns implementation flow,
|
||||
internal failure categories, and test ownership. Architecture policy does not
|
||||
need a new package or dependency rule for this feature.
|
||||
|
||||
## Documentation End State
|
||||
|
||||
Canonical documentation at the target state has these responsibilities:
|
||||
|
||||
- the `GenerationError` declaration and GoDoc define the exact public methods,
|
||||
formatting, unwrapping, zero-value behavior, and trust boundary;
|
||||
- `Engine.Run` and `Engine.RunPrepared` GoDoc identify the typed error without
|
||||
duplicating its accessor contract;
|
||||
- the consumer guide includes one short `errors.As` example and links to the
|
||||
public declaration;
|
||||
- the integration document replaces its status-only description with the
|
||||
bounded envelope contract; and
|
||||
- the internal model-client document describes parsing, conversion ownership,
|
||||
and narrow test owners.
|
||||
|
||||
The architecture policy, framework format reference, and built-in backend
|
||||
catalog do not duplicate this API or wire contract.
|
||||
|
||||
## Verification Expectations
|
||||
|
||||
Verification protects each behavior at its narrowest stable owner:
|
||||
|
||||
- internal model-client tests cover recognized string and numeric codes,
|
||||
independent optional-field handling, unknown fields, empty and malformed
|
||||
envelopes, single-document framing, read failures, declared and streamed
|
||||
size boundaries, body closure, normalization, field limits, and absence of
|
||||
raw provider content from rendered errors;
|
||||
- root error-boundary tests cover conversion to the immutable public type,
|
||||
every accessor, `errors.Is`, `errors.As`, and safe `%v`, `%+v`, and `%#v`
|
||||
formatting;
|
||||
- one representative ordinary run and one prepared run prove that the built-in
|
||||
transport contract crosses the assembled engine boundary, without repeating
|
||||
the complete parser matrix;
|
||||
- existing injected-client tests continue to prove preservation of consumer
|
||||
error identity without fabricated provider details; and
|
||||
- all tests use controlled transports or local servers and never contact a
|
||||
live or paid provider.
|
||||
|
||||
Security limits and their exact boundaries are contractual enough to warrant
|
||||
literal boundary tests. Higher-level tests should remain representative and
|
||||
must not duplicate the internal transport matrix.
|
||||
|
||||
## Acceptance Criteria
|
||||
|
||||
- A downstream consumer can distinguish a provider HTTP 400 from other
|
||||
generation failures and obtain a bounded provider explanation when present.
|
||||
- The typed error still satisfies `errors.Is(err, ErrLLMGenerate)`.
|
||||
- Existing cancellation, capacity, validation, and injected-client error
|
||||
identities remain unchanged.
|
||||
- Tests cover recognized string and numeric provider codes, absent and malformed
|
||||
envelopes, oversized bodies and fields, control characters, and error-chain
|
||||
behavior without making live provider requests.
|
||||
- Current-state GoDoc and the OpenAI-compatible integration and internal-client
|
||||
documents are updated only when the implementation lands.
|
||||
- A consumer can distinguish an HTTP 400 from other generation failures and
|
||||
deliberately obtain a bounded provider explanation when one is available.
|
||||
- The same typed error remains available through ordinary and prepared
|
||||
execution and still satisfies `errors.Is(err, ErrLLMGenerate)`.
|
||||
- Default and Go-syntax error formatting cannot disclose any provider-derived
|
||||
string or raw response content.
|
||||
- Empty, malformed, unreadable, unrecognized, and oversized bodies preserve a
|
||||
typed status-only error.
|
||||
- No read, retained field, or formatted representation can exceed its stated
|
||||
bound, and the body is closed on every outcome.
|
||||
- Existing success, cancellation, capacity, validation, repair, and injected-
|
||||
client contracts remain unchanged.
|
||||
- Current-state documentation changes only when the implementation exists and
|
||||
follows the repository's canonical ownership policy.
|
||||
|
||||
## Non-Goals
|
||||
|
||||
This feature does not add:
|
||||
|
||||
- retryability classification, retry loops, backoff, failover, or routing;
|
||||
- parsing of success bodies as errors or changes to successful-response limits;
|
||||
- provider-specific envelope variants beyond the conventional top-level
|
||||
`error` object;
|
||||
- response headers such as `Retry-After`, raw bodies, request data, endpoints,
|
||||
credentials, schema documents, generated content, or provider metadata;
|
||||
- logging, telemetry, redaction policy for downstream applications, HTTP status
|
||||
mapping for consumer servers, or user-facing presentation;
|
||||
- translation or enrichment of arbitrary injected-client errors; or
|
||||
- a new public package, public constructor, mutable error value, or transport
|
||||
type in the domain model.
|
||||
|
||||
## Open Questions
|
||||
|
||||
None. The public type direction, accessor surface, formatting and error-chain
|
||||
behavior, envelope scope, normalization, safety limits, fallback behavior,
|
||||
layer ownership, compatibility boundaries, documentation ownership, and test
|
||||
boundaries are fixed by this roadmap.
|
||||
|
||||
30
engine.go
30
engine.go
@@ -53,9 +53,9 @@ var (
|
||||
// an execution profile or resolve its backend, except for the profile
|
||||
// not-found case represented by ErrProfileNotFound.
|
||||
ErrProfileLoad = errors.New("failed to load execution profile")
|
||||
// ErrAPIKeyEnvMissing identifies an APIKeyEnv whose environment variable is
|
||||
// unset or empty when no direct RunRequest.APIKey takes precedence. Such an
|
||||
// error also matches ErrInvalidRequest.
|
||||
// ErrAPIKeyEnvMissing identifies an explicitly required APIKeyEnv whose
|
||||
// environment variable is unset or empty after direct RunRequest.APIKey
|
||||
// precedence is applied. Such an error also matches ErrInvalidRequest.
|
||||
ErrAPIKeyEnvMissing = errors.New("api_key_env points to an unset environment variable")
|
||||
// ErrArtifactLoad identifies a failure to resolve an input artifact. Errors
|
||||
// returned by an injected ArtifactReader remain available through errors.Is.
|
||||
@@ -69,8 +69,9 @@ var (
|
||||
// request, an LLM or provider rate-limit response, or ErrLLMGenerate.
|
||||
ErrCapacityExceeded = errors.New("backend capacity exceeded")
|
||||
// ErrLLMGenerate identifies a model-client failure or a nil successful
|
||||
// response. Errors returned by an injected LLMClient remain available
|
||||
// through errors.Is.
|
||||
// response. A built-in OpenAI-compatible non-2xx response is available as a
|
||||
// [GenerationError]. Errors returned by an injected LLMClient remain
|
||||
// available through errors.Is.
|
||||
ErrLLMGenerate = errors.New("failed to generate output")
|
||||
// ErrValidation identifies an operational failure to load or compile a
|
||||
// schema or validate output. A completed validation whose Status is
|
||||
@@ -646,11 +647,13 @@ func (e *Engine) PrepareExecution(ctx context.Context, req RunRequest) (*Prepare
|
||||
// discoverable as [CapacityError] and still matches ErrCapacityExceeded. It
|
||||
// occurs before artifacts, schemas, rendering, or model generation because the
|
||||
// selected backend's admission capacity is full; it does not match
|
||||
// ErrInvalidRequest or ErrLLMGenerate. Errors from injected clients remain
|
||||
// available through errors.Is. Cancellation while waiting for model-generation
|
||||
// capacity matches both ErrLLMGenerate and the context error. Cancellation
|
||||
// otherwise follows the active collaborator's documented behavior. A nil
|
||||
// Engine returns ErrInvalidConfig. Run returns no partial result on error.
|
||||
// ErrInvalidRequest or ErrLLMGenerate. A built-in OpenAI-compatible non-2xx
|
||||
// response is discoverable as [GenerationError]. Errors from injected clients
|
||||
// remain available through errors.Is. Cancellation while waiting for
|
||||
// model-generation capacity matches both ErrLLMGenerate and the context error.
|
||||
// Cancellation otherwise follows the active collaborator's documented
|
||||
// behavior. A nil Engine returns ErrInvalidConfig. Run returns no partial
|
||||
// result on error.
|
||||
func (e *Engine) Run(ctx context.Context, req RunRequest) (*RunResult, error) {
|
||||
if e == nil || e.runner == nil {
|
||||
return nil, fmt.Errorf("%w: engine is nil", ErrInvalidConfig)
|
||||
@@ -688,9 +691,10 @@ func (e *Engine) Run(ctx context.Context, req RunRequest) (*RunResult, error) {
|
||||
// ErrCapacityExceeded, ErrLLMGenerate, or ErrValidation as applicable while
|
||||
// preserving documented collaborator and context identities. An engine
|
||||
// admission rejection is discoverable as [CapacityError] and still matches
|
||||
// ErrCapacityExceeded. A completed content-validation rejection is returned
|
||||
// in RunResult, not as an operational error. An operational error returns no
|
||||
// partial RunResult.
|
||||
// ErrCapacityExceeded. A built-in OpenAI-compatible non-2xx response is
|
||||
// discoverable as [GenerationError]. A completed content-validation rejection
|
||||
// is returned in RunResult, not as an operational error. An operational error
|
||||
// returns no partial RunResult.
|
||||
func (e *Engine) RunPrepared(ctx context.Context, prepared *PreparedExecution) (*RunResult, error) {
|
||||
if e == nil || e.runner == nil {
|
||||
return nil, fmt.Errorf("%w: engine is nil", ErrInvalidConfig)
|
||||
|
||||
168
engine_test.go
168
engine_test.go
@@ -977,37 +977,72 @@ func TestPrepareDirectAPIKeyBypassesMissingEnvWithoutLeakingOrHashing(t *testing
|
||||
}
|
||||
}
|
||||
|
||||
func TestMissingCredentialsFailClearlyWhenProfileRequiresAuth(t *testing.T) {
|
||||
func TestOptionalMissingCredentialsReachUpstream(t *testing.T) {
|
||||
const missingEnv = "PROMPTKIT_PUBLIC_AUTH_MISSING"
|
||||
const providerBody = `{"error":{"message":"authentication failed","type":"authentication_error","code":"invalid_api_key"}}`
|
||||
t.Setenv(missingEnv, "")
|
||||
|
||||
profileDir := t.TempDir()
|
||||
writePublicProfileFileWithAPIKeyEnv(t, profileDir, "requires-auth", "http://localhost:8000/v1", "test-model", missingEnv)
|
||||
engine, err := promptkit.NewEngine(promptkit.Config{
|
||||
PromptDir: frameworkPromptDir,
|
||||
ProfileDir: profileDir,
|
||||
SchemaDir: frameworkSchemaDir,
|
||||
})
|
||||
called := false
|
||||
config := promptkit.Config{
|
||||
PromptDir: frameworkPromptDir,
|
||||
SchemaDir: frameworkSchemaDir,
|
||||
HTTPClient: &http.Client{Transport: roundTripFunc(func(req *http.Request) (*http.Response, error) {
|
||||
called = true
|
||||
if values := req.Header.Values("Authorization"); len(values) != 0 {
|
||||
t.Fatalf("Authorization values = %q, want absent", values)
|
||||
}
|
||||
return &http.Response{
|
||||
StatusCode: http.StatusUnauthorized,
|
||||
ContentLength: int64(len(providerBody)),
|
||||
Body: io.NopCloser(strings.NewReader(providerBody)),
|
||||
}, nil
|
||||
})},
|
||||
}
|
||||
engine, err := promptkit.NewEngine(config,
|
||||
promptkit.WithBackend(promptkit.Backend{
|
||||
ID: "optional-auth",
|
||||
Endpoint: "http://provider.test/v1",
|
||||
APIKeyEnv: missingEnv,
|
||||
}),
|
||||
promptkit.WithProfiles(promptkit.Profile{
|
||||
ID: "optional-auth-profile",
|
||||
BackendID: "optional-auth",
|
||||
Model: "test-model",
|
||||
}),
|
||||
)
|
||||
if err != nil {
|
||||
t.Fatalf("expected engine construction to succeed, got %v", err)
|
||||
}
|
||||
|
||||
_, err = engine.Prepare(context.Background(), promptkit.RunRequest{
|
||||
result, err := engine.Run(context.Background(), promptkit.RunRequest{
|
||||
PromptID: frameworkMarkdownSummaryPromptID,
|
||||
ProfileID: "requires-auth",
|
||||
ProfileID: "optional-auth-profile",
|
||||
Inputs: map[string]promptkit.ArtifactRef{
|
||||
"transcript": promptkit.Inline("Rin opens the gate."),
|
||||
"glossary": promptkit.Inline("gate: A guarded passage."),
|
||||
},
|
||||
})
|
||||
if !errors.Is(err, promptkit.ErrInvalidRequest) {
|
||||
t.Fatalf("expected invalid request for missing credentials, got %v", err)
|
||||
if !called {
|
||||
t.Fatal("optional missing credential did not reach upstream")
|
||||
}
|
||||
if !errors.Is(err, promptkit.ErrAPIKeyEnvMissing) {
|
||||
t.Fatalf("expected missing credential environment error, got %v", err)
|
||||
if result != nil {
|
||||
t.Fatalf("result = %+v, want nil", result)
|
||||
}
|
||||
if err == nil || !strings.Contains(err.Error(), missingEnv) {
|
||||
t.Fatalf("expected missing env name in error, got %v", err)
|
||||
if errors.Is(err, promptkit.ErrInvalidRequest) || errors.Is(err, promptkit.ErrAPIKeyEnvMissing) {
|
||||
t.Fatalf("error = %v, want upstream generation error without credential identities", err)
|
||||
}
|
||||
if !errors.Is(err, promptkit.ErrLLMGenerate) {
|
||||
t.Fatalf("error = %v, want ErrLLMGenerate", err)
|
||||
}
|
||||
var generationErr *promptkit.GenerationError
|
||||
if !errors.As(err, &generationErr) {
|
||||
t.Fatalf("error = %v, want GenerationError", err)
|
||||
}
|
||||
if generationErr.StatusCode() != http.StatusUnauthorized ||
|
||||
generationErr.ProviderType() != "authentication_error" ||
|
||||
generationErr.ProviderCode() != "invalid_api_key" ||
|
||||
generationErr.ProviderMessage() != "authentication failed" {
|
||||
t.Fatalf("GenerationError = %+v, want structured upstream authentication failure", generationErr)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1164,8 +1199,9 @@ func TestArtifactReaderFailuresPreserveArtifactLoadErrors(t *testing.T) {
|
||||
}
|
||||
|
||||
func TestRunAddsLLMGenerateToCollaboratorPublicError(t *testing.T) {
|
||||
injectedErr := errors.New("injected model client failure")
|
||||
engine := newContractEngineWithOptions(t, frameworkSchemaDir,
|
||||
promptkit.WithLLMClient(&fakeLLMClient{err: promptkit.ErrArtifactLoad}),
|
||||
promptkit.WithLLMClient(&fakeLLMClient{err: injectedErr}),
|
||||
)
|
||||
|
||||
_, err := engine.Run(context.Background(), promptkit.RunRequest{
|
||||
@@ -1178,8 +1214,12 @@ func TestRunAddsLLMGenerateToCollaboratorPublicError(t *testing.T) {
|
||||
if !errors.Is(err, promptkit.ErrLLMGenerate) {
|
||||
t.Fatalf("expected ErrLLMGenerate, got %v", err)
|
||||
}
|
||||
if !errors.Is(err, promptkit.ErrArtifactLoad) {
|
||||
t.Fatalf("expected preserved ErrArtifactLoad, got %v", err)
|
||||
if !errors.Is(err, injectedErr) {
|
||||
t.Fatalf("expected preserved injected error, got %v", err)
|
||||
}
|
||||
var generationErr *promptkit.GenerationError
|
||||
if errors.As(err, &generationErr) {
|
||||
t.Fatalf("injected error became GenerationError: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1492,43 +1532,63 @@ func TestSelectedProfileRepositoryReadFailureMapsToProfileLoad(t *testing.T) {
|
||||
}
|
||||
|
||||
func TestPrepareUsesBuiltInProfileWithoutProfileDir(t *testing.T) {
|
||||
t.Setenv("OPENROUTER_API_KEY", "test-key")
|
||||
engine, err := promptkit.NewEngine(promptkit.Config{
|
||||
PromptDir: frameworkPromptDir,
|
||||
SchemaDir: frameworkSchemaDir,
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("expected engine construction to succeed, got %v", err)
|
||||
tests := []struct {
|
||||
name string
|
||||
profileID string
|
||||
backendID string
|
||||
endpoint string
|
||||
apiKeyEnv string
|
||||
model string
|
||||
}{
|
||||
{
|
||||
name: "OpenRouter",
|
||||
profileID: "mistral-small-3",
|
||||
backendID: promptkit.BackendOpenRouter,
|
||||
endpoint: "https://openrouter.ai/api/v1",
|
||||
apiKeyEnv: "OPENROUTER_API_KEY",
|
||||
model: "mistralai/mistral-small-3.2-24b-instruct",
|
||||
},
|
||||
{
|
||||
name: "Rakestrawhome",
|
||||
profileID: "rakestrawhome-gemma-4-31b",
|
||||
backendID: promptkit.BackendRakestrawHome,
|
||||
endpoint: "https://inference.ai.rakestrawhome.com/v1",
|
||||
apiKeyEnv: "RAKESTRAWHOME_INFERENCE_API_KEY",
|
||||
model: "google/gemma-4-31b-it",
|
||||
},
|
||||
}
|
||||
|
||||
prepared, err := engine.Prepare(context.Background(), promptkit.RunRequest{
|
||||
PromptID: frameworkMarkdownSummaryPromptID,
|
||||
ProfileID: "mistral-small-3",
|
||||
Inputs: map[string]promptkit.ArtifactRef{
|
||||
"transcript": promptkit.Inline("Rin opens the gate."),
|
||||
"glossary": promptkit.Inline("gate: A guarded passage."),
|
||||
},
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("expected built-in profile prepare to succeed, got %v", err)
|
||||
}
|
||||
if prepared.SelectedProfileID != "mistral-small-3" {
|
||||
t.Fatalf("unexpected selected profile: %q", prepared.SelectedProfileID)
|
||||
}
|
||||
if prepared.SelectedBackendID != promptkit.BackendOpenRouter {
|
||||
t.Fatalf("unexpected selected backend: %q", prepared.SelectedBackendID)
|
||||
}
|
||||
if prepared.EffectiveModelParams.BackendID != promptkit.BackendOpenRouter {
|
||||
t.Fatalf("unexpected effective backend: %q", prepared.EffectiveModelParams.BackendID)
|
||||
}
|
||||
if prepared.EffectiveModelParams.Endpoint != "https://openrouter.ai/api/v1" {
|
||||
t.Fatalf("unexpected built-in endpoint: %q", prepared.EffectiveModelParams.Endpoint)
|
||||
}
|
||||
if prepared.EffectiveModelParams.APIKeyEnv != "OPENROUTER_API_KEY" {
|
||||
t.Fatalf("unexpected built-in api key environment name: %q", prepared.EffectiveModelParams.APIKeyEnv)
|
||||
}
|
||||
if prepared.EffectiveModelParams.Model != "mistralai/mistral-small-3.2-24b-instruct" {
|
||||
t.Fatalf("unexpected built-in model: %q", prepared.EffectiveModelParams.Model)
|
||||
for _, tc := range tests {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
t.Setenv(tc.apiKeyEnv, "test-key")
|
||||
engine, err := promptkit.NewEngine(promptkit.Config{
|
||||
PromptDir: frameworkPromptDir,
|
||||
SchemaDir: frameworkSchemaDir,
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("expected engine construction to succeed, got %v", err)
|
||||
}
|
||||
|
||||
prepared, err := engine.Prepare(context.Background(), promptkit.RunRequest{
|
||||
PromptID: frameworkMarkdownSummaryPromptID,
|
||||
ProfileID: tc.profileID,
|
||||
Inputs: map[string]promptkit.ArtifactRef{
|
||||
"transcript": promptkit.Inline("Rin opens the gate."),
|
||||
"glossary": promptkit.Inline("gate: A guarded passage."),
|
||||
},
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("expected built-in profile prepare to succeed, got %v", err)
|
||||
}
|
||||
if prepared.SelectedProfileID != tc.profileID ||
|
||||
prepared.SelectedBackendID != tc.backendID ||
|
||||
prepared.EffectiveModelParams.BackendID != tc.backendID ||
|
||||
prepared.EffectiveModelParams.Endpoint != tc.endpoint ||
|
||||
prepared.EffectiveModelParams.APIKeyEnv != tc.apiKeyEnv ||
|
||||
prepared.EffectiveModelParams.Model != tc.model {
|
||||
t.Fatalf("unexpected built-in preparation: %#v", prepared)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
14
errors.go
14
errors.go
@@ -6,6 +6,7 @@ import (
|
||||
"strings"
|
||||
|
||||
"gitea.maximumdirect.net/eric/promptkit/internal/capacity"
|
||||
"gitea.maximumdirect.net/eric/promptkit/internal/llm"
|
||||
"gitea.maximumdirect.net/eric/promptkit/internal/profile"
|
||||
"gitea.maximumdirect.net/eric/promptkit/internal/promptdef"
|
||||
"gitea.maximumdirect.net/eric/promptkit/internal/usecase"
|
||||
@@ -21,6 +22,19 @@ func mapPublicError(err error) error {
|
||||
return &CapacityError{BackendID: internalCapacityError.BackendID}
|
||||
}
|
||||
publicErr := publicErrorFor(err)
|
||||
var providerHTTPError *llm.ProviderHTTPError
|
||||
if errors.As(err, &providerHTTPError) && providerHTTPError != nil {
|
||||
generationErr := newGenerationError(
|
||||
providerHTTPError.StatusCode(),
|
||||
providerHTTPError.ProviderCode(),
|
||||
providerHTTPError.ProviderType(),
|
||||
providerHTTPError.ProviderMessage(),
|
||||
)
|
||||
if publicErr != nil && !errors.Is(publicErr, ErrLLMGenerate) {
|
||||
return fmt.Errorf("%w: %w", publicErr, generationErr)
|
||||
}
|
||||
return generationErr
|
||||
}
|
||||
if publicErr == nil {
|
||||
return err
|
||||
}
|
||||
|
||||
@@ -6,6 +6,7 @@ import (
|
||||
"fmt"
|
||||
"testing"
|
||||
|
||||
"gitea.maximumdirect.net/eric/promptkit/internal/llm"
|
||||
"gitea.maximumdirect.net/eric/promptkit/internal/usecase"
|
||||
)
|
||||
|
||||
@@ -48,3 +49,27 @@ func TestMapPublicErrorTranslatesCapacityError(t *testing.T) {
|
||||
t.Fatalf("mapped backend ID changed with source error: %q", publicErr.BackendID)
|
||||
}
|
||||
}
|
||||
|
||||
func TestMapPublicErrorPreservesValidationAroundGenerationError(t *testing.T) {
|
||||
internalErr := fmt.Errorf(
|
||||
"%w: %w",
|
||||
usecase.ErrValidation,
|
||||
&llm.ProviderHTTPError{},
|
||||
)
|
||||
|
||||
err := mapPublicError(internalErr)
|
||||
if !errors.Is(err, ErrValidation) {
|
||||
t.Fatalf("mapped error=%v, want ErrValidation", err)
|
||||
}
|
||||
if !errors.Is(err, ErrLLMGenerate) {
|
||||
t.Fatalf("mapped error=%v, want ErrLLMGenerate", err)
|
||||
}
|
||||
var generationErr *GenerationError
|
||||
if !errors.As(err, &generationErr) || generationErr == nil {
|
||||
t.Fatalf("mapped error=%v, want GenerationError", err)
|
||||
}
|
||||
var leakedInternalErr *llm.ProviderHTTPError
|
||||
if errors.As(err, &leakedInternalErr) {
|
||||
t.Fatalf("mapped error exposes internal ProviderHTTPError: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
87
generation_error.go
Normal file
87
generation_error.go
Normal file
@@ -0,0 +1,87 @@
|
||||
package promptkit
|
||||
|
||||
import "fmt"
|
||||
|
||||
// GenerationError reports a non-2xx response from Promptkit's built-in
|
||||
// OpenAI-compatible client during [Engine.Run] or [Engine.RunPrepared].
|
||||
//
|
||||
// Engine-produced values are immutable, caller-owned values. Use errors.Is to
|
||||
// match [ErrLLMGenerate] and errors.As with a *GenerationError target to obtain
|
||||
// this type. The four provider accessors expose untrusted provider-controlled
|
||||
// values that can contain sensitive request or schema fragments. Applications
|
||||
// must apply their own disclosure policy before logging, displaying, or
|
||||
// returning them to another caller.
|
||||
//
|
||||
// Accessors, Error, GoString, and Unwrap are safe on a nil receiver and a zero
|
||||
// value. Default and Go-syntax formatting deliberately redact provider details.
|
||||
// GenerationError has no stable JSON representation.
|
||||
type GenerationError struct {
|
||||
statusCode int
|
||||
providerCode string
|
||||
providerType string
|
||||
providerMessage string
|
||||
}
|
||||
|
||||
func newGenerationError(statusCode int, providerCode, providerType, providerMessage string) *GenerationError {
|
||||
return &GenerationError{
|
||||
statusCode: statusCode,
|
||||
providerCode: providerCode,
|
||||
providerType: providerType,
|
||||
providerMessage: providerMessage,
|
||||
}
|
||||
}
|
||||
|
||||
// StatusCode returns the received provider HTTP status code, or zero for a nil
|
||||
// receiver or zero value.
|
||||
func (e *GenerationError) StatusCode() int {
|
||||
if e == nil {
|
||||
return 0
|
||||
}
|
||||
return e.statusCode
|
||||
}
|
||||
|
||||
// ProviderCode returns the normalized provider error code, if present. Its
|
||||
// value is untrusted and may contain sensitive data.
|
||||
func (e *GenerationError) ProviderCode() string {
|
||||
if e == nil {
|
||||
return ""
|
||||
}
|
||||
return e.providerCode
|
||||
}
|
||||
|
||||
// ProviderType returns the normalized provider error type, if present. Its
|
||||
// value is untrusted and may contain sensitive data.
|
||||
func (e *GenerationError) ProviderType() string {
|
||||
if e == nil {
|
||||
return ""
|
||||
}
|
||||
return e.providerType
|
||||
}
|
||||
|
||||
// ProviderMessage returns the bounded normalized provider diagnostic, if
|
||||
// present. Its value is untrusted and may contain sensitive data.
|
||||
func (e *GenerationError) ProviderMessage() string {
|
||||
if e == nil {
|
||||
return ""
|
||||
}
|
||||
return e.providerMessage
|
||||
}
|
||||
|
||||
// Error returns a redacted diagnostic that is not a parsing contract.
|
||||
func (e *GenerationError) Error() string {
|
||||
if e == nil || e.statusCode == 0 {
|
||||
return ErrLLMGenerate.Error()
|
||||
}
|
||||
return fmt.Sprintf("%s: provider returned HTTP status %d", ErrLLMGenerate, e.statusCode)
|
||||
}
|
||||
|
||||
// GoString returns the same redacted diagnostic as Error.
|
||||
func (e *GenerationError) GoString() string {
|
||||
return e.Error()
|
||||
}
|
||||
|
||||
// Unwrap returns ErrLLMGenerate. It is safe to call on a nil receiver or zero
|
||||
// value.
|
||||
func (e *GenerationError) Unwrap() error {
|
||||
return ErrLLMGenerate
|
||||
}
|
||||
95
generation_error_contract_test.go
Normal file
95
generation_error_contract_test.go
Normal file
@@ -0,0 +1,95 @@
|
||||
package promptkit_test
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"gitea.maximumdirect.net/eric/promptkit"
|
||||
)
|
||||
|
||||
func TestBuiltInGenerationError(t *testing.T) {
|
||||
const (
|
||||
codeMarker = "provider-code-marker"
|
||||
typeMarker = "provider-type-marker"
|
||||
messageMarker = "provider-message-marker"
|
||||
)
|
||||
engine := newBuiltInGenerationErrorEngine(t, http.StatusUnprocessableEntity,
|
||||
`{"error":{"code":"`+codeMarker+`","type":"`+typeMarker+`","message":"`+messageMarker+`"}}`)
|
||||
|
||||
result, err := engine.Run(context.Background(), generationErrorRunRequest())
|
||||
if result != nil {
|
||||
t.Fatalf("Run result = %#v, want nil", result)
|
||||
}
|
||||
assertGenerationError(t, err, http.StatusUnprocessableEntity, codeMarker, typeMarker, messageMarker)
|
||||
|
||||
preparedEngine := newBuiltInGenerationErrorEngine(t, http.StatusServiceUnavailable, `{"error":{}}`)
|
||||
prepared, err := preparedEngine.PrepareExecution(context.Background(), generationErrorRunRequest())
|
||||
if err != nil {
|
||||
t.Fatalf("PrepareExecution: %v", err)
|
||||
}
|
||||
result, err = preparedEngine.RunPrepared(context.Background(), prepared)
|
||||
if result != nil {
|
||||
t.Fatalf("RunPrepared result = %#v, want nil", result)
|
||||
}
|
||||
assertGenerationError(t, err, http.StatusServiceUnavailable, "", "", "")
|
||||
}
|
||||
|
||||
func assertGenerationError(t *testing.T, err error, statusCode int, code, providerType, message string) {
|
||||
t.Helper()
|
||||
|
||||
if !errors.Is(err, promptkit.ErrLLMGenerate) {
|
||||
t.Fatalf("errors.Is(%v, ErrLLMGenerate) = false", err)
|
||||
}
|
||||
var generationErr *promptkit.GenerationError
|
||||
if !errors.As(err, &generationErr) || generationErr == nil {
|
||||
t.Fatalf("error = %T, want *GenerationError", err)
|
||||
}
|
||||
if generationErr.StatusCode() != statusCode || generationErr.ProviderCode() != code || generationErr.ProviderType() != providerType || generationErr.ProviderMessage() != message {
|
||||
t.Fatalf("GenerationError = %#v", generationErr)
|
||||
}
|
||||
|
||||
wantFormatted := fmt.Sprintf("failed to generate output: provider returned HTTP status %d", statusCode)
|
||||
for _, rendered := range []string{fmt.Sprintf("%v", generationErr), fmt.Sprintf("%+v", generationErr), fmt.Sprintf("%#v", generationErr)} {
|
||||
if rendered != wantFormatted {
|
||||
t.Fatalf("formatted error = %q, want %q", rendered, wantFormatted)
|
||||
}
|
||||
for _, marker := range []string{code, providerType, message} {
|
||||
if marker != "" && strings.Contains(rendered, marker) {
|
||||
t.Fatalf("formatted error exposed provider marker %q: %q", marker, rendered)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func newBuiltInGenerationErrorEngine(t *testing.T, statusCode int, body string) *promptkit.Engine {
|
||||
t.Helper()
|
||||
|
||||
config := contractConfig(frameworkSchemaDir)
|
||||
config.HTTPClient = &http.Client{Transport: roundTripFunc(func(*http.Request) (*http.Response, error) {
|
||||
return &http.Response{
|
||||
StatusCode: statusCode,
|
||||
ContentLength: int64(len(body)),
|
||||
Body: io.NopCloser(strings.NewReader(body)),
|
||||
}, nil
|
||||
})}
|
||||
engine, err := promptkit.NewEngine(config)
|
||||
if err != nil {
|
||||
t.Fatalf("NewEngine: %v", err)
|
||||
}
|
||||
return engine
|
||||
}
|
||||
|
||||
func generationErrorRunRequest() promptkit.RunRequest {
|
||||
return promptkit.RunRequest{
|
||||
PromptID: frameworkMarkdownSummaryPromptID,
|
||||
Inputs: map[string]promptkit.ArtifactRef{
|
||||
"transcript": promptkit.Inline("Rin opens the gate."),
|
||||
"glossary": promptkit.Inline("gate: A guarded passage."),
|
||||
},
|
||||
}
|
||||
}
|
||||
32
generation_error_internal_test.go
Normal file
32
generation_error_internal_test.go
Normal file
@@ -0,0 +1,32 @@
|
||||
package promptkit
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestGenerationErrorNilAndZeroValue(t *testing.T) {
|
||||
var nilError *GenerationError
|
||||
zeroError := &GenerationError{}
|
||||
|
||||
for name, err := range map[string]*GenerationError{
|
||||
"nil": nilError,
|
||||
"zero": zeroError,
|
||||
} {
|
||||
t.Run(name, func(t *testing.T) {
|
||||
if err.StatusCode() != 0 || err.ProviderCode() != "" || err.ProviderType() != "" || err.ProviderMessage() != "" {
|
||||
t.Fatalf("accessors returned provider details: %#v", err)
|
||||
}
|
||||
if err.Error() != "failed to generate output" || err.GoString() != "failed to generate output" {
|
||||
t.Fatalf("redacted formatting = (%q, %q)", err.Error(), err.GoString())
|
||||
}
|
||||
if fmt.Sprintf("%v", err) != "failed to generate output" || fmt.Sprintf("%#v", err) != "failed to generate output" {
|
||||
t.Fatalf("formatted error = (%q, %q)", fmt.Sprintf("%v", err), fmt.Sprintf("%#v", err))
|
||||
}
|
||||
if !errors.Is(err, ErrLLMGenerate) {
|
||||
t.Fatalf("errors.Is(%v, ErrLLMGenerate) = false", err)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -18,12 +18,20 @@ const (
|
||||
// OpenRouterID is the reserved ID of Promptkit's built-in OpenRouter
|
||||
// backend.
|
||||
OpenRouterID = "openrouter"
|
||||
// RakestrawHomeID is the reserved ID of Promptkit's built-in Rakestrawhome
|
||||
// backend.
|
||||
RakestrawHomeID = "rakestrawhome"
|
||||
|
||||
openRouterEndpoint = "https://openrouter.ai/api/v1"
|
||||
openRouterAPIKeyEnv = "OPENROUTER_API_KEY"
|
||||
|
||||
openRouterConcurrencyLimit = 16
|
||||
defaultQueueCapacity = 1024
|
||||
|
||||
rakestrawHomeEndpoint = "https://inference.ai.rakestrawhome.com/v1"
|
||||
rakestrawHomeAPIKeyEnv = "RAKESTRAWHOME_INFERENCE_API_KEY"
|
||||
|
||||
rakestrawHomeConcurrencyLimit = 4
|
||||
defaultQueueCapacity = 1024
|
||||
)
|
||||
|
||||
// ErrBackendNotFound identifies a registry lookup for an unknown backend ID.
|
||||
@@ -36,20 +44,16 @@ type Registry struct {
|
||||
backends map[string]domain.Backend
|
||||
}
|
||||
|
||||
// NewRegistry constructs a registry containing the built-in OpenRouter
|
||||
// definition followed by the supplied additions. Every ID must be unique.
|
||||
// NewRegistry constructs a registry containing the built-in definitions
|
||||
// followed by the supplied additions. Every ID must be unique.
|
||||
func NewRegistry(additions []domain.Backend) (*Registry, error) {
|
||||
builtIns := builtInBackends()
|
||||
registry := &Registry{
|
||||
backends: make(map[string]domain.Backend, len(additions)+1),
|
||||
backends: make(map[string]domain.Backend, len(builtIns)+len(additions)),
|
||||
}
|
||||
|
||||
definitions := make([]domain.Backend, 0, len(additions)+1)
|
||||
definitions = append(definitions, domain.Backend{
|
||||
ID: OpenRouterID,
|
||||
Endpoint: openRouterEndpoint,
|
||||
APIKeyEnv: openRouterAPIKeyEnv,
|
||||
ConcurrencyLimit: openRouterConcurrencyLimit,
|
||||
})
|
||||
definitions := make([]domain.Backend, 0, len(builtIns)+len(additions))
|
||||
definitions = append(definitions, builtIns...)
|
||||
definitions = append(definitions, additions...)
|
||||
|
||||
for _, definition := range definitions {
|
||||
@@ -71,6 +75,23 @@ func NewRegistry(additions []domain.Backend) (*Registry, error) {
|
||||
return registry, nil
|
||||
}
|
||||
|
||||
func builtInBackends() []domain.Backend {
|
||||
return []domain.Backend{
|
||||
{
|
||||
ID: OpenRouterID,
|
||||
Endpoint: openRouterEndpoint,
|
||||
APIKeyEnv: openRouterAPIKeyEnv,
|
||||
ConcurrencyLimit: openRouterConcurrencyLimit,
|
||||
},
|
||||
{
|
||||
ID: RakestrawHomeID,
|
||||
Endpoint: rakestrawHomeEndpoint,
|
||||
APIKeyEnv: rakestrawHomeAPIKeyEnv,
|
||||
ConcurrencyLimit: rakestrawHomeConcurrencyLimit,
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// GetBackend returns a defensive copy of the backend registered with id.
|
||||
func (r *Registry) GetBackend(id string) (domain.Backend, error) {
|
||||
if r == nil {
|
||||
|
||||
@@ -11,32 +11,63 @@ import (
|
||||
|
||||
const validEndpoint = "https://backend.example/v1"
|
||||
|
||||
func TestRegistryIncludesExactOpenRouterDefinition(t *testing.T) {
|
||||
func TestRegistryIncludesExactBuiltInDefinitions(t *testing.T) {
|
||||
registry, err := backend.NewRegistry(nil)
|
||||
if err != nil {
|
||||
t.Fatalf("construct registry: %v", err)
|
||||
}
|
||||
|
||||
definition, err := registry.GetBackend(backend.OpenRouterID)
|
||||
if err != nil {
|
||||
t.Fatalf("look up OpenRouter: %v", err)
|
||||
tests := []struct {
|
||||
name string
|
||||
id string
|
||||
endpoint string
|
||||
apiKeyEnv string
|
||||
concurrent int
|
||||
}{
|
||||
{
|
||||
name: "OpenRouter",
|
||||
id: backend.OpenRouterID,
|
||||
endpoint: "https://openrouter.ai/api/v1",
|
||||
apiKeyEnv: "OPENROUTER_API_KEY",
|
||||
concurrent: 16,
|
||||
},
|
||||
{
|
||||
name: "Rakestrawhome",
|
||||
id: backend.RakestrawHomeID,
|
||||
endpoint: "https://inference.ai.rakestrawhome.com/v1",
|
||||
apiKeyEnv: "RAKESTRAWHOME_INFERENCE_API_KEY",
|
||||
concurrent: 4,
|
||||
},
|
||||
}
|
||||
if definition.ID != "openrouter" ||
|
||||
definition.Endpoint != "https://openrouter.ai/api/v1" ||
|
||||
definition.APIKeyEnv != "OPENROUTER_API_KEY" ||
|
||||
definition.ConcurrencyLimit != 16 ||
|
||||
definition.QueueCapacity != 1024 ||
|
||||
!definition.QueueCapacitySet ||
|
||||
definition.ExtraParams != nil {
|
||||
t.Fatalf("unexpected OpenRouter definition: %#v", definition)
|
||||
for _, tc := range tests {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
definition, err := registry.GetBackend(tc.id)
|
||||
if err != nil {
|
||||
t.Fatalf("look up built-in: %v", err)
|
||||
}
|
||||
if definition.ID != tc.id ||
|
||||
definition.Endpoint != tc.endpoint ||
|
||||
definition.APIKeyEnv != tc.apiKeyEnv ||
|
||||
definition.ConcurrencyLimit != tc.concurrent ||
|
||||
definition.QueueCapacity != 1024 ||
|
||||
!definition.QueueCapacitySet ||
|
||||
definition.ExtraParams != nil {
|
||||
t.Fatalf("unexpected built-in definition: %#v", definition)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
policies := registry.CapacityPolicies()
|
||||
if len(policies) != 1 ||
|
||||
policies["openrouter"] != (domain.BackendCapacityPolicy{
|
||||
if len(policies) != 2 ||
|
||||
policies[backend.OpenRouterID] != (domain.BackendCapacityPolicy{
|
||||
ConcurrencyLimit: 16,
|
||||
QueueCapacity: 1024,
|
||||
}) ||
|
||||
policies[backend.RakestrawHomeID] != (domain.BackendCapacityPolicy{
|
||||
ConcurrencyLimit: 4,
|
||||
QueueCapacity: 1024,
|
||||
}) {
|
||||
t.Fatalf("unexpected OpenRouter capacity policies: %#v", policies)
|
||||
t.Fatalf("unexpected built-in capacity policies: %#v", policies)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -109,11 +140,12 @@ func TestRegistryNormalizesUniqueAdditionsAndIsolatesMutations(t *testing.T) {
|
||||
}
|
||||
|
||||
policies := registry.CapacityPolicies()
|
||||
if len(policies) != 2 {
|
||||
if len(policies) != 3 {
|
||||
t.Fatalf("unexpected capacity policy count: %#v", policies)
|
||||
}
|
||||
policies["custom"] = domain.BackendCapacityPolicy{}
|
||||
delete(policies, backend.OpenRouterID)
|
||||
delete(policies, backend.RakestrawHomeID)
|
||||
againPolicies := registry.CapacityPolicies()
|
||||
if againPolicies["custom"] != (domain.BackendCapacityPolicy{
|
||||
ConcurrencyLimit: 3,
|
||||
@@ -122,7 +154,10 @@ func TestRegistryNormalizesUniqueAdditionsAndIsolatesMutations(t *testing.T) {
|
||||
t.Fatalf("capacity policy map mutated registry state: %#v", againPolicies)
|
||||
}
|
||||
if _, ok := againPolicies[backend.OpenRouterID]; !ok {
|
||||
t.Fatalf("capacity policy deletion mutated registry state: %#v", againPolicies)
|
||||
t.Fatalf("OpenRouter capacity policy deletion mutated registry state: %#v", againPolicies)
|
||||
}
|
||||
if _, ok := againPolicies[backend.RakestrawHomeID]; !ok {
|
||||
t.Fatalf("Rakestrawhome capacity policy deletion mutated registry state: %#v", againPolicies)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -242,11 +277,14 @@ func TestNewRegistryRejectsDuplicateIDs(t *testing.T) {
|
||||
wantID string
|
||||
}{
|
||||
{
|
||||
name: "built-in collision after normalization",
|
||||
additions: []domain.Backend{{
|
||||
ID: " openrouter ",
|
||||
}},
|
||||
wantID: "openrouter",
|
||||
name: "OpenRouter collision after normalization",
|
||||
additions: []domain.Backend{{ID: " openrouter "}},
|
||||
wantID: backend.OpenRouterID,
|
||||
},
|
||||
{
|
||||
name: "Rakestrawhome collision after normalization",
|
||||
additions: []domain.Backend{{ID: " rakestrawhome "}},
|
||||
wantID: backend.RakestrawHomeID,
|
||||
},
|
||||
{
|
||||
name: "consumer collision after normalization",
|
||||
|
||||
@@ -133,13 +133,18 @@ func (c *OpenAICompatibleClient) Generate(ctx context.Context, req domain.Genera
|
||||
return nil, fmt.Errorf("%w: failed to create request: %v", ErrRequestFailed, err)
|
||||
}
|
||||
httpReq.Header.Set("Content-Type", "application/json")
|
||||
if apiKey := strings.TrimSpace(req.Target.APIKey); apiKey != "" {
|
||||
httpReq.Header.Set("Authorization", "Bearer "+apiKey)
|
||||
} else if envName := strings.TrimSpace(req.Target.APIKeyEnv); envName != "" {
|
||||
apiKey := strings.TrimSpace(os.Getenv(envName))
|
||||
if apiKey == "" {
|
||||
apiKey := strings.TrimSpace(req.Target.APIKey)
|
||||
envName := strings.TrimSpace(req.Target.APIKeyEnv)
|
||||
if apiKey == "" && envName != "" {
|
||||
apiKey = strings.TrimSpace(os.Getenv(envName))
|
||||
}
|
||||
if apiKey == "" && req.Target.APIKeyRequired {
|
||||
if envName != "" {
|
||||
return nil, fmt.Errorf("%w: api key environment variable %q is not set", ErrInvalidRequest, envName)
|
||||
}
|
||||
return nil, fmt.Errorf("%w: api key is required", ErrInvalidRequest)
|
||||
}
|
||||
if apiKey != "" {
|
||||
httpReq.Header.Set("Authorization", "Bearer "+apiKey)
|
||||
}
|
||||
|
||||
@@ -155,8 +160,11 @@ func (c *OpenAICompatibleClient) Generate(ctx context.Context, req domain.Genera
|
||||
defer httpResp.Body.Close()
|
||||
|
||||
if httpResp.StatusCode < 200 || httpResp.StatusCode >= 300 {
|
||||
_, _ = io.Copy(io.Discard, io.LimitReader(httpResp.Body, 4096))
|
||||
return nil, fmt.Errorf("%w: status=%d", ErrUnexpectedStatus, httpResp.StatusCode)
|
||||
return nil, providerHTTPErrorFromBody(
|
||||
httpResp.StatusCode,
|
||||
httpResp.ContentLength,
|
||||
httpResp.Body,
|
||||
)
|
||||
}
|
||||
if httpResp.ContentLength > maxOpenAIChatResponseBytes {
|
||||
return nil, openAIChatResponseTooLargeError()
|
||||
|
||||
@@ -501,36 +501,61 @@ func checkCompleteRequestAndResponseMapping(t *testing.T) {
|
||||
|
||||
func TestOpenAICompatibleClientAuthentication(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
configureEnv func(*testing.T)
|
||||
target domain.ExecutionTarget
|
||||
wantAuth string
|
||||
wantErr error
|
||||
wantCallCount int
|
||||
name string
|
||||
configureEnv func(*testing.T)
|
||||
target domain.ExecutionTarget
|
||||
wantAuthorization string
|
||||
wantError error
|
||||
wantCallCount int
|
||||
}{
|
||||
{
|
||||
name: "direct key takes precedence over environment",
|
||||
configureEnv: func(t *testing.T) {
|
||||
t.Setenv("PROMPTKIT_TEST_API_KEY", "env-key")
|
||||
t.Setenv("PROMPTKIT_TEST_API_KEY", " env-key ")
|
||||
},
|
||||
target: domain.ExecutionTarget{
|
||||
APIKeyEnv: "PROMPTKIT_TEST_API_KEY",
|
||||
APIKey: "direct-llm-key",
|
||||
APIKey: " direct-llm-key ",
|
||||
},
|
||||
wantAuth: "Bearer direct-llm-key",
|
||||
wantCallCount: 1,
|
||||
wantAuthorization: "Bearer direct-llm-key",
|
||||
wantCallCount: 1,
|
||||
},
|
||||
{
|
||||
name: "environment key supplies authorization",
|
||||
configureEnv: func(t *testing.T) {
|
||||
t.Setenv("PROMPTKIT_TEST_API_KEY", " env-key ")
|
||||
},
|
||||
target: domain.ExecutionTarget{APIKeyEnv: "PROMPTKIT_TEST_API_KEY"},
|
||||
wantAuthorization: "Bearer env-key",
|
||||
wantCallCount: 1,
|
||||
},
|
||||
{
|
||||
name: "no key omits authorization",
|
||||
wantCallCount: 1,
|
||||
},
|
||||
{
|
||||
name: "missing environment key fails before transport",
|
||||
name: "optional missing environment omits authorization",
|
||||
configureEnv: func(t *testing.T) {
|
||||
t.Setenv("PROMPTKIT_MISSING_KEY", "")
|
||||
},
|
||||
target: domain.ExecutionTarget{APIKeyEnv: "PROMPTKIT_MISSING_KEY"},
|
||||
wantErr: ErrInvalidRequest,
|
||||
target: domain.ExecutionTarget{APIKeyEnv: "PROMPTKIT_MISSING_KEY"},
|
||||
wantCallCount: 1,
|
||||
},
|
||||
{
|
||||
name: "required missing environment fails before transport",
|
||||
configureEnv: func(t *testing.T) {
|
||||
t.Setenv("PROMPTKIT_MISSING_KEY", "")
|
||||
},
|
||||
target: domain.ExecutionTarget{
|
||||
APIKeyEnv: "PROMPTKIT_MISSING_KEY",
|
||||
APIKeyRequired: true,
|
||||
},
|
||||
wantError: ErrInvalidRequest,
|
||||
},
|
||||
{
|
||||
name: "required target without source fails before transport",
|
||||
target: domain.ExecutionTarget{APIKeyRequired: true},
|
||||
wantError: ErrInvalidRequest,
|
||||
},
|
||||
}
|
||||
|
||||
@@ -545,9 +570,9 @@ func TestOpenAICompatibleClientAuthentication(t *testing.T) {
|
||||
request.Target = tc.target
|
||||
|
||||
_, err := client.Generate(context.Background(), request)
|
||||
if tc.wantErr != nil {
|
||||
if !errors.Is(err, tc.wantErr) {
|
||||
t.Fatalf("error = %v, want %v", err, tc.wantErr)
|
||||
if tc.wantError != nil {
|
||||
if !errors.Is(err, tc.wantError) {
|
||||
t.Fatalf("error = %v, want %v", err, tc.wantError)
|
||||
}
|
||||
} else if err != nil {
|
||||
t.Fatalf("generate: %v", err)
|
||||
@@ -556,8 +581,13 @@ func TestOpenAICompatibleClientAuthentication(t *testing.T) {
|
||||
t.Fatalf("provider calls = %d, want %d", got, tc.wantCallCount)
|
||||
}
|
||||
if tc.wantCallCount == 1 {
|
||||
if got := provider.lastRequest(t).header.Get("Authorization"); got != tc.wantAuth {
|
||||
t.Fatalf("Authorization = %q, want %q", got, tc.wantAuth)
|
||||
values := provider.lastRequest(t).header.Values("Authorization")
|
||||
if tc.wantAuthorization == "" {
|
||||
if len(values) != 0 {
|
||||
t.Fatalf("Authorization values = %q, want absent", values)
|
||||
}
|
||||
} else if len(values) != 1 || values[0] != tc.wantAuthorization {
|
||||
t.Fatalf("Authorization values = %q, want [%q]", values, tc.wantAuthorization)
|
||||
}
|
||||
}
|
||||
})
|
||||
@@ -1114,6 +1144,15 @@ func checkCommonResponseFailures(t *testing.T) {
|
||||
if !errors.Is(err, tc.wantErr) {
|
||||
t.Fatalf("error = %v, want %v", err, tc.wantErr)
|
||||
}
|
||||
if tc.statusCode < http.StatusOK || tc.statusCode >= http.StatusMultipleChoices {
|
||||
var providerHTTPError *ProviderHTTPError
|
||||
if !errors.As(err, &providerHTTPError) {
|
||||
t.Fatalf("error = %T, want *ProviderHTTPError", err)
|
||||
}
|
||||
if got := providerHTTPError.StatusCode(); got != tc.statusCode {
|
||||
t.Fatalf("provider status = %d, want %d", got, tc.statusCode)
|
||||
}
|
||||
}
|
||||
if tc.wantText != "" && !strings.Contains(err.Error(), tc.wantText) {
|
||||
t.Fatalf("error %q does not contain %q", err, tc.wantText)
|
||||
}
|
||||
|
||||
189
internal/llm/provider_http_error.go
Normal file
189
internal/llm/provider_http_error.go
Normal file
@@ -0,0 +1,189 @@
|
||||
package llm
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"strings"
|
||||
"unicode"
|
||||
)
|
||||
|
||||
const (
|
||||
maxProviderErrorResponseBytes int64 = 64 << 10
|
||||
maxProviderErrorIdentifierRunes = 256
|
||||
maxProviderErrorMessageRunes = 4096
|
||||
)
|
||||
|
||||
// ProviderHTTPError describes a non-success response from an LLM provider.
|
||||
type ProviderHTTPError struct {
|
||||
statusCode int
|
||||
providerCode string
|
||||
providerType string
|
||||
providerMessage string
|
||||
}
|
||||
|
||||
func (e *ProviderHTTPError) StatusCode() int {
|
||||
if e == nil {
|
||||
return 0
|
||||
}
|
||||
return e.statusCode
|
||||
}
|
||||
|
||||
func (e *ProviderHTTPError) ProviderCode() string {
|
||||
if e == nil {
|
||||
return ""
|
||||
}
|
||||
return e.providerCode
|
||||
}
|
||||
|
||||
func (e *ProviderHTTPError) ProviderType() string {
|
||||
if e == nil {
|
||||
return ""
|
||||
}
|
||||
return e.providerType
|
||||
}
|
||||
|
||||
func (e *ProviderHTTPError) ProviderMessage() string {
|
||||
if e == nil {
|
||||
return ""
|
||||
}
|
||||
return e.providerMessage
|
||||
}
|
||||
|
||||
func (e *ProviderHTTPError) Error() string {
|
||||
if e == nil || e.statusCode == 0 {
|
||||
return ErrUnexpectedStatus.Error()
|
||||
}
|
||||
return fmt.Sprintf("%s: status=%d", ErrUnexpectedStatus, e.statusCode)
|
||||
}
|
||||
|
||||
func (e *ProviderHTTPError) GoString() string {
|
||||
return e.Error()
|
||||
}
|
||||
|
||||
func (e *ProviderHTTPError) Unwrap() error {
|
||||
return ErrUnexpectedStatus
|
||||
}
|
||||
|
||||
type providerErrorDetails struct {
|
||||
providerCode string
|
||||
providerType string
|
||||
providerMessage string
|
||||
}
|
||||
|
||||
func newProviderHTTPError(statusCode int, details providerErrorDetails) *ProviderHTTPError {
|
||||
return &ProviderHTTPError{
|
||||
statusCode: statusCode,
|
||||
providerCode: details.providerCode,
|
||||
providerType: details.providerType,
|
||||
providerMessage: details.providerMessage,
|
||||
}
|
||||
}
|
||||
|
||||
func providerHTTPErrorFromBody(statusCode int, contentLength int64, body io.Reader) *ProviderHTTPError {
|
||||
if contentLength > maxProviderErrorResponseBytes {
|
||||
return newProviderHTTPError(statusCode, providerErrorDetails{})
|
||||
}
|
||||
|
||||
limited := &io.LimitedReader{
|
||||
R: body,
|
||||
N: maxProviderErrorResponseBytes + 1,
|
||||
}
|
||||
contents, err := io.ReadAll(limited)
|
||||
if err != nil || limited.N == 0 {
|
||||
return newProviderHTTPError(statusCode, providerErrorDetails{})
|
||||
}
|
||||
return newProviderHTTPError(statusCode, parseProviderErrorEnvelope(contents))
|
||||
}
|
||||
|
||||
func parseProviderErrorEnvelope(body []byte) providerErrorDetails {
|
||||
decoder := json.NewDecoder(strings.NewReader(string(body)))
|
||||
decoder.UseNumber()
|
||||
|
||||
var envelope map[string]json.RawMessage
|
||||
if err := decoder.Decode(&envelope); err != nil {
|
||||
return providerErrorDetails{}
|
||||
}
|
||||
var trailing any
|
||||
if err := decoder.Decode(&trailing); !errors.Is(err, io.EOF) {
|
||||
return providerErrorDetails{}
|
||||
}
|
||||
|
||||
rawError, ok := envelope["error"]
|
||||
if !ok {
|
||||
return providerErrorDetails{}
|
||||
}
|
||||
var providerError map[string]json.RawMessage
|
||||
if err := json.Unmarshal(rawError, &providerError); err != nil || providerError == nil {
|
||||
return providerErrorDetails{}
|
||||
}
|
||||
|
||||
var details providerErrorDetails
|
||||
if raw, ok := providerError["message"]; ok {
|
||||
var value string
|
||||
if json.Unmarshal(raw, &value) == nil {
|
||||
details.providerMessage = normalizeProviderErrorMessage(value)
|
||||
}
|
||||
}
|
||||
if raw, ok := providerError["type"]; ok {
|
||||
var value string
|
||||
if json.Unmarshal(raw, &value) == nil {
|
||||
details.providerType = normalizeProviderErrorIdentifier(value)
|
||||
}
|
||||
}
|
||||
if raw, ok := providerError["code"]; ok {
|
||||
var value any
|
||||
fieldDecoder := json.NewDecoder(strings.NewReader(string(raw)))
|
||||
fieldDecoder.UseNumber()
|
||||
if fieldDecoder.Decode(&value) == nil {
|
||||
switch value := value.(type) {
|
||||
case string:
|
||||
details.providerCode = normalizeProviderErrorIdentifier(value)
|
||||
case json.Number:
|
||||
details.providerCode = normalizeProviderErrorIdentifier(value.String())
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return details
|
||||
}
|
||||
|
||||
func normalizeProviderErrorIdentifier(value string) string {
|
||||
normalized := normalizeProviderErrorText(value)
|
||||
if len([]rune(normalized)) > maxProviderErrorIdentifierRunes {
|
||||
return ""
|
||||
}
|
||||
return normalized
|
||||
}
|
||||
|
||||
func normalizeProviderErrorMessage(value string) string {
|
||||
normalized := normalizeProviderErrorText(value)
|
||||
runes := []rune(normalized)
|
||||
if len(runes) <= maxProviderErrorMessageRunes {
|
||||
return normalized
|
||||
}
|
||||
return string(runes[:maxProviderErrorMessageRunes-1]) + "…"
|
||||
}
|
||||
|
||||
func normalizeProviderErrorText(value string) string {
|
||||
value = strings.ToValidUTF8(value, "<22>")
|
||||
|
||||
var result strings.Builder
|
||||
result.Grow(len(value))
|
||||
separatorPending := false
|
||||
for _, r := range value {
|
||||
if unicode.IsSpace(r) || unicode.IsControl(r) || unicode.In(r, unicode.Cf) {
|
||||
if result.Len() > 0 {
|
||||
separatorPending = true
|
||||
}
|
||||
continue
|
||||
}
|
||||
if separatorPending {
|
||||
result.WriteByte(' ')
|
||||
separatorPending = false
|
||||
}
|
||||
result.WriteRune(r)
|
||||
}
|
||||
return result.String()
|
||||
}
|
||||
255
internal/llm/provider_http_error_test.go
Normal file
255
internal/llm/provider_http_error_test.go
Normal file
@@ -0,0 +1,255 @@
|
||||
package llm
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"reflect"
|
||||
"strings"
|
||||
"testing"
|
||||
"unicode/utf8"
|
||||
)
|
||||
|
||||
type guardedReader struct {
|
||||
reader io.Reader
|
||||
remaining int64
|
||||
bytes int64
|
||||
violated bool
|
||||
}
|
||||
|
||||
func (r *guardedReader) Read(buffer []byte) (int, error) {
|
||||
if int64(len(buffer)) > r.remaining {
|
||||
r.violated = true
|
||||
return 0, errors.New("reader was read past its allowed boundary")
|
||||
}
|
||||
n, err := r.reader.Read(buffer)
|
||||
r.bytes += int64(n)
|
||||
r.remaining -= int64(n)
|
||||
return n, err
|
||||
}
|
||||
|
||||
type failingReader struct {
|
||||
err error
|
||||
}
|
||||
|
||||
func (r failingReader) Read([]byte) (int, error) {
|
||||
return 0, r.err
|
||||
}
|
||||
|
||||
func TestProviderHTTPErrorEnvelopeParsing(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
body string
|
||||
want providerErrorDetails
|
||||
}{
|
||||
{
|
||||
name: "all supported string fields",
|
||||
body: `{"error":{"message":"diagnostic","type":"invalid_request_error","code":"unsupported_parameter"}}`,
|
||||
want: providerErrorDetails{providerMessage: "diagnostic", providerType: "invalid_request_error", providerCode: "unsupported_parameter"},
|
||||
},
|
||||
{
|
||||
name: "integer code",
|
||||
body: `{"error":{"code":17}}`,
|
||||
want: providerErrorDetails{providerCode: "17"},
|
||||
},
|
||||
{
|
||||
name: "fractional code",
|
||||
body: `{"error":{"code":1.25}}`,
|
||||
want: providerErrorDetails{providerCode: "1.25"},
|
||||
},
|
||||
{
|
||||
name: "exponent code",
|
||||
body: `{"error":{"code":6.02e+23}}`,
|
||||
want: providerErrorDetails{providerCode: "6.02e+23"},
|
||||
},
|
||||
{
|
||||
name: "invalid fields do not discard valid fields",
|
||||
body: `{"error":{"message":null,"type":"invalid_request_error","code":false}}`,
|
||||
want: providerErrorDetails{providerType: "invalid_request_error"},
|
||||
},
|
||||
{
|
||||
name: "unknown fields are ignored",
|
||||
body: `{"trace":"do not retain","error":{"param":"temperature","metadata":{"secret":"x"}}}`,
|
||||
want: providerErrorDetails{},
|
||||
},
|
||||
{name: "missing error", body: `{}`, want: providerErrorDetails{}},
|
||||
{name: "null error", body: `{"error":null}`, want: providerErrorDetails{}},
|
||||
{name: "scalar error", body: `{"error":"nope"}`, want: providerErrorDetails{}},
|
||||
{name: "empty error", body: `{"error":{}}`, want: providerErrorDetails{}},
|
||||
{name: "malformed", body: `{"error":`, want: providerErrorDetails{}},
|
||||
{name: "truncated", body: `{"error":{"message":"x"`, want: providerErrorDetails{}},
|
||||
{name: "trailing garbage", body: `{"error":{"message":"x"}} garbage`, want: providerErrorDetails{}},
|
||||
{name: "second document", body: `{"error":{"message":"x"}} {}`, want: providerErrorDetails{}},
|
||||
}
|
||||
|
||||
for _, tc := range tests {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
if got := parseProviderErrorEnvelope([]byte(tc.body)); !reflect.DeepEqual(got, tc.want) {
|
||||
t.Fatalf("parseProviderErrorEnvelope() = %#v, want %#v", got, tc.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestProviderErrorTextNormalizationAndLimits(t *testing.T) {
|
||||
validIdentifier := strings.Repeat("界", maxProviderErrorIdentifierRunes)
|
||||
validMessage := strings.Repeat("界", maxProviderErrorMessageRunes)
|
||||
tests := []struct {
|
||||
name string
|
||||
got string
|
||||
want string
|
||||
}{
|
||||
{name: "multibyte text", got: "Grüße 世界", want: "Grüße 世界"},
|
||||
{name: "invalid UTF-8", got: string([]byte{'a', 0xff, 'b'}), want: "a<>b"},
|
||||
{name: "whitespace control and format runs", got: " \n\talpha\x00\u200b\u200bbeta \r ", want: "alpha beta"},
|
||||
{name: "blank normalization", got: "\t\u200b\n", want: ""},
|
||||
}
|
||||
for _, tc := range tests {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
if got := normalizeProviderErrorText(tc.got); got != tc.want {
|
||||
t.Fatalf("normalizeProviderErrorText() = %q, want %q", got, tc.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
if got := normalizeProviderErrorIdentifier(validIdentifier); got != validIdentifier {
|
||||
t.Fatalf("exact identifier boundary = %q, want retained value", got)
|
||||
}
|
||||
if got := normalizeProviderErrorIdentifier(validIdentifier + "界"); got != "" {
|
||||
t.Fatalf("overlong identifier = %q, want empty", got)
|
||||
}
|
||||
if got := normalizeProviderErrorMessage(validMessage); got != validMessage {
|
||||
t.Fatalf("exact message boundary = %q, want retained value", got)
|
||||
}
|
||||
wantTruncatedMessage := strings.Repeat("界", maxProviderErrorMessageRunes-1) + "…"
|
||||
if got := normalizeProviderErrorMessage(validMessage + "界"); got != wantTruncatedMessage {
|
||||
t.Fatalf("overlong message length = %d, want %d", utf8.RuneCountInString(got), maxProviderErrorMessageRunes)
|
||||
}
|
||||
}
|
||||
|
||||
func TestProviderHTTPErrorIdentityAndFormatting(t *testing.T) {
|
||||
const marker = "provider-secret-marker"
|
||||
err := newProviderHTTPError(429, providerErrorDetails{
|
||||
providerCode: marker + "-code",
|
||||
providerType: marker + "-type",
|
||||
providerMessage: marker + "-message",
|
||||
})
|
||||
|
||||
if err.StatusCode() != 429 || err.ProviderCode() != marker+"-code" || err.ProviderType() != marker+"-type" || err.ProviderMessage() != marker+"-message" {
|
||||
t.Fatalf("accessors returned unexpected values: %#v", err)
|
||||
}
|
||||
if !errors.Is(err, ErrUnexpectedStatus) {
|
||||
t.Fatalf("errors.Is(%v, ErrUnexpectedStatus) = false", err)
|
||||
}
|
||||
for _, rendered := range []string{fmt.Sprintf("%v", err), fmt.Sprintf("%+v", err), fmt.Sprintf("%#v", err)} {
|
||||
if rendered != "llm returned non-success status: status=429" {
|
||||
t.Fatalf("formatted error = %q", rendered)
|
||||
}
|
||||
if strings.Contains(rendered, marker) {
|
||||
t.Fatalf("formatted error exposed provider marker: %q", rendered)
|
||||
}
|
||||
}
|
||||
|
||||
var nilError *ProviderHTTPError
|
||||
if nilError.StatusCode() != 0 || nilError.ProviderCode() != "" || nilError.ProviderType() != "" || nilError.ProviderMessage() != "" {
|
||||
t.Fatal("nil accessors returned provider values")
|
||||
}
|
||||
if nilError.Error() != "llm returned non-success status" || nilError.GoString() != "llm returned non-success status" || !errors.Is(nilError, ErrUnexpectedStatus) {
|
||||
t.Fatalf("nil error behavior is not safe: %v", nilError)
|
||||
}
|
||||
|
||||
zero := &ProviderHTTPError{}
|
||||
if zero.Error() != "llm returned non-success status" || zero.GoString() != "llm returned non-success status" || !errors.Is(zero, ErrUnexpectedStatus) {
|
||||
t.Fatalf("zero error behavior is not safe: %v", zero)
|
||||
}
|
||||
}
|
||||
|
||||
func TestProviderHTTPErrorBodyBounds(t *testing.T) {
|
||||
const (
|
||||
statusCode = 502
|
||||
marker = "provider-body-marker"
|
||||
)
|
||||
|
||||
ordinaryBody := `{"error":{"message":"` + marker + `"}}`
|
||||
exactLimitBody := ordinaryBody + strings.Repeat(" ", int(maxProviderErrorResponseBytes)-len(ordinaryBody))
|
||||
overLimitBody := ordinaryBody + strings.Repeat(" ", int(maxProviderErrorResponseBytes)+1-len(ordinaryBody))
|
||||
tests := []struct {
|
||||
name string
|
||||
contentLength int64
|
||||
reader io.Reader
|
||||
wantRead int64
|
||||
wantMessage string
|
||||
}{
|
||||
{
|
||||
name: "recognized envelope",
|
||||
contentLength: int64(len(ordinaryBody)),
|
||||
reader: strings.NewReader(ordinaryBody),
|
||||
wantRead: int64(len(ordinaryBody)),
|
||||
wantMessage: marker,
|
||||
},
|
||||
{
|
||||
name: "exact limit",
|
||||
contentLength: maxProviderErrorResponseBytes,
|
||||
reader: strings.NewReader(exactLimitBody),
|
||||
wantRead: maxProviderErrorResponseBytes,
|
||||
wantMessage: marker,
|
||||
},
|
||||
{
|
||||
name: "declared oversize does not read",
|
||||
contentLength: maxProviderErrorResponseBytes + 1,
|
||||
reader: strings.NewReader(ordinaryBody),
|
||||
wantRead: 0,
|
||||
},
|
||||
{
|
||||
name: "unknown length oversize",
|
||||
contentLength: -1,
|
||||
reader: strings.NewReader(overLimitBody),
|
||||
wantRead: maxProviderErrorResponseBytes + 1,
|
||||
},
|
||||
{
|
||||
name: "underreported oversize",
|
||||
contentLength: maxProviderErrorResponseBytes,
|
||||
reader: strings.NewReader(overLimitBody),
|
||||
wantRead: maxProviderErrorResponseBytes + 1,
|
||||
},
|
||||
{
|
||||
name: "read failure",
|
||||
contentLength: -1,
|
||||
reader: failingReader{err: errors.New("read failure")},
|
||||
wantRead: 0,
|
||||
},
|
||||
{
|
||||
name: "empty body",
|
||||
contentLength: 0,
|
||||
reader: strings.NewReader(""),
|
||||
wantRead: 0,
|
||||
},
|
||||
}
|
||||
|
||||
for _, tc := range tests {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
reader := &guardedReader{
|
||||
reader: tc.reader,
|
||||
remaining: maxProviderErrorResponseBytes + 1,
|
||||
}
|
||||
err := providerHTTPErrorFromBody(statusCode, tc.contentLength, reader)
|
||||
if err == nil || err.StatusCode() != statusCode {
|
||||
t.Fatalf("error status = %v, want %d", err, statusCode)
|
||||
}
|
||||
if reader.bytes != tc.wantRead {
|
||||
t.Fatalf("body bytes read = %d, want %d", reader.bytes, tc.wantRead)
|
||||
}
|
||||
if reader.violated {
|
||||
t.Fatal("body reader was asked to read beyond the overflow probe")
|
||||
}
|
||||
if got := err.ProviderMessage(); got != tc.wantMessage {
|
||||
t.Fatalf("provider message = %q, want %q", got, tc.wantMessage)
|
||||
}
|
||||
if tc.wantMessage == "" {
|
||||
if err.ProviderCode() != "" || err.ProviderType() != "" || strings.Contains(err.Error(), marker) {
|
||||
t.Fatalf("discarded details were retained: %#v", err)
|
||||
}
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
145
internal/llm/provider_http_error_transport_test.go
Normal file
145
internal/llm/provider_http_error_transport_test.go
Normal file
@@ -0,0 +1,145 @@
|
||||
package llm
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"io"
|
||||
"net/http"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestOpenAICompatibleClientStructuredNonSuccessResponse(t *testing.T) {
|
||||
body := `{"error":{"message":" provider\nmessage\u200b","type":"invalid\ttype","code":1.5e+4}}`
|
||||
responseBody := &countingReadCloser{reader: strings.NewReader(body)}
|
||||
client := newNonSuccessResponseClient(t, http.StatusBadRequest, int64(len(body)), responseBody)
|
||||
|
||||
response, err := client.Generate(context.Background(), ordinaryGenerateRequest())
|
||||
if response != nil {
|
||||
t.Fatalf("response = %#v, want nil", response)
|
||||
}
|
||||
if !errors.Is(err, ErrUnexpectedStatus) {
|
||||
t.Fatalf("errors.Is(%v, ErrUnexpectedStatus) = false", err)
|
||||
}
|
||||
var providerHTTPError *ProviderHTTPError
|
||||
if !errors.As(err, &providerHTTPError) {
|
||||
t.Fatalf("error = %T, want *ProviderHTTPError", err)
|
||||
}
|
||||
if providerHTTPError.StatusCode() != http.StatusBadRequest || providerHTTPError.ProviderCode() != "1.5e+4" || providerHTTPError.ProviderType() != "invalid type" || providerHTTPError.ProviderMessage() != "provider message" {
|
||||
t.Fatalf("provider error = %#v", providerHTTPError)
|
||||
}
|
||||
if !responseBody.closed {
|
||||
t.Fatal("non-success response body was not closed")
|
||||
}
|
||||
}
|
||||
|
||||
func TestOpenAICompatibleClientNonSuccessBodyOwnership(t *testing.T) {
|
||||
const marker = "provider-body-marker"
|
||||
normalBody := `{"error":{"message":"` + marker + `"}}`
|
||||
overLimitBody := normalBody + strings.Repeat(" ", int(maxProviderErrorResponseBytes)+1-len(normalBody))
|
||||
tests := []struct {
|
||||
name string
|
||||
contentLength int64
|
||||
reader io.Reader
|
||||
wantRead int64
|
||||
wantMessage string
|
||||
}{
|
||||
{
|
||||
name: "normal",
|
||||
contentLength: int64(len(normalBody)),
|
||||
reader: strings.NewReader(normalBody),
|
||||
wantRead: int64(len(normalBody)),
|
||||
wantMessage: marker,
|
||||
},
|
||||
{
|
||||
name: "declared oversize",
|
||||
contentLength: maxProviderErrorResponseBytes + 1,
|
||||
reader: strings.NewReader(normalBody),
|
||||
wantRead: 0,
|
||||
},
|
||||
{
|
||||
name: "streamed oversize",
|
||||
contentLength: -1,
|
||||
reader: &guardedReader{
|
||||
reader: strings.NewReader(overLimitBody),
|
||||
remaining: maxProviderErrorResponseBytes + 1,
|
||||
},
|
||||
wantRead: maxProviderErrorResponseBytes + 1,
|
||||
},
|
||||
{
|
||||
name: "underreported oversize",
|
||||
contentLength: maxProviderErrorResponseBytes,
|
||||
reader: &guardedReader{
|
||||
reader: strings.NewReader(overLimitBody),
|
||||
remaining: maxProviderErrorResponseBytes + 1,
|
||||
},
|
||||
wantRead: maxProviderErrorResponseBytes + 1,
|
||||
},
|
||||
{
|
||||
name: "malformed",
|
||||
contentLength: 1,
|
||||
reader: strings.NewReader("{"),
|
||||
wantRead: 1,
|
||||
},
|
||||
{
|
||||
name: "read failure",
|
||||
contentLength: -1,
|
||||
reader: failingReader{err: errors.New("response read failed")},
|
||||
wantRead: 0,
|
||||
},
|
||||
}
|
||||
|
||||
for _, tc := range tests {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
body := &countingReadCloser{reader: tc.reader}
|
||||
client := newNonSuccessResponseClient(t, http.StatusBadGateway, tc.contentLength, body)
|
||||
|
||||
response, err := client.Generate(context.Background(), ordinaryGenerateRequest())
|
||||
if response != nil {
|
||||
t.Fatalf("response = %#v, want nil", response)
|
||||
}
|
||||
var providerHTTPError *ProviderHTTPError
|
||||
if !errors.As(err, &providerHTTPError) {
|
||||
t.Fatalf("error = %T, want *ProviderHTTPError", err)
|
||||
}
|
||||
if !body.closed {
|
||||
t.Fatal("response body was not closed")
|
||||
}
|
||||
if body.bytesRead != tc.wantRead {
|
||||
t.Fatalf("body bytes read = %d, want %d", body.bytesRead, tc.wantRead)
|
||||
}
|
||||
if body.bytesRead > maxProviderErrorResponseBytes+1 {
|
||||
t.Fatalf("body bytes read = %d, exceeds overflow probe", body.bytesRead)
|
||||
}
|
||||
if guarded, ok := tc.reader.(*guardedReader); ok && guarded.violated {
|
||||
t.Fatal("body reader was asked to read beyond the overflow probe")
|
||||
}
|
||||
if got := providerHTTPError.ProviderMessage(); got != tc.wantMessage {
|
||||
t.Fatalf("provider message = %q, want %q", got, tc.wantMessage)
|
||||
}
|
||||
if tc.wantMessage == "" && (providerHTTPError.ProviderCode() != "" || providerHTTPError.ProviderType() != "" || strings.Contains(providerHTTPError.Error(), marker)) {
|
||||
t.Fatalf("discarded details were retained: %#v", providerHTTPError)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func newNonSuccessResponseClient(t *testing.T, statusCode int, contentLength int64, body io.ReadCloser) *OpenAICompatibleClient {
|
||||
t.Helper()
|
||||
|
||||
client, err := NewOpenAICompatibleClient(OpenAICompatibleConfig{
|
||||
BaseURL: "https://provider.example/v1",
|
||||
Model: "m",
|
||||
HTTPClient: &http.Client{Transport: roundTripFunc(func(*http.Request) (*http.Response, error) {
|
||||
return &http.Response{
|
||||
StatusCode: statusCode,
|
||||
ContentLength: contentLength,
|
||||
Body: body,
|
||||
}, nil
|
||||
})},
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("construct client: %v", err)
|
||||
}
|
||||
return client
|
||||
}
|
||||
@@ -0,0 +1,3 @@
|
||||
id: rakestrawhome-gemma-4-31b
|
||||
backend: rakestrawhome
|
||||
model: google/gemma-4-31b-it
|
||||
@@ -26,8 +26,8 @@ func TestBuiltInProfilesValidateThroughRepository(t *testing.T) {
|
||||
if p.ID != id {
|
||||
t.Fatalf("expected profile id %q, got %q", id, p.ID)
|
||||
}
|
||||
if p.BackendID != backend.OpenRouterID {
|
||||
t.Fatalf("expected profile %q to select %q, got %q", id, backend.OpenRouterID, p.BackendID)
|
||||
if !builtInBackendIDs[p.BackendID] {
|
||||
t.Fatalf("expected profile %q to select a maintained built-in, got %q", id, p.BackendID)
|
||||
}
|
||||
if p.Endpoint != "" || p.APIKeyEnv != "" {
|
||||
t.Fatalf("expected profile %q to inherit backend connection settings, got endpoint=%q api_key_env=%q", id, p.Endpoint, p.APIKeyEnv)
|
||||
@@ -40,6 +40,33 @@ func TestBuiltInProfilesDoNotContainDuplicateIDsOrRawAPIKeys(t *testing.T) {
|
||||
loadBuiltInProfileIDs(t)
|
||||
}
|
||||
|
||||
func TestRakestrawhomeGemmaProfileUsesNativeDefaults(t *testing.T) {
|
||||
p, err := NewRepository().GetProfile(context.Background(), "rakestrawhome-gemma-4-31b")
|
||||
if err != nil {
|
||||
t.Fatalf("load Rakestrawhome Gemma profile: %v", err)
|
||||
}
|
||||
if p.ID != "rakestrawhome-gemma-4-31b" ||
|
||||
p.BackendID != backend.RakestrawHomeID ||
|
||||
p.Model != "google/gemma-4-31b-it" ||
|
||||
p.Endpoint != "" ||
|
||||
p.Temperature != 0 ||
|
||||
p.MaxTokens != 0 ||
|
||||
p.TopP != 0 ||
|
||||
p.TimeoutSeconds != 0 ||
|
||||
p.ServiceTier != "" ||
|
||||
p.ReasoningEffort != "" ||
|
||||
p.APIKeyEnv != "" ||
|
||||
p.APIKeyRequired ||
|
||||
p.ExtraParams != nil {
|
||||
t.Fatalf("unexpected Rakestrawhome Gemma profile: %#v", p)
|
||||
}
|
||||
}
|
||||
|
||||
var builtInBackendIDs = map[string]bool{
|
||||
backend.OpenRouterID: true,
|
||||
backend.RakestrawHomeID: true,
|
||||
}
|
||||
|
||||
func loadBuiltInProfileIDs(t *testing.T) map[string]string {
|
||||
t.Helper()
|
||||
|
||||
@@ -64,8 +91,9 @@ func loadBuiltInProfileIDs(t *testing.T) map[string]string {
|
||||
if _, ok := raw["api_key"]; ok {
|
||||
t.Fatalf("built-in profile %s contains raw api_key", name)
|
||||
}
|
||||
if raw["backend"] != backend.OpenRouterID {
|
||||
t.Fatalf("built-in profile %s does not select %q", name, backend.OpenRouterID)
|
||||
backendID, ok := raw["backend"].(string)
|
||||
if !ok || !builtInBackendIDs[backendID] {
|
||||
t.Fatalf("built-in profile %s does not select a maintained built-in: %#v", name, raw["backend"])
|
||||
}
|
||||
if _, ok := raw["endpoint"]; ok {
|
||||
t.Fatalf("built-in profile %s repeats endpoint", name)
|
||||
|
||||
@@ -3,7 +3,6 @@ package usecase
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"os"
|
||||
"reflect"
|
||||
"testing"
|
||||
|
||||
@@ -241,49 +240,76 @@ func excessivelyDeepPreparedJSONValue() any {
|
||||
return value
|
||||
}
|
||||
|
||||
func TestRunnerRunPreparedRechecksEnvironmentCredentialBeforeAdmission(t *testing.T) {
|
||||
func TestRunnerRunPreparedCredentialAvailabilityBeforeAdmission(t *testing.T) {
|
||||
const environmentName = "PROMPTKIT_PREPARED_EXECUTION_TEST_KEY"
|
||||
t.Setenv(environmentName, "available-during-preparation")
|
||||
|
||||
profile := defaultExecutionProfile()
|
||||
profile.APIKeyEnv = environmentName
|
||||
validator := &recordingValidationPreparer{plan: &recordingPreparedValidation{}}
|
||||
admitter := &fakeRunAdmitter{}
|
||||
llmClient := &fakeLLM{resp: &domain.GenerateResponse{Content: "unexpected"}}
|
||||
runner := NewRunner(
|
||||
&fakePromptRepo{def: promptDef(domain.FormatText, domain.ValidationNone, 0)},
|
||||
&fakeExecutionProfileRepo{profiles: map[string]*domain.ExecutionProfile{"exec": profile}},
|
||||
nil,
|
||||
defaultArtifactReader(),
|
||||
defaultRenderer(),
|
||||
llmClient,
|
||||
validator,
|
||||
admitter,
|
||||
)
|
||||
prepared, err := runner.PrepareExecution(context.Background(), domain.RunRequest{
|
||||
PromptID: "p",
|
||||
ProfileID: "exec",
|
||||
Inputs: singleInputRef(),
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("prepare execution: %v", err)
|
||||
}
|
||||
if err := os.Unsetenv(environmentName); err != nil {
|
||||
t.Fatalf("unset credential environment: %v", err)
|
||||
tests := []struct {
|
||||
name string
|
||||
apiKeyRequired bool
|
||||
profileEnv bool
|
||||
overrideEnv bool
|
||||
wantFailure bool
|
||||
}{
|
||||
{name: "optional environment becomes unavailable", profileEnv: true},
|
||||
{
|
||||
name: "required request environment becomes unavailable",
|
||||
apiKeyRequired: true,
|
||||
overrideEnv: true,
|
||||
wantFailure: true,
|
||||
},
|
||||
}
|
||||
|
||||
result, err := runner.RunPrepared(context.Background(), prepared)
|
||||
if result != nil {
|
||||
t.Fatalf("credential failure returned partial result: %+v", result)
|
||||
}
|
||||
if !errors.Is(err, ErrInvalidRequest) || !errors.Is(err, ErrAPIKeyEnvMissing) {
|
||||
t.Fatalf("credential error identities are missing: %v", err)
|
||||
}
|
||||
if len(admitter.backendIDs) != 0 || llmClient.calls != 0 {
|
||||
t.Fatalf("credential failure reached admission or generation: admission=%v generation=%d", admitter.backendIDs, llmClient.calls)
|
||||
}
|
||||
if _, err := runner.RunPrepared(context.Background(), prepared); !errors.Is(err, ErrInvalidRequest) {
|
||||
t.Fatalf("credential failure did not consume execution: %v", err)
|
||||
for _, tc := range tests {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
t.Setenv(environmentName, "available-during-preparation")
|
||||
|
||||
profile := defaultExecutionProfile()
|
||||
profile.APIKeyRequired = tc.apiKeyRequired
|
||||
if tc.profileEnv {
|
||||
profile.APIKeyEnv = environmentName
|
||||
}
|
||||
validator := &recordingValidationPreparer{plan: &recordingPreparedValidation{}}
|
||||
admitter := &fakeRunAdmitter{}
|
||||
llmClient := &fakeLLM{resp: &domain.GenerateResponse{Content: "ok"}}
|
||||
runner := NewRunner(
|
||||
&fakePromptRepo{def: promptDef(domain.FormatText, domain.ValidationNone, 0)},
|
||||
&fakeExecutionProfileRepo{profiles: map[string]*domain.ExecutionProfile{"exec": profile}},
|
||||
nil,
|
||||
defaultArtifactReader(),
|
||||
defaultRenderer(),
|
||||
llmClient,
|
||||
validator,
|
||||
admitter,
|
||||
)
|
||||
request := domain.RunRequest{PromptID: "p", ProfileID: "exec", Inputs: singleInputRef()}
|
||||
if tc.overrideEnv {
|
||||
request.Execution = &domain.ExecutionTargetOverride{APIKeyEnv: environmentName}
|
||||
}
|
||||
prepared, err := runner.PrepareExecution(context.Background(), request)
|
||||
if err != nil {
|
||||
t.Fatalf("prepare execution: %v", err)
|
||||
}
|
||||
t.Setenv(environmentName, "")
|
||||
|
||||
result, err := runner.RunPrepared(context.Background(), prepared)
|
||||
if tc.wantFailure {
|
||||
if result != nil || !errors.Is(err, ErrInvalidRequest) || !errors.Is(err, ErrAPIKeyEnvMissing) {
|
||||
t.Fatalf("required credential result = (%+v, %v)", result, err)
|
||||
}
|
||||
if len(admitter.backendIDs) != 0 || llmClient.calls != 0 {
|
||||
t.Fatalf("required credential reached admission or generation: admission=%v generation=%d", admitter.backendIDs, llmClient.calls)
|
||||
}
|
||||
} else {
|
||||
if result == nil || err != nil {
|
||||
t.Fatalf("optional credential result = (%+v, %v), want success", result, err)
|
||||
}
|
||||
if len(admitter.backendIDs) != 1 || llmClient.calls != 1 {
|
||||
t.Fatalf("optional credential admission=%v generation=%d, want one each", admitter.backendIDs, llmClient.calls)
|
||||
}
|
||||
}
|
||||
if _, err := runner.RunPrepared(context.Background(), prepared); !errors.Is(err, ErrInvalidRequest) {
|
||||
t.Fatalf("execution outcome did not consume handle: %v", err)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -624,12 +624,12 @@ func validateAPIKey(apiKeyEnv string, apiKey string, apiKeyRequired bool) error
|
||||
if strings.TrimSpace(apiKey) != "" {
|
||||
return nil
|
||||
}
|
||||
if !apiKeyRequired {
|
||||
return nil
|
||||
}
|
||||
envName := strings.TrimSpace(apiKeyEnv)
|
||||
if envName == "" {
|
||||
if apiKeyRequired {
|
||||
return ErrAPIKeyRequired
|
||||
}
|
||||
return nil
|
||||
return ErrAPIKeyRequired
|
||||
}
|
||||
if strings.TrimSpace(os.Getenv(envName)) == "" {
|
||||
return fmt.Errorf("%w: api key environment variable %q is not set", ErrAPIKeyEnvMissing, envName)
|
||||
|
||||
@@ -1787,22 +1787,26 @@ func TestRunnerRunAPIKeyEnvResolvesFromEnvironment(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunnerRunAPIKeyEnvMissingEnvironmentValueFailsClearly(t *testing.T) {
|
||||
func TestRunnerRunOptionalAPIKeyEnvMissingEnvironmentValueReachesLLM(t *testing.T) {
|
||||
const environmentName = "PROMPTKIT_MISSING_KEY"
|
||||
t.Setenv(environmentName, "")
|
||||
|
||||
promptRepo := &fakePromptRepo{def: promptDef(domain.FormatText, domain.ValidationNone, 0)}
|
||||
execRepo := &fakeExecutionProfileRepo{profiles: map[string]*domain.ExecutionProfile{
|
||||
"exec": {ID: "exec", Endpoint: "http://profile/v1", Model: "profile-model", APIKeyEnv: "PROMPTKIT_MISSING_KEY"},
|
||||
"exec": {ID: "exec", Endpoint: "http://profile/v1", Model: "profile-model", APIKeyEnv: environmentName},
|
||||
}}
|
||||
runner := NewRunner(promptRepo, execRepo, nil, defaultArtifactReader(), defaultRenderer(), &fakeLLM{resp: &domain.GenerateResponse{Content: "ok"}}, nil, nil)
|
||||
llmClient := &fakeLLM{resp: &domain.GenerateResponse{Content: "ok"}}
|
||||
runner := NewRunner(promptRepo, execRepo, nil, defaultArtifactReader(), defaultRenderer(), llmClient, nil, nil)
|
||||
|
||||
_, err := runner.Run(context.Background(), domain.RunRequest{PromptID: "p", ProfileID: "exec", Inputs: singleInputRef()})
|
||||
if !errors.Is(err, ErrInvalidRequest) {
|
||||
t.Fatalf("expected ErrInvalidRequest, got %v", err)
|
||||
result, err := runner.Run(context.Background(), domain.RunRequest{PromptID: "p", ProfileID: "exec", Inputs: singleInputRef()})
|
||||
if err != nil || result == nil {
|
||||
t.Fatalf("optional credential run = (%+v, %v), want success", result, err)
|
||||
}
|
||||
if !errors.Is(err, ErrAPIKeyEnvMissing) {
|
||||
t.Fatalf("expected ErrAPIKeyEnvMissing, got %v", err)
|
||||
if llmClient.calls != 1 {
|
||||
t.Fatalf("LLM calls = %d, want 1", llmClient.calls)
|
||||
}
|
||||
if !strings.Contains(err.Error(), "PROMPTKIT_MISSING_KEY") {
|
||||
t.Fatalf("expected missing env name in error, got %v", err)
|
||||
if llmClient.lastReq.Target.APIKeyEnv != environmentName {
|
||||
t.Fatalf("LLM api_key_env = %q, want %q", llmClient.lastReq.Target.APIKeyEnv, environmentName)
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -5,7 +5,6 @@ import (
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"os"
|
||||
"reflect"
|
||||
"strings"
|
||||
"sync"
|
||||
@@ -524,19 +523,25 @@ func TestPreparedExecutionCredentialCapacityAndTimingBoundaries(t *testing.T) {
|
||||
engine, err := promptkit.NewEngine(
|
||||
promptkit.Config{},
|
||||
promptkit.WithPromptFS(contractPromptFS("prepared", "profile", "content"), "."),
|
||||
promptkit.WithProfileFS(preparedCredentialProfileSource(environmentName), "."),
|
||||
promptkit.WithProfiles(promptkit.Profile{
|
||||
ID: "profile",
|
||||
Endpoint: "http://example.test/v1",
|
||||
Model: "model",
|
||||
APIKeyRequired: true,
|
||||
}),
|
||||
promptkit.WithLLMClient(client),
|
||||
)
|
||||
if err != nil {
|
||||
t.Fatalf("construct credential engine: %v", err)
|
||||
}
|
||||
prepared, err := engine.PrepareExecution(context.Background(), promptkit.RunRequest{PromptID: "prepared"})
|
||||
prepared, err := engine.PrepareExecution(context.Background(), promptkit.RunRequest{
|
||||
PromptID: "prepared",
|
||||
Execution: &promptkit.ExecutionTargetOverride{APIKeyEnv: environmentName},
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("prepare credential execution: %v", err)
|
||||
}
|
||||
if err := os.Unsetenv(environmentName); err != nil {
|
||||
t.Fatalf("unset credential environment: %v", err)
|
||||
}
|
||||
t.Setenv(environmentName, "")
|
||||
|
||||
result, err := engine.RunPrepared(context.Background(), prepared)
|
||||
if result != nil ||
|
||||
@@ -755,16 +760,6 @@ model: ` + model + `
|
||||
}
|
||||
}
|
||||
|
||||
func preparedCredentialProfileSource(environmentName string) fstest.MapFS {
|
||||
return fstest.MapFS{
|
||||
"profile.yaml": &fstest.MapFile{Data: []byte(`id: profile
|
||||
endpoint: http://example.test/v1
|
||||
model: model
|
||||
api_key_env: ` + environmentName + `
|
||||
`)},
|
||||
}
|
||||
}
|
||||
|
||||
func preparedSchemaSource() fstest.MapFS {
|
||||
return fstest.MapFS{
|
||||
"schema.json": &fstest.MapFile{Data: []byte(`{
|
||||
|
||||
@@ -784,9 +784,12 @@ func TestBackendRegistrationRejectsInvalidAndDuplicateDefinitions(t *testing.T)
|
||||
{ID: " custom ", Endpoint: "http://one.example/v1"},
|
||||
{ID: "custom", Endpoint: "http://two.example/v1"},
|
||||
}},
|
||||
{name: "reserved built-in id", backends: []promptkit.Backend{{
|
||||
{name: "reserved OpenRouter ID", backends: []promptkit.Backend{{
|
||||
ID: promptkit.BackendOpenRouter, Endpoint: "http://replacement.example/v1",
|
||||
}}},
|
||||
{name: "reserved Rakestrawhome ID", backends: []promptkit.Backend{{
|
||||
ID: promptkit.BackendRakestrawHome, Endpoint: "http://replacement.example/v1",
|
||||
}}},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
|
||||
23
types.go
23
types.go
@@ -322,7 +322,10 @@ type ExecutionTarget struct {
|
||||
// ReasoningEffort is the effective opaque provider-specific reasoning
|
||||
// setting. An empty value instructs model clients to omit reasoning.
|
||||
ReasoningEffort string `json:"reasoning_effort"`
|
||||
// APIKeyEnv is an environment-variable name, not its credential value.
|
||||
// APIKeyEnv is the resolved name of an optional environment lookup source,
|
||||
// not its credential value. The built-in client omits Authorization when no
|
||||
// usable direct or environment credential is available; injected clients may
|
||||
// resolve this metadata differently.
|
||||
APIKeyEnv string `json:"api_key_env"`
|
||||
// ExtraParams contains copied JSON-compatible provider parameters.
|
||||
ExtraParams map[string]any `json:"extra_params"`
|
||||
@@ -426,9 +429,10 @@ type ExecutionTargetOverride struct {
|
||||
// inherited value and disables reasoning for this run. Non-blank values
|
||||
// are opaque and are not validated against a fixed vocabulary.
|
||||
ReasoningEffort *string
|
||||
// APIKeyEnv replaces the profile or backend environment-variable name when
|
||||
// non-blank. A direct RunRequest.APIKey still takes precedence over
|
||||
// environment lookup.
|
||||
// APIKeyEnv replaces the profile or backend optional environment lookup
|
||||
// source when non-blank. A direct RunRequest.APIKey still takes precedence.
|
||||
// The built-in client omits Authorization when neither source has a usable
|
||||
// value; injected clients may resolve this metadata differently.
|
||||
APIKeyEnv string
|
||||
// ExtraParams, when non-empty, replaces the complete profile or backend map.
|
||||
// Values must be JSON-compatible: nil, booleans, finite numbers, strings,
|
||||
@@ -481,7 +485,8 @@ type Profile struct {
|
||||
ReasoningEffort string
|
||||
// APIKeyRequired clears a backend's inherited API-key environment name and
|
||||
// requires a non-blank RunRequest.APIKey unless the request explicitly
|
||||
// supplies ExecutionTargetOverride.APIKeyEnv. It does not store a credential.
|
||||
// supplies ExecutionTargetOverride.APIKeyEnv. When false, a named
|
||||
// environment source remains optional. It does not store a credential.
|
||||
APIKeyRequired bool
|
||||
// ExtraParams contains provider-specific JSON-compatible values. An empty
|
||||
// map inherits backend request defaults, when any. WithProfiles validates
|
||||
@@ -665,10 +670,10 @@ type StructuredOutputJSONSpec struct {
|
||||
// and retained copies. It is responsible for the cancellation behavior of any
|
||||
// work it starts and for synchronizing access to retained or shared data.
|
||||
//
|
||||
// A returned error makes Run or RunPrepared return ErrLLMGenerate while
|
||||
// preserving the client error through errors.Is. A nil response with a nil
|
||||
// error also produces ErrLLMGenerate. Promptkit copies the non-nil response
|
||||
// before returning from either method.
|
||||
// An arbitrary returned error makes Run or RunPrepared return ErrLLMGenerate
|
||||
// while preserving the client error through errors.Is rather than translating
|
||||
// it. A nil response with a nil error also produces ErrLLMGenerate. Promptkit
|
||||
// copies the non-nil response before returning from either method.
|
||||
type LLMClient interface {
|
||||
Generate(context.Context, GenerateRequest) (*GenerateResponse, error)
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user