Translate LLM backend capacity failures
This commit is contained in:
@@ -19,7 +19,8 @@ Notarius relies on the root `promptkit` package to:
|
|||||||
- return rendered debug material, validated structured output, selected
|
- return rendered debug material, validated structured output, selected
|
||||||
profile, backend, effective model metadata, and token usage;
|
profile, backend, effective model metadata, and token usage;
|
||||||
- distinguish structured-output validation failure from execution failure; and
|
- distinguish structured-output validation failure from execution failure; and
|
||||||
- identify a missing explicit profile through `ErrProfileNotFound`.
|
- identify a missing explicit profile through `ErrProfileNotFound` and backend
|
||||||
|
admission exhaustion through `ErrCapacityExceeded`.
|
||||||
|
|
||||||
Notarius does not use PromptKit's optional `ArtifactReader`. It materializes
|
Notarius does not use PromptKit's optional `ArtifactReader`. It materializes
|
||||||
source and reference content itself and supplies owned inline artifacts at the
|
source and reference content itself and supplies owned inline artifacts at the
|
||||||
@@ -40,6 +41,13 @@ PromptKit's stable lower-case `effective_model_params` JSON, which may include
|
|||||||
`backend_id`. Notarius production configuration does not expose user-defined
|
`backend_id`. Notarius production configuration does not expose user-defined
|
||||||
PromptKit backend registration.
|
PromptKit backend registration.
|
||||||
|
|
||||||
|
Notarius retains its application-wide scheduled client around the PromptKit
|
||||||
|
adapter. PromptKit may apply a narrower limit for the selected backend;
|
||||||
|
endpoint-only profiles have no such backend limit. The adapter translates
|
||||||
|
PromptKit capacity rejection into the provider-neutral Notarius
|
||||||
|
`ErrLLMCapacityExceeded` contract and leaves retries to the calling pipeline
|
||||||
|
stage.
|
||||||
|
|
||||||
## Notarius Ownership
|
## Notarius Ownership
|
||||||
|
|
||||||
[LLM Runtime Internals](../internal/llm.md) describes how Notarius mounts
|
[LLM Runtime Internals](../internal/llm.md) describes how Notarius mounts
|
||||||
|
|||||||
@@ -81,6 +81,13 @@ worker counts cannot exceed the configured LLM limit. The configuration field
|
|||||||
and its effective default are owned by
|
and its effective default are owned by
|
||||||
[Configuration](../config.md#concurrency-output-cache-and-debug).
|
[Configuration](../config.md#concurrency-output-cache-and-debug).
|
||||||
|
|
||||||
|
PromptKit applies a second, independent admission limit when the selected
|
||||||
|
profile names a limited backend. It sits beneath the Notarius scheduled client,
|
||||||
|
so it may narrow but cannot expand the application-wide limit. Built-in
|
||||||
|
OpenRouter profiles select PromptKit's reserved backend and its upstream
|
||||||
|
capacity policy. Endpoint-only profiles do not select a PromptKit backend and
|
||||||
|
remain limited only by the Notarius scheduler.
|
||||||
|
|
||||||
## Prompt And Schema Assets
|
## Prompt And Schema Assets
|
||||||
|
|
||||||
An `AssetRegistry` collects prompt and schema filesystems from production module
|
An `AssetRegistry` collects prompt and schema filesystems from production module
|
||||||
@@ -132,6 +139,14 @@ failure, empty structured body, or decode failure as
|
|||||||
material when they exist. Provider failures remain operational errors rather
|
material when they exist. Provider failures remain operational errors rather
|
||||||
than output-validation failures.
|
than output-validation failures.
|
||||||
|
|
||||||
|
When PromptKit rejects backend admission before generation, the adapter maps
|
||||||
|
`promptkit.ErrCapacityExceeded` to
|
||||||
|
`contracts.ErrLLMCapacityExceeded`, retaining prompt context and a redacted
|
||||||
|
upstream diagnostic without exposing the PromptKit sentinel as a framework
|
||||||
|
contract. A canceled caller context takes precedence. The adapter does not
|
||||||
|
retry capacity failures; the pipeline's existing binding attempt policy sees
|
||||||
|
the operational error and decides whether to rerun the complete operation.
|
||||||
|
|
||||||
Prompt-declared repair is executed within PromptKit’s structured-output flow.
|
Prompt-declared repair is executed within PromptKit’s structured-output flow.
|
||||||
The current production D&D prompt manifests set repair attempts to zero. That
|
The current production D&D prompt manifests set repair attempts to zero. That
|
||||||
setting does not replace pipeline retry behavior: a binding’s configured retry
|
setting does not replace pipeline retry behavior: a binding’s configured retry
|
||||||
@@ -184,6 +199,8 @@ contents safe for general logging.
|
|||||||
sources, invalid asset registration, or a non-positive scheduler limit.
|
sources, invalid asset registration, or a non-positive scheduler limit.
|
||||||
- Preparation failures, unavailable explicit profiles, provider failures, and
|
- Preparation failures, unavailable explicit profiles, provider failures, and
|
||||||
context cancellation propagate to the calling stage with context.
|
context cancellation propagate to the calling stage with context.
|
||||||
|
- Backend admission exhaustion is a provider-neutral operational error and is
|
||||||
|
not classified as invalid structured output or validator rejection.
|
||||||
- Malformed or schema-invalid provider output is classified separately as
|
- Malformed or schema-invalid provider output is classified separately as
|
||||||
invalid structured output so the module or pipeline can apply its own retry
|
invalid structured output so the module or pipeline can apply its own retry
|
||||||
and rejection policy.
|
and rejection policy.
|
||||||
|
|||||||
@@ -217,9 +217,9 @@ Provider execution settings and the generation timeout come from the selected
|
|||||||
PromptKit profile. The invocation-only **--reasoning-effort** and
|
PromptKit profile. The invocation-only **--reasoning-effort** and
|
||||||
**--clear-reasoning-effort** controls may replace or clear that profile setting
|
**--clear-reasoning-effort** controls may replace or clear that profile setting
|
||||||
for all LLM-backed calls in one run without changing the profile. PromptKit
|
for all LLM-backed calls in one run without changing the profile. PromptKit
|
||||||
v0.2.0 does not add a provider retry loop;
|
v0.2.0 does not add a provider retry loop. Notarius binding retries rerun the
|
||||||
Notarius binding retries rerun the complete module operation and validation
|
complete module operation and validation chain as defined by
|
||||||
chain as defined by [module bindings](config.md#module-bindings-and-validators).
|
[module bindings](config.md#module-bindings-and-validators).
|
||||||
|
|
||||||
Timeouts are layered. Caller cancellation is the outer authority. A positive
|
Timeouts are layered. Caller cancellation is the outer authority. A positive
|
||||||
effective generation timeout adds an inner request deadline, while zero
|
effective generation timeout adds an inner request deadline, while zero
|
||||||
@@ -228,7 +228,19 @@ transport-wide cap. Notarius does not add another timeout around PromptKit.
|
|||||||
The pinned upstream boundary and profile-format links are in
|
The pinned upstream boundary and profile-format links are in
|
||||||
[PromptKit Integration](integrations/pkg-promptkit.md).
|
[PromptKit Integration](integrations/pkg-promptkit.md).
|
||||||
|
|
||||||
Concurrency limits are configuration contracts; see
|
Concurrency has two independent layers. Notarius **total_llm** is the
|
||||||
|
application-wide provider-call limit shared by all backends, modules, retries,
|
||||||
|
and validators. PromptKit may impose a narrower admission limit for the
|
||||||
|
selected backend. The effective active-generation bound is the intersection of
|
||||||
|
both limits and can therefore be lower than **total_llm**. Built-in OpenRouter
|
||||||
|
profiles use PromptKit's upstream backend limit; endpoint-only profiles have no
|
||||||
|
PromptKit backend limit and remain bounded by Notarius.
|
||||||
|
|
||||||
|
When a PromptKit backend has admitted all active and queued work, a new call
|
||||||
|
fails as capacity exhaustion before generation. The adapter does not retry it.
|
||||||
|
The calling stage's configured retry policy applies normally, and the run fails
|
||||||
|
if those attempts are exhausted. Caller cancellation remains authoritative.
|
||||||
|
Configuration contracts are documented under
|
||||||
[concurrency](config.md#concurrency-output-cache-and-debug). Extract-worker
|
[concurrency](config.md#concurrency-output-cache-and-debug). Extract-worker
|
||||||
limits and actual provider-call limits are independent. Notarius writes local
|
limits and actual provider-call limits are independent. Notarius writes local
|
||||||
filesystem state only; remote storage, archival, and retention automation are
|
filesystem state only; remote storage, archival, and retention automation are
|
||||||
|
|||||||
@@ -177,7 +177,8 @@ individual modules.
|
|||||||
The application-wide LLM scheduler bounds actual provider calls independently
|
The application-wide LLM scheduler bounds actual provider calls independently
|
||||||
of framework worker limits. Every LLM-backed module, retry, and validator uses
|
of framework worker limits. Every LLM-backed module, retry, and validator uses
|
||||||
the single injected scheduled client, including work performed by overlapping
|
the single injected scheduled client, including work performed by overlapping
|
||||||
lanes.
|
lanes. Provider runtime adapters may enforce a narrower backend-specific limit
|
||||||
|
beneath this mandatory application-wide scheduler.
|
||||||
|
|
||||||
## Configuration And Provenance
|
## Configuration And Provenance
|
||||||
|
|
||||||
|
|||||||
@@ -5,3 +5,7 @@ import "errors"
|
|||||||
// ErrInvalidStructuredOutput identifies a provider response that cannot satisfy
|
// ErrInvalidStructuredOutput identifies a provider response that cannot satisfy
|
||||||
// the caller's declared structured-output contract.
|
// the caller's declared structured-output contract.
|
||||||
var ErrInvalidStructuredOutput = errors.New("invalid structured output")
|
var ErrInvalidStructuredOutput = errors.New("invalid structured output")
|
||||||
|
|
||||||
|
// ErrLLMCapacityExceeded identifies backend admission exhaustion before model
|
||||||
|
// generation begins.
|
||||||
|
var ErrLLMCapacityExceeded = errors.New("LLM capacity exceeded")
|
||||||
|
|||||||
@@ -3,6 +3,7 @@ package llm
|
|||||||
import (
|
import (
|
||||||
"context"
|
"context"
|
||||||
"encoding/json"
|
"encoding/json"
|
||||||
|
"errors"
|
||||||
"fmt"
|
"fmt"
|
||||||
"net/http"
|
"net/http"
|
||||||
"regexp"
|
"regexp"
|
||||||
@@ -131,6 +132,14 @@ func (c *PromptKitClient) CompleteStructured(ctx context.Context, req contracts.
|
|||||||
if ctxErr := ctx.Err(); ctxErr != nil {
|
if ctxErr := ctx.Err(); ctxErr != nil {
|
||||||
return contracts.StructuredCompletionResponse{}, ctxErr
|
return contracts.StructuredCompletionResponse{}, ctxErr
|
||||||
}
|
}
|
||||||
|
if errors.Is(err, promptkit.ErrCapacityExceeded) {
|
||||||
|
return contracts.StructuredCompletionResponse{}, fmt.Errorf(
|
||||||
|
"run PromptKit prompt %q: %w: %v",
|
||||||
|
promptID,
|
||||||
|
contracts.ErrLLMCapacityExceeded,
|
||||||
|
redactPromptKitError(err),
|
||||||
|
)
|
||||||
|
}
|
||||||
return contracts.StructuredCompletionResponse{}, fmt.Errorf("run PromptKit prompt %q: %w", promptID, redactPromptKitError(err))
|
return contracts.StructuredCompletionResponse{}, fmt.Errorf("run PromptKit prompt %q: %w", promptID, redactPromptKitError(err))
|
||||||
}
|
}
|
||||||
if result == nil {
|
if result == nil {
|
||||||
|
|||||||
@@ -429,6 +429,89 @@ func TestPromptKitClientProviderFailureIncludesContextAndRedactsBearerToken(t *t
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func TestPromptKitClientTranslatesBackendCapacityExhaustion(t *testing.T) {
|
||||||
|
queueCapacity := 0
|
||||||
|
fake := &fakePromptKitLLM{
|
||||||
|
err: errors.New("provider failed with Bearer secret-token"),
|
||||||
|
block: make(chan struct{}),
|
||||||
|
}
|
||||||
|
client, err := NewPromptKitClient(PromptKitClientConfig{
|
||||||
|
Assets: newTestPromptKitAssets(t),
|
||||||
|
EngineOptions: []promptkit.Option{
|
||||||
|
promptkit.WithBackend(promptkit.Backend{
|
||||||
|
ID: "limited-backend",
|
||||||
|
Endpoint: "http://127.0.0.1:1/v1",
|
||||||
|
ConcurrencyLimit: 1,
|
||||||
|
QueueCapacity: &queueCapacity,
|
||||||
|
}),
|
||||||
|
promptkit.WithProfiles(promptkit.OpenAICompatibleProfile(promptkit.OpenAICompatibleProfileConfig{
|
||||||
|
ID: "limited-profile",
|
||||||
|
BackendID: "limited-backend",
|
||||||
|
Model: "limited-model",
|
||||||
|
})),
|
||||||
|
promptkit.WithLLMClient(fake),
|
||||||
|
},
|
||||||
|
})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("NewPromptKitClient() error = %v, want nil", err)
|
||||||
|
}
|
||||||
|
defer func() {
|
||||||
|
select {
|
||||||
|
case <-fake.block:
|
||||||
|
default:
|
||||||
|
close(fake.block)
|
||||||
|
}
|
||||||
|
}()
|
||||||
|
|
||||||
|
request := contracts.StructuredCompletionRequest{
|
||||||
|
PromptID: "adapter.direct-session",
|
||||||
|
ProfileID: "limited-profile",
|
||||||
|
Inputs: contracts.LLMInputSet{
|
||||||
|
"transcript": contracts.NewLLMInputMaterial("transcript", "application/json", []byte(`{"source":true}`), "", ""),
|
||||||
|
},
|
||||||
|
Vars: map[string]any{"custom": "value"},
|
||||||
|
}
|
||||||
|
firstResult := make(chan error, 1)
|
||||||
|
go func() {
|
||||||
|
var out map[string]any
|
||||||
|
_, callErr := client.CompleteStructured(context.Background(), request, &out)
|
||||||
|
firstResult <- callErr
|
||||||
|
}()
|
||||||
|
waitForAtomicAtLeast(t, &fake.calls, 1)
|
||||||
|
|
||||||
|
var out map[string]any
|
||||||
|
response, capacityErr := client.CompleteStructured(context.Background(), request, &out)
|
||||||
|
if len(response.Content) != 0 {
|
||||||
|
t.Fatalf("capacity response = %#v, want empty", response)
|
||||||
|
}
|
||||||
|
if !errors.Is(capacityErr, contracts.ErrLLMCapacityExceeded) {
|
||||||
|
t.Fatalf("capacity error = %v, want ErrLLMCapacityExceeded", capacityErr)
|
||||||
|
}
|
||||||
|
if errors.Is(capacityErr, contracts.ErrInvalidStructuredOutput) {
|
||||||
|
t.Fatalf("capacity error = %v, must not be invalid structured output", capacityErr)
|
||||||
|
}
|
||||||
|
if errors.Is(capacityErr, promptkit.ErrCapacityExceeded) {
|
||||||
|
t.Fatalf("capacity error exposes PromptKit sentinel: %v", capacityErr)
|
||||||
|
}
|
||||||
|
if !strings.Contains(capacityErr.Error(), `run PromptKit prompt "adapter.direct-session"`) ||
|
||||||
|
!strings.Contains(capacityErr.Error(), "backend capacity exceeded") {
|
||||||
|
t.Fatalf("capacity error = %q, want prompt context and upstream diagnostic", capacityErr)
|
||||||
|
}
|
||||||
|
if calls := atomic.LoadInt32(&fake.calls); calls != 1 {
|
||||||
|
t.Fatalf("provider calls after capacity rejection = %d, want 1", calls)
|
||||||
|
}
|
||||||
|
|
||||||
|
close(fake.block)
|
||||||
|
firstErr := <-firstResult
|
||||||
|
if firstErr == nil || strings.Contains(firstErr.Error(), "secret-token") ||
|
||||||
|
!strings.Contains(firstErr.Error(), "Bearer [REDACTED]") {
|
||||||
|
t.Fatalf("admitted provider error = %v, want redacted diagnostic", firstErr)
|
||||||
|
}
|
||||||
|
if calls := atomic.LoadInt32(&fake.calls); calls != 1 {
|
||||||
|
t.Fatalf("provider calls after release = %d, want no adapter retry", calls)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func TestPromptKitClientContextCancellationIsRespected(t *testing.T) {
|
func TestPromptKitClientContextCancellationIsRespected(t *testing.T) {
|
||||||
ctx, cancel := context.WithCancel(context.Background())
|
ctx, cancel := context.WithCancel(context.Background())
|
||||||
cancel()
|
cancel()
|
||||||
|
|||||||
Reference in New Issue
Block a user