Enable structural output repair by default
This commit is contained in:
@@ -42,4 +42,4 @@ output:
|
|||||||
format: json
|
format: json
|
||||||
validation_mode: json_schema
|
validation_mode: json_schema
|
||||||
schema_path: dnd_combat_turns_llm.v1.json
|
schema_path: dnd_combat_turns_llm.v1.json
|
||||||
repair_attempts: 0
|
repair_attempts: 1
|
||||||
|
|||||||
@@ -50,4 +50,4 @@ output:
|
|||||||
format: json
|
format: json
|
||||||
validation_mode: json_schema
|
validation_mode: json_schema
|
||||||
schema_path: dnd_enemy_events_llm.v1.json
|
schema_path: dnd_enemy_events_llm.v1.json
|
||||||
repair_attempts: 0
|
repair_attempts: 1
|
||||||
|
|||||||
@@ -42,4 +42,4 @@ output:
|
|||||||
format: json
|
format: json
|
||||||
validation_mode: json_schema
|
validation_mode: json_schema
|
||||||
schema_path: dnd_item_occurrences_llm.v1.json
|
schema_path: dnd_item_occurrences_llm.v1.json
|
||||||
repair_attempts: 0
|
repair_attempts: 1
|
||||||
|
|||||||
@@ -37,4 +37,4 @@ output:
|
|||||||
format: json
|
format: json
|
||||||
validation_mode: json_schema
|
validation_mode: json_schema
|
||||||
schema_path: dnd_item_registry_llm.v1.json
|
schema_path: dnd_item_registry_llm.v1.json
|
||||||
repair_attempts: 0
|
repair_attempts: 1
|
||||||
|
|||||||
@@ -27,4 +27,4 @@ output:
|
|||||||
format: json
|
format: json
|
||||||
validation_mode: json_schema
|
validation_mode: json_schema
|
||||||
schema_path: semantic_reconciliation_llm.v1.json
|
schema_path: semantic_reconciliation_llm.v1.json
|
||||||
repair_attempts: 0
|
repair_attempts: 1
|
||||||
|
|||||||
@@ -42,4 +42,4 @@ output:
|
|||||||
format: json
|
format: json
|
||||||
validation_mode: json_schema
|
validation_mode: json_schema
|
||||||
schema_path: dnd_location_occurrences_llm.v1.json
|
schema_path: dnd_location_occurrences_llm.v1.json
|
||||||
repair_attempts: 0
|
repair_attempts: 1
|
||||||
|
|||||||
@@ -37,4 +37,4 @@ output:
|
|||||||
format: json
|
format: json
|
||||||
validation_mode: json_schema
|
validation_mode: json_schema
|
||||||
schema_path: dnd_location_registry_llm.v1.json
|
schema_path: dnd_location_registry_llm.v1.json
|
||||||
repair_attempts: 0
|
repair_attempts: 1
|
||||||
|
|||||||
@@ -27,4 +27,4 @@ output:
|
|||||||
format: json
|
format: json
|
||||||
validation_mode: json_schema
|
validation_mode: json_schema
|
||||||
schema_path: semantic_reconciliation_llm.v1.json
|
schema_path: semantic_reconciliation_llm.v1.json
|
||||||
repair_attempts: 0
|
repair_attempts: 1
|
||||||
|
|||||||
@@ -42,4 +42,4 @@ output:
|
|||||||
format: json
|
format: json
|
||||||
validation_mode: json_schema
|
validation_mode: json_schema
|
||||||
schema_path: dnd_npc_occurrences_llm.v1.json
|
schema_path: dnd_npc_occurrences_llm.v1.json
|
||||||
repair_attempts: 0
|
repair_attempts: 1
|
||||||
|
|||||||
@@ -37,4 +37,4 @@ output:
|
|||||||
format: json
|
format: json
|
||||||
validation_mode: json_schema
|
validation_mode: json_schema
|
||||||
schema_path: dnd_npc_registry_llm.v1.json
|
schema_path: dnd_npc_registry_llm.v1.json
|
||||||
repair_attempts: 0
|
repair_attempts: 1
|
||||||
|
|||||||
@@ -27,4 +27,4 @@ output:
|
|||||||
format: json
|
format: json
|
||||||
validation_mode: json_schema
|
validation_mode: json_schema
|
||||||
schema_path: semantic_reconciliation_llm.v1.json
|
schema_path: semantic_reconciliation_llm.v1.json
|
||||||
repair_attempts: 0
|
repair_attempts: 1
|
||||||
|
|||||||
@@ -35,4 +35,4 @@ output:
|
|||||||
format: json
|
format: json
|
||||||
validation_mode: json_schema
|
validation_mode: json_schema
|
||||||
schema_path: dnd_scene_descriptions_llm.v1.json
|
schema_path: dnd_scene_descriptions_llm.v1.json
|
||||||
repair_attempts: 0
|
repair_attempts: 1
|
||||||
|
|||||||
@@ -31,4 +31,4 @@ output:
|
|||||||
format: json
|
format: json
|
||||||
validation_mode: json_schema
|
validation_mode: json_schema
|
||||||
schema_path: dnd_scenes_llm.v1.json
|
schema_path: dnd_scenes_llm.v1.json
|
||||||
repair_attempts: 0
|
repair_attempts: 1
|
||||||
|
|||||||
@@ -47,4 +47,4 @@ output:
|
|||||||
format: json
|
format: json
|
||||||
validation_mode: json_schema
|
validation_mode: json_schema
|
||||||
schema_path: dnd_spells_llm.v1.json
|
schema_path: dnd_spells_llm.v1.json
|
||||||
repair_attempts: 0
|
repair_attempts: 1
|
||||||
|
|||||||
@@ -24,4 +24,4 @@ output:
|
|||||||
format: json
|
format: json
|
||||||
validation_mode: json_schema
|
validation_mode: json_schema
|
||||||
schema_path: semantic_reconciliation_llm.v1.json
|
schema_path: semantic_reconciliation_llm.v1.json
|
||||||
repair_attempts: 0
|
repair_attempts: 1
|
||||||
|
|||||||
@@ -222,7 +222,7 @@ pipelines:
|
|||||||
| Field | Type | Default | Rules |
|
| Field | Type | Default | Rules |
|
||||||
| --- | --- | --- | --- |
|
| --- | --- | --- | --- |
|
||||||
| **llm_profile** | string | none | Optional non-empty default PromptKit profile ID for selected LLM-backed bindings and validators. An explicitly present blank value is invalid. |
|
| **llm_profile** | string | none | Optional non-empty default PromptKit profile ID for selected LLM-backed bindings and validators. An explicitly present blank value is invalid. |
|
||||||
| **structured_output_repair_attempts** | integer | prompt-owned | Optional structural-repair limit from 0 through 3 for selected LLM-backed bindings and validators. Omission leaves the prompt's declared policy in control; explicit 0 disables structural repair at that scope. |
|
| **structured_output_repair_attempts** | integer | prompt-owned (1 in maintained production prompts) | Optional structural-repair limit from 0 through 3 for selected LLM-backed bindings and validators. Omission leaves the prompt's declared policy in control; explicit 0 disables structural repair at that scope. |
|
||||||
| **input** | module binding | none | Required. |
|
| **input** | module binding | none | Required. |
|
||||||
| **chunk** | module binding | **generic** | Optional. |
|
| **chunk** | module binding | **generic** | Optional. |
|
||||||
| **output** | module binding | **json** | Optional. |
|
| **output** | module binding | **json** | Optional. |
|
||||||
@@ -288,7 +288,7 @@ extract:
|
|||||||
| --- | --- | --- | --- |
|
| --- | --- | --- | --- |
|
||||||
| **module** | string | none | Required for an object binding. Must be a registered compatible key. |
|
| **module** | string | none | Required for an object binding. Must be a registered compatible key. |
|
||||||
| **llm_profile** | string | none | Optional non-empty PromptKit profile ID for an LLM-backed binding. It overrides the pipeline default unless the run supplies **--llm-profile**. |
|
| **llm_profile** | string | none | Optional non-empty PromptKit profile ID for an LLM-backed binding. It overrides the pipeline default unless the run supplies **--llm-profile**. |
|
||||||
| **structured_output_repair_attempts** | integer | pipeline or prompt-owned | Optional structural-repair limit from 0 through 3 for an LLM-backed binding. It overrides the pipeline value; explicit 0 disables structural repair. |
|
| **structured_output_repair_attempts** | integer | pipeline or prompt-owned (1 in maintained production prompts) | Optional structural-repair limit from 0 through 3 for an LLM-backed binding. It overrides the pipeline value; explicit 0 disables structural repair. |
|
||||||
| **retries** | integer | 0 | Non-negative additional attempts for chunk, extract, merge, and normalize bindings. |
|
| **retries** | integer | 0 | Non-negative additional attempts for chunk, extract, merge, and normalize bindings. |
|
||||||
| **options** | object | none | Must satisfy the selected module. |
|
| **options** | object | none | Must satisfy the selected module. |
|
||||||
| **references** | map | none | Valid only on chunk, extract, merge, and normalize bindings. |
|
| **references** | map | none | Valid only on chunk, extract, merge, and normalize bindings. |
|
||||||
|
|||||||
@@ -115,7 +115,7 @@ the conventional local backend.
|
|||||||
|
|
||||||
PromptKit API or format changes outside this boundary are not implicitly
|
PromptKit API or format changes outside this boundary are not implicitly
|
||||||
supported. Updating the pinned version requires reviewing the adapter and
|
supported. Updating the pinned version requires reviewing the adapter and
|
||||||
profile/configuration contracts against the upstream documentation.
|
profile/configuration contracts against the upstream documentation. Maintained
|
||||||
|
production prompts use PromptKit's bounded structural-repair contract; their
|
||||||
This dependency update does not change any production prompt's declared
|
current declaration is one additional repair attempt. Notarius retains the
|
||||||
`repair_attempts` budget.
|
transport-neutral boundary and does not expose PromptKit types to modules.
|
||||||
|
|||||||
@@ -210,14 +210,29 @@ caller context takes precedence. The adapter does not retry capacity failures;
|
|||||||
the pipeline's existing binding attempt policy sees the operational error and
|
the pipeline's existing binding attempt policy sees the operational error and
|
||||||
decides whether to rerun the complete operation.
|
decides whether to rerun the complete operation.
|
||||||
|
|
||||||
Prompt-declared repair is executed within PromptKit’s structured-output flow.
|
PromptKit executes structural repair within its structured-output flow. The
|
||||||
The current production D&D prompt manifests set repair attempts to zero. That
|
maintained production prompt manifests declare one additional repair attempt.
|
||||||
setting does not replace pipeline retry behavior: a binding’s configured retry
|
When a resolved binding supplies a repair value, the adapter inspects the
|
||||||
count reruns its stage attempt after an error or rejection, and an exhausted
|
prompt, copies its complete output contract, changes only the repair limit, and
|
||||||
rejection is a recorded output rather than a provider error. The pipeline owns
|
passes that complete replacement contract to PromptKit. This preserves the
|
||||||
attempt lifecycle, validation chains, and retry diagnostics; see
|
prompt's output format, validation mode, schema, and provider structured-output
|
||||||
[Pipeline Internals](pipeline.md#validation-retries-and-output) and the
|
settings.
|
||||||
[binding reference](../config.md#module-bindings-and-validators).
|
|
||||||
|
A successful repair is an ordinary successful completion, not a warning. The
|
||||||
|
adapter reports PromptKit's actual repair count and its cumulative usage
|
||||||
|
directly, without adding the initial and corrective counts again. Debug prompt
|
||||||
|
material records the configured complete contract; debug response material
|
||||||
|
records the repaired response and actual validation result. If the repair
|
||||||
|
budget is exhausted, the adapter retains the final raw bytes and debug material
|
||||||
|
and reports `ErrInvalidStructuredOutput`. Generation failures during an initial
|
||||||
|
or corrective call remain provider-neutral operational errors with the same
|
||||||
|
redaction boundary.
|
||||||
|
|
||||||
|
Structural repair does not replace pipeline retry behavior: a binding's
|
||||||
|
configured retry count reruns its complete stage attempt after an error or
|
||||||
|
rejection. The pipeline owns attempt lifecycle, validation chains, and retry
|
||||||
|
diagnostics; see [Pipeline Internals](pipeline.md#validation-retries-and-output)
|
||||||
|
and the [binding reference](../config.md#module-bindings-and-validators).
|
||||||
|
|
||||||
## Timeout Ownership
|
## Timeout Ownership
|
||||||
|
|
||||||
|
|||||||
@@ -43,6 +43,9 @@ leaves the prompt-owned policy intact. An explicit repair value on a
|
|||||||
deterministic binding is rejected. Resolved bindings own copied repair values,
|
deterministic binding is rejected. Resolved bindings own copied repair values,
|
||||||
and these effective values are part of the digest, so execution and checkpoint
|
and these effective values are part of the digest, so execution and checkpoint
|
||||||
consumers do not repeat profile inheritance or configuration resolution.
|
consumers do not repeat profile inheritance or configuration resolution.
|
||||||
|
Each LLM request receives its own copy of that resolved value. PromptKit spends
|
||||||
|
it only for structural correction inside one completion; the runner's binding
|
||||||
|
retry policy remains the separate outer budget for complete stage attempts.
|
||||||
Configuration resolution supplies the selected profile and catalog; see
|
Configuration resolution supplies the selected profile and catalog; see
|
||||||
[Configuration Internals](configuration.md).
|
[Configuration Internals](configuration.md).
|
||||||
|
|
||||||
|
|||||||
@@ -281,12 +281,23 @@ selected binding, then the pipeline, then the prompt declaration; see
|
|||||||
[module bindings](config.md#module-bindings-and-validators). This is distinct
|
[module bindings](config.md#module-bindings-and-validators). This is distinct
|
||||||
from Notarius binding **retries**, which rerun the complete module operation
|
from Notarius binding **retries**, which rerun the complete module operation
|
||||||
and validation chain and do not consume or replenish the structural-repair
|
and validation chain and do not consume or replenish the structural-repair
|
||||||
limit.
|
limit. The maintained production prompts declare one repair attempt, paid only
|
||||||
|
after a structural failure. One structured completion with repair budget **R**
|
||||||
|
makes at most **R + 1** serial provider calls. If one stage attempt performs
|
||||||
|
**C** structured completions, a binding with **retries: N** has a maximum of
|
||||||
|
**(N + 1) * C * (R + 1)** provider calls; LLM-backed validators have their own
|
||||||
|
corresponding invocation counts and budgets. This is an upper bound, not a
|
||||||
|
promise that every call reaches a provider.
|
||||||
|
|
||||||
Timeouts are layered. Caller cancellation is the outer authority. A positive
|
Timeouts are layered. Caller cancellation is the outer authority. A positive
|
||||||
effective generation timeout adds an inner request deadline, while zero
|
effective generation timeout adds an inner request deadline, while zero
|
||||||
disables only that generation deadline. The HTTP client timeout remains a
|
disables only that generation deadline. The HTTP client timeout remains a
|
||||||
transport-wide cap. Notarius does not add another timeout around PromptKit.
|
transport-wide cap. Notarius does not add another timeout around PromptKit.
|
||||||
|
Repairs are serial within the same caller context, so their worst-case latency
|
||||||
|
and cost follow the provider-call bound above; provision run deadlines and
|
||||||
|
provider budgets accordingly. Credentials remain optional unless the selected
|
||||||
|
PromptKit profile requires one, in which case preparation fails before a
|
||||||
|
provider call when its configured credential is unavailable.
|
||||||
The pinned upstream boundary and profile-format links are in
|
The pinned upstream boundary and profile-format links are in
|
||||||
[PromptKit Integration](integrations/pkg-promptkit.md).
|
[PromptKit Integration](integrations/pkg-promptkit.md).
|
||||||
|
|
||||||
@@ -303,6 +314,11 @@ positive value makes the effective active local-generation bound the smaller
|
|||||||
of **total_llm** and that local limit, so a local limit of four permits no more
|
of **total_llm** and that local limit, so a local limit of four permits no more
|
||||||
than four active local generations.
|
than four active local generations.
|
||||||
|
|
||||||
|
The Notarius scheduler admits one logical structured completion and holds that
|
||||||
|
permit while PromptKit performs its serial corrective calls. PromptKit applies
|
||||||
|
its selected-backend admission to each provider call; Notarius does not
|
||||||
|
reacquire a permit or add another scheduler for a repair.
|
||||||
|
|
||||||
For a positive local limit, PromptKit owns its default waiting capacity and
|
For a positive local limit, PromptKit owns its default waiting capacity and
|
||||||
admission behavior. When a PromptKit backend has admitted all active and queued
|
admission behavior. When a PromptKit backend has admitted all active and queued
|
||||||
work, a new call fails as capacity exhaustion before generation. The adapter
|
work, a new call fails as capacity exhaustion before generation. The adapter
|
||||||
|
|||||||
@@ -195,6 +195,10 @@ The caller of the LLM owns prompt selection, prompt inputs, response schema,
|
|||||||
and interpretation of structured output. Provider adapters do not own source-
|
and interpretation of structured output. Provider adapters do not own source-
|
||||||
or domain-specific prompt logic.
|
or domain-specific prompt logic.
|
||||||
|
|
||||||
|
PromptKit owns bounded structural correction within one structured completion.
|
||||||
|
Notarius owns outer stage attempts, semantic validation, and acceptance policy;
|
||||||
|
the two budgets must remain separate.
|
||||||
|
|
||||||
When a model selects an application entity, callers must supply a contextual
|
When a model selects an application entity, callers must supply a contextual
|
||||||
selection and deterministically attach the opaque application identity whenever
|
selection and deterministically attach the opaque application identity whenever
|
||||||
the selection resolves exactly. Models do not receive or reproduce opaque
|
the selection resolves exactly. Models do not receive or reproduce opaque
|
||||||
|
|||||||
@@ -676,7 +676,7 @@ git diff --check
|
|||||||
- Current configuration, pipeline, and operations documentation matches the
|
- Current configuration, pipeline, and operations documentation matches the
|
||||||
implemented behavior.
|
implemented behavior.
|
||||||
|
|
||||||
## Stage 10: Enable The Default, Finish Documentation, And Verify The Feature
|
## Stage 10: Enable The Default, Finish Documentation, And Verify The Feature ✅
|
||||||
|
|
||||||
### Goal
|
### Goal
|
||||||
|
|
||||||
|
|||||||
@@ -19,6 +19,8 @@ import (
|
|||||||
"testing/fstest"
|
"testing/fstest"
|
||||||
"time"
|
"time"
|
||||||
|
|
||||||
|
"gopkg.in/yaml.v3"
|
||||||
|
|
||||||
"gitea.maximumdirect.net/eric/notarius/internal/core/artifacts"
|
"gitea.maximumdirect.net/eric/notarius/internal/core/artifacts"
|
||||||
"gitea.maximumdirect.net/eric/notarius/internal/core/config"
|
"gitea.maximumdirect.net/eric/notarius/internal/core/config"
|
||||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/chunkmap"
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/chunkmap"
|
||||||
@@ -331,27 +333,53 @@ func TestProductionPromptAssetsPrepareWithoutProviderCredentials(t *testing.T) {
|
|||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("production prompt engine: %v", err)
|
t.Fatalf("production prompt engine: %v", err)
|
||||||
}
|
}
|
||||||
inputs := map[string]promptkit.ArtifactRef{
|
promptFS, err := components.assets.PromptFS()
|
||||||
"candidates": promptkit.Inline(`{"candidates":[{"candidate_id":1,"label":"Alias","source_refs":[{"start_unit_id":1,"end_unit_id":1}]}]}`),
|
if err != nil {
|
||||||
"transcript": promptkit.Inline(`{"windows":[{"units":[]}]}`),
|
t.Fatalf("production prompt assets: %v", err)
|
||||||
}
|
}
|
||||||
for _, prompt := range []struct {
|
type manifest struct {
|
||||||
id string
|
ID string `yaml:"id"`
|
||||||
version string
|
Version string `yaml:"version"`
|
||||||
}{
|
Inputs []struct {
|
||||||
{id: npcnormalize.PromptID, version: npcnormalize.PromptVersion},
|
Name string `yaml:"name"`
|
||||||
{id: itemregistrynormalize.PromptID, version: itemregistrynormalize.PromptVersion},
|
} `yaml:"inputs"`
|
||||||
{id: locationnormalize.PromptID, version: locationnormalize.PromptVersion},
|
}
|
||||||
} {
|
preparedPrompts := 0
|
||||||
|
if err := fs.WalkDir(promptFS, ".", func(path string, entry fs.DirEntry, walkErr error) error {
|
||||||
|
if walkErr != nil {
|
||||||
|
return walkErr
|
||||||
|
}
|
||||||
|
if entry.IsDir() || filepath.Base(path) != "prompt.yaml" {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
data, err := fs.ReadFile(promptFS, path)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
var prompt manifest
|
||||||
|
if err := yaml.Unmarshal(data, &prompt); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
inputs := make(map[string]promptkit.ArtifactRef, len(prompt.Inputs))
|
||||||
|
for _, input := range prompt.Inputs {
|
||||||
|
inputs[input.Name] = promptkit.Inline(`{}`)
|
||||||
|
}
|
||||||
prepared, err := engine.Prepare(context.Background(), promptkit.RunRequest{
|
prepared, err := engine.Prepare(context.Background(), promptkit.RunRequest{
|
||||||
PromptID: prompt.id, PromptVersion: prompt.version, ProfileID: "assembled-prompt-test", Inputs: inputs,
|
PromptID: prompt.ID, PromptVersion: prompt.Version, ProfileID: "assembled-prompt-test", Inputs: inputs,
|
||||||
})
|
})
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("prepare production prompt %q: %v", prompt.id, err)
|
return fmt.Errorf("prepare production prompt %q: %w", prompt.ID, err)
|
||||||
}
|
}
|
||||||
if prepared.OutputContract.SchemaPath != filepath.Base(semanticreconcile.SchemaAssetPath) {
|
if prepared.OutputContract.RepairAttempts != 1 {
|
||||||
t.Fatalf("prompt %q schema = %q, want generic reconciliation schema", prompt.id, prepared.OutputContract.SchemaPath)
|
return fmt.Errorf("prompt %q repair attempts = %d, want 1", prompt.ID, prepared.OutputContract.RepairAttempts)
|
||||||
}
|
}
|
||||||
|
preparedPrompts++
|
||||||
|
return nil
|
||||||
|
}); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if preparedPrompts == 0 {
|
||||||
|
t.Fatal("prepared no production prompts")
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -4,6 +4,7 @@ import (
|
|||||||
"context"
|
"context"
|
||||||
"encoding/json"
|
"encoding/json"
|
||||||
"errors"
|
"errors"
|
||||||
|
"fmt"
|
||||||
"io"
|
"io"
|
||||||
"io/fs"
|
"io/fs"
|
||||||
"net/http"
|
"net/http"
|
||||||
@@ -922,12 +923,15 @@ func TestLLMProfileRecorderDistinguishesEffectiveTargets(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
func TestPromptKitClientValidationFailureReturnsError(t *testing.T) {
|
func TestPromptKitClientValidationFailureReturnsError(t *testing.T) {
|
||||||
client := newTestPromptKitClient(t, &fakePromptKitLLM{content: `{"bad":true}`})
|
attempts := 1
|
||||||
|
fake := &fakePromptKitLLM{responses: []promptkit.GenerateResponse{{Content: `{"bad":true}`}, {Content: `{"bad":true}`}}}
|
||||||
|
client := newTestPromptKitClient(t, fake)
|
||||||
|
|
||||||
var out map[string]any
|
var out map[string]any
|
||||||
resp, err := client.CompleteStructured(context.Background(), contracts.StructuredCompletionRequest{
|
resp, err := client.CompleteStructured(context.Background(), contracts.StructuredCompletionRequest{
|
||||||
PromptID: "adapter.test",
|
PromptID: "adapter.test",
|
||||||
SessionID: "session-123",
|
SessionID: "session-123",
|
||||||
|
StructuredOutputRepairAttempts: &attempts,
|
||||||
Inputs: contracts.LLMInputSet{
|
Inputs: contracts.LLMInputSet{
|
||||||
"transcript": contracts.NewLLMInputMaterial("transcript", "application/json", []byte(`{"source":true}`), "", ""),
|
"transcript": contracts.NewLLMInputMaterial("transcript", "application/json", []byte(`{"source":true}`), "", ""),
|
||||||
},
|
},
|
||||||
@@ -935,6 +939,9 @@ func TestPromptKitClientValidationFailureReturnsError(t *testing.T) {
|
|||||||
if err == nil || !errors.Is(err, contracts.ErrInvalidStructuredOutput) || !strings.Contains(err.Error(), "validation failed") {
|
if err == nil || !errors.Is(err, contracts.ErrInvalidStructuredOutput) || !strings.Contains(err.Error(), "validation failed") {
|
||||||
t.Fatalf("CompleteStructured() error = %v, want validation failure", err)
|
t.Fatalf("CompleteStructured() error = %v, want validation failure", err)
|
||||||
}
|
}
|
||||||
|
if got := atomic.LoadInt32(&fake.calls); got != 2 {
|
||||||
|
t.Fatalf("provider calls = %d, want exhausted repair budget", got)
|
||||||
|
}
|
||||||
if got := string(resp.Content); got != `{"bad":true}` {
|
if got := string(resp.Content); got != `{"bad":true}` {
|
||||||
t.Fatalf("response content = %q, want raw failed output", got)
|
t.Fatalf("response content = %q, want raw failed output", got)
|
||||||
}
|
}
|
||||||
@@ -946,6 +953,45 @@ func TestPromptKitClientValidationFailureReturnsError(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func TestPromptKitClientRepairsStructuredOutputAndReportsCumulativeUsage(t *testing.T) {
|
||||||
|
attempts := 1
|
||||||
|
fake := &fakePromptKitLLM{responses: []promptkit.GenerateResponse{
|
||||||
|
{Content: `{"bad":true}`, Usage: promptkit.TokenUsage{PromptTokens: 3, CompletionTokens: 5, TotalTokens: 8}},
|
||||||
|
{Content: `{"ok":true}`, Usage: promptkit.TokenUsage{PromptTokens: 7, CompletionTokens: 11, TotalTokens: 18}},
|
||||||
|
}}
|
||||||
|
client := newTestPromptKitClient(t, fake)
|
||||||
|
|
||||||
|
var out struct {
|
||||||
|
OK bool `json:"ok"`
|
||||||
|
}
|
||||||
|
response, err := client.CompleteStructured(context.Background(), contracts.StructuredCompletionRequest{
|
||||||
|
PromptID: "adapter.test",
|
||||||
|
StructuredOutputRepairAttempts: &attempts,
|
||||||
|
SessionID: "repair-test",
|
||||||
|
Inputs: contracts.LLMInputSet{
|
||||||
|
"transcript": contracts.NewLLMInputMaterial("transcript", "application/json", []byte(`{"source":true}`), "", ""),
|
||||||
|
},
|
||||||
|
}, &out)
|
||||||
|
if err != nil || !out.OK {
|
||||||
|
t.Fatalf("CompleteStructured() = (%#v, %v), want repaired success", response, err)
|
||||||
|
}
|
||||||
|
if got := atomic.LoadInt32(&fake.calls); got != 2 {
|
||||||
|
t.Fatalf("provider calls = %d, want initial generation and one repair", got)
|
||||||
|
}
|
||||||
|
if response.RepairAttempts != 1 || response.PromptTokens != 10 || response.CompletionTokens != 16 || response.TotalTokens != 26 {
|
||||||
|
t.Fatalf("response repair and usage = %#v, want one repair and PromptKit cumulative usage", response)
|
||||||
|
}
|
||||||
|
if response.Debug == nil || response.Debug.Prompt == nil || response.Debug.Response == nil {
|
||||||
|
t.Fatalf("debug = %#v, want prompt and response details", response.Debug)
|
||||||
|
}
|
||||||
|
if response.Debug.Prompt.OutputContract["repair_attempts"] != float64(1) ||
|
||||||
|
response.Debug.Response.Validation["repair_attempts"] != float64(1) ||
|
||||||
|
response.Debug.Response.Content != `{"ok":true}` ||
|
||||||
|
response.Debug.Response.Usage.TotalTokens != 26 {
|
||||||
|
t.Fatalf("debug repair result = %#v, want configured contract and repaired response", response.Debug)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func TestPromptKitClientDecodeFailureReturnsRawResponse(t *testing.T) {
|
func TestPromptKitClientDecodeFailureReturnsRawResponse(t *testing.T) {
|
||||||
client := newTestPromptKitClient(t, &fakePromptKitLLM{content: `{"ok":true}`})
|
client := newTestPromptKitClient(t, &fakePromptKitLLM{content: `{"ok":true}`})
|
||||||
|
|
||||||
@@ -1221,6 +1267,66 @@ func TestScheduledPromptKitClientBoundsConcurrentCalls(t *testing.T) {
|
|||||||
wg.Wait()
|
wg.Wait()
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func TestScheduledPromptKitClientHoldsPermitAcrossStructuredOutputRepair(t *testing.T) {
|
||||||
|
attempts := 1
|
||||||
|
fake := &fakePromptKitLLM{
|
||||||
|
responses: []promptkit.GenerateResponse{
|
||||||
|
{Content: `{"bad":true}`},
|
||||||
|
{Content: `{"ok":true}`},
|
||||||
|
{Content: `{"ok":true}`},
|
||||||
|
},
|
||||||
|
block: make(chan struct{}),
|
||||||
|
}
|
||||||
|
scheduler, err := NewScheduler(1)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("NewScheduler() error = %v", err)
|
||||||
|
}
|
||||||
|
scheduled := NewScheduledClient(newTestPromptKitClient(t, fake), scheduler)
|
||||||
|
|
||||||
|
firstDone := make(chan error, 1)
|
||||||
|
go func() {
|
||||||
|
var out map[string]any
|
||||||
|
_, callErr := scheduled.CompleteStructured(context.Background(), contracts.StructuredCompletionRequest{
|
||||||
|
PromptID: "adapter.test",
|
||||||
|
SessionID: "repair-session",
|
||||||
|
StructuredOutputRepairAttempts: &attempts,
|
||||||
|
Inputs: contracts.LLMInputSet{
|
||||||
|
"transcript": contracts.NewLLMInputMaterial("transcript", "application/json", []byte(`{"source":true}`), "", ""),
|
||||||
|
},
|
||||||
|
}, &out)
|
||||||
|
firstDone <- callErr
|
||||||
|
}()
|
||||||
|
waitForAtomicAtLeast(t, &fake.calls, 1)
|
||||||
|
|
||||||
|
secondDone := make(chan error, 1)
|
||||||
|
go func() {
|
||||||
|
var out map[string]any
|
||||||
|
_, callErr := scheduled.CompleteStructured(context.Background(), contracts.StructuredCompletionRequest{
|
||||||
|
PromptID: "adapter.test",
|
||||||
|
SessionID: "queued-session",
|
||||||
|
Inputs: contracts.LLMInputSet{
|
||||||
|
"transcript": contracts.NewLLMInputMaterial("transcript", "application/json", []byte(`{"source":true}`), "", ""),
|
||||||
|
},
|
||||||
|
}, &out)
|
||||||
|
secondDone <- callErr
|
||||||
|
}()
|
||||||
|
close(fake.block)
|
||||||
|
if err := <-firstDone; err != nil {
|
||||||
|
t.Fatalf("repaired completion error = %v", err)
|
||||||
|
}
|
||||||
|
if err := <-secondDone; err != nil {
|
||||||
|
t.Fatalf("queued completion error = %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
requests := fake.requestsSnapshot()
|
||||||
|
if len(requests) != 3 || requests[1].Prompt.SessionID != "repair-session" || requests[2].Prompt.SessionID != "queued-session" {
|
||||||
|
t.Fatalf("generation order = %#v, want repair before queued completion", requests)
|
||||||
|
}
|
||||||
|
if got := atomic.LoadInt32(&fake.maxInFlight); got > 1 {
|
||||||
|
t.Fatalf("max in-flight calls = %d, want one scheduled logical completion", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func TestPromptKitClientValidatesRequest(t *testing.T) {
|
func TestPromptKitClientValidatesRequest(t *testing.T) {
|
||||||
client := newTestPromptKitClient(t, &fakePromptKitLLM{content: `{"ok":true}`})
|
client := newTestPromptKitClient(t, &fakePromptKitLLM{content: `{"ok":true}`})
|
||||||
var out map[string]any
|
var out map[string]any
|
||||||
@@ -1375,11 +1481,13 @@ func (f *switchingPromptFS) resumePromptRead() {
|
|||||||
|
|
||||||
type fakePromptKitLLM struct {
|
type fakePromptKitLLM struct {
|
||||||
content string
|
content string
|
||||||
|
responses []promptkit.GenerateResponse
|
||||||
allowEmpty bool
|
allowEmpty bool
|
||||||
err error
|
err error
|
||||||
block chan struct{}
|
block chan struct{}
|
||||||
mu sync.Mutex
|
mu sync.Mutex
|
||||||
last promptkit.GenerateRequest
|
last promptkit.GenerateRequest
|
||||||
|
requests []promptkit.GenerateRequest
|
||||||
calls int32
|
calls int32
|
||||||
inFlight int32
|
inFlight int32
|
||||||
maxInFlight int32
|
maxInFlight int32
|
||||||
@@ -1394,8 +1502,9 @@ func (*credentialBearingProviderError) Error() string {
|
|||||||
func (f *fakePromptKitLLM) Generate(ctx context.Context, req promptkit.GenerateRequest) (*promptkit.GenerateResponse, error) {
|
func (f *fakePromptKitLLM) Generate(ctx context.Context, req promptkit.GenerateRequest) (*promptkit.GenerateResponse, error) {
|
||||||
f.mu.Lock()
|
f.mu.Lock()
|
||||||
f.last = req
|
f.last = req
|
||||||
|
f.requests = append(f.requests, req)
|
||||||
f.mu.Unlock()
|
f.mu.Unlock()
|
||||||
atomic.AddInt32(&f.calls, 1)
|
call := atomic.AddInt32(&f.calls, 1)
|
||||||
current := atomic.AddInt32(&f.inFlight, 1)
|
current := atomic.AddInt32(&f.inFlight, 1)
|
||||||
for {
|
for {
|
||||||
seen := atomic.LoadInt32(&f.maxInFlight)
|
seen := atomic.LoadInt32(&f.maxInFlight)
|
||||||
@@ -1414,6 +1523,14 @@ func (f *fakePromptKitLLM) Generate(ctx context.Context, req promptkit.GenerateR
|
|||||||
if f.err != nil {
|
if f.err != nil {
|
||||||
return nil, f.err
|
return nil, f.err
|
||||||
}
|
}
|
||||||
|
if len(f.responses) > 0 {
|
||||||
|
index := int(call - 1)
|
||||||
|
if index >= len(f.responses) {
|
||||||
|
return nil, fmt.Errorf("unexpected provider call %d", call)
|
||||||
|
}
|
||||||
|
response := f.responses[index]
|
||||||
|
return &response, nil
|
||||||
|
}
|
||||||
content := f.content
|
content := f.content
|
||||||
if content == "" && !f.allowEmpty {
|
if content == "" && !f.allowEmpty {
|
||||||
content = `{"ok":true}`
|
content = `{"ok":true}`
|
||||||
@@ -1444,3 +1561,9 @@ func (f *fakePromptKitLLM) lastRequest() promptkit.GenerateRequest {
|
|||||||
defer f.mu.Unlock()
|
defer f.mu.Unlock()
|
||||||
return f.last
|
return f.last
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func (f *fakePromptKitLLM) requestsSnapshot() []promptkit.GenerateRequest {
|
||||||
|
f.mu.Lock()
|
||||||
|
defer f.mu.Unlock()
|
||||||
|
return append([]promptkit.GenerateRequest(nil), f.requests...)
|
||||||
|
}
|
||||||
|
|||||||
@@ -45,8 +45,8 @@ func TestRegisterAssetsPreparesGenericPromptOffline(t *testing.T) {
|
|||||||
if prepared.SelectedProfileID != "semantic-reconciliation-test" {
|
if prepared.SelectedProfileID != "semantic-reconciliation-test" {
|
||||||
t.Fatalf("selected profile = %q, want explicit test profile", prepared.SelectedProfileID)
|
t.Fatalf("selected profile = %q, want explicit test profile", prepared.SelectedProfileID)
|
||||||
}
|
}
|
||||||
if contract := prepared.OutputContract; contract.SchemaPath != "semantic_reconciliation_llm.v1.json" || contract.RepairAttempts != 0 {
|
if contract := prepared.OutputContract; contract.SchemaPath != "semantic_reconciliation_llm.v1.json" || contract.RepairAttempts != 1 {
|
||||||
t.Fatalf("output contract = %#v, want generic schema without repair", contract)
|
t.Fatalf("output contract = %#v, want generic schema with one repair", contract)
|
||||||
}
|
}
|
||||||
if len(prepared.Messages) != 5 || prepared.Messages[0].Role != "system" {
|
if len(prepared.Messages) != 5 || prepared.Messages[0].Role != "system" {
|
||||||
t.Fatalf("prepared messages = %#v, want five ordered messages beginning with system", prepared.Messages)
|
t.Fatalf("prepared messages = %#v, want five ordered messages beginning with system", prepared.Messages)
|
||||||
|
|||||||
Reference in New Issue
Block a user