Added support for the service_tier key in profiles

This commit is contained in:
2026-05-26 07:44:49 -05:00
parent 3f4fd230b9
commit c6c5e3cb69
16 changed files with 65 additions and 3 deletions

View File

@@ -157,6 +157,7 @@ max_tokens: 500
top_p: 1.0
timeout_seconds: 90
api_key_env: SCRIPTORIUM_API_KEY
service_tier: priority
```
Field reference:
@@ -168,6 +169,7 @@ Field reference:
- `max_tokens` (optional): `>= 0`
- `top_p` (optional): range `0..1`
- `timeout_seconds` (optional): `>= 0`
- `service_tier` (optional): provider-specific request tier such as OpenRouter `flex` or `priority`
- `reasoning_effort` (optional)
- `api_key_env` (optional)
- `extra_params` (optional map of strings)
@@ -181,7 +183,7 @@ Profile rules:
Current outbound request behavior:
- The OpenAI-compatible client currently serializes: `model`, `messages`, `temperature`, `max_tokens`, `top_p`, and optional `response_format` for `json_schema` prompts.
- The OpenAI-compatible client currently serializes: `model`, `messages`, `temperature`, `max_tokens`, `top_p`, `service_tier`, and optional `response_format` for `json_schema` prompts.
- `reasoning_effort` and `extra_params` are parsed and carried in effective settings, but are not currently serialized into outbound chat-completions requests.
## Schema Behavior

View File

@@ -46,6 +46,7 @@ Copyable request example file:
"max_tokens": 800,
"top_p": 1.0,
"timeout_seconds": 120,
"service_tier": "priority",
"reasoning_effort": "medium",
"api_key_env": "SCRIPTORIUM_API_KEY",
"extra_params": {
@@ -114,6 +115,7 @@ Response shape:
"max_tokens": 800,
"top_p": 1,
"timeout_seconds": 120,
"service_tier": "priority",
"reasoning_effort": "medium",
"api_key_env": "SCRIPTORIUM_API_KEY",
"extra_params": {

View File

@@ -30,8 +30,11 @@ Serialized JSON fields:
- `temperature` (only when non-zero)
- `max_tokens` (only when non-zero)
- `top_p` (only when non-zero)
- `service_tier` (only when non-empty)
- `response_format` (only when structured output is provided)
`service_tier` is provider-specific. OpenRouter currently documents request values such as `flex` and `priority`; Scriptorium forwards any non-empty configured value and lets the backend validate support.
Structured output is currently `json_schema` only, serialized as:
```json

View File

@@ -66,7 +66,7 @@ Primary app settings consumed by adapters:
Execution profile/request settings used through runner:
- `endpoint`, `model`, `temperature`, `max_tokens`, `top_p`, `timeout_seconds`, `api_key_env`, `reasoning_effort`, `extra_params`
- `endpoint`, `model`, `temperature`, `max_tokens`, `top_p`, `timeout_seconds`, `service_tier`, `api_key_env`, `reasoning_effort`, `extra_params`
## External Dependencies

View File

@@ -27,6 +27,7 @@ type modelOverrideRequestDTO struct {
MaxTokens int `json:"max_tokens,omitempty"`
TopP float64 `json:"top_p,omitempty"`
TimeoutSeconds int `json:"timeout_seconds,omitempty"`
ServiceTier string `json:"service_tier,omitempty"`
ReasoningEffort string `json:"reasoning_effort,omitempty"`
APIKeyEnv string `json:"api_key_env,omitempty"`
ExtraParams map[string]string `json:"extra_params,omitempty"`
@@ -75,6 +76,7 @@ type modelParamsDTO struct {
MaxTokens int `json:"max_tokens"`
TopP float64 `json:"top_p"`
TimeoutSeconds int `json:"timeout_seconds"`
ServiceTier string `json:"service_tier,omitempty"`
ReasoningEffort string `json:"reasoning_effort,omitempty"`
APIKeyEnv string `json:"api_key_env,omitempty"`
ExtraParams map[string]string `json:"extra_params,omitempty"`

View File

@@ -70,6 +70,7 @@ func (h *Handler) ServeHTTP(w http.ResponseWriter, r *http.Request) {
MaxTokens: req.Model.MaxTokens,
TopP: req.Model.TopP,
TimeoutSeconds: req.Model.TimeoutSeconds,
ServiceTier: req.Model.ServiceTier,
ReasoningEffort: req.Model.ReasoningEffort,
APIKeyEnv: req.Model.APIKeyEnv,
ExtraParams: req.Model.ExtraParams,
@@ -116,6 +117,7 @@ func (h *Handler) ServeHTTP(w http.ResponseWriter, r *http.Request) {
MaxTokens: res.EffectiveModelParams.MaxTokens,
TopP: res.EffectiveModelParams.TopP,
TimeoutSeconds: res.EffectiveModelParams.TimeoutSeconds,
ServiceTier: res.EffectiveModelParams.ServiceTier,
ReasoningEffort: res.EffectiveModelParams.ReasoningEffort,
APIKeyEnv: res.EffectiveModelParams.APIKeyEnv,
ExtraParams: res.EffectiveModelParams.ExtraParams,

View File

@@ -62,6 +62,7 @@ func TestHandlerPostRunsSuccessWithExplicitProfileID(t *testing.T) {
MaxTokens: 42,
TopP: 0.9,
TimeoutSeconds: 120,
ServiceTier: "priority",
APIKeyEnv: envName,
},
InputHashes: map[string]string{"transcript": "h1"},
@@ -81,7 +82,7 @@ func TestHandlerPostRunsSuccessWithExplicitProfileID(t *testing.T) {
"transcript": {"type": "file", "uri": "./t.md"}
},
"vars": {"k": "v"},
"model": {"model": "gpt-x", "timeout_seconds": 120, "api_key_env": "SCRIPTORIUM_API_KEY"}
"model": {"model": "gpt-x", "timeout_seconds": 120, "service_tier": "flex", "api_key_env": "SCRIPTORIUM_API_KEY"}
}`)
req := httptest.NewRequest(http.MethodPost, "/v1/runs", bytes.NewReader(body))
w := httptest.NewRecorder()
@@ -114,6 +115,9 @@ func TestHandlerPostRunsSuccessWithExplicitProfileID(t *testing.T) {
if modelParams["api_key_env"] != envName {
t.Fatalf("expected model_params.api_key_env=%q, got %#v", envName, modelParams["api_key_env"])
}
if modelParams["service_tier"] != "priority" {
t.Fatalf("expected model_params.service_tier=priority, got %#v", modelParams["service_tier"])
}
if strings.Contains(w.Body.String(), secret) {
t.Fatalf("response leaked raw API key value: %s", w.Body.String())
}
@@ -133,6 +137,9 @@ func TestHandlerPostRunsSuccessWithExplicitProfileID(t *testing.T) {
if r.last.Execution.TimeoutSeconds != 120 {
t.Fatalf("expected timeout_seconds override 120, got %#v", r.last.Execution)
}
if r.last.Execution.ServiceTier != "flex" {
t.Fatalf("expected service_tier override flex, got %#v", r.last.Execution)
}
}
func TestHandlerPostRunsSuccessUsingPromptDefaultProfile(t *testing.T) {

View File

@@ -146,6 +146,7 @@ type ExecutionProfile struct {
MaxTokens int `yaml:"max_tokens"`
TopP float64 `yaml:"top_p"`
TimeoutSeconds int `yaml:"timeout_seconds"`
ServiceTier string `yaml:"service_tier"`
ReasoningEffort string `yaml:"reasoning_effort"`
APIKeyEnv string `yaml:"api_key_env"`
ExtraParams map[string]string `yaml:"extra_params"`
@@ -159,6 +160,7 @@ type ExecutionTarget struct {
MaxTokens int `yaml:"max_tokens" json:"max_tokens"`
TopP float64 `yaml:"top_p" json:"top_p"`
TimeoutSeconds int `yaml:"timeout_seconds" json:"timeout_seconds"`
ServiceTier string `yaml:"service_tier" json:"service_tier"`
ReasoningEffort string `yaml:"reasoning_effort" json:"reasoning_effort"`
APIKeyEnv string `yaml:"api_key_env" json:"api_key_env"`
ExtraParams map[string]string `yaml:"extra_params" json:"extra_params"`

View File

@@ -106,6 +106,9 @@ func (textPreparedRunFormatter) Format(prepared *domain.PreparedRun) ([]byte, er
fmt.Fprintf(&b, " max_tokens: %d\n", target.MaxTokens)
fmt.Fprintf(&b, " top_p: %g\n", target.TopP)
fmt.Fprintf(&b, " timeout_seconds: %d\n", target.TimeoutSeconds)
if target.ServiceTier != "" {
fmt.Fprintf(&b, " service_tier: %s\n", target.ServiceTier)
}
if target.ReasoningEffort != "" {
fmt.Fprintf(&b, " reasoning_effort: %s\n", target.ReasoningEffort)
}

View File

@@ -28,6 +28,7 @@ func TestTextFormatterIncludesPreparedRunDetails(t *testing.T) {
"max_tokens: 256",
"top_p: 0.8",
"timeout_seconds: 45",
"service_tier: priority",
"reasoning_effort: medium",
"api_key_env: SCRIPTORIUM_API_KEY",
"prompt_hash: prompt-hash",
@@ -168,6 +169,7 @@ func samplePreparedRun() *domain.PreparedRun {
MaxTokens: 256,
TopP: 0.8,
TimeoutSeconds: 45,
ServiceTier: "priority",
ReasoningEffort: "medium",
APIKeyEnv: "SCRIPTORIUM_API_KEY",
},

View File

@@ -113,6 +113,9 @@ func (c *OpenAICompatibleClient) Generate(ctx context.Context, req domain.Genera
if req.Target.TopP != 0 {
wireReq.TopP = &req.Target.TopP
}
if strings.TrimSpace(req.Target.ServiceTier) != "" {
wireReq.ServiceTier = req.Target.ServiceTier
}
if req.StructuredOutput != nil {
responseFormat, err := toOpenAIResponseFormat(req.StructuredOutput)
if err != nil {
@@ -193,6 +196,7 @@ type openAIChatRequest struct {
Temperature *float64 `json:"temperature,omitempty"`
MaxTokens *int `json:"max_tokens,omitempty"`
TopP *float64 `json:"top_p,omitempty"`
ServiceTier string `json:"service_tier,omitempty"`
ResponseFormat *openAIResponseFormat `json:"response_format,omitempty"`
}

View File

@@ -61,6 +61,7 @@ func TestOpenAICompatibleClientGenerateSuccess(t *testing.T) {
Temperature: 0.4,
MaxTokens: 123,
TopP: 0.7,
ServiceTier: "priority",
APIKeyEnv: "SCRIPTORIUM_TEST_API_KEY",
},
StructuredOutput: &domain.StructuredOutputSpec{
@@ -95,6 +96,9 @@ func TestOpenAICompatibleClientGenerateSuccess(t *testing.T) {
if got, ok := obs.Body["model"].(string); !ok || got != "gpt-test" {
t.Fatalf("unexpected model payload: %#v", obs.Body["model"])
}
if got, ok := obs.Body["service_tier"].(string); !ok || got != "priority" {
t.Fatalf("unexpected service_tier payload: %#v", obs.Body["service_tier"])
}
msgs, ok := obs.Body["messages"].([]any)
if !ok || len(msgs) != 2 {
@@ -157,6 +161,9 @@ func TestOpenAICompatibleClientOmitsResponseFormatWhenNoStructuredOutput(t *test
if _, exists := observedBody["response_format"]; exists {
t.Fatalf("expected response_format omitted, got %#v", observedBody["response_format"])
}
if _, exists := observedBody["service_tier"]; exists {
t.Fatalf("expected service_tier omitted, got %#v", observedBody["service_tier"])
}
}
func TestOpenAICompatibleClientNoAuthorizationHeaderWhenNoAPIKey(t *testing.T) {

View File

@@ -59,6 +59,9 @@ func TestFilesystemRepository_GetProfile(t *testing.T) {
if p.ReasoningEffort != "medium" {
t.Fatalf("unexpected reasoning_effort: %q", p.ReasoningEffort)
}
if p.ServiceTier != "priority" {
t.Fatalf("unexpected service_tier: %q", p.ServiceTier)
}
})
t.Run("valid nested profile", func(t *testing.T) {

View File

@@ -2,6 +2,7 @@ id: local-secure
endpoint: http://localhost:8000/v1
model: gpt-4o-mini
api_key_env: SCRIPTORIUM_API_KEY
service_tier: priority
reasoning_effort: medium
extra_params:
provider: local

View File

@@ -342,6 +342,9 @@ func mergeExecutionTarget(base domain.ExecutionTarget, override domain.Execution
if override.TimeoutSeconds != 0 {
out.TimeoutSeconds = override.TimeoutSeconds
}
if strings.TrimSpace(override.ServiceTier) != "" {
out.ServiceTier = override.ServiceTier
}
if strings.TrimSpace(override.ReasoningEffort) != "" {
out.ReasoningEffort = override.ReasoningEffort
}
@@ -396,6 +399,7 @@ func executionProfileToTarget(p *domain.ExecutionProfile) domain.ExecutionTarget
MaxTokens: p.MaxTokens,
TopP: p.TopP,
TimeoutSeconds: p.TimeoutSeconds,
ServiceTier: p.ServiceTier,
ReasoningEffort: p.ReasoningEffort,
APIKeyEnv: p.APIKeyEnv,
ExtraParams: cp,

View File

@@ -257,6 +257,7 @@ func TestRunnerPrepareRuntimeOverrideBeatsSelectedProfileValue(t *testing.T) {
MaxTokens: 500,
TopP: 0.9,
TimeoutSeconds: 120,
ServiceTier: "priority",
},
}}
runner := NewRunner(promptRepo, execRepo, defaultArtifactReader(), defaultRenderer(), &fakeLLM{forbid: true}, nil)
@@ -270,6 +271,7 @@ func TestRunnerPrepareRuntimeOverrideBeatsSelectedProfileValue(t *testing.T) {
Model: "override-model",
Temperature: 0.7,
TimeoutSeconds: 30,
ServiceTier: "flex",
},
})
if err != nil {
@@ -281,6 +283,9 @@ func TestRunnerPrepareRuntimeOverrideBeatsSelectedProfileValue(t *testing.T) {
if prepared.EffectiveModelParams.TopP != 0.9 {
t.Fatalf("expected profile top_p to remain, got %v", prepared.EffectiveModelParams.TopP)
}
if prepared.EffectiveModelParams.ServiceTier != "flex" {
t.Fatalf("expected service_tier override to win, got %q", prepared.EffectiveModelParams.ServiceTier)
}
}
func TestRunnerPrepareSelectedProfileBeatsBuiltInDefault(t *testing.T) {
@@ -292,6 +297,7 @@ func TestRunnerPrepareSelectedProfileBeatsBuiltInDefault(t *testing.T) {
Model: "profile-model",
TopP: 0.8,
TimeoutSeconds: 90,
ServiceTier: "priority",
},
}}
runner := NewRunner(promptRepo, execRepo, defaultArtifactReader(), defaultRenderer(), &fakeLLM{forbid: true}, nil)
@@ -310,6 +316,9 @@ func TestRunnerPrepareSelectedProfileBeatsBuiltInDefault(t *testing.T) {
if prepared.EffectiveModelParams.TimeoutSeconds != 90 {
t.Fatalf("expected profile timeout to beat default, got %d", prepared.EffectiveModelParams.TimeoutSeconds)
}
if prepared.EffectiveModelParams.ServiceTier != "priority" {
t.Fatalf("expected profile service_tier to beat default, got %q", prepared.EffectiveModelParams.ServiceTier)
}
}
func TestRunnerPrepareFileBackedPromptBodiesRenderCorrectly(t *testing.T) {
@@ -726,6 +735,7 @@ func TestRunnerRunExplicitRuntimeOverrideBeatsSelectedProfileValue(t *testing.T)
MaxTokens: 500,
TopP: 0.9,
TimeoutSeconds: 120,
ServiceTier: "priority",
},
}}
llmClient := &fakeLLM{resp: &domain.GenerateResponse{Content: "ok"}}
@@ -740,6 +750,7 @@ func TestRunnerRunExplicitRuntimeOverrideBeatsSelectedProfileValue(t *testing.T)
Model: "override-model",
Temperature: 0.7,
TimeoutSeconds: 30,
ServiceTier: "flex",
},
})
if err != nil {
@@ -754,6 +765,9 @@ func TestRunnerRunExplicitRuntimeOverrideBeatsSelectedProfileValue(t *testing.T)
if res.EffectiveModelParams.TopP != 0.9 {
t.Fatalf("expected non-overridden profile top_p to remain, got %v", res.EffectiveModelParams.TopP)
}
if res.EffectiveModelParams.ServiceTier != "flex" {
t.Fatalf("expected service_tier override to win, got %q", res.EffectiveModelParams.ServiceTier)
}
}
func TestRunnerRunSelectedProfileBeatsBuiltInDefault(t *testing.T) {
@@ -765,6 +779,7 @@ func TestRunnerRunSelectedProfileBeatsBuiltInDefault(t *testing.T) {
Model: "profile-model",
TopP: 0.8,
TimeoutSeconds: 90,
ServiceTier: "priority",
},
}}
llmClient := &fakeLLM{resp: &domain.GenerateResponse{Content: "ok"}}
@@ -784,6 +799,9 @@ func TestRunnerRunSelectedProfileBeatsBuiltInDefault(t *testing.T) {
if res.EffectiveModelParams.TimeoutSeconds != 90 {
t.Fatalf("expected profile timeout to beat default, got %d", res.EffectiveModelParams.TimeoutSeconds)
}
if res.EffectiveModelParams.ServiceTier != "priority" {
t.Fatalf("expected profile service_tier to beat default, got %q", res.EffectiveModelParams.ServiceTier)
}
}
func TestRunnerRunBuiltInDefaultsUsedWhenProfileOmitsOptionalFields(t *testing.T) {