This commit is contained in:
2026-08-23 12:40:59 +02:00
parent bd762d1ccf
commit 5340158228
11 changed files with 342 additions and 51 deletions
+13 -4
View File
@@ -30,8 +30,17 @@ var PopularCatalog = []PopularProvider{
Name: "openai", Name: "openai",
Label: "OpenAI", Label: "OpenAI",
BaseURL: "https://api.openai.com/v1", BaseURL: "https://api.openai.com/v1",
DefaultModel: "gpt-4o-mini", DefaultModel: "gpt-5.6-luna",
Models: []string{"gpt-4o-mini", "gpt-4o", "gpt-4.1-mini", "gpt-4.1"}, Models: []string{
"gpt-5.6-luna",
"gpt-5.6-terra",
"gpt-5.6-sol",
"gpt-5.6",
"gpt-4o-mini",
"gpt-4o",
"gpt-4.1-mini",
"gpt-4.1",
},
}, },
{ {
Name: "google", Name: "google",
@@ -65,8 +74,8 @@ var PopularCatalog = []PopularProvider{
Name: "openrouter", Name: "openrouter",
Label: "OpenRouter", Label: "OpenRouter",
BaseURL: "https://openrouter.ai/api/v1", BaseURL: "https://openrouter.ai/api/v1",
DefaultModel: "openai/gpt-4o-mini", DefaultModel: "openai/gpt-5.6-luna",
Models: []string{"openai/gpt-4o-mini", "anthropic/claude-sonnet-4", "google/gemini-2.0-flash-001"}, Models: []string{"openai/gpt-5.6-luna", "openai/gpt-5.6-terra", "openai/gpt-4o-mini", "anthropic/claude-sonnet-4", "google/gemini-2.0-flash-001"},
}, },
} }
+16 -5
View File
@@ -15,13 +15,18 @@ const (
// spend thousands of tokens in reasoning_content before message.content. // spend thousands of tokens in reasoning_content before message.content.
// Formula HTML JSON often needs a large completion budget; 4096 hit // Formula HTML JSON often needs a large completion budget; 4096 hit
// finish_reason=length with empty/truncated content in live enhance. // finish_reason=length with empty/truncated content in live enhance.
MaxTokensEnhance = 16384 // GPT-5.6 max_completion_tokens includes hidden reasoning tokens — keep
// this high so multi-section HTML descriptions are not truncated.
MaxTokensEnhance = 24576
// MaxTokensEnhanceRetry is the one-shot length-cap bump ceiling used by // MaxTokensEnhanceRetry is the one-shot length-cap bump ceiling used by
// OpenAIClient when finish_reason=length yields empty or unparseable JSON. // OpenAIClient when finish_reason=length yields empty or unparseable JSON.
MaxTokensEnhanceRetry = 32768 MaxTokensEnhanceRetry = 40960
MaxTokensSEO = 180 MaxTokensSEO = 180
MaxTokensCampaign = 650 MaxTokensCampaign = 650
MaxProductDescRunes = 400 // MaxProductDescRunes caps source description text fed into enhance prompts.
// Kept large enough for existing long/HTML descriptions so the model can
// rewrite factually instead of inventing from a truncated stub.
MaxProductDescRunes = 3500
MaxAttrKeys = 10 MaxAttrKeys = 10
MaxAttrValueRunes = 60 MaxAttrValueRunes = 60
MaxBrandInjectRunes = 500 MaxBrandInjectRunes = 500
@@ -32,7 +37,11 @@ const (
// CompleteOptions tunes a single chat completion for structured tasks. // CompleteOptions tunes a single chat completion for structured tasks.
type CompleteOptions struct { type CompleteOptions struct {
MaxTokens int MaxTokens int
Temperature float64 // 0 → client default (≤0.3 for structured) Temperature float64 // 0 → client default (≤0.3 for structured); omitted on GPT-5/o-series
// ReasoningEffort is Chat Completions reasoning_effort for GPT-5 / o-series.
// Empty → auto: "none" for tiny/probe budgets, "low" for content-heavy calls
// (long formula HTML) so visible tokens are not eaten by deep reasoning.
ReasoningEffort string
} }
// CompleterWithOptions is optional; OpenAIClient implements it. // CompleterWithOptions is optional; OpenAIClient implements it.
@@ -205,7 +214,9 @@ func CompactBrandPrompt(block string) string {
return truncateRunes(SanitizeText(block), MaxBrandInjectRunes) return truncateRunes(SanitizeText(block), MaxBrandInjectRunes)
} }
// ProductEnhanceUser builds a short user prompt for title/description enhance. // ProductEnhanceUser builds the compact user prompt for title/description enhance.
// Source description is capped at MaxProductDescRunes so long/HTML inputs stay available
// for factual rewrite (not invented from a short stub).
func ProductEnhanceUser(category, name, description string, attrs map[string]any) string { func ProductEnhanceUser(category, name, description string, attrs map[string]any) string {
var b strings.Builder var b strings.Builder
b.WriteString("Category: ") b.WriteString("Category: ")
+14 -2
View File
@@ -109,11 +109,23 @@ func TestCompleteJSON_returnsRetryError(t *testing.T) {
} }
func TestProductEnhanceUser_truncates(t *testing.T) { func TestProductEnhanceUser_truncates(t *testing.T) {
long := strings.Repeat("x", 2000) long := strings.Repeat("x", MaxProductDescRunes+500)
u := ProductEnhanceUser("Cat", "Name", long, map[string]any{"brand": "B"}) u := ProductEnhanceUser("Cat", "Name", long, map[string]any{"brand": "B"})
if len([]rune(u)) > 900 { // Category/Name/Attrs overhead + MaxProductDescRunes of description.
if len([]rune(u)) > MaxProductDescRunes+200 {
t.Fatalf("user too long: %d", len([]rune(u))) t.Fatalf("user too long: %d", len([]rune(u)))
} }
descIdx := strings.Index(u, "Description: ")
if descIdx < 0 {
t.Fatalf("missing Description: in %q", u)
}
descPart := u[descIdx+len("Description: "):]
if i := strings.Index(descPart, "\n"); i >= 0 {
descPart = descPart[:i]
}
if len([]rune(descPart)) > MaxProductDescRunes {
t.Fatalf("description runes=%d want <= %d", len([]rune(descPart)), MaxProductDescRunes)
}
if !strings.Contains(u, "brand") { if !strings.Contains(u, "brand") {
t.Fatalf("%s", u) t.Fatalf("%s", u)
} }
+5 -16
View File
@@ -223,13 +223,6 @@ func classifyOpenAITransportErr(model string, err error) (error, bool) {
return err, true return err, true
} }
type chatRequest struct {
Model string `json:"model"`
Messages []chatMessage `json:"messages"`
Temperature float64 `json:"temperature"`
MaxTokens int `json:"max_tokens,omitempty"`
}
type chatMessage struct { type chatMessage struct {
Role string `json:"role"` Role string `json:"role"`
Content string `json:"content"` Content string `json:"content"`
@@ -283,6 +276,7 @@ func (c *OpenAIClient) CompleteWithOptions(ctx context.Context, system, user str
temp = 0.3 temp = 0.3
} }
maxTok := opts.MaxTokens maxTok := opts.MaxTokens
effort := opts.ReasoningEffort
var lastErr error var lastErr error
for attempt := 0; attempt <= c.MaxRetries; attempt++ { for attempt := 0; attempt <= c.MaxRetries; attempt++ {
@@ -298,7 +292,7 @@ func (c *OpenAIClient) CompleteWithOptions(ctx context.Context, system, user str
if err := c.waitRate(ctx); err != nil { if err := c.waitRate(ctx); err != nil {
return Completion{}, err return Completion{}, err
} }
comp, retryable, err := c.doComplete(ctx, system, user, temp, maxTok) comp, retryable, err := c.doComplete(ctx, system, user, temp, maxTok, effort)
if err == nil { if err == nil {
return comp, nil return comp, nil
} }
@@ -471,16 +465,11 @@ func (c *OpenAIClient) doEmbed(ctx context.Context, texts []string) ([][]float32
return out, false, nil return out, false, nil
} }
func (c *OpenAIClient) doComplete(ctx context.Context, system, user string, temperature float64, maxTokens int) (Completion, bool, error) { func (c *OpenAIClient) doComplete(ctx context.Context, system, user string, temperature float64, maxTokens int, reasoningEffort string) (Completion, bool, error) {
reqBody := chatRequest{ reqBody := buildChatCompletionBody(c.Model, []chatMessage{
Model: c.Model,
Messages: []chatMessage{
{Role: "system", Content: system}, {Role: "system", Content: system},
{Role: "user", Content: user}, {Role: "user", Content: user},
}, }, temperature, maxTokens, reasoningEffort)
Temperature: temperature,
MaxTokens: maxTokens,
}
body, err := json.Marshal(reqBody) body, err := json.Marshal(reqBody)
if err != nil { if err != nil {
return Completion{}, false, err return Completion{}, false, err
@@ -0,0 +1,97 @@
package processing
import (
"strings"
)
// openAIChatUsesMaxCompletionTokens reports models that reject legacy max_tokens
// on Chat Completions (o-series + GPT-5 family including GPT-5.6).
// See https://developers.openai.com/api/reference/resources/chat
func openAIChatUsesMaxCompletionTokens(model string) bool {
return openAIReasoningChatModel(model)
}
// openAIChatOmitsCustomTemperature reports models that reject non-default
// temperature (only the API default of 1 is accepted, if temperature is sent).
func openAIChatOmitsCustomTemperature(model string) bool {
return openAIReasoningChatModel(model)
}
// openAIChatSupportsReasoningEffort reports Chat Completions models that accept
// top-level reasoning_effort (Responses API uses reasoning.effort instead).
func openAIChatSupportsReasoningEffort(model string) bool {
return openAIReasoningChatModel(model)
}
// openAIReasoningChatModel matches OpenAI reasoning / GPT-5(+).5/5.6 chat models
// and common snapshot aliases (e.g. gpt-5.6-luna-2026-…).
func openAIReasoningChatModel(model string) bool {
m := strings.ToLower(strings.TrimSpace(model))
if m == "" {
return false
}
// Strip provider prefixes used by OpenRouter-style gateways.
if i := strings.LastIndex(m, "/"); i >= 0 {
m = m[i+1:]
}
switch {
case strings.HasPrefix(m, "o1"),
strings.HasPrefix(m, "o3"),
strings.HasPrefix(m, "o4"),
strings.HasPrefix(m, "gpt-5"),
strings.HasPrefix(m, "chatgpt-o"):
return true
default:
return false
}
}
// defaultReasoningEffort picks Chat Completions reasoning_effort when unset.
// Long formula-HTML enhance uses "low" so max_completion_tokens prefers visible
// HTML over deep hidden reasoning. Tiny probes use "none" for speed/cost.
func defaultReasoningEffort(model string, maxTokens int, explicit string) string {
explicit = strings.ToLower(strings.TrimSpace(explicit))
if explicit != "" {
return explicit
}
if !openAIChatSupportsReasoningEffort(model) {
return ""
}
if maxTokens <= 0 || maxTokens < 512 {
return "none"
}
return "low"
}
// chatCompletionBody is the Chat Completions JSON body. Temperature is a pointer
// so reasoning models can omit it entirely (omitempty).
type chatCompletionBody struct {
Model string `json:"model"`
Messages []chatMessage `json:"messages"`
Temperature *float64 `json:"temperature,omitempty"`
MaxTokens int `json:"max_tokens,omitempty"`
MaxCompletionTokens int `json:"max_completion_tokens,omitempty"`
ReasoningEffort string `json:"reasoning_effort,omitempty"`
}
// buildChatCompletionBody maps CompleteOptions onto the correct Chat Completions
// fields for classic vs GPT-5 / o-series models.
func buildChatCompletionBody(model string, messages []chatMessage, temperature float64, maxTokens int, reasoningEffort string) chatCompletionBody {
body := chatCompletionBody{
Model: model,
Messages: messages,
}
if openAIChatUsesMaxCompletionTokens(model) {
if maxTokens > 0 {
body.MaxCompletionTokens = maxTokens
}
} else if maxTokens > 0 {
body.MaxTokens = maxTokens
}
if !openAIChatOmitsCustomTemperature(model) {
t := temperature
body.Temperature = &t
}
body.ReasoningEffort = defaultReasoningEffort(model, maxTokens, reasoningEffort)
return body
}
@@ -0,0 +1,123 @@
package processing
import (
"encoding/json"
"testing"
)
func TestOpenAIReasoningChatModel(t *testing.T) {
t.Parallel()
cases := []struct {
model string
want bool
}{
{"gpt-5.6-luna", true},
{"gpt-5.6-terra", true},
{"gpt-5.6-sol", true},
{"gpt-5.6", true},
{"GPT-5.4", true},
{"openai/gpt-5.6-luna", true},
{"o1-mini", true},
{"o3", true},
{"o4-mini", true},
{"gpt-4o-mini", false},
{"gpt-4.1", false},
{"mock-llm", false},
{"", false},
}
for _, tc := range cases {
if got := openAIReasoningChatModel(tc.model); got != tc.want {
t.Fatalf("openAIReasoningChatModel(%q)=%v want %v", tc.model, got, tc.want)
}
}
}
func TestBuildChatCompletionBody_gpt56Luna(t *testing.T) {
t.Parallel()
body := buildChatCompletionBody("gpt-5.6-luna", []chatMessage{
{Role: "system", Content: "sys"},
{Role: "user", Content: "ping"},
}, 0.2, MaxTokensEnhance, "low")
raw, err := json.Marshal(body)
if err != nil {
t.Fatal(err)
}
var m map[string]any
if err := json.Unmarshal(raw, &m); err != nil {
t.Fatal(err)
}
if _, ok := m["temperature"]; ok {
t.Fatalf("temperature must be omitted for gpt-5.6-luna, got %v", m["temperature"])
}
if _, ok := m["max_tokens"]; ok {
t.Fatalf("max_tokens must be omitted for gpt-5.6-luna, got %v", m["max_tokens"])
}
gotMax, _ := m["max_completion_tokens"].(float64)
if int(gotMax) != MaxTokensEnhance {
t.Fatalf("max_completion_tokens=%v want %d", gotMax, MaxTokensEnhance)
}
if m["reasoning_effort"] != "low" {
t.Fatalf("reasoning_effort=%v want low", m["reasoning_effort"])
}
}
func TestBuildChatCompletionBody_classicModel(t *testing.T) {
t.Parallel()
body := buildChatCompletionBody("gpt-4o-mini", []chatMessage{
{Role: "user", Content: "hi"},
}, 0.2, 350, "")
raw, err := json.Marshal(body)
if err != nil {
t.Fatal(err)
}
var m map[string]any
if err := json.Unmarshal(raw, &m); err != nil {
t.Fatal(err)
}
if _, ok := m["max_completion_tokens"]; ok {
t.Fatalf("max_completion_tokens unexpected for classic model: %v", m)
}
gotMax, _ := m["max_tokens"].(float64)
if int(gotMax) != 350 {
t.Fatalf("max_tokens=%v want 350", gotMax)
}
gotTemp, _ := m["temperature"].(float64)
if gotTemp != 0.2 {
t.Fatalf("temperature=%v want 0.2", gotTemp)
}
if _, ok := m["reasoning_effort"]; ok {
t.Fatalf("reasoning_effort unexpected for classic model: %v", m)
}
}
func TestBuildChatCompletionBody_probeDefaultsNoneEffort(t *testing.T) {
t.Parallel()
body := buildChatCompletionBody("gpt-5.6-luna", []chatMessage{
{Role: "user", Content: "ping"},
}, 0.2, 0, "")
if body.ReasoningEffort != "none" {
t.Fatalf("reasoning_effort=%q want none for probe", body.ReasoningEffort)
}
if body.MaxCompletionTokens != 0 || body.MaxTokens != 0 {
t.Fatalf("expected no token cap on probe, got %+v", body)
}
if body.Temperature != nil {
t.Fatalf("temperature must be nil for probe on gpt-5.6")
}
}
func TestDefaultReasoningEffort(t *testing.T) {
t.Parallel()
if got := defaultReasoningEffort("gpt-5.6-luna", MaxTokensEnhance, ""); got != "low" {
t.Fatalf("enhance default=%q", got)
}
if got := defaultReasoningEffort("gpt-5.6-luna", 0, ""); got != "none" {
t.Fatalf("probe default=%q", got)
}
if got := defaultReasoningEffort("gpt-5.6-luna", 100, "medium"); got != "medium" {
t.Fatalf("explicit=%q", got)
}
if got := defaultReasoningEffort("gpt-4o", 1000, ""); got != "" {
t.Fatalf("classic=%q", got)
}
}
@@ -179,6 +179,52 @@ func TestOpenAIClient_doComplete_emptyContentWithReasoningJSON(t *testing.T) {
} }
} }
func TestOpenAIClient_doComplete_gpt56LunaUsesMaxCompletionTokens(t *testing.T) {
t.Parallel()
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
var req map[string]any
_ = json.NewDecoder(r.Body).Decode(&req)
if _, ok := req["temperature"]; ok {
t.Errorf("temperature must be omitted for gpt-5.6-luna")
}
if _, ok := req["max_tokens"]; ok {
t.Errorf("max_tokens must be omitted for gpt-5.6-luna")
}
gotMax, _ := req["max_completion_tokens"].(float64)
if int(gotMax) != MaxTokensEnhance {
t.Errorf("max_completion_tokens=%v want %d", gotMax, MaxTokensEnhance)
}
if req["reasoning_effort"] != "low" {
t.Errorf("reasoning_effort=%v want low", req["reasoning_effort"])
}
_ = json.NewEncoder(w).Encode(map[string]any{
"model": "gpt-5.6-luna",
"choices": []map[string]any{{
"finish_reason": "stop",
"message": map[string]any{
"role": "assistant",
"content": `{"name":"Acme Monitor","description":"<h1>Acme Monitor</h1><p>IPS panel for office use.</p>"}`,
},
}},
"usage": map[string]int{"prompt_tokens": 20, "completion_tokens": 40, "total_tokens": 60},
})
}))
defer srv.Close()
c := NewOpenAIClient("test-key", srv.URL, "gpt-5.6-luna", 0, 1)
c.HTTPClient = srv.Client()
comp, err := c.CompleteWithOptions(context.Background(), "sys", "user", CompleteOptions{
MaxTokens: MaxTokensEnhance,
Temperature: DefaultStructuredTemp,
ReasoningEffort: "low",
})
if err != nil {
t.Fatal(err)
}
if !strings.Contains(comp.Text, "Acme Monitor") {
t.Fatalf("text=%q", comp.Text)
}
}
func TestOpenAIClient_doComplete_emptyContentLengthRetriesWithBudget(t *testing.T) { func TestOpenAIClient_doComplete_emptyContentLengthRetriesWithBudget(t *testing.T) {
t.Parallel() t.Parallel()
calls := 0 calls := 0
+1 -1
View File
@@ -11,7 +11,7 @@ import (
const maxPromptFieldRunes = 4000 const maxPromptFieldRunes = 4000
// maxOutputFieldRunes bounds stored / polled model text (titles + multi-section // maxOutputFieldRunes bounds stored / polled model text (titles + multi-section
// formula HTML descriptions). Must stay well above MaxTokensEnhance (~16k tokens) // formula HTML descriptions). Must stay well above MaxTokensEnhance (~24k tokens)
// so SanitizeOutput does not chop JSON completion bodies or A1 HTML mid-string. // so SanitizeOutput does not chop JSON completion bodies or A1 HTML mid-string.
const maxOutputFieldRunes = 120000 const maxOutputFieldRunes = 120000
+2
View File
@@ -802,6 +802,7 @@ func (e *Engine) enhance(ctx context.Context, in ProductInput, category string,
comp, obj, err := CompleteJSON(ctx, e.Completer, system, user, CompleteOptions{ comp, obj, err := CompleteJSON(ctx, e.Completer, system, user, CompleteOptions{
MaxTokens: MaxTokensEnhance, MaxTokens: MaxTokensEnhance,
Temperature: DefaultStructuredTemp, Temperature: DefaultStructuredTemp,
ReasoningEffort: "low",
}) })
elapsed := time.Since(started) elapsed := time.Since(started)
if err != nil { if err != nil {
@@ -893,6 +894,7 @@ func (e *Engine) enhance(ctx context.Context, in ProductInput, category string,
comp2, obj2, err2 := CompleteJSON(ctx, e.Completer, system, retryUser, CompleteOptions{ comp2, obj2, err2 := CompleteJSON(ctx, e.Completer, system, retryUser, CompleteOptions{
MaxTokens: MaxTokensEnhance, MaxTokens: MaxTokensEnhance,
Temperature: DefaultStructuredTemp, Temperature: DefaultStructuredTemp,
ReasoningEffort: "low",
}) })
elapsed += time.Since(retryStarted) elapsed += time.Since(retryStarted)
comp.PromptTokens += comp2.PromptTokens comp.PromptTokens += comp2.PromptTokens
+3 -3
View File
@@ -31,7 +31,7 @@ export const AI_ROLE_META: Record<
processing: { processing: {
label: "Processing", label: "Processing",
description: "Product pipeline chat/completions (titles, descriptions, enhance).", description: "Product pipeline chat/completions (titles, descriptions, enhance).",
modelPlaceholder: "gpt-4o-mini" modelPlaceholder: "gpt-5.6-luna"
}, },
vectorization: { vectorization: {
label: "Vectorization", label: "Vectorization",
@@ -42,13 +42,13 @@ export const AI_ROLE_META: Record<
label: "Docs / API", label: "Docs / API",
description: description:
"Future docs/API assistant slot — /docs Ask stays rule-based and must never call this role.", "Future docs/API assistant slot — /docs Ask stays rule-based and must never call this role.",
modelPlaceholder: "gpt-4o-mini" modelPlaceholder: "gpt-5.6-luna"
}, },
support: { support: {
label: "Support", label: "Support",
description: description:
"Ticket auto-reply AI fallback — configure provider/key/model here; enable delivery in Support knowledge → Auto-reply.", "Ticket auto-reply AI fallback — configure provider/key/model here; enable delivery in Support knowledge → Auto-reply.",
modelPlaceholder: "gpt-4o-mini" modelPlaceholder: "gpt-5.6-luna"
} }
}; };
+11 -9
View File
@@ -16,27 +16,29 @@ Restart **both** `api` and `worker` after changing `OPENAI_*`. Worker logs shoul
## Product enhance model choice ## Product enhance model choice
Admin AI role for **product enhance** should prefer a **fast non-reasoning chat model** that emits `message.content` (JSON) under budget. Admin AI role for **product enhance** should prefer a model that emits `message.content` (JSON) under budget. Recommended OpenAI: **`gpt-5.6-luna`** (high-volume / cost-efficient GPT-5.6 tier).
Reasoning / “code-fast” style models often spend the entire `max_tokens` budget in `reasoning_content`, then return `finish_reason=length` with empty or truncated JSON. Descrybe retries once at `MaxTokensEnhanceRetry` and may synthesize formula HTML as salvage — that is **not** live LLM formula compliance. Reasoning / “code-fast” style models often spend the entire completion budget in `reasoning_content`, then return `finish_reason=length` with empty or truncated JSON. Descrybe retries once at `MaxTokensEnhanceRetry` and may synthesize formula HTML as salvage — that is **not** live LLM formula compliance.
If enhance logs show `finish_reason=length` with empty/truncated content even after the retry, switch Admin enhance to a non-reasoning model. For GPT-5 / o-series Chat Completions, the client sends `max_completion_tokens` (not `max_tokens`), omits custom `temperature`, and sets `reasoning_effort` (`none` for probes, `low` for long formula-HTML enhance) so visible HTML is not eaten by deep reasoning.
If enhance logs show `finish_reason=length` with empty/truncated content even after the retry, switch Admin enhance to a lower-reasoning or classic chat model (e.g. `gpt-4o-mini`).
## Hardening defaults (code) ## Hardening defaults (code)
| Constant | Value | Where | | Constant | Value | Where |
|----------|------:|-------| |----------|------:|-------|
| `DefaultStructuredTemp` | `0.2` (capped ≤0.3) | `processing/llm_json.go`, `openai.go` | | `DefaultStructuredTemp` | `0.2` (capped ≤0.3; omitted on GPT-5/o-series) | `processing/llm_json.go`, `openai.go` |
| `MaxTokensEnhance` | `16384` | product title/description (reasoning + formula HTML JSON) | | `MaxTokensEnhance` | `24576` | product title + multi-section HTML description JSON |
| `MaxTokensEnhanceRetry` | `32768` | one-shot bump on `finish_reason=length` empty/truncated JSON | | `MaxTokensEnhanceRetry` | `40960` | one-shot bump on `finish_reason=length` empty/truncated JSON |
| `MaxTokensSEO` | `180` | meta title/description | | `MaxTokensSEO` | `180` | meta title/description |
| `MaxTokensCampaign` | `650` | email JSON | | `MaxTokensCampaign` | `650` | email JSON |
| `MaxProductDescRunes` | `400` | user context | | `MaxProductDescRunes` | `3500` | source description in enhance user context |
| `MaxAttrKeys` / `MaxAttrValueRunes` | `10` / `60` | attrs in enhance prompt | | `MaxAttrKeys` / `MaxAttrValueRunes` | `10` / `60` | attrs in enhance prompt |
| `MaxBrandInjectRunes` | `500` | brand kit block | | `MaxBrandInjectRunes` | `500` | brand kit block |
| `MaxCampaignProducts` | `8` | campaign product list | | `MaxCampaignProducts` | `8` | campaign product list |
OpenAI-compatible chat completions here use a single `max_tokens` completion budget (no separate reasoning vs content split on typical gateways). Length-cap handling lives in `OpenAIClient.doComplete` / `CompleteWithOptions`. Classic OpenAI-compatible chat completions use `max_tokens`. GPT-5 / o-series use `max_completion_tokens` (budget includes hidden reasoning tokens). Length-cap handling lives in `OpenAIClient.doComplete` / `CompleteWithOptions`.
## Prompt style ## Prompt style
@@ -49,7 +51,7 @@ OpenAI-compatible chat completions here use a single `max_tokens` completion bud
`processing.CompleteJSON`: `processing.CompleteJSON`:
1. Call Completer with `max_tokens` + low temperature 1. Call Completer with completion budget + low temperature (classic) / GPT-5.6 param mapping
2. `StripJSONFences` / isolate `{…}` 2. `StripJSONFences` / isolate `{…}`
3. On parse fail → **one retry** with `INVALID. Reply with ONLY one JSON object…` 3. On parse fail → **one retry** with `INVALID. Reply with ONLY one JSON object…`
4. Call sites fall back to originals/templates instead of storing garbage prose as titles 4. Call sites fall back to originals/templates instead of storing garbage prose as titles