fix
This commit is contained in:
@@ -30,8 +30,17 @@ var PopularCatalog = []PopularProvider{
|
|||||||
Name: "openai",
|
Name: "openai",
|
||||||
Label: "OpenAI",
|
Label: "OpenAI",
|
||||||
BaseURL: "https://api.openai.com/v1",
|
BaseURL: "https://api.openai.com/v1",
|
||||||
DefaultModel: "gpt-4o-mini",
|
DefaultModel: "gpt-5.6-luna",
|
||||||
Models: []string{"gpt-4o-mini", "gpt-4o", "gpt-4.1-mini", "gpt-4.1"},
|
Models: []string{
|
||||||
|
"gpt-5.6-luna",
|
||||||
|
"gpt-5.6-terra",
|
||||||
|
"gpt-5.6-sol",
|
||||||
|
"gpt-5.6",
|
||||||
|
"gpt-4o-mini",
|
||||||
|
"gpt-4o",
|
||||||
|
"gpt-4.1-mini",
|
||||||
|
"gpt-4.1",
|
||||||
|
},
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
Name: "google",
|
Name: "google",
|
||||||
@@ -65,8 +74,8 @@ var PopularCatalog = []PopularProvider{
|
|||||||
Name: "openrouter",
|
Name: "openrouter",
|
||||||
Label: "OpenRouter",
|
Label: "OpenRouter",
|
||||||
BaseURL: "https://openrouter.ai/api/v1",
|
BaseURL: "https://openrouter.ai/api/v1",
|
||||||
DefaultModel: "openai/gpt-4o-mini",
|
DefaultModel: "openai/gpt-5.6-luna",
|
||||||
Models: []string{"openai/gpt-4o-mini", "anthropic/claude-sonnet-4", "google/gemini-2.0-flash-001"},
|
Models: []string{"openai/gpt-5.6-luna", "openai/gpt-5.6-terra", "openai/gpt-4o-mini", "anthropic/claude-sonnet-4", "google/gemini-2.0-flash-001"},
|
||||||
},
|
},
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -15,24 +15,33 @@ const (
|
|||||||
// spend thousands of tokens in reasoning_content before message.content.
|
// spend thousands of tokens in reasoning_content before message.content.
|
||||||
// Formula HTML JSON often needs a large completion budget; 4096 hit
|
// Formula HTML JSON often needs a large completion budget; 4096 hit
|
||||||
// finish_reason=length with empty/truncated content in live enhance.
|
// finish_reason=length with empty/truncated content in live enhance.
|
||||||
MaxTokensEnhance = 16384
|
// GPT-5.6 max_completion_tokens includes hidden reasoning tokens — keep
|
||||||
|
// this high so multi-section HTML descriptions are not truncated.
|
||||||
|
MaxTokensEnhance = 24576
|
||||||
// MaxTokensEnhanceRetry is the one-shot length-cap bump ceiling used by
|
// MaxTokensEnhanceRetry is the one-shot length-cap bump ceiling used by
|
||||||
// OpenAIClient when finish_reason=length yields empty or unparseable JSON.
|
// OpenAIClient when finish_reason=length yields empty or unparseable JSON.
|
||||||
MaxTokensEnhanceRetry = 32768
|
MaxTokensEnhanceRetry = 40960
|
||||||
MaxTokensSEO = 180
|
MaxTokensSEO = 180
|
||||||
MaxTokensCampaign = 650
|
MaxTokensCampaign = 650
|
||||||
MaxProductDescRunes = 400
|
// MaxProductDescRunes caps source description text fed into enhance prompts.
|
||||||
MaxAttrKeys = 10
|
// Kept large enough for existing long/HTML descriptions so the model can
|
||||||
MaxAttrValueRunes = 60
|
// rewrite factually instead of inventing from a truncated stub.
|
||||||
MaxBrandInjectRunes = 500
|
MaxProductDescRunes = 3500
|
||||||
MaxCampaignProducts = 8
|
MaxAttrKeys = 10
|
||||||
MaxCampaignNameRunes = 80
|
MaxAttrValueRunes = 60
|
||||||
|
MaxBrandInjectRunes = 500
|
||||||
|
MaxCampaignProducts = 8
|
||||||
|
MaxCampaignNameRunes = 80
|
||||||
)
|
)
|
||||||
|
|
||||||
// CompleteOptions tunes a single chat completion for structured tasks.
|
// CompleteOptions tunes a single chat completion for structured tasks.
|
||||||
type CompleteOptions struct {
|
type CompleteOptions struct {
|
||||||
MaxTokens int
|
MaxTokens int
|
||||||
Temperature float64 // 0 → client default (≤0.3 for structured)
|
Temperature float64 // 0 → client default (≤0.3 for structured); omitted on GPT-5/o-series
|
||||||
|
// ReasoningEffort is Chat Completions reasoning_effort for GPT-5 / o-series.
|
||||||
|
// Empty → auto: "none" for tiny/probe budgets, "low" for content-heavy calls
|
||||||
|
// (long formula HTML) so visible tokens are not eaten by deep reasoning.
|
||||||
|
ReasoningEffort string
|
||||||
}
|
}
|
||||||
|
|
||||||
// CompleterWithOptions is optional; OpenAIClient implements it.
|
// CompleterWithOptions is optional; OpenAIClient implements it.
|
||||||
@@ -205,7 +214,9 @@ func CompactBrandPrompt(block string) string {
|
|||||||
return truncateRunes(SanitizeText(block), MaxBrandInjectRunes)
|
return truncateRunes(SanitizeText(block), MaxBrandInjectRunes)
|
||||||
}
|
}
|
||||||
|
|
||||||
// ProductEnhanceUser builds a short user prompt for title/description enhance.
|
// ProductEnhanceUser builds the compact user prompt for title/description enhance.
|
||||||
|
// Source description is capped at MaxProductDescRunes so long/HTML inputs stay available
|
||||||
|
// for factual rewrite (not invented from a short stub).
|
||||||
func ProductEnhanceUser(category, name, description string, attrs map[string]any) string {
|
func ProductEnhanceUser(category, name, description string, attrs map[string]any) string {
|
||||||
var b strings.Builder
|
var b strings.Builder
|
||||||
b.WriteString("Category: ")
|
b.WriteString("Category: ")
|
||||||
|
|||||||
@@ -109,11 +109,23 @@ func TestCompleteJSON_returnsRetryError(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
func TestProductEnhanceUser_truncates(t *testing.T) {
|
func TestProductEnhanceUser_truncates(t *testing.T) {
|
||||||
long := strings.Repeat("x", 2000)
|
long := strings.Repeat("x", MaxProductDescRunes+500)
|
||||||
u := ProductEnhanceUser("Cat", "Name", long, map[string]any{"brand": "B"})
|
u := ProductEnhanceUser("Cat", "Name", long, map[string]any{"brand": "B"})
|
||||||
if len([]rune(u)) > 900 {
|
// Category/Name/Attrs overhead + MaxProductDescRunes of description.
|
||||||
|
if len([]rune(u)) > MaxProductDescRunes+200 {
|
||||||
t.Fatalf("user too long: %d", len([]rune(u)))
|
t.Fatalf("user too long: %d", len([]rune(u)))
|
||||||
}
|
}
|
||||||
|
descIdx := strings.Index(u, "Description: ")
|
||||||
|
if descIdx < 0 {
|
||||||
|
t.Fatalf("missing Description: in %q", u)
|
||||||
|
}
|
||||||
|
descPart := u[descIdx+len("Description: "):]
|
||||||
|
if i := strings.Index(descPart, "\n"); i >= 0 {
|
||||||
|
descPart = descPart[:i]
|
||||||
|
}
|
||||||
|
if len([]rune(descPart)) > MaxProductDescRunes {
|
||||||
|
t.Fatalf("description runes=%d want <= %d", len([]rune(descPart)), MaxProductDescRunes)
|
||||||
|
}
|
||||||
if !strings.Contains(u, "brand") {
|
if !strings.Contains(u, "brand") {
|
||||||
t.Fatalf("%s", u)
|
t.Fatalf("%s", u)
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -223,13 +223,6 @@ func classifyOpenAITransportErr(model string, err error) (error, bool) {
|
|||||||
return err, true
|
return err, true
|
||||||
}
|
}
|
||||||
|
|
||||||
type chatRequest struct {
|
|
||||||
Model string `json:"model"`
|
|
||||||
Messages []chatMessage `json:"messages"`
|
|
||||||
Temperature float64 `json:"temperature"`
|
|
||||||
MaxTokens int `json:"max_tokens,omitempty"`
|
|
||||||
}
|
|
||||||
|
|
||||||
type chatMessage struct {
|
type chatMessage struct {
|
||||||
Role string `json:"role"`
|
Role string `json:"role"`
|
||||||
Content string `json:"content"`
|
Content string `json:"content"`
|
||||||
@@ -283,6 +276,7 @@ func (c *OpenAIClient) CompleteWithOptions(ctx context.Context, system, user str
|
|||||||
temp = 0.3
|
temp = 0.3
|
||||||
}
|
}
|
||||||
maxTok := opts.MaxTokens
|
maxTok := opts.MaxTokens
|
||||||
|
effort := opts.ReasoningEffort
|
||||||
|
|
||||||
var lastErr error
|
var lastErr error
|
||||||
for attempt := 0; attempt <= c.MaxRetries; attempt++ {
|
for attempt := 0; attempt <= c.MaxRetries; attempt++ {
|
||||||
@@ -298,7 +292,7 @@ func (c *OpenAIClient) CompleteWithOptions(ctx context.Context, system, user str
|
|||||||
if err := c.waitRate(ctx); err != nil {
|
if err := c.waitRate(ctx); err != nil {
|
||||||
return Completion{}, err
|
return Completion{}, err
|
||||||
}
|
}
|
||||||
comp, retryable, err := c.doComplete(ctx, system, user, temp, maxTok)
|
comp, retryable, err := c.doComplete(ctx, system, user, temp, maxTok, effort)
|
||||||
if err == nil {
|
if err == nil {
|
||||||
return comp, nil
|
return comp, nil
|
||||||
}
|
}
|
||||||
@@ -471,16 +465,11 @@ func (c *OpenAIClient) doEmbed(ctx context.Context, texts []string) ([][]float32
|
|||||||
return out, false, nil
|
return out, false, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
func (c *OpenAIClient) doComplete(ctx context.Context, system, user string, temperature float64, maxTokens int) (Completion, bool, error) {
|
func (c *OpenAIClient) doComplete(ctx context.Context, system, user string, temperature float64, maxTokens int, reasoningEffort string) (Completion, bool, error) {
|
||||||
reqBody := chatRequest{
|
reqBody := buildChatCompletionBody(c.Model, []chatMessage{
|
||||||
Model: c.Model,
|
{Role: "system", Content: system},
|
||||||
Messages: []chatMessage{
|
{Role: "user", Content: user},
|
||||||
{Role: "system", Content: system},
|
}, temperature, maxTokens, reasoningEffort)
|
||||||
{Role: "user", Content: user},
|
|
||||||
},
|
|
||||||
Temperature: temperature,
|
|
||||||
MaxTokens: maxTokens,
|
|
||||||
}
|
|
||||||
body, err := json.Marshal(reqBody)
|
body, err := json.Marshal(reqBody)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return Completion{}, false, err
|
return Completion{}, false, err
|
||||||
|
|||||||
@@ -0,0 +1,97 @@
|
|||||||
|
package processing
|
||||||
|
|
||||||
|
import (
|
||||||
|
"strings"
|
||||||
|
)
|
||||||
|
|
||||||
|
// openAIChatUsesMaxCompletionTokens reports models that reject legacy max_tokens
|
||||||
|
// on Chat Completions (o-series + GPT-5 family including GPT-5.6).
|
||||||
|
// See https://developers.openai.com/api/reference/resources/chat
|
||||||
|
func openAIChatUsesMaxCompletionTokens(model string) bool {
|
||||||
|
return openAIReasoningChatModel(model)
|
||||||
|
}
|
||||||
|
|
||||||
|
// openAIChatOmitsCustomTemperature reports models that reject non-default
|
||||||
|
// temperature (only the API default of 1 is accepted, if temperature is sent).
|
||||||
|
func openAIChatOmitsCustomTemperature(model string) bool {
|
||||||
|
return openAIReasoningChatModel(model)
|
||||||
|
}
|
||||||
|
|
||||||
|
// openAIChatSupportsReasoningEffort reports Chat Completions models that accept
|
||||||
|
// top-level reasoning_effort (Responses API uses reasoning.effort instead).
|
||||||
|
func openAIChatSupportsReasoningEffort(model string) bool {
|
||||||
|
return openAIReasoningChatModel(model)
|
||||||
|
}
|
||||||
|
|
||||||
|
// openAIReasoningChatModel matches OpenAI reasoning / GPT-5(+).5/5.6 chat models
|
||||||
|
// and common snapshot aliases (e.g. gpt-5.6-luna-2026-…).
|
||||||
|
func openAIReasoningChatModel(model string) bool {
|
||||||
|
m := strings.ToLower(strings.TrimSpace(model))
|
||||||
|
if m == "" {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
// Strip provider prefixes used by OpenRouter-style gateways.
|
||||||
|
if i := strings.LastIndex(m, "/"); i >= 0 {
|
||||||
|
m = m[i+1:]
|
||||||
|
}
|
||||||
|
switch {
|
||||||
|
case strings.HasPrefix(m, "o1"),
|
||||||
|
strings.HasPrefix(m, "o3"),
|
||||||
|
strings.HasPrefix(m, "o4"),
|
||||||
|
strings.HasPrefix(m, "gpt-5"),
|
||||||
|
strings.HasPrefix(m, "chatgpt-o"):
|
||||||
|
return true
|
||||||
|
default:
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// defaultReasoningEffort picks Chat Completions reasoning_effort when unset.
|
||||||
|
// Long formula-HTML enhance uses "low" so max_completion_tokens prefers visible
|
||||||
|
// HTML over deep hidden reasoning. Tiny probes use "none" for speed/cost.
|
||||||
|
func defaultReasoningEffort(model string, maxTokens int, explicit string) string {
|
||||||
|
explicit = strings.ToLower(strings.TrimSpace(explicit))
|
||||||
|
if explicit != "" {
|
||||||
|
return explicit
|
||||||
|
}
|
||||||
|
if !openAIChatSupportsReasoningEffort(model) {
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
if maxTokens <= 0 || maxTokens < 512 {
|
||||||
|
return "none"
|
||||||
|
}
|
||||||
|
return "low"
|
||||||
|
}
|
||||||
|
|
||||||
|
// chatCompletionBody is the Chat Completions JSON body. Temperature is a pointer
|
||||||
|
// so reasoning models can omit it entirely (omitempty).
|
||||||
|
type chatCompletionBody struct {
|
||||||
|
Model string `json:"model"`
|
||||||
|
Messages []chatMessage `json:"messages"`
|
||||||
|
Temperature *float64 `json:"temperature,omitempty"`
|
||||||
|
MaxTokens int `json:"max_tokens,omitempty"`
|
||||||
|
MaxCompletionTokens int `json:"max_completion_tokens,omitempty"`
|
||||||
|
ReasoningEffort string `json:"reasoning_effort,omitempty"`
|
||||||
|
}
|
||||||
|
|
||||||
|
// buildChatCompletionBody maps CompleteOptions onto the correct Chat Completions
|
||||||
|
// fields for classic vs GPT-5 / o-series models.
|
||||||
|
func buildChatCompletionBody(model string, messages []chatMessage, temperature float64, maxTokens int, reasoningEffort string) chatCompletionBody {
|
||||||
|
body := chatCompletionBody{
|
||||||
|
Model: model,
|
||||||
|
Messages: messages,
|
||||||
|
}
|
||||||
|
if openAIChatUsesMaxCompletionTokens(model) {
|
||||||
|
if maxTokens > 0 {
|
||||||
|
body.MaxCompletionTokens = maxTokens
|
||||||
|
}
|
||||||
|
} else if maxTokens > 0 {
|
||||||
|
body.MaxTokens = maxTokens
|
||||||
|
}
|
||||||
|
if !openAIChatOmitsCustomTemperature(model) {
|
||||||
|
t := temperature
|
||||||
|
body.Temperature = &t
|
||||||
|
}
|
||||||
|
body.ReasoningEffort = defaultReasoningEffort(model, maxTokens, reasoningEffort)
|
||||||
|
return body
|
||||||
|
}
|
||||||
@@ -0,0 +1,123 @@
|
|||||||
|
package processing
|
||||||
|
|
||||||
|
import (
|
||||||
|
"encoding/json"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestOpenAIReasoningChatModel(t *testing.T) {
|
||||||
|
t.Parallel()
|
||||||
|
cases := []struct {
|
||||||
|
model string
|
||||||
|
want bool
|
||||||
|
}{
|
||||||
|
{"gpt-5.6-luna", true},
|
||||||
|
{"gpt-5.6-terra", true},
|
||||||
|
{"gpt-5.6-sol", true},
|
||||||
|
{"gpt-5.6", true},
|
||||||
|
{"GPT-5.4", true},
|
||||||
|
{"openai/gpt-5.6-luna", true},
|
||||||
|
{"o1-mini", true},
|
||||||
|
{"o3", true},
|
||||||
|
{"o4-mini", true},
|
||||||
|
{"gpt-4o-mini", false},
|
||||||
|
{"gpt-4.1", false},
|
||||||
|
{"mock-llm", false},
|
||||||
|
{"", false},
|
||||||
|
}
|
||||||
|
for _, tc := range cases {
|
||||||
|
if got := openAIReasoningChatModel(tc.model); got != tc.want {
|
||||||
|
t.Fatalf("openAIReasoningChatModel(%q)=%v want %v", tc.model, got, tc.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestBuildChatCompletionBody_gpt56Luna(t *testing.T) {
|
||||||
|
t.Parallel()
|
||||||
|
body := buildChatCompletionBody("gpt-5.6-luna", []chatMessage{
|
||||||
|
{Role: "system", Content: "sys"},
|
||||||
|
{Role: "user", Content: "ping"},
|
||||||
|
}, 0.2, MaxTokensEnhance, "low")
|
||||||
|
raw, err := json.Marshal(body)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
var m map[string]any
|
||||||
|
if err := json.Unmarshal(raw, &m); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if _, ok := m["temperature"]; ok {
|
||||||
|
t.Fatalf("temperature must be omitted for gpt-5.6-luna, got %v", m["temperature"])
|
||||||
|
}
|
||||||
|
if _, ok := m["max_tokens"]; ok {
|
||||||
|
t.Fatalf("max_tokens must be omitted for gpt-5.6-luna, got %v", m["max_tokens"])
|
||||||
|
}
|
||||||
|
gotMax, _ := m["max_completion_tokens"].(float64)
|
||||||
|
if int(gotMax) != MaxTokensEnhance {
|
||||||
|
t.Fatalf("max_completion_tokens=%v want %d", gotMax, MaxTokensEnhance)
|
||||||
|
}
|
||||||
|
if m["reasoning_effort"] != "low" {
|
||||||
|
t.Fatalf("reasoning_effort=%v want low", m["reasoning_effort"])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestBuildChatCompletionBody_classicModel(t *testing.T) {
|
||||||
|
t.Parallel()
|
||||||
|
body := buildChatCompletionBody("gpt-4o-mini", []chatMessage{
|
||||||
|
{Role: "user", Content: "hi"},
|
||||||
|
}, 0.2, 350, "")
|
||||||
|
raw, err := json.Marshal(body)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
var m map[string]any
|
||||||
|
if err := json.Unmarshal(raw, &m); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if _, ok := m["max_completion_tokens"]; ok {
|
||||||
|
t.Fatalf("max_completion_tokens unexpected for classic model: %v", m)
|
||||||
|
}
|
||||||
|
gotMax, _ := m["max_tokens"].(float64)
|
||||||
|
if int(gotMax) != 350 {
|
||||||
|
t.Fatalf("max_tokens=%v want 350", gotMax)
|
||||||
|
}
|
||||||
|
gotTemp, _ := m["temperature"].(float64)
|
||||||
|
if gotTemp != 0.2 {
|
||||||
|
t.Fatalf("temperature=%v want 0.2", gotTemp)
|
||||||
|
}
|
||||||
|
if _, ok := m["reasoning_effort"]; ok {
|
||||||
|
t.Fatalf("reasoning_effort unexpected for classic model: %v", m)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestBuildChatCompletionBody_probeDefaultsNoneEffort(t *testing.T) {
|
||||||
|
t.Parallel()
|
||||||
|
body := buildChatCompletionBody("gpt-5.6-luna", []chatMessage{
|
||||||
|
{Role: "user", Content: "ping"},
|
||||||
|
}, 0.2, 0, "")
|
||||||
|
if body.ReasoningEffort != "none" {
|
||||||
|
t.Fatalf("reasoning_effort=%q want none for probe", body.ReasoningEffort)
|
||||||
|
}
|
||||||
|
if body.MaxCompletionTokens != 0 || body.MaxTokens != 0 {
|
||||||
|
t.Fatalf("expected no token cap on probe, got %+v", body)
|
||||||
|
}
|
||||||
|
if body.Temperature != nil {
|
||||||
|
t.Fatalf("temperature must be nil for probe on gpt-5.6")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestDefaultReasoningEffort(t *testing.T) {
|
||||||
|
t.Parallel()
|
||||||
|
if got := defaultReasoningEffort("gpt-5.6-luna", MaxTokensEnhance, ""); got != "low" {
|
||||||
|
t.Fatalf("enhance default=%q", got)
|
||||||
|
}
|
||||||
|
if got := defaultReasoningEffort("gpt-5.6-luna", 0, ""); got != "none" {
|
||||||
|
t.Fatalf("probe default=%q", got)
|
||||||
|
}
|
||||||
|
if got := defaultReasoningEffort("gpt-5.6-luna", 100, "medium"); got != "medium" {
|
||||||
|
t.Fatalf("explicit=%q", got)
|
||||||
|
}
|
||||||
|
if got := defaultReasoningEffort("gpt-4o", 1000, ""); got != "" {
|
||||||
|
t.Fatalf("classic=%q", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -179,6 +179,52 @@ func TestOpenAIClient_doComplete_emptyContentWithReasoningJSON(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func TestOpenAIClient_doComplete_gpt56LunaUsesMaxCompletionTokens(t *testing.T) {
|
||||||
|
t.Parallel()
|
||||||
|
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
var req map[string]any
|
||||||
|
_ = json.NewDecoder(r.Body).Decode(&req)
|
||||||
|
if _, ok := req["temperature"]; ok {
|
||||||
|
t.Errorf("temperature must be omitted for gpt-5.6-luna")
|
||||||
|
}
|
||||||
|
if _, ok := req["max_tokens"]; ok {
|
||||||
|
t.Errorf("max_tokens must be omitted for gpt-5.6-luna")
|
||||||
|
}
|
||||||
|
gotMax, _ := req["max_completion_tokens"].(float64)
|
||||||
|
if int(gotMax) != MaxTokensEnhance {
|
||||||
|
t.Errorf("max_completion_tokens=%v want %d", gotMax, MaxTokensEnhance)
|
||||||
|
}
|
||||||
|
if req["reasoning_effort"] != "low" {
|
||||||
|
t.Errorf("reasoning_effort=%v want low", req["reasoning_effort"])
|
||||||
|
}
|
||||||
|
_ = json.NewEncoder(w).Encode(map[string]any{
|
||||||
|
"model": "gpt-5.6-luna",
|
||||||
|
"choices": []map[string]any{{
|
||||||
|
"finish_reason": "stop",
|
||||||
|
"message": map[string]any{
|
||||||
|
"role": "assistant",
|
||||||
|
"content": `{"name":"Acme Monitor","description":"<h1>Acme Monitor</h1><p>IPS panel for office use.</p>"}`,
|
||||||
|
},
|
||||||
|
}},
|
||||||
|
"usage": map[string]int{"prompt_tokens": 20, "completion_tokens": 40, "total_tokens": 60},
|
||||||
|
})
|
||||||
|
}))
|
||||||
|
defer srv.Close()
|
||||||
|
c := NewOpenAIClient("test-key", srv.URL, "gpt-5.6-luna", 0, 1)
|
||||||
|
c.HTTPClient = srv.Client()
|
||||||
|
comp, err := c.CompleteWithOptions(context.Background(), "sys", "user", CompleteOptions{
|
||||||
|
MaxTokens: MaxTokensEnhance,
|
||||||
|
Temperature: DefaultStructuredTemp,
|
||||||
|
ReasoningEffort: "low",
|
||||||
|
})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if !strings.Contains(comp.Text, "Acme Monitor") {
|
||||||
|
t.Fatalf("text=%q", comp.Text)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func TestOpenAIClient_doComplete_emptyContentLengthRetriesWithBudget(t *testing.T) {
|
func TestOpenAIClient_doComplete_emptyContentLengthRetriesWithBudget(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
calls := 0
|
calls := 0
|
||||||
|
|||||||
@@ -11,7 +11,7 @@ import (
|
|||||||
const maxPromptFieldRunes = 4000
|
const maxPromptFieldRunes = 4000
|
||||||
|
|
||||||
// maxOutputFieldRunes bounds stored / polled model text (titles + multi-section
|
// maxOutputFieldRunes bounds stored / polled model text (titles + multi-section
|
||||||
// formula HTML descriptions). Must stay well above MaxTokensEnhance (~16k tokens)
|
// formula HTML descriptions). Must stay well above MaxTokensEnhance (~24k tokens)
|
||||||
// so SanitizeOutput does not chop JSON completion bodies or A1 HTML mid-string.
|
// so SanitizeOutput does not chop JSON completion bodies or A1 HTML mid-string.
|
||||||
const maxOutputFieldRunes = 120000
|
const maxOutputFieldRunes = 120000
|
||||||
|
|
||||||
|
|||||||
@@ -800,8 +800,9 @@ func (e *Engine) enhance(ctx context.Context, in ProductInput, category string,
|
|||||||
logEnhancePrompt(in, catUID, catName, system, user)
|
logEnhancePrompt(in, catUID, catName, system, user)
|
||||||
started := time.Now()
|
started := time.Now()
|
||||||
comp, obj, err := CompleteJSON(ctx, e.Completer, system, user, CompleteOptions{
|
comp, obj, err := CompleteJSON(ctx, e.Completer, system, user, CompleteOptions{
|
||||||
MaxTokens: MaxTokensEnhance,
|
MaxTokens: MaxTokensEnhance,
|
||||||
Temperature: DefaultStructuredTemp,
|
Temperature: DefaultStructuredTemp,
|
||||||
|
ReasoningEffort: "low",
|
||||||
})
|
})
|
||||||
elapsed := time.Since(started)
|
elapsed := time.Since(started)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
@@ -891,8 +892,9 @@ func (e *Engine) enhance(ctx context.Context, in ProductInput, category string,
|
|||||||
retryUser := user + descriptionFormulaRetrySuffix
|
retryUser := user + descriptionFormulaRetrySuffix
|
||||||
retryStarted := time.Now()
|
retryStarted := time.Now()
|
||||||
comp2, obj2, err2 := CompleteJSON(ctx, e.Completer, system, retryUser, CompleteOptions{
|
comp2, obj2, err2 := CompleteJSON(ctx, e.Completer, system, retryUser, CompleteOptions{
|
||||||
MaxTokens: MaxTokensEnhance,
|
MaxTokens: MaxTokensEnhance,
|
||||||
Temperature: DefaultStructuredTemp,
|
Temperature: DefaultStructuredTemp,
|
||||||
|
ReasoningEffort: "low",
|
||||||
})
|
})
|
||||||
elapsed += time.Since(retryStarted)
|
elapsed += time.Since(retryStarted)
|
||||||
comp.PromptTokens += comp2.PromptTokens
|
comp.PromptTokens += comp2.PromptTokens
|
||||||
|
|||||||
@@ -31,7 +31,7 @@ export const AI_ROLE_META: Record<
|
|||||||
processing: {
|
processing: {
|
||||||
label: "Processing",
|
label: "Processing",
|
||||||
description: "Product pipeline chat/completions (titles, descriptions, enhance).",
|
description: "Product pipeline chat/completions (titles, descriptions, enhance).",
|
||||||
modelPlaceholder: "gpt-4o-mini"
|
modelPlaceholder: "gpt-5.6-luna"
|
||||||
},
|
},
|
||||||
vectorization: {
|
vectorization: {
|
||||||
label: "Vectorization",
|
label: "Vectorization",
|
||||||
@@ -42,13 +42,13 @@ export const AI_ROLE_META: Record<
|
|||||||
label: "Docs / API",
|
label: "Docs / API",
|
||||||
description:
|
description:
|
||||||
"Future docs/API assistant slot — /docs Ask stays rule-based and must never call this role.",
|
"Future docs/API assistant slot — /docs Ask stays rule-based and must never call this role.",
|
||||||
modelPlaceholder: "gpt-4o-mini"
|
modelPlaceholder: "gpt-5.6-luna"
|
||||||
},
|
},
|
||||||
support: {
|
support: {
|
||||||
label: "Support",
|
label: "Support",
|
||||||
description:
|
description:
|
||||||
"Ticket auto-reply AI fallback — configure provider/key/model here; enable delivery in Support knowledge → Auto-reply.",
|
"Ticket auto-reply AI fallback — configure provider/key/model here; enable delivery in Support knowledge → Auto-reply.",
|
||||||
modelPlaceholder: "gpt-4o-mini"
|
modelPlaceholder: "gpt-5.6-luna"
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
|
|||||||
@@ -16,27 +16,29 @@ Restart **both** `api` and `worker` after changing `OPENAI_*`. Worker logs shoul
|
|||||||
|
|
||||||
## Product enhance model choice
|
## Product enhance model choice
|
||||||
|
|
||||||
Admin AI role for **product enhance** should prefer a **fast non-reasoning chat model** that emits `message.content` (JSON) under budget.
|
Admin AI role for **product enhance** should prefer a model that emits `message.content` (JSON) under budget. Recommended OpenAI: **`gpt-5.6-luna`** (high-volume / cost-efficient GPT-5.6 tier).
|
||||||
|
|
||||||
Reasoning / “code-fast” style models often spend the entire `max_tokens` budget in `reasoning_content`, then return `finish_reason=length` with empty or truncated JSON. Descrybe retries once at `MaxTokensEnhanceRetry` and may synthesize formula HTML as salvage — that is **not** live LLM formula compliance.
|
Reasoning / “code-fast” style models often spend the entire completion budget in `reasoning_content`, then return `finish_reason=length` with empty or truncated JSON. Descrybe retries once at `MaxTokensEnhanceRetry` and may synthesize formula HTML as salvage — that is **not** live LLM formula compliance.
|
||||||
|
|
||||||
If enhance logs show `finish_reason=length` with empty/truncated content even after the retry, switch Admin enhance to a non-reasoning model.
|
For GPT-5 / o-series Chat Completions, the client sends `max_completion_tokens` (not `max_tokens`), omits custom `temperature`, and sets `reasoning_effort` (`none` for probes, `low` for long formula-HTML enhance) so visible HTML is not eaten by deep reasoning.
|
||||||
|
|
||||||
|
If enhance logs show `finish_reason=length` with empty/truncated content even after the retry, switch Admin enhance to a lower-reasoning or classic chat model (e.g. `gpt-4o-mini`).
|
||||||
|
|
||||||
## Hardening defaults (code)
|
## Hardening defaults (code)
|
||||||
|
|
||||||
| Constant | Value | Where |
|
| Constant | Value | Where |
|
||||||
|----------|------:|-------|
|
|----------|------:|-------|
|
||||||
| `DefaultStructuredTemp` | `0.2` (capped ≤0.3) | `processing/llm_json.go`, `openai.go` |
|
| `DefaultStructuredTemp` | `0.2` (capped ≤0.3; omitted on GPT-5/o-series) | `processing/llm_json.go`, `openai.go` |
|
||||||
| `MaxTokensEnhance` | `16384` | product title/description (reasoning + formula HTML JSON) |
|
| `MaxTokensEnhance` | `24576` | product title + multi-section HTML description JSON |
|
||||||
| `MaxTokensEnhanceRetry` | `32768` | one-shot bump on `finish_reason=length` empty/truncated JSON |
|
| `MaxTokensEnhanceRetry` | `40960` | one-shot bump on `finish_reason=length` empty/truncated JSON |
|
||||||
| `MaxTokensSEO` | `180` | meta title/description |
|
| `MaxTokensSEO` | `180` | meta title/description |
|
||||||
| `MaxTokensCampaign` | `650` | email JSON |
|
| `MaxTokensCampaign` | `650` | email JSON |
|
||||||
| `MaxProductDescRunes` | `400` | user context |
|
| `MaxProductDescRunes` | `3500` | source description in enhance user context |
|
||||||
| `MaxAttrKeys` / `MaxAttrValueRunes` | `10` / `60` | attrs in enhance prompt |
|
| `MaxAttrKeys` / `MaxAttrValueRunes` | `10` / `60` | attrs in enhance prompt |
|
||||||
| `MaxBrandInjectRunes` | `500` | brand kit block |
|
| `MaxBrandInjectRunes` | `500` | brand kit block |
|
||||||
| `MaxCampaignProducts` | `8` | campaign product list |
|
| `MaxCampaignProducts` | `8` | campaign product list |
|
||||||
|
|
||||||
OpenAI-compatible chat completions here use a single `max_tokens` completion budget (no separate reasoning vs content split on typical gateways). Length-cap handling lives in `OpenAIClient.doComplete` / `CompleteWithOptions`.
|
Classic OpenAI-compatible chat completions use `max_tokens`. GPT-5 / o-series use `max_completion_tokens` (budget includes hidden reasoning tokens). Length-cap handling lives in `OpenAIClient.doComplete` / `CompleteWithOptions`.
|
||||||
|
|
||||||
## Prompt style
|
## Prompt style
|
||||||
|
|
||||||
@@ -49,7 +51,7 @@ OpenAI-compatible chat completions here use a single `max_tokens` completion bud
|
|||||||
|
|
||||||
`processing.CompleteJSON`:
|
`processing.CompleteJSON`:
|
||||||
|
|
||||||
1. Call Completer with `max_tokens` + low temperature
|
1. Call Completer with completion budget + low temperature (classic) / GPT-5.6 param mapping
|
||||||
2. `StripJSONFences` / isolate `{…}`
|
2. `StripJSONFences` / isolate `{…}`
|
||||||
3. On parse fail → **one retry** with `INVALID. Reply with ONLY one JSON object…`
|
3. On parse fail → **one retry** with `INVALID. Reply with ONLY one JSON object…`
|
||||||
4. Call sites fall back to originals/templates instead of storing garbage prose as titles
|
4. Call sites fall back to originals/templates instead of storing garbage prose as titles
|
||||||
|
|||||||
Reference in New Issue
Block a user