package processing import ( "context" "encoding/json" "fmt" "sort" "strings" "github.com/descrybe/descrybe-v2/apps/api/internal/company" "github.com/descrybe/descrybe-v2/apps/api/internal/eprel" ) // StepPolicy controls entitlement-gated steps (AI / EPREL). type StepPolicy struct { AllowAI bool AllowEPREL bool } // RunSteps executes the multi-step product pipeline. // OpenAI enhance runs only when Completer is configured, Enabled(), and policy.AllowAI. // EPREL runs only when enricher enabled and policy.AllowEPREL. func (e *Engine) RunSteps(ctx context.Context, companyID string, in ProductInput, processingType string, categoryNames []string, policy StepPolicy) (StepResult, error) { steps := resolveSteps(processingType) out := StepResult{ Attributes: map[string]any{}, ProcessedAttributes: map[string]any{}, FieldSources: map[string]any{}, EPREL: map[string]any{}, GPTResponse: map[string]any{"steps": []any{}}, Notes: []string{}, } normalized := map[string]any{} attrs := map[string]any{} for _, step := range steps { switch step { case StepNormalize: normalized = NormalizeMapped(in.Mapped, in.Raw) out.Name = preferredProductTitle(in.GTIN, stringFromAny(normalized["name"]), stringFromAny(normalized["title"]), in.Name, in.PriorProcessedName, ) out.Description = preferredProductDescription(out.Name, stringFromAny(normalized["description"]), in.Description, in.PriorProcessedDescription, ) // mapped_data.category / category_unique_id (unique_id codes) win. applyCategoryFromMapped(&out, normalized, in.Mapped, in.Raw) // otherwise keep existing processed category so enhance_only / reprocess // cannot blank A1 legacy categories. preserveCategoryIfEmpty(&out, in.PriorCategory) out.FieldSources["normalize"] = "mapped+raw" appendStepLog(out.GPTResponse, StepNormalize, map[string]any{ "keys": len(normalized), }) case StepParseSpecs: specVal := normalized["specifications"] if specVal == nil { specVal = in.Mapped["specifications"] } if specVal == nil { specVal = in.Raw["specifications"] } parsed := ParseSpecifications(specVal) // Also accept pre-mapped attributes map if am, ok := normalized["attributes"].(map[string]any); ok { for k, v := range ParseSpecifications(am) { if _, exists := parsed[k]; !exists { parsed[k] = v } } } attrs = SanitizeProductAttributes(parsed) // Map feed/spec labels onto category_attributes (formula) keys early. if allowed := enhanceAllowedAttrKeys(in, out.Category); allowed != nil { attrs = MapAttrsOntoAllowedKeys(attrs, allowed) } out.Attributes = attrs out.ProcessedAttributes = attrs out.FieldSources["attributes"] = "specifications" appendStepLog(out.GPTResponse, StepParseSpecs, map[string]any{ "count": len(attrs), }) case StepFillFields: normalized = FillMissingFields(normalized, attrs) if len(in.StandardFields) > 0 { normalized = FillMissingStandardFields(normalized, in.Raw, in.StandardFields) } out.Name = preferredProductTitle(in.GTIN, stringFromAny(normalized["name"]), stringFromAny(normalized["title"]), out.Name, in.Name, in.PriorProcessedName, ) out.Description = preferredProductDescription(out.Name, stringFromAny(normalized["description"]), out.Description, in.Description, ) applyCategoryFromMapped(&out, normalized, in.Mapped, in.Raw) // Promote characteristic fields only — never core product identity/content // (those belong on the V1 item root: title, description, ean, images, …). promote := []string{"brand", "width", "height", "depth", "weight", "product_model", "warranty"} for _, f := range in.StandardFields { k := strings.TrimSpace(f.Key) if k == "" || isReservedProductKey(k) || isInvalidAttributeKey(k) { continue } promote = append(promote, k) } seen := map[string]bool{} for _, k := range promote { if seen[k] { continue } seen[k] = true if v := stringFromAny(normalized[k]); v != "" { if _, exists := attrs[k]; !exists { attrs[k] = v } out.FieldSources[k] = "fill_fields" } } attrs = SanitizeProductAttributes(attrs) // Remap again after category may have resolved in applyCategoryFromMapped. if allowed := enhanceAllowedAttrKeys(in, out.Category); allowed != nil { attrs = MapAttrsOntoAllowedKeys(attrs, allowed) } out.Attributes = attrs out.ProcessedAttributes = attrs appendStepLog(out.GPTResponse, StepFillFields, map[string]any{ "brand": stringFromAny(normalized["brand"]), }) preserveCategoryIfEmpty(&out, in.PriorCategory) case StepEPREL: if !policy.AllowEPREL { out.Notes = append(out.Notes, "eprel: skipped (not allowed for this job)") appendStepLog(out.GPTResponse, StepEPREL, map[string]any{ "status": "skipped", "reason": "entitlement_can_use_eprel", }) break } // Also scan parsed attrs — eprel_id may only appear after parse_specs. id := eprel.ExtractID(normalized, in.Mapped, in.Raw, attrs) if id == "" { out.Notes = append(out.Notes, "eprel: no id") appendStepLog(out.GPTResponse, StepEPREL, map[string]any{"status": "skipped", "reason": "no_id"}) break } enricher := e.EPREL if enricher == nil { enricher = eprel.Disabled{} } if !enricher.Enabled() { out.Notes = append(out.Notes, "eprel: enricher disabled") out.EPREL = map[string]any{"id": id, "status": "skipped"} appendStepLog(out.GPTResponse, StepEPREL, map[string]any{"status": "skipped", "eprel_id": id, "reason": "disabled"}) break } data, err := enricher.Fetch(ctx, id) if err != nil { out.Notes = append(out.Notes, "eprel: "+TruncateError(err)) attrs["eprel_id"] = id out.Attributes = attrs out.ProcessedAttributes = attrs appendStepLog(out.GPTResponse, StepEPREL, map[string]any{"status": "failed", "error": TruncateError(err)}) // Non-fatal: continue pipeline break } if data == nil { attrs["eprel_id"] = id out.Attributes = attrs out.ProcessedAttributes = attrs out.EPREL = map[string]any{"id": id, "status": "empty"} appendStepLog(out.GPTResponse, StepEPREL, map[string]any{"status": "empty", "eprel_id": id}) break } attrs = eprel.MergeInto(attrs, data) // Promote energy class onto the characteristic attr key when missing // (feeds often omit it; EPREL API is the source of truth). if data.EnergyClass != "" { if cur := strings.TrimSpace(fmt.Sprint(attrs["energy_class"])); cur == "" || cur == "" { attrs["energy_class"] = data.EnergyClass } } out.Attributes = attrs out.ProcessedAttributes = attrs out.EPREL = map[string]any{ "id": data.ID, "label": data.Label, "pdf": data.PDF, "energy_class": data.EnergyClass, "energy_scale": data.EnergyScale, } out.FieldSources["eprel"] = "eprel_api" appendStepLog(out.GPTResponse, StepEPREL, map[string]any{"status": "ok", "eprel_id": data.ID}) case StepAIEnhance: preservePriorEnhanceHash := func() { if in.PriorEnhanceHash != "" { out.FieldSources[FieldEnhanceInputHash] = in.PriorEnhanceHash } } if !policy.AllowAI { out.ProcessedName = out.Name out.ProcessedDescription = out.Description out.Notes = append(out.Notes, "ai_enhance: skipped (Free plan — upgrade for AI titles/descriptions)") preservePriorEnhanceHash() appendStepLog(out.GPTResponse, StepAIEnhance, map[string]any{ "status": "skipped", "reason": "entitlement_can_use_ai", }) break } if !e.CompleterEnabled() { out.ProcessedName = out.Name out.ProcessedDescription = out.Description out.Notes = append(out.Notes, "ai_enhance: skipped (platform OpenAI unset; configure admin settings or company BYOK)") preservePriorEnhanceHash() appendStepLog(out.GPTResponse, StepAIEnhance, map[string]any{ "status": "skipped", "reason": "openai_not_configured", }) break } langs := in.ContentLanguages if len(langs) == 0 { langs = []string{in.Language} } if len(langs) == 0 { langs = []string{company.DefaultLanguage} } primary := in.Language if primary == "" { primary = langs[0] } syncCategoryName(&out, in.CategoryNamesByUID) displayCat := categoryDisplayLabel(out) ensureEnergyClassFromEPREL(attrs) enhanceAttrs := AttrsForEnhance(attrs, enhanceAllowedAttrKeys(in, out.Category)) localized := company.LocalizedContent{} if in.PriorLocalized != nil { for k, v := range in.PriorLocalized { localized[k] = v } } anyFailed := false anyOK := false allUnchanged := true langMetas := make([]any, 0, len(langs)) for _, lang := range langs { tpl := in.EnhanceByLang[lang] if tpl.System == "" && tpl.User == "" && lang == primary { tpl = PromptTemplates{System: in.EnhanceSystemTemplate, User: in.EnhanceUserTemplate} } catPrompt := in.CategoryEnhancePrompt if lang != primary || catPrompt == "" { catPrompt = categoryEnhancePromptFor(in.CategoryPromptsByLang, out.Category, lang, primary) } titleTpl, descTpl := categoryFormulasFor(in, out.Category) priorFields := company.FieldsForLanguage(in.PriorLocalized, lang) priorHash := priorFields.EnhanceInputHash priorName := priorFields.ProcessedName priorDesc := priorFields.ProcessedDescription if lang == primary { if priorHash == "" { priorHash = in.PriorEnhanceHash } if priorName == "" { priorName = in.PriorProcessedName } if priorDesc == "" { priorDesc = in.PriorProcessedDescription } } name, desc, tokens, raw, err := e.enhance(ctx, ProductInput{ GTIN: in.GTIN, Name: out.Name, Description: out.Description, Mapped: normalized, BrandPrompt: in.BrandPrompt, Language: lang, EnhanceSystemTemplate: tpl.System, EnhanceUserTemplate: tpl.User, CategoryEnhancePrompt: catPrompt, TitleTemplate: titleTpl, DescriptionTemplate: descTpl, PriorEnhanceHash: priorHash, PriorProcessedName: priorName, PriorProcessedDescription: priorDesc, }, displayCat, enhanceAttrs) out.TotalTokens += tokens status := enhanceStatusFromMeta(raw) meta := map[string]any{"language": lang, "raw": raw} if err != nil { anyFailed = true allUnchanged = false meta["status"] = "failed" meta["error"] = TruncateError(err) out.Notes = append(out.Notes, "ai_enhance: "+TruncateError(err)) if lang == primary { name, desc = out.Name, out.Description } else if priorName != "" || priorDesc != "" { name, desc = priorName, priorDesc } else { langMetas = append(langMetas, meta) continue } } else if status == "unchanged" { meta["status"] = "unchanged" } else { allUnchanged = false anyOK = true meta["status"] = status } // Prefer title-aware selection + synthesize before deciding hash persistence. name = preferredProductTitle(in.GTIN, name, out.Name, priorName, in.Name) desc = preferredProductDescription(name, desc, out.Description, priorDesc) outerSynth := false if isWeakPriorEnhanceDescription(desc, name) { if synth := synthesizeProductDescription(name, displayCat, lang, enhanceAttrs, descTpl); synth != "" { desc = synth outerSynth = true } } weakDesc := isWeakPriorEnhanceDescription(desc, name) // Only persist enhance_input_hash for quality ok / hash-skip unchanged. // Never copy input_hash from error/passthrough/thin/synthesized meta. persistHash := "" rawHash := enhanceHashFromMeta(raw) switch { case err != nil: persistHash = "" case status == "synthesized" || outerSynth: persistHash = "" allUnchanged = false case status == "unchanged" && !weakDesc: persistHash = rawHash case status == "ok" && !weakDesc: persistHash = rawHash default: // thin ok / parse_failed / skipped — do not poison reprocess skip persistHash = "" if status == "ok" && weakDesc { allUnchanged = false } } localized[lang] = company.LocalizedFields{ ProcessedName: name, ProcessedDescription: desc, EnhanceInputHash: persistHash, MetaTitle: company.FieldsForLanguage(localized, lang).MetaTitle, MetaDescription: company.FieldsForLanguage(localized, lang).MetaDescription, } // Preserve existing meta when re-enhancing titles only. // Never keep a bare "| " poisoned meta_title. // When dropping poisoned title, also drop empty/weak stub meta_description. if prev := company.FieldsForLanguage(in.PriorLocalized, lang); prev.MetaTitle != "" || prev.MetaDescription != "" { f := localized[lang] poisonedTitle := isPoisonedMetaTitle(prev.MetaTitle) if f.MetaTitle == "" && !poisonedTitle { f.MetaTitle = prev.MetaTitle } if f.MetaDescription == "" { weakStub := poisonedTitle && isWeakPriorEnhanceDescription(prev.MetaDescription, name, priorName) if !weakStub { f.MetaDescription = prev.MetaDescription } } localized[lang] = f } langMetas = append(langMetas, meta) if lang == primary { out.ProcessedName = name out.ProcessedDescription = desc if name != "" { out.Name = name } if desc != "" { out.Description = desc } if persistHash != "" { out.FieldSources[FieldEnhanceInputHash] = persistHash } else { delete(out.FieldSources, FieldEnhanceInputHash) } } } out.LocalizedContent = localized if anyFailed && !anyOK { if out.ProcessedName == "" { out.ProcessedName = out.Name } if out.ProcessedDescription == "" { out.ProcessedDescription = out.Description } // Timeout/error/empty: always synthesize a factual fallback when a title exists. if out.ProcessedName != "" && (out.ProcessedDescription == "" || isWeakPriorEnhanceDescription(out.ProcessedDescription, out.ProcessedName, out.Name)) { _, failDescTpl := categoryFormulasFor(in, out.Category) if synth := synthesizeProductDescription(out.ProcessedName, displayCat, primary, enhanceAttrs, failDescTpl); synth != "" { out.ProcessedDescription = synth if out.Description == "" || isWeakPriorEnhanceDescription(out.Description, out.Name, out.ProcessedName) { out.Description = synth } if lf, ok := localized[primary]; ok { lf.ProcessedDescription = synth if lf.ProcessedName == "" { lf.ProcessedName = out.ProcessedName } localized[primary] = lf } } } // Attempted enhance (even on timeout) — record provider mode, not misleading unknown. out.AIProviderMode = e.EngineProviderMode() out.FieldSources["name"] = "ai_enhance_failed" out.FieldSources["description"] = "ai_enhance_failed" // Failed enhance must not leave a skippable hash behind. delete(out.FieldSources, FieldEnhanceInputHash) for lang, lf := range localized { lf.EnhanceInputHash = "" localized[lang] = lf } out.LocalizedContent = localized errNote := "" for _, m := range langMetas { if mm, ok := m.(map[string]any); ok { if e, ok := mm["error"].(string); ok && e != "" { errNote = e break } } } appendStepLog(out.GPTResponse, StepAIEnhance, map[string]any{ "status": "failed", "error": errNote, "languages": langMetas, }) break } if allUnchanged { out.SkipCreditDebit = true out.FieldSources["name"] = "ai_enhance_unchanged" out.FieldSources["description"] = "ai_enhance_unchanged" out.Notes = append(out.Notes, "ai_enhance: skipped (inputs unchanged)") } else { out.AIProviderMode = e.EngineProviderMode() out.FieldSources["name"] = "ai_enhance" out.FieldSources["description"] = "ai_enhance" } appendStepLog(out.GPTResponse, StepAIEnhance, map[string]any{ "status": map[string]any{"unchanged": allUnchanged, "ok": anyOK, "failed": anyFailed}, "languages": langMetas, }) default: appendStepLog(out.GPTResponse, step, map[string]any{"status": "unknown"}) } } // Paths without fill_fields (enhance_only / normalize_only) still get vector // categorize when AllowAI + embeddings are available. tryVectorCategorize(ctx, e, companyID, &out, categoryNames, policy) preserveCategoryIfEmpty(&out, in.PriorCategory) noteMissingCategory(&out, policy, e != nil && e.Vector != nil && e.Vector.Enabled()) syncCategoryName(&out, in.CategoryNamesByUID) displayCat := categoryDisplayLabel(out) out.Name = preferredProductTitle(in.GTIN, out.Name, out.ProcessedName, in.Name, in.PriorProcessedName) out.ProcessedName = preferredProductTitle(in.GTIN, out.ProcessedName, out.Name, in.PriorProcessedName, in.Name) out.Description = preferredProductDescription(out.Name, out.Description, out.ProcessedDescription, in.Description) out.ProcessedDescription = preferredProductDescription(out.ProcessedName, out.ProcessedDescription, out.Description, in.PriorProcessedDescription) if out.ProcessedName == "" { out.ProcessedName = out.Name } _, finalDescTpl := categoryFormulasFor(in, out.Category) if out.ProcessedDescription == "" || isWeakPriorEnhanceDescription(out.ProcessedDescription, out.ProcessedName, out.Name) { if synth := synthesizeProductDescription(out.ProcessedName, displayCat, in.Language, out.Attributes, finalDescTpl); synth != "" { out.ProcessedDescription = synth } } if out.Description == "" || isWeakPriorEnhanceDescription(out.Description, out.Name, out.ProcessedName) { if out.ProcessedDescription != "" && !isWeakPriorEnhanceDescription(out.ProcessedDescription, out.Name, out.ProcessedName) { out.Description = out.ProcessedDescription } else if synth := synthesizeProductDescription(out.Name, displayCat, in.Language, out.Attributes, finalDescTpl); synth != "" { out.Description = synth if out.ProcessedDescription == "" || isWeakPriorEnhanceDescription(out.ProcessedDescription, out.ProcessedName, out.Name) { out.ProcessedDescription = synth } } } if out.Attributes == nil { out.Attributes = map[string]any{} } // Persist the same sanitize + category allowlist used by enhance (poll already // projects clean attrs; DB must not keep feed junk like zavora/vzmetenje). allowed := enhanceAllowedAttrKeys(in, out.Category) out.Attributes = AttrsForPersist(out.Attributes, allowed) out.ProcessedAttributes = AttrsForPersist(out.ProcessedAttributes, allowed) if len(out.ProcessedAttributes) == 0 { out.ProcessedAttributes = out.Attributes } // Prefer EngineProviderMode (job/completer label) over stamping "unknown" on // 0-token paths (hash-skip / AI skipped). processOne also treats unknown as empty. if isUnknownProviderMode(out.AIProviderMode) { out.AIProviderMode = preferKnownProviderMode(e.EngineProviderMode(), out.AIProviderMode) } if len(out.Notes) > 0 { out.GPTResponse["notes"] = out.Notes } return out, nil } // preserveCategoryIfEmpty keeps an existing processed category when normalize/AI // left Category empty (common for A1 feeds where category lives only on processed). func preserveCategoryIfEmpty(out *StepResult, prior string) { if out == nil || strings.TrimSpace(out.Category) != "" { return } prior = strings.TrimSpace(prior) if prior == "" { return } out.Category = SanitizeText(prior) if out.FieldSources == nil { out.FieldSources = map[string]any{} } out.FieldSources["category"] = "prior_processed" } // tryVectorCategorize sets Category from embeddings when mapped unique_id is absent. // Requires policy.AllowAI and a configured/enabled VectorCategorizer. func tryVectorCategorize(ctx context.Context, e *Engine, companyID string, out *StepResult, categoryNames []string, policy StepPolicy) { if out == nil || strings.TrimSpace(out.Category) != "" { return } if !policy.AllowAI { return } if e == nil || e.Vector == nil || !e.Vector.Enabled() { return } text := strings.TrimSpace(out.Name + " " + out.Description) if text == "" { return } cat, err := e.Vector.SuggestCategory(ctx, companyID, text, categoryNames) if err != nil { out.Notes = append(out.Notes, "vector_categorize: "+TruncateError(err)) appendStepLog(out.GPTResponse, "vector_categorize", map[string]any{ "status": "failed", "error": TruncateError(err), }) return } if strings.TrimSpace(cat) == "" { return } out.Category = SanitizeOutput(cat) if out.FieldSources == nil { out.FieldSources = map[string]any{} } out.FieldSources["category"] = "vector" out.Notes = append(out.Notes, "category: vector") appendStepLog(out.GPTResponse, "vector_categorize", map[string]any{ "status": "ok", "category": out.Category, }) } // noteMissingCategory records why Category stayed empty (mapped absent; vector skipped or failed). func noteMissingCategory(out *StepResult, policy StepPolicy, vectorEnabled bool) { if out == nil || strings.TrimSpace(out.Category) != "" { return } for _, n := range out.Notes { if strings.HasPrefix(n, "category: unset") || strings.HasPrefix(n, "category: vector") { return } } switch { case !policy.AllowAI: out.Notes = append(out.Notes, "category: unset (no mapped unique_id; AI/vector not allowed)") case !vectorEnabled: out.Notes = append(out.Notes, "category: unset (no mapped unique_id; vector embeddings unavailable)") default: out.Notes = append(out.Notes, "category: unset (no mapped unique_id; vector did not match)") } } func resolveSteps(processingType string) []string { switch strings.ToLower(strings.TrimSpace(processingType)) { case "enhance", "enhance_only", "enhance-only", "title", "description": return []string{StepNormalize, StepAIEnhance} case "attributes", "attributes_only", "specs", "specifications": return []string{StepNormalize, StepParseSpecs, StepFillFields} case "eprel", "eprel_only": // parse_specs first so eprel_id buried in specifications/attributes is visible. return []string{StepNormalize, StepParseSpecs, StepEPREL} case "normalize_only": return []string{StepNormalize} case "categorize", "categorize_only", "categorize_enhance": // Legacy aliases → full deterministic + optional AI return append([]string{}, CanonicalSteps...) default: // full return append([]string{}, CanonicalSteps...) } } // InitialStepProgress builds pending step_progress rows for a job. func InitialStepProgress(processingType string) []StepProgress { steps := resolveSteps(processingType) out := make([]StepProgress, 0, len(steps)) for _, s := range steps { out = append(out, StepProgress{Step: s, Status: "pending"}) } return out } func appendStepLog(gpt map[string]any, name string, raw any) { steps, _ := gpt["steps"].([]any) gpt["steps"] = append(steps, map[string]any{"step": name, "raw": raw}) } func (e *Engine) enhance(ctx context.Context, in ProductInput, category string, attrs map[string]any) (string, string, int, any, error) { sysTpl, userTpl := resolveProductPromptTemplates(in) hash := HashEnhanceInput(category, in.Name, in.Description, in.BrandPrompt, in.Language, sysTpl, userTpl, attrs) if e == nil || e.Completer == nil { name := preferredProductTitle(in.GTIN, in.Name, in.PriorProcessedName) return name, preferredProductDescription(name, in.Description, in.PriorProcessedDescription), 0, map[string]any{"status": "skipped", "input_hash": hash}, nil } // Skip LLM when inputs match the last successful enhance (before any credit debit), // but never reuse a thin / title-echo prior description, or a prompt-leakage title. if in.PriorEnhanceHash != "" && in.PriorEnhanceHash == hash && (in.PriorProcessedName != "" || in.PriorProcessedDescription != "") && !isPromptLabelTitle(in.PriorProcessedName) && !isWeakPriorEnhanceDescription(in.PriorProcessedDescription, in.PriorProcessedName, in.Name) { name := preferredProductTitle(in.GTIN, in.PriorProcessedName, in.Name) desc := preferredProductDescription(name, in.PriorProcessedDescription, in.Description) return name, desc, 0, map[string]any{ "status": "unchanged", "input_hash": hash, }, nil } system, user := RenderProductEnhancePrompts(sysTpl, userTpl, category, in.Name, in.Description, in.GTIN, in.BrandPrompt, in.Language, attrs) comp, obj, err := CompleteJSON(ctx, e.Completer, system, user, CompleteOptions{ MaxTokens: MaxTokensEnhance, Temperature: DefaultStructuredTemp, }) if err != nil { // Network/provider failure vs parse failure after retry name := preferredProductTitle(in.GTIN, in.Name, in.PriorProcessedName) desc := preferredProductDescription(name, in.Description, in.PriorProcessedDescription) if isWeakPriorEnhanceDescription(desc, name) { if synth := synthesizeProductDescription(name, category, in.Language, attrs, in.DescriptionTemplate); synth != "" { desc = synth } } if obj == nil && comp.Text == "" { return name, desc, 0, map[string]any{ "provider": "passthrough", "error": TruncateError(err), }, err } // Parse failed after retry — keep usable copy; synthesize when empty/weak. return name, desc, comp.TotalTokens, map[string]any{ "status": "parse_failed", "error": "AI returned invalid JSON; kept original title/description", "raw": truncateRunes(comp.Text, 200), }, nil } name := preferredProductTitle(in.GTIN, SanitizeOutput(fmt.Sprint(obj["name"])), in.Name, in.PriorProcessedName) desc := preferredProductDescription(name, SanitizeOutput(fmt.Sprint(obj["description"])), in.Description, in.PriorProcessedDescription) synthesized := false if isWeakPriorEnhanceDescription(desc, name) { if synth := synthesizeProductDescription(name, category, in.Language, attrs, in.DescriptionTemplate); synth != "" { desc = synth synthesized = true } } meta := map[string]any{ "status": "ok", "raw": comp.Raw, } // Heuristic synthesize must not poison enhance_input_hash (would hash-skip // formula-aware LLM copy on reprocess). Only persist hash for real LLM quality. if synthesized { meta["status"] = "synthesized" } else if !isWeakPriorEnhanceDescription(desc, name) { meta["input_hash"] = hash } return name, desc, comp.TotalTokens, meta, nil } func sanitizeJSON(v any) string { if v == nil { return "{}" } b, err := json.Marshal(v) if err != nil { return "{}" } return SanitizeText(string(b)) } func firstLine(s string) string { s = strings.TrimSpace(s) if i := strings.IndexByte(s, '\n'); i >= 0 { s = s[:i] } return SanitizeOutput(strings.Trim(s, "\"'` ")) } // labeledPromptValue returns the first usable line after any of the given labels // (case-insensitive), e.g. "Name:" / "Desc:" from ProductEnhanceUser. // Skips matches whose value is prompt-label / formula-leakage text so instruction // bullets like "- name: short retail title; follow any Title formula…" do not // win over the later "Name: " line in CategoryEnhanceUserTemplate. func labeledPromptValue(user string, labels ...string) string { lower := strings.ToLower(user) type hit struct { at int label string } var hits []hit for _, label := range labels { label = strings.ToLower(strings.TrimSpace(label)) if label == "" { continue } searchFrom := 0 for { rel := strings.Index(lower[searchFrom:], label) if rel < 0 { break } at := searchFrom + rel hits = append(hits, hit{at: at, label: label}) searchFrom = at + len(label) } } if len(hits) == 0 { return "" } sort.Slice(hits, func(i, j int) bool { return hits[i].at < hits[j].at }) for _, h := range hits { rest := user[h.at+len(h.label):] if j := strings.Index(strings.ToLower(rest), "attrs:"); j >= 0 { rest = rest[:j] } if j := strings.Index(strings.ToLower(rest), "attributes:"); j >= 0 { rest = rest[:j] } val := firstLine(rest) if val == "" || isPromptLabelTitle(val) { continue } return val } return "" } // isPromptLabelTitle detects enhance pollution where the model echoed a // prompt header ("Category:" / "Category: 120") as the product title, or // leaked instruction scaffolding from CategoryEnhanceUserTemplate / // AppendFormulaConstraints into name. func isPromptLabelTitle(s string) bool { s = strings.TrimSpace(s) if s == "" || s == "" { return false } if isPromptLeakageTitle(s) { return true } lower := strings.ToLower(s) for _, label := range []string{ "category", "name", "desc", "description", "attrs", "attributes", "current name", "current description", } { if lower == label || lower == label+":" { return true } if strings.HasPrefix(lower, label+":") || strings.HasPrefix(lower, label+" :") { return true } } return false } // isPromptLeakageTitle detects when an LLM echoed enhance-prompt instructions // (Title formula / short retail title / Schema / Reply with ONLY JSON / …) // as the product name instead of a real title. func isPromptLeakageTitle(s string) bool { s = strings.TrimSpace(s) if s == "" || s == "" { return false } lower := strings.ToLower(s) for _, phrase := range promptLeakagePhrases { if strings.Contains(lower, phrase) { return true } } // Long dumps of the enhance template: any formula/schema keyword is enough. if len([]rune(s)) > 120 { for _, kw := range promptLeakageLongKeywords { if strings.Contains(lower, kw) { return true } } } return false } // Phrases copied from aiprompts.CategoryEnhanceUserTemplate, BuiltInDefaults, // and processing.AppendFormulaConstraints / FormatTitleFormulaConstraint. var promptLeakagePhrases = []string{ "title formula", "follow any", "constraints that follow", "short retail title", "use attrs", "write name in", "schema:", "reply with only json", "your reply is parsed as json", "description formula", "build name from attrs", "prefer attrs values", "order matters; join with", "do not hardcode a language", } var promptLeakageLongKeywords = []string{ "formula", "constraints", "schema", "json", "attrs", "retail title", } // preferredProductTitle picks the first usable title, skipping empty values, // prompt-label echoes like "Category:", and instruction-text leakage. func preferredProductTitle(gtin string, candidates ...string) string { for _, c := range candidates { c = strings.TrimSpace(c) if c == "" || c == "" || isPromptLabelTitle(c) { continue } return SanitizeOutput(c) } if strings.TrimSpace(gtin) != "" { return SanitizeText("Product " + strings.TrimSpace(gtin)) } return "Product" } // Thin wrappers keep processing call sites stable; logic lives in company so // catalog.RepairWeakEnhanceHashes can reuse it without an import cycle. func isWeakPriorEnhanceDescription(priorDesc string, titles ...string) bool { return company.IsWeakPriorEnhanceDescription(priorDesc, titles...) } func containsWeakFillerPhrase(desc string) bool { return company.ContainsWeakFillerPhrase(desc) } func descriptionEchoesTitle(desc, title string) bool { return company.DescriptionEchoesTitle(desc, title) } // preferredProductDescription picks the first usable description, skipping empty // values, prompt-label echoes, weak filler phrases, and copy that merely repeats // the product title. func preferredProductDescription(title string, candidates ...string) string { for _, c := range candidates { c = strings.TrimSpace(c) if c == "" || c == "" || isPromptLabelTitle(c) { continue } if descriptionEchoesTitle(c, title) { continue } if containsWeakFillerPhrase(c) { continue } return SanitizeOutput(c) } return "" } func attrLookupCI(attrs map[string]any, keys ...string) string { if attrs == nil { return "" } for _, want := range keys { want = strings.TrimSpace(want) if want == "" { continue } if v := strings.TrimSpace(stringFromAny(attrs[want])); v != "" { return v } for k, raw := range attrs { if strings.EqualFold(strings.TrimSpace(k), want) { if v := strings.TrimSpace(stringFromAny(raw)); v != "" { return v } } } } return "" } func formatAttrDimParts(attrs map[string]any, maxParts int) []string { if attrs == nil || maxParts <= 0 { return nil } prefer := []string{ "width", "height", "depth", "weight", "max_load", "max load", "load_capacity", "vesa", "screen_size", "diagonal", "color", "material", "size", } parts := make([]string, 0, maxParts) seen := map[string]struct{}{} add := func(k, v string) { k = strings.TrimSpace(k) v = strings.TrimSpace(v) if k == "" || v == "" { return } lk := strings.ToLower(k) if _, ok := seen[lk]; ok { return } seen[lk] = struct{}{} parts = append(parts, fmt.Sprintf("%s %s", k, v)) } for _, k := range prefer { if len(parts) >= maxParts { break } if v := attrLookupCI(attrs, k); v != "" { add(k, v) } } return parts } // synthesizeProductDescription prefers a category description_template skeleton // (HTML section types) when present; otherwise falls back to plain title synthesize. func synthesizeProductDescription(title, category, language string, attrs map[string]any, descriptionTemplate any) string { if sections, ok := parseDescriptionFormulaSections(descriptionTemplate); ok && len(sections) > 0 { if out := synthesizeDescriptionFromFormula(title, category, language, attrs, sections); out != "" { return out } } return synthesizeDescriptionFromTitle(title, category, language, attrs) } // synthesizeDescriptionFromFormula builds a minimal HTML description matching // category description_template section types so timeout/fallback still respects // A1 category structure (unlike plain title synthesize). func synthesizeDescriptionFromFormula(title, category, language string, attrs map[string]any, sections []descriptionFormulaSection) string { title = strings.TrimSpace(title) if title == "" || title == "" || isPromptLabelTitle(title) { return "" } base := synthesizeDescriptionFromTitle(title, category, language, attrs) if base == "" { return "" } dims := formatAttrDimParts(attrs, 6) var b strings.Builder paraUsed := false listUsed := false for _, s := range sections { typ := strings.ToLower(strings.TrimSpace(s.Type)) switch typ { case "h1", "h2", "h3", "h4": heading := title if typ != "h1" { if isSlovenianContentLanguage(language) { heading = "Ključne lastnosti" } else { heading = "Key features" } } fmt.Fprintf(&b, "<%s>%s", typ, SanitizeOutput(heading), typ) case "ul": b.WriteString("
    ") items := dims if len(items) == 0 { items = []string{base} } for _, it := range items { fmt.Fprintf(&b, "
  • %s
  • ", SanitizeOutput(it)) } b.WriteString("
") listUsed = true default: // p and unknown → paragraph body := base if paraUsed && len(dims) > 0 && !listUsed { if isSlovenianContentLanguage(language) { body = "Ključne specifikacije: " + strings.Join(dims, ", ") + "." } else { body = "Key specs: " + strings.Join(dims, ", ") + "." } } fmt.Fprintf(&b, "

%s

", SanitizeOutput(body)) paraUsed = true } } return SanitizeOutput(b.String()) } // synthesizeDescriptionFromTitle builds a short factual fallback when enhance // returns empty/title-echo/filler copy. Uses title, category, brand, model, and // key dims. language is a content-language code (en/sl/…) or English label. func synthesizeDescriptionFromTitle(title, category, language string, attrs map[string]any) string { title = strings.TrimSpace(title) if title == "" || title == "" || isPromptLabelTitle(title) { return "" } cat := strings.TrimSpace(category) if strings.EqualFold(cat, "general") { cat = "" } brand := attrLookupCI(attrs, "brand") model := attrLookupCI(attrs, "product_model", "model", "sku") dims := formatAttrDimParts(attrs, 3) sl := isSlovenianContentLanguage(language) var b strings.Builder if sl { b.WriteString(title) switch { case cat != "" && brand != "": fmt.Fprintf(&b, " je izdelek v kategoriji %s znamke %s", cat, brand) case cat != "": fmt.Fprintf(&b, " je izdelek v kategoriji %s", cat) case brand != "": fmt.Fprintf(&b, " je izdelek znamke %s", brand) default: b.WriteString(" je katalogski izdelek z znanimi atributi") } if model != "" && !strings.Contains(strings.ToLower(title), strings.ToLower(model)) { fmt.Fprintf(&b, " (model %s)", model) } if len(dims) > 0 { fmt.Fprintf(&b, ". Ključne specifikacije: %s", strings.Join(dims, ", ")) } b.WriteByte('.') } else { b.WriteString(title) switch { case cat != "" && brand != "": fmt.Fprintf(&b, " is a %s product from %s", cat, brand) case cat != "": fmt.Fprintf(&b, " is listed in the %s category", cat) case brand != "": fmt.Fprintf(&b, " is a product from %s", brand) default: b.WriteString(" is a catalog product with the known attributes") } if model != "" && !strings.Contains(strings.ToLower(title), strings.ToLower(model)) { fmt.Fprintf(&b, " (model %s)", model) } if len(dims) > 0 { fmt.Fprintf(&b, ". Key specs: %s", strings.Join(dims, ", ")) } b.WriteByte('.') } out := SanitizeOutput(b.String()) // Never emit sole retail-filler when title/attrs exist — strip legacy phrase if any helper reintroduces it. if containsWeakFillerPhrase(out) { out = strings.TrimSpace(strings.ReplaceAll(out, "Ready for retail listing.", "")) out = strings.TrimSpace(strings.ReplaceAll(out, "ready for retail listing.", "")) out = strings.TrimSpace(strings.Trim(out, ".")) + "." } return out } func isSlovenianContentLanguage(raw string) bool { raw = strings.TrimSpace(raw) if raw == "" { return false } if code, err := company.ParseLanguage(raw, false); err == nil && code == "sl" { return true } lower := strings.ToLower(raw) return lower == "slovenian" || strings.Contains(lower, "slovenian") }