From 93a175d04f2f88d2afcc0243d06b061a8adf0c08 Mon Sep 17 00:00:00 2001 From: QiuSW Date: Tue, 15 Sep 2026 14:23:02 +0800 Subject: [PATCH] =?UTF-8?q?fix(syb):=20=E7=A9=BA=E6=A0=BC=E5=88=86?= =?UTF-8?q?=E9=9A=94=E7=9A=84=E9=A2=9C=E8=89=B2=E5=B0=BA=E7=A0=81=E6=8C=89?= =?UTF-8?q?=E9=80=97=E5=8F=B7=E5=90=8C=E8=A7=84=E5=88=99=E6=8B=86=E5=88=86?= =?UTF-8?q?=20(#284)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit #274 删除 ambiguousPattern 以接受描述性颜色(黑色+白色 簡約親膚), 意图正确,但副作用是「黑色 XL」也被整条当成颜色、尺码丢失且判为 success。success 的行不进 AI 解析队列(ai_parse_batch.go 只选 uncertain/failed),因此没有任何机制会纠正它们。2026-09-15 线上有 152 条 parse_status='success' 且 target_size='' 的明细。 不恢复被删的旧闸——那会退回 #274 之前重新拒绝描述性颜色。改为把 空格当作与逗号同等的分隔符,复用逗号分支已有的规则:恰好一个空格 分隔 token 带明确尺码特征时才拆分,不构成猜测。 - 多于一个 token 命中尺码特征时不拆,对齐逗号分支「两侧均像尺码则 无法安全识别颜色」。 - 没有尺码 token 时原样保留为颜色,#274 的意图不受影响。 - 一并修正 parse.go 中停留在旧 uncertain 描述的文档注释。 新增 parse_whitespace_split_test.go 锁住完整对照表,含 #274 描述性 颜色不被退回的回归。 `[必须]` app/goauto/sybimport 存在 12 个先于本工单的失败用例,来自 #272/#274 改变契约后未更新的旧断言,已记录为 #285。本次提交前后失败 集合逐条比对完全一致,未新增。 Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01NTDbDcwbDw1TSAcE6wfh2F --- server/app/goauto/sybimport/parse.go | 60 +++++++++++--- .../sybimport/parse_whitespace_split_test.go | 80 +++++++++++++++++++ 2 files changed, 131 insertions(+), 9 deletions(-) create mode 100644 server/app/goauto/sybimport/parse_whitespace_split_test.go diff --git a/server/app/goauto/sybimport/parse.go b/server/app/goauto/sybimport/parse.go index 0b56c81..c339869 100644 --- a/server/app/goauto/sybimport/parse.go +++ b/server/app/goauto/sybimport/parse.go @@ -20,11 +20,46 @@ var bracketPattern = regexp.MustCompile(`【[^】]*】`) // deliberately narrow pattern. var explicitSizePattern = regexp.MustCompile(`(?i)^(?:均(?:码|碼|号|號)|one\s*size|free\s*size|x{0,4}[sml]|[2-9]xl|(?:加大|大|中|小)(?:码|碼|号|號)|\d+(?:\.\d+)?(?:cm|mm|m|码|碼|号|號|公分)|\d+(?:\.\d+)?(?:[-~~至到]\d+(?:\.\d+)?)?(?:斤|公斤|千克|kg))$`) -// ambiguousPattern flags leftover separators or multi-token noise after -// bracket stripping — the signal that a "clean" split still isn't reliable. -// Observed in the real SYB sample: "黑色+白色【純棉兩件裝】 簡約親膚" strips -// its bracket to "黑色+白色 簡約親膚", which still carries a '+' and internal -// whitespace, so it must not be reported as a confident match. +// splitOnWhitespace applies the comma rule to a spec that has no comma. +// +// `[必须]` #274 removed the old ambiguity guard so that descriptive colors +// like "黑色+白色 簡約親膚" are accepted instead of rejected. That is right, but +// it also let "黑色 XL" through as one confident **color**, dropping the size and +// marking the row success — and success rows never reach the AI parse queue +// (ai_parse_batch.go selects parse_status IN ('uncertain','failed')), so nothing +// ever corrects them. 152 such rows existed in production on 2026-09-15. +// +// Whitespace is therefore treated exactly like the comma: split only when +// **exactly one** whitespace-delimited token carries an explicit size signal. +// That is the same non-guessing rule the comma branch already uses, so +// "黑色 XL" recovers as color+size while "黑色+白色 簡約親膚" (no size token) +// keeps #274's descriptive-color behaviour untouched. +func splitOnWhitespace(value string) (color string, size string, ok bool) { + fields := strings.Fields(value) + if len(fields) < 2 { + return "", "", false + } + sizeIndex := -1 + for i, field := range fields { + if explicitSizePattern.MatchString(field) { + if sizeIndex >= 0 { + // More than one size-looking token: no safe colour decision, + // same as the comma branch's both-sides-are-size case. + return "", "", false + } + sizeIndex = i + } + } + if sizeIndex < 0 { + return "", "", false + } + remainder := append(append([]string{}, fields[:sizeIndex]...), fields[sizeIndex+1:]...) + color = strings.Join(remainder, " ") + if color == "" { + return "", "", false + } + return color, fields[sizeIndex], true +} // ParseResult is the color/size candidate extracted from one productSpec // string, plus how much the caller should trust it. @@ -43,10 +78,11 @@ type ParseResult struct { // Rules, derived from real SYB samples (demo/shunyunbaoerp_stock_list.har) // plus the boundary cases already confirmed in the #41 prototype: // - empty/whitespace-only input -> failed, nothing to extract. -// - no comma present (e.g. "均碼") -> uncertain: the whole string, with -// brackets stripped, becomes the size candidate; color stays empty. A -// single token with no separator cannot be split into two dimensions -// without guessing which one it is. +// - no comma present -> a single token becomes size when it carries an +// explicit size signal ("均碼") and colour otherwise (#274's descriptive +// colours). Multiple whitespace-delimited tokens are split by the same +// one-explicit-size rule the comma branch uses, so "黑色 XL" yields both +// dimensions instead of one over-long colour (see splitOnWhitespace). // - comma present and exactly one side has an explicit size signal -> that // side is size and the other side is color. This supports both observed // SYB orders without allowing AI or a fuzzy color dictionary to swap roles. @@ -75,6 +111,9 @@ func Parse(raw string) ParseResult { if explicitSizePattern.MatchString(size) { return ParseResult{Size: size, Status: models.SYBParseStatusSuccess, Note: "仅识别到尺码"} } + if color, sizeToken, ok := splitOnWhitespace(size); ok { + return ParseResult{Color: color, Size: sizeToken, Status: models.SYBParseStatusSuccess} + } return ParseResult{Color: size, Status: models.SYBParseStatusSuccess, Note: "仅识别到颜色"} } @@ -90,6 +129,9 @@ func Parse(raw string) ParseResult { if explicitSizePattern.MatchString(only) { return ParseResult{Size: only, Status: models.SYBParseStatusSuccess, Note: "仅识别到尺码"} } + if color, sizeToken, ok := splitOnWhitespace(only); ok { + return ParseResult{Color: color, Size: sizeToken, Status: models.SYBParseStatusSuccess} + } if only != "" { return ParseResult{Color: only, Status: models.SYBParseStatusSuccess, Note: "仅识别到颜色"} } diff --git a/server/app/goauto/sybimport/parse_whitespace_split_test.go b/server/app/goauto/sybimport/parse_whitespace_split_test.go new file mode 100644 index 0000000..f0fce30 --- /dev/null +++ b/server/app/goauto/sybimport/parse_whitespace_split_test.go @@ -0,0 +1,80 @@ +package sybimport_test + +import ( + "testing" + + "go-admin/app/goauto/models" + "go-admin/app/goauto/sybimport" +) + +// #284: 空格分隔的「颜色 尺码」曾被整条当成颜色,尺码丢失且状态判为 +// success。success 的行不进 AI 解析队列(ai_parse_batch.go 只选 uncertain/failed), +// 因此没有任何机制会纠正它们。本组用例锁住修复后的完整对照表,包括 +// #274 “接受描述性颜色”的意图必须保持不变。 +func TestParseSplitsWhitespaceWhenExactlyOneTokenIsASize(t *testing.T) { + cases := []struct { + name string + raw string + color string + size string + }{ + {"颜色在前", "黑色 XL", "黑色", "XL"}, + {"单字母尺码", "紅色 M", "紅色", "M"}, + // 与逗号分支一致:角色反转只看哪一侧带明确尺码特征,不看位置。 + {"尺码在前", "XL 黑色", "黑色", "XL"}, + {"颜色含多个词", "黑色 XL 加厚", "黑色 加厚", "XL"}, + } + for _, item := range cases { + t.Run(item.name, func(t *testing.T) { + result := sybimport.Parse(item.raw) + if result.Status != models.SYBParseStatusSuccess { + t.Fatalf("status = %q, want success", result.Status) + } + if result.Color != item.color || result.Size != item.size { + t.Fatalf("got color=%q size=%q, want color=%q size=%q", + result.Color, result.Size, item.color, item.size) + } + }) + } +} + +// `[必须]` #274 的意图是让描述性颜色被接受而不是被拒绝。#284 只能在 +// “恰好一个 token 带明确尺码特征”时介入;没有尺码 token 的字串必须原样保留为 +// 颜色,否则就把 #274 退回去了。 +func TestParseKeepsDescriptiveColorsWithoutASizeToken(t *testing.T) { + // 来自真实 SYB 样本:“黑色+白色【純棉兩件裝】 簡約親膚”剥离备注后的形态。 + result := sybimport.Parse("黑色+白色 簡約親膚") + if result.Status != models.SYBParseStatusSuccess { + t.Fatalf("status = %q, want success", result.Status) + } + if result.Color != "黑色+白色 簡約親膚" || result.Size != "" { + t.Fatalf("descriptive colour was altered: color=%q size=%q", result.Color, result.Size) + } +} + +// 多个 token 都像尺码时无法安全地挑出颜色,与逗号分支「两侧均具尺码特征」 +// 同一道理,保守处理:不拆。实际数据中该形态的占比尚无证据(见 #284 风险节)。 +func TestParseDoesNotSplitWhenSeveralTokensLookLikeSizes(t *testing.T) { + result := sybimport.Parse("S M L") + if result.Size != "" { + t.Fatalf("ambiguous multi-size spec must not be split, got size=%q", result.Size) + } +} + +// 单 token 的行为是 #274 定下的,#284 不得触碰。 +func TestParseSingleTokenBehaviourIsUnchanged(t *testing.T) { + if result := sybimport.Parse("均碼"); result.Size != "均碼" || result.Color != "" { + t.Fatalf("size-only token changed: color=%q size=%q", result.Color, result.Size) + } + if result := sybimport.Parse("黑色"); result.Color != "黑色" || result.Size != "" { + t.Fatalf("colour-only token changed: color=%q size=%q", result.Color, result.Size) + } +} + +// 逗号仍然优先,空格规则不得干扰已有的逗号拆分。 +func TestParseCommaStillTakesPrecedenceOverWhitespace(t *testing.T) { + result := sybimport.Parse("黑色,XL") + if result.Color != "黑色" || result.Size != "XL" { + t.Fatalf("comma split changed: color=%q size=%q", result.Color, result.Size) + } +}