From 4a8bb1b72ae8f4d48fffa0a66d527b8dc52a9889 Mon Sep 17 00:00:00 2001 From: Jack Date: Tue, 11 Aug 2026 18:00:25 +0800 Subject: [PATCH] test(parser): add shared golden-doc + alignment helpers (#18098) Centralize the shared golden-doc + alignment helpers in `align_test.go` so the format-specific PRs (text&code, markdown golden, HTML) reuse one implementation instead of each carrying their own copy of the scaffolding. --- internal/parser/parser/align_test.go | 218 +++++++++++++++++++++++++-- internal/parser/parser/delimiter.go | 34 +++++ 2 files changed, 243 insertions(+), 9 deletions(-) create mode 100644 internal/parser/parser/delimiter.go diff --git a/internal/parser/parser/align_test.go b/internal/parser/parser/align_test.go index ac4f991c07..0699948c87 100644 --- a/internal/parser/parser/align_test.go +++ b/internal/parser/parser/align_test.go @@ -22,6 +22,7 @@ package parser import ( "encoding/json" + "fmt" "os" "regexp" "strings" @@ -188,19 +189,110 @@ func diffReport(g, p string) string { return "alignment mismatch after normalization:\n--- GO ---\n" + g + "\n--- PY ---\n" + p } -// LoadGolden reads a Python golden JSON file (a JSON list of item objects) -// produced by the Python flow parser for the same input. +// GoldenDoc is a {meta, items} Python golden baseline. Meta records how the +// baseline was produced (generator, sample, delimiter, accepted divergences) +// so it stays reproducible without a committed generator script; Items is the +// list of parsed output items compared against Go's parser. +type GoldenDoc struct { + Meta map[string]any + Items []map[string]any +} + +// parseGolden unmarshals a golden file that may be either a bare JSON array of +// items (legacy format) or a {meta, items} document (current format). It +// returns the full document either way. Tolerant parsing keeps older +// callers/tests working after the format gained a meta block. +func parseGolden(t *testing.T, data []byte) (*GoldenDoc, error) { + t.Helper() + var doc GoldenDoc + if err := json.Unmarshal(data, &doc); err == nil && doc.Items != nil { + return &doc, nil + } + // Legacy flat-array format: treat the whole file as the items list. + var items []map[string]any + if err := json.Unmarshal(data, &items); err != nil { + return nil, fmt.Errorf("parse golden: %w", err) + } + return &GoldenDoc{Items: items}, nil +} + +// LoadGolden reads a Python golden JSON file and returns its items. The file +// may be a bare array (legacy) or a {meta, items} document; either way only +// the items are returned, so existing callers keep working unchanged. func LoadGolden(t *testing.T, path string) []map[string]any { t.Helper() data, err := os.ReadFile(path) if err != nil { t.Fatalf("load golden %s: %v", path, err) } - var items []map[string]any - if err := json.Unmarshal(data, &items); err != nil { - t.Fatalf("parse golden %s: %v", path, err) + doc, err := parseGolden(t, data) + if err != nil { + t.Fatalf("load golden %s: %v", path, err) } - return items + if len(doc.Items) == 0 { + t.Fatalf("golden %s has no items", path) + } + return doc.Items +} + +// LoadGoldenDoc reads a Python golden JSON file and returns the full +// {meta, items} document, including the meta block. Used by tests that drive +// behavior from the golden's metadata (e.g. accepted_divergences). +func LoadGoldenDoc(t *testing.T, path string) *GoldenDoc { + t.Helper() + data, err := os.ReadFile(path) + if err != nil { + t.Fatalf("load golden %s: %v", path, err) + } + doc, err := parseGolden(t, data) + if err != nil { + t.Fatalf("load golden %s: %v", path, err) + } + if len(doc.Items) == 0 { + t.Fatalf("golden %s has no items", path) + } + return doc +} + +// AcceptedDivergences returns the doc_type_kwd values the golden baseline +// declares as accepted representation differences (e.g. "table"/"image"), so +// the comparison can ignore them on both sides. Driven entirely by the +// golden's meta block — the test holds no hardcoded divergence list. +func AcceptedDivergences(meta map[string]any) []string { + raw, ok := meta["accepted_divergences"] + if !ok { + return nil + } + list, ok := raw.([]any) + if !ok { + return nil + } + out := make([]string, 0, len(list)) + for _, e := range list { + if s, ok := e.(string); ok { + out = append(out, s) + } + } + return out +} + +// FilterOutDocTypes returns the items whose doc_type_kwd is NOT in drop. Used +// to exclude the meta-declared accepted divergences from the comparison. +func FilterOutDocTypes(items []map[string]any, drop []string) []map[string]any { + if len(drop) == 0 { + return items + } + banned := make(map[string]bool, len(drop)) + for _, d := range drop { + banned[d] = true + } + out := make([]map[string]any, 0, len(items)) + for _, it := range items { + if v, _ := it["doc_type_kwd"].(string); !banned[v] { + out = append(out, it) + } + } + return out } // MarkdownAlignOptions returns the normalizer preset for Markdown. The order @@ -232,6 +324,114 @@ func MarkdownAlignOptions(delimiter string) AlignOptions { } } -// DefaultMarkdownDelimiter is the flow parser's default Markdown delimiter -// set, used when generating/loading the golden baseline. -const DefaultMarkdownDelimiter = "\n!?;。;!?" +// TextCodeAlignOptions returns the normalizer preset for the text&code family. +// Unlike markdown it has no syntax or HTML markup to strip, so only the +// delimiter-set replacement and whitespace collapse run: +// - WithDelimiterStrip: replace the delimiter runes Python consumes at split +// points (kept inline on the Go side via keep_delimiters=True) with a space +// so both sides keep the same token separation. +// - CollapseWhitespace last: folds the introduced spaces and any inter-segment +// gaps into single spaces. +// +// Reused by every text&code alignment test; shares CompareAlignment with the +// other format presets (MarkdownAlignOptions). +func TextCodeAlignOptions(delimiter string) AlignOptions { + return AlignOptions{ + Normalizers: []Normalizer{ + WithDelimiterStrip(delimiter), + CollapseWhitespace(), + }, + ItemKey: "text", + } +} + +// htmlHeadingMarkerRE matches a leading ATX heading marker so Python's +// "# Title" (deepdoc merge_block_text prefixes h1–h6 with "# ") can be +// normalized to Go's clean heading text. +var htmlHeadingMarkerRE = regexp.MustCompile(`(?m)^#{1,6}\s+`) + +// StripHTMLHeadingMarker returns a Normalizer that removes a leading ATX +// heading marker ("#"/"##"/…) from a line. Python's HTML flow parser +// (deepdoc parser.py merge_block_text) prefixes h1–h6 sections with "# ", +// while the Go HTML parser emits clean heading text. This is a representation +// difference, not a content divergence, so it is stripped before comparing. +// It must run before CollapseWhitespace because the marker relies on the line +// start. +func StripHTMLHeadingMarker() Normalizer { + return func(s string) string { + return htmlHeadingMarkerRE.ReplaceAllString(s, "") + } +} + +// HTMLAlignOptions returns the normalizer preset for HTML. Order matters: +// - StripHTMLHeadingMarker first: drops the "# " Python prefixes from h1–h6 +// sections (relies on the line start, so before CollapseWhitespace). +// - StripHTMLTags next: replace table/HTML tags with a space (not delete) so +// adjacent cell text does not fuse, e.g. "AB" → "A B". +// - CollapseWhitespace last: folds all remaining internal whitespace (the +// inter-tag gaps of a table, the space from any heading-marker removal) +// into single spaces. +// +// No WithDelimiterStrip: HTML is emitted one item per block (no delimiter +// split), and the Python flow chunks HTML at 512 tokens — boundaries differ, +// but CompareAlignment concatenates normalized text boundary-agnostically, so +// only the concatenated content (order preserved) is compared. +// +// Reused by the HTML alignment test; shares CompareAlignment with the other +// format presets (MarkdownAlignOptions, TextCodeAlignOptions). +func HTMLAlignOptions() AlignOptions { + return AlignOptions{ + Normalizers: []Normalizer{ + StripHTMLHeadingMarker(), + StripHTMLTags(), + CollapseWhitespace(), + }, + ItemKey: "text", + } +} + +// TestParseGolden locks the two tolerated golden formats (legacy bare array +// and {meta, items}) plus the corrupt-input failure path, so future format +// evolution of the golden files cannot silently change parse behavior. +func TestParseGolden(t *testing.T) { + // Legacy bare-array format: parsed as the items list, no meta. + legacy := []byte(`[{"text":"hello"},{"text":"world"}]`) + doc, err := parseGolden(t, legacy) + if err != nil { + t.Fatalf("legacy: parse golden: %v", err) + } + if doc.Meta != nil { + t.Fatalf("legacy: want nil meta, got %v", doc.Meta) + } + if len(doc.Items) != 2 { + t.Fatalf("legacy: want 2 items, got %d", len(doc.Items)) + } + if got := doc.Items[0]["text"]; got != "hello" { + t.Fatalf("legacy: item0 text = %v, want hello", got) + } + + // {meta, items} format: meta preserved, items parsed. + meta := `{"meta":{"generator":"py","sample":"in.txt","accepted_divergences":["table"]},"items":[{"text":"a"},{"text":"b"}]}` + doc2, err := parseGolden(t, []byte(meta)) + if err != nil { + t.Fatalf("{meta,items}: parse golden: %v", err) + } + if doc2.Meta == nil { + t.Fatalf("{meta,items}: want non-nil meta") + } + if got := doc2.Meta["generator"]; got != "py" { + t.Fatalf("{meta,items}: meta.generator = %v, want py", got) + } + if len(doc2.Items) != 2 { + t.Fatalf("{meta,items}: want 2 items, got %d", len(doc2.Items)) + } + if got := doc2.Items[1]["text"]; got != "b" { + t.Fatalf("{meta,items}: item1 text = %v, want b", got) + } + + // Corrupt input must fail loudly: a broken golden should error, not + // silently fall through to an empty baseline. + if _, err := parseGolden(t, []byte(`{not valid json`)); err == nil { + t.Fatalf("corrupt input: expected parseGolden to fail, but it returned") + } +} diff --git a/internal/parser/parser/delimiter.go b/internal/parser/parser/delimiter.go new file mode 100644 index 0000000000..9ab0e62e1d --- /dev/null +++ b/internal/parser/parser/delimiter.go @@ -0,0 +1,34 @@ +// +// Copyright 2026 The InfiniFlow Authors. All Rights Reserved. +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// Warranties, INCLUDING THE WARRANTIES OF MERCHANTABILITY AND +// FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE +// AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING +// FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER +// DEALINGS IN THE SOFTWARE. +// + +package parser + +// DefaultMarkdownDelimiter is the flow parser's default markdown delimiter +// set, used when generating/loading the golden baseline and when the markdown +// alignment test normalizes delimiters. It is a non-test symbol so both the +// production parsers and the alignment tests share one source of truth. +const DefaultMarkdownDelimiter = "\n!?;。;!?" + +// DefaultTextCodeDelimiter is the flow parser's default text&code delimiter +// set, used by TextParser and when the text&code alignment test normalizes +// delimiters. It is a non-test symbol so both the production parser and the +// alignment tests share one source of truth (no duplicate hard-coded copy in +// text_parser.go, which would otherwise drift silently). +const DefaultTextCodeDelimiter = "\n!?;。;!?"