mirror of
https://github.com/infiniflow/ragflow.git
synced 2026-08-12 11:43:39 +08:00
test(parser): add shared golden-doc + alignment helpers (#18098)
Centralize the shared golden-doc + alignment helpers in `align_test.go` so the format-specific PRs (text&code, markdown golden, HTML) reuse one implementation instead of each carrying their own copy of the scaffolding.
This commit is contained in:
@@ -22,6 +22,7 @@ package parser
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"os"
|
||||
"regexp"
|
||||
"strings"
|
||||
@@ -188,19 +189,110 @@ func diffReport(g, p string) string {
|
||||
return "alignment mismatch after normalization:\n--- GO ---\n" + g + "\n--- PY ---\n" + p
|
||||
}
|
||||
|
||||
// LoadGolden reads a Python golden JSON file (a JSON list of item objects)
|
||||
// produced by the Python flow parser for the same input.
|
||||
// GoldenDoc is a {meta, items} Python golden baseline. Meta records how the
|
||||
// baseline was produced (generator, sample, delimiter, accepted divergences)
|
||||
// so it stays reproducible without a committed generator script; Items is the
|
||||
// list of parsed output items compared against Go's parser.
|
||||
type GoldenDoc struct {
|
||||
Meta map[string]any
|
||||
Items []map[string]any
|
||||
}
|
||||
|
||||
// parseGolden unmarshals a golden file that may be either a bare JSON array of
|
||||
// items (legacy format) or a {meta, items} document (current format). It
|
||||
// returns the full document either way. Tolerant parsing keeps older
|
||||
// callers/tests working after the format gained a meta block.
|
||||
func parseGolden(t *testing.T, data []byte) (*GoldenDoc, error) {
|
||||
t.Helper()
|
||||
var doc GoldenDoc
|
||||
if err := json.Unmarshal(data, &doc); err == nil && doc.Items != nil {
|
||||
return &doc, nil
|
||||
}
|
||||
// Legacy flat-array format: treat the whole file as the items list.
|
||||
var items []map[string]any
|
||||
if err := json.Unmarshal(data, &items); err != nil {
|
||||
return nil, fmt.Errorf("parse golden: %w", err)
|
||||
}
|
||||
return &GoldenDoc{Items: items}, nil
|
||||
}
|
||||
|
||||
// LoadGolden reads a Python golden JSON file and returns its items. The file
|
||||
// may be a bare array (legacy) or a {meta, items} document; either way only
|
||||
// the items are returned, so existing callers keep working unchanged.
|
||||
func LoadGolden(t *testing.T, path string) []map[string]any {
|
||||
t.Helper()
|
||||
data, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
t.Fatalf("load golden %s: %v", path, err)
|
||||
}
|
||||
var items []map[string]any
|
||||
if err := json.Unmarshal(data, &items); err != nil {
|
||||
t.Fatalf("parse golden %s: %v", path, err)
|
||||
doc, err := parseGolden(t, data)
|
||||
if err != nil {
|
||||
t.Fatalf("load golden %s: %v", path, err)
|
||||
}
|
||||
return items
|
||||
if len(doc.Items) == 0 {
|
||||
t.Fatalf("golden %s has no items", path)
|
||||
}
|
||||
return doc.Items
|
||||
}
|
||||
|
||||
// LoadGoldenDoc reads a Python golden JSON file and returns the full
|
||||
// {meta, items} document, including the meta block. Used by tests that drive
|
||||
// behavior from the golden's metadata (e.g. accepted_divergences).
|
||||
func LoadGoldenDoc(t *testing.T, path string) *GoldenDoc {
|
||||
t.Helper()
|
||||
data, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
t.Fatalf("load golden %s: %v", path, err)
|
||||
}
|
||||
doc, err := parseGolden(t, data)
|
||||
if err != nil {
|
||||
t.Fatalf("load golden %s: %v", path, err)
|
||||
}
|
||||
if len(doc.Items) == 0 {
|
||||
t.Fatalf("golden %s has no items", path)
|
||||
}
|
||||
return doc
|
||||
}
|
||||
|
||||
// AcceptedDivergences returns the doc_type_kwd values the golden baseline
|
||||
// declares as accepted representation differences (e.g. "table"/"image"), so
|
||||
// the comparison can ignore them on both sides. Driven entirely by the
|
||||
// golden's meta block — the test holds no hardcoded divergence list.
|
||||
func AcceptedDivergences(meta map[string]any) []string {
|
||||
raw, ok := meta["accepted_divergences"]
|
||||
if !ok {
|
||||
return nil
|
||||
}
|
||||
list, ok := raw.([]any)
|
||||
if !ok {
|
||||
return nil
|
||||
}
|
||||
out := make([]string, 0, len(list))
|
||||
for _, e := range list {
|
||||
if s, ok := e.(string); ok {
|
||||
out = append(out, s)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// FilterOutDocTypes returns the items whose doc_type_kwd is NOT in drop. Used
|
||||
// to exclude the meta-declared accepted divergences from the comparison.
|
||||
func FilterOutDocTypes(items []map[string]any, drop []string) []map[string]any {
|
||||
if len(drop) == 0 {
|
||||
return items
|
||||
}
|
||||
banned := make(map[string]bool, len(drop))
|
||||
for _, d := range drop {
|
||||
banned[d] = true
|
||||
}
|
||||
out := make([]map[string]any, 0, len(items))
|
||||
for _, it := range items {
|
||||
if v, _ := it["doc_type_kwd"].(string); !banned[v] {
|
||||
out = append(out, it)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// MarkdownAlignOptions returns the normalizer preset for Markdown. The order
|
||||
@@ -232,6 +324,114 @@ func MarkdownAlignOptions(delimiter string) AlignOptions {
|
||||
}
|
||||
}
|
||||
|
||||
// DefaultMarkdownDelimiter is the flow parser's default Markdown delimiter
|
||||
// set, used when generating/loading the golden baseline.
|
||||
const DefaultMarkdownDelimiter = "\n!?;。;!?"
|
||||
// TextCodeAlignOptions returns the normalizer preset for the text&code family.
|
||||
// Unlike markdown it has no syntax or HTML markup to strip, so only the
|
||||
// delimiter-set replacement and whitespace collapse run:
|
||||
// - WithDelimiterStrip: replace the delimiter runes Python consumes at split
|
||||
// points (kept inline on the Go side via keep_delimiters=True) with a space
|
||||
// so both sides keep the same token separation.
|
||||
// - CollapseWhitespace last: folds the introduced spaces and any inter-segment
|
||||
// gaps into single spaces.
|
||||
//
|
||||
// Reused by every text&code alignment test; shares CompareAlignment with the
|
||||
// other format presets (MarkdownAlignOptions).
|
||||
func TextCodeAlignOptions(delimiter string) AlignOptions {
|
||||
return AlignOptions{
|
||||
Normalizers: []Normalizer{
|
||||
WithDelimiterStrip(delimiter),
|
||||
CollapseWhitespace(),
|
||||
},
|
||||
ItemKey: "text",
|
||||
}
|
||||
}
|
||||
|
||||
// htmlHeadingMarkerRE matches a leading ATX heading marker so Python's
|
||||
// "# Title" (deepdoc merge_block_text prefixes h1–h6 with "# ") can be
|
||||
// normalized to Go's clean heading text.
|
||||
var htmlHeadingMarkerRE = regexp.MustCompile(`(?m)^#{1,6}\s+`)
|
||||
|
||||
// StripHTMLHeadingMarker returns a Normalizer that removes a leading ATX
|
||||
// heading marker ("#"/"##"/…) from a line. Python's HTML flow parser
|
||||
// (deepdoc parser.py merge_block_text) prefixes h1–h6 sections with "# ",
|
||||
// while the Go HTML parser emits clean heading text. This is a representation
|
||||
// difference, not a content divergence, so it is stripped before comparing.
|
||||
// It must run before CollapseWhitespace because the marker relies on the line
|
||||
// start.
|
||||
func StripHTMLHeadingMarker() Normalizer {
|
||||
return func(s string) string {
|
||||
return htmlHeadingMarkerRE.ReplaceAllString(s, "")
|
||||
}
|
||||
}
|
||||
|
||||
// HTMLAlignOptions returns the normalizer preset for HTML. Order matters:
|
||||
// - StripHTMLHeadingMarker first: drops the "# " Python prefixes from h1–h6
|
||||
// sections (relies on the line start, so before CollapseWhitespace).
|
||||
// - StripHTMLTags next: replace table/HTML tags with a space (not delete) so
|
||||
// adjacent cell text does not fuse, e.g. "<td>A</td><td>B</td>" → "A B".
|
||||
// - CollapseWhitespace last: folds all remaining internal whitespace (the
|
||||
// inter-tag gaps of a table, the space from any heading-marker removal)
|
||||
// into single spaces.
|
||||
//
|
||||
// No WithDelimiterStrip: HTML is emitted one item per block (no delimiter
|
||||
// split), and the Python flow chunks HTML at 512 tokens — boundaries differ,
|
||||
// but CompareAlignment concatenates normalized text boundary-agnostically, so
|
||||
// only the concatenated content (order preserved) is compared.
|
||||
//
|
||||
// Reused by the HTML alignment test; shares CompareAlignment with the other
|
||||
// format presets (MarkdownAlignOptions, TextCodeAlignOptions).
|
||||
func HTMLAlignOptions() AlignOptions {
|
||||
return AlignOptions{
|
||||
Normalizers: []Normalizer{
|
||||
StripHTMLHeadingMarker(),
|
||||
StripHTMLTags(),
|
||||
CollapseWhitespace(),
|
||||
},
|
||||
ItemKey: "text",
|
||||
}
|
||||
}
|
||||
|
||||
// TestParseGolden locks the two tolerated golden formats (legacy bare array
|
||||
// and {meta, items}) plus the corrupt-input failure path, so future format
|
||||
// evolution of the golden files cannot silently change parse behavior.
|
||||
func TestParseGolden(t *testing.T) {
|
||||
// Legacy bare-array format: parsed as the items list, no meta.
|
||||
legacy := []byte(`[{"text":"hello"},{"text":"world"}]`)
|
||||
doc, err := parseGolden(t, legacy)
|
||||
if err != nil {
|
||||
t.Fatalf("legacy: parse golden: %v", err)
|
||||
}
|
||||
if doc.Meta != nil {
|
||||
t.Fatalf("legacy: want nil meta, got %v", doc.Meta)
|
||||
}
|
||||
if len(doc.Items) != 2 {
|
||||
t.Fatalf("legacy: want 2 items, got %d", len(doc.Items))
|
||||
}
|
||||
if got := doc.Items[0]["text"]; got != "hello" {
|
||||
t.Fatalf("legacy: item0 text = %v, want hello", got)
|
||||
}
|
||||
|
||||
// {meta, items} format: meta preserved, items parsed.
|
||||
meta := `{"meta":{"generator":"py","sample":"in.txt","accepted_divergences":["table"]},"items":[{"text":"a"},{"text":"b"}]}`
|
||||
doc2, err := parseGolden(t, []byte(meta))
|
||||
if err != nil {
|
||||
t.Fatalf("{meta,items}: parse golden: %v", err)
|
||||
}
|
||||
if doc2.Meta == nil {
|
||||
t.Fatalf("{meta,items}: want non-nil meta")
|
||||
}
|
||||
if got := doc2.Meta["generator"]; got != "py" {
|
||||
t.Fatalf("{meta,items}: meta.generator = %v, want py", got)
|
||||
}
|
||||
if len(doc2.Items) != 2 {
|
||||
t.Fatalf("{meta,items}: want 2 items, got %d", len(doc2.Items))
|
||||
}
|
||||
if got := doc2.Items[1]["text"]; got != "b" {
|
||||
t.Fatalf("{meta,items}: item1 text = %v, want b", got)
|
||||
}
|
||||
|
||||
// Corrupt input must fail loudly: a broken golden should error, not
|
||||
// silently fall through to an empty baseline.
|
||||
if _, err := parseGolden(t, []byte(`{not valid json`)); err == nil {
|
||||
t.Fatalf("corrupt input: expected parseGolden to fail, but it returned")
|
||||
}
|
||||
}
|
||||
|
||||
34
internal/parser/parser/delimiter.go
Normal file
34
internal/parser/parser/delimiter.go
Normal file
@@ -0,0 +1,34 @@
|
||||
//
|
||||
// Copyright 2026 The InfiniFlow Authors. All Rights Reserved.
|
||||
//
|
||||
// Licensed under the Apache License, Version 2.0 (the "License");
|
||||
// you may not use this file except in compliance with the License.
|
||||
// You may obtain a copy of the License at
|
||||
//
|
||||
// http://www.apache.org/licenses/LICENSE-2.0
|
||||
//
|
||||
// Unless required by applicable law or agreed to in writing, software
|
||||
// distributed under the License is distributed on an "AS IS" BASIS,
|
||||
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
// Warranties, INCLUDING THE WARRANTIES OF MERCHANTABILITY AND
|
||||
// FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE
|
||||
// AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING
|
||||
// FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
|
||||
// DEALINGS IN THE SOFTWARE.
|
||||
//
|
||||
|
||||
package parser
|
||||
|
||||
// DefaultMarkdownDelimiter is the flow parser's default markdown delimiter
|
||||
// set, used when generating/loading the golden baseline and when the markdown
|
||||
// alignment test normalizes delimiters. It is a non-test symbol so both the
|
||||
// production parsers and the alignment tests share one source of truth.
|
||||
const DefaultMarkdownDelimiter = "\n!?;。;!?"
|
||||
|
||||
// DefaultTextCodeDelimiter is the flow parser's default text&code delimiter
|
||||
// set, used by TextParser and when the text&code alignment test normalizes
|
||||
// delimiters. It is a non-test symbol so both the production parser and the
|
||||
// alignment tests share one source of truth (no duplicate hard-coded copy in
|
||||
// text_parser.go, which would otherwise drift silently).
|
||||
const DefaultTextCodeDelimiter = "\n!?;。;!?"
|
||||
Reference in New Issue
Block a user