diff --git a/baidu/parse_html.go b/baidu/parse_html.go index bbfe00f..caa78b4 100644 --- a/baidu/parse_html.go +++ b/baidu/parse_html.go @@ -41,9 +41,7 @@ func baiduResultSelector() string { func parseBaiduSelection(sel *goquery.Selection) []core.SearchResult { var results []core.SearchResult - rank := 1 - adRank := 1 - absoluteRank := 1 + rank := core.NewRankState(0) sel.Each(func(_ int, item *goquery.Selection) { isAd := baiduSelectionHasAdMarker(item) @@ -120,14 +118,7 @@ func parseBaiduSelection(sel *goquery.Selection) []core.SearchResult { desc = strings.TrimSpace(strings.Replace(full, title, "", 1)) } - resultRank := rank - if isAd { - resultRank = adRank - adRank++ - } else { - rank++ - } - + resultRank, absoluteRank := rank.Next(isAd) results = append(results, core.SearchResult{ Rank: resultRank, AbsoluteRank: absoluteRank, @@ -136,7 +127,6 @@ func parseBaiduSelection(sel *goquery.Selection) []core.SearchResult { Description: desc, Ad: isAd, }) - absoluteRank++ }) // Re-rank sequentially after dedup so callers get a clean 1..N sequence diff --git a/bing/parse_html.go b/bing/parse_html.go index 1050f7b..32ac356 100644 --- a/bing/parse_html.go +++ b/bing/parse_html.go @@ -21,9 +21,7 @@ func ParseHTML(r io.Reader) ([]core.SearchResult, error) { func parseBingDocument(doc *goquery.Document) []core.SearchResult { var results []core.SearchResult - rank := 1 - adRank := 1 - absoluteRank := 1 + rank := core.NewRankState(0) doc.Find(Selectors.ResultItems).Each(func(_ int, item *goquery.Selection) { isAd := item.Is(Selectors.Ads) @@ -41,99 +39,89 @@ func parseBingDocument(doc *goquery.Document) []core.SearchResult { if titleTag.Length() == 0 { return } + href, _ := titleTag.Attr("href") + title := bingDocumentTitle(item, titleTag) + desc := bingDocumentDescription(item, title) - link, exists := titleTag.Attr("href") - if !exists || link == "" || link == "#" { - return + if res, ok := assembleBingRow(href, title, desc, isAd, rank); ok { + results = append(results, res) } - - title := firstNonEmptyAttr(titleTag, "aria-label", "title") - if title == "" { - title = normalizeWhitespace(titleTag.Text()) - } - if title == "" { - title = extractFirstText(item, Selectors.TitleFallbacks) - } - if title == "" { - return - } - - desc := descriptionFromItem(item, title) - - resultRank := rank - if isAd { - resultRank = adRank - adRank++ - } else { - rank++ - } - - results = append(results, core.SearchResult{ - Rank: resultRank, - AbsoluteRank: absoluteRank, - URL: link, - Title: title, - Description: desc, - Ad: isAd, - }) - absoluteRank++ }) return core.AttachFeaturesToFirstResult(core.DeduplicateResults(results), extractBingFeatures(doc)) } -func extractFirstText(item *goquery.Selection, selectors []string) string { - for _, selector := range selectors { - if tag := item.Find(selector).First(); tag.Length() > 0 { - if text := normalizeWhitespace(tag.Text()); text != "" { - return text - } - if label := firstNonEmptyAttr(tag, "aria-label", "title"); label != "" { - return label - } +// assembleBingRow validates an already-extracted Bing row and assigns ranks. +// Shared by the rod (browser) and goquery (raw / parse) parsers, which differ +// only in how they pull title/href/desc out of the DOM. +func assembleBingRow(href, title, desc string, isAd bool, rank *core.RankState) (core.SearchResult, bool) { + url := strings.TrimSpace(href) + if url == "" || url == "#" || strings.HasPrefix(url, "javascript:") { + return core.SearchResult{}, false + } + if title == "" { + return core.SearchResult{}, false + } + + resultRank, absoluteRank := rank.Next(isAd) + return core.SearchResult{ + Rank: resultRank, + AbsoluteRank: absoluteRank, + URL: url, + Title: title, + Description: desc, + Ad: isAd, + }, true +} + +// bingDocumentTitle reproduces the rod path's title fallback for goquery: the +// title anchor's aria-label/title attribute, then its text, then any fallback +// selector's text or aria-label. +func bingDocumentTitle(item, titleTag *goquery.Selection) string { + if title := firstNonEmptyAttr(titleTag, "aria-label", "title"); title != "" { + return title + } + if title := core.NormalizeWhitespace(titleTag.Text()); title != "" { + return title + } + for _, selector := range Selectors.TitleFallbacks { + tag := item.Find(selector).First() + if tag.Length() == 0 { + continue + } + if text := core.NormalizeWhitespace(tag.Text()); text != "" { + return text + } + if label := firstNonEmptyAttr(tag, "aria-label", "title"); label != "" { + return label } } return "" } +// bingDocumentDescription reproduces the rod path's 3-selector description +// fallback plus the strip-title structural fallback. Bing renders snippet text +// with heavy source-indentation whitespace, so each candidate is collapsed. +func bingDocumentDescription(item *goquery.Selection, title string) string { + for _, selector := range []string{Selectors.DescPrimary, Selectors.DescFallback, Selectors.DescAny} { + if tag := item.Find(selector).First(); tag.Length() > 0 { + if text := core.NormalizeWhitespace(tag.Text()); text != "" { + return text + } + } + } + return core.NormalizeWhitespace(strings.Replace(item.Text(), title, "", 1)) +} + func firstNonEmptyAttr(item *goquery.Selection, attrs ...string) string { for _, attr := range attrs { value, exists := item.Attr(attr) if !exists { continue } - if value = normalizeWhitespace(value); value != "" { + if value = core.NormalizeWhitespace(value); value != "" { return value } } return "" } - -// descriptionFromItem extracts a description using the same 4-step fallback -// chain as the rod-based browser parser. Bing renders snippet text with heavy -// source-indentation whitespace, so each candidate is whitespace-collapsed. -func descriptionFromItem(item *goquery.Selection, title string) string { - if descTag := item.Find(Selectors.DescPrimary).First(); descTag.Length() > 0 { - if text := normalizeWhitespace(descTag.Text()); text != "" { - return text - } - } - if descTag := item.Find(Selectors.DescFallback).First(); descTag.Length() > 0 { - if text := normalizeWhitespace(descTag.Text()); text != "" { - return text - } - } - if descTag := item.Find(Selectors.DescAny).First(); descTag.Length() > 0 { - if text := normalizeWhitespace(descTag.Text()); text != "" { - return text - } - } - // Structural fallback: strip title from full text - return normalizeWhitespace(strings.Replace(item.Text(), title, "", 1)) -} - -// normalizeWhitespace collapses Bing's snippet-markup whitespace into single -// spaces (see core.NormalizeWhitespace). -func normalizeWhitespace(s string) string { - return core.NormalizeWhitespace(s) -} diff --git a/bing/search.go b/bing/search.go index 47b1e03..ea020f7 100644 --- a/bing/search.go +++ b/bing/search.go @@ -90,7 +90,7 @@ func bingElementMatches(el *rod.Element, selector string) bool { return err == nil && matches } -func (bing *Bing) parseResultElement(el *rod.Element, isAd bool, rank, absoluteRank int) (core.SearchResult, bool) { +func (bing *Bing) parseResultElement(el *rod.Element, isAd bool, rank *core.RankState) (core.SearchResult, bool) { titleSelector := Selectors.Title if isAd { titleSelector = Selectors.AdTitle @@ -101,16 +101,11 @@ func (bing *Bing) parseResultElement(el *rod.Element, isAd bool, rank, absoluteR bing.logger.Debug("Missing title") return core.SearchResult{}, false } - href, err := titleElem.Property("href") if err != nil { bing.logger.Debug("Missing URL") return core.SearchResult{}, false } - url := strings.TrimSpace(href.String()) - if url == "" || url == "#" || strings.HasPrefix(url, "javascript:") { - return core.SearchResult{}, false - } title := core.ElementAttribute(titleElem, "aria-label", "title") if title == "" { @@ -122,10 +117,6 @@ func (bing *Bing) parseResultElement(el *rod.Element, isAd bool, rank, absoluteR if title == "" { title = core.FirstNonEmptyAttribute(el, "aria-label", Selectors.TitleFallbacks...) } - if title == "" { - bing.logger.Debug("Missing title text") - return core.SearchResult{}, false - } desc := core.FirstNonEmptyText(el, Selectors.DescPrimary, Selectors.DescFallback, Selectors.DescAny) if desc == "" { @@ -133,14 +124,7 @@ func (bing *Bing) parseResultElement(el *rod.Element, isAd bool, rank, absoluteR desc = core.NormalizeWhitespace(strings.Replace(fullText, title, "", 1)) } - return core.SearchResult{ - Rank: rank, - AbsoluteRank: absoluteRank, - URL: url, - Title: title, - Description: desc, - Ad: isAd, - }, true + return assembleBingRow(href.String(), title, desc, isAd, rank) } // Search executes a Bing web search and returns normalized search results. @@ -192,9 +176,7 @@ func (bing *Bing) Search(ctx context.Context, query core.Query) (results []core. } bing.logger.Info("Found %d organic result containers", totalResults) - rank := query.Start - adRank := 1 - absoluteRank := query.Start + 1 + rank := core.NewRankStateAt(query.Start, query.Start+1) for _, result := range resultElements { isAd := bingElementMatches(result, Selectors.Ads) isOrganic := bingElementMatches(result, Selectors.Results) @@ -202,21 +184,11 @@ func (bing *Bing) Search(ctx context.Context, query core.Query) (results []core. continue } - resultRank := rank + 1 - if isAd { - resultRank = adRank - } - srchRes, ok := bing.parseResultElement(result, isAd, resultRank, absoluteRank) + srchRes, ok := bing.parseResultElement(result, isAd, rank) if !ok { continue } searchResults = append(searchResults, srchRes) - absoluteRank++ - if isAd { - adRank++ - } else { - rank++ - } } // Deduplicate results diff --git a/core/proxy_test.go b/core/proxy_test.go index a9c6649..d8277fe 100644 --- a/core/proxy_test.go +++ b/core/proxy_test.go @@ -15,7 +15,6 @@ import ( "time" socks5 "github.com/armon/go-socks5" - xcontext "golang.org/x/net/context" ) type timeoutTestError struct{} @@ -438,7 +437,7 @@ type staticResolver struct { ip net.IP } -func (r staticResolver) Resolve(ctx xcontext.Context, name string) (xcontext.Context, net.IP, error) { +func (r staticResolver) Resolve(ctx context.Context, name string) (context.Context, net.IP, error) { if name == r.host { return ctx, r.ip, nil } diff --git a/core/rank_state.go b/core/rank_state.go new file mode 100644 index 0000000..ed4d552 --- /dev/null +++ b/core/rank_state.go @@ -0,0 +1,54 @@ +package core + +// RankState tracks organic, ad, and absolute ranks for mixed SERP rows. +type RankState struct { + organicRank int + adRank int + absoluteRank int +} + +// NewRankState seeds ranks for a 0-based page on a 10-results-per-page engine. +func NewRankState(pageNum int) *RankState { + return NewRankStateAt(pageNum*10, pageNum*10+1) +} + +// NewRankStateAt seeds ranks from explicit organic and absolute bases. +func NewRankStateAt(organicBase, absoluteBase int) *RankState { + return &RankState{ + organicRank: organicBase, + adRank: 1, + absoluteRank: absoluteBase, + } +} + +// Next reserves the rank pair for the next emitted row. +func (r *RankState) Next(isAd bool) (rank, absoluteRank int) { + absoluteRank = r.absoluteRank + r.absoluteRank++ + if isAd { + rank = r.adRank + r.adRank++ + return rank, absoluteRank + } + r.organicRank++ + return r.organicRank, absoluteRank +} + +// SetSeparatedAdAbsoluteRanks gives separated ad/organic passes one mixed order. +func SetSeparatedAdAbsoluteRanks(results []SearchResult, start int) { + adCount := 0 + for i := range results { + if results[i].Ad { + adCount++ + results[i].AbsoluteRank = start + results[i].Rank + } + } + organicAbsoluteRank := start + adCount + 1 + for i := range results { + if results[i].Ad { + continue + } + results[i].AbsoluteRank = organicAbsoluteRank + organicAbsoluteRank++ + } +} diff --git a/core/result_rank_test.go b/core/result_rank_test.go index 47fc03f..da1f535 100644 --- a/core/result_rank_test.go +++ b/core/result_rank_test.go @@ -101,6 +101,54 @@ func TestEnrichResultUsesExplicitResultType(t *testing.T) { } } +func TestRankStateInterleavesAdsAndSeedsPages(t *testing.T) { + t.Parallel() + + // Page 1 (0-based): organic ranks continue at 11, ad ranks always restart at + // 1, and the absolute rank counts every emitted row regardless of kind. + rank := NewRankState(1) + + steps := []struct { + isAd bool + rank, absolute int + }{ + {true, 1, 11}, // ad + {false, 11, 12}, // organic (seeded from page*10) + {false, 12, 13}, // organic + {true, 2, 14}, // ad interleaved after organics + {false, 13, 15}, // organic + } + for i, s := range steps { + gotRank, gotAbs := rank.Next(s.isAd) + if gotRank != s.rank || gotAbs != s.absolute { + t.Fatalf("step %d (ad=%v): got rank=%d absolute=%d, want rank=%d absolute=%d", + i, s.isAd, gotRank, gotAbs, s.rank, s.absolute) + } + } +} + +func TestSetSeparatedAdAbsoluteRanks(t *testing.T) { + t.Parallel() + + // Ecosia collects ads and organics in separate passes (each rank-1-based), + // then this assigns one mixed absolute order: ads first, then organics. + results := []SearchResult{ + {Rank: 1, URL: "https://organic.example.com/one"}, + {Rank: 2, URL: "https://organic.example.com/two"}, + {Rank: 1, Ad: true, URL: "https://ads.example.com/one"}, + {Rank: 2, Ad: true, URL: "https://ads.example.com/two"}, + } + + SetSeparatedAdAbsoluteRanks(results, 0) + + want := []int{3, 4, 1, 2} // ads (passes 1,2) precede organics (3,4) in absolute order + for i, r := range results { + if r.AbsoluteRank != want[i] { + t.Fatalf("%s absolute rank = %d, want %d", r.URL, r.AbsoluteRank, want[i]) + } + } +} + func TestEnvelopePaginationCountsOrganicResults(t *testing.T) { t.Parallel() diff --git a/core/server.go b/core/server.go index 5922acc..3751dba 100644 --- a/core/server.go +++ b/core/server.go @@ -195,10 +195,7 @@ func NewServerWithOptions(host string, port int, opts ServerOptions, searchEngin for _, engine := range searchEngines { locEngine := engine - endpointName := strings.ToLower(locEngine.Name()) - if endpointName == "duckduckgo" { - endpointName = "duck" - } + endpointName := engineEndpointName(locEngine.Name()) serv.app.Get(fmt.Sprintf("/%s/search", endpointName), func(c *fiber.Ctx) error { return serv.handleDedicatedEndpoint(c, locEngine, false) @@ -215,15 +212,15 @@ func NewServerWithOptions(host string, port int, opts ServerOptions, searchEngin continue } locParser := parser - parserEndpointName := strings.ToLower(parser.Name()) - serv.app.Post(fmt.Sprintf("/%s/parse", parserEndpointName), - func(c *fiber.Ctx) error { - return serv.handleParseEndpoint(c, locParser) - }) - if parserEndpointName == "duckduckgo" { - serv.app.Post("/duck/parse", func(c *fiber.Ctx) error { - return serv.handleParseEndpoint(c, locParser) - }) + // Parse registers the canonical-name path, plus the endpoint slug when it + // differs (e.g. /duckduckgo/parse and /duck/parse both resolve). + parserName := strings.ToLower(parser.Name()) + parseHandler := func(c *fiber.Ctx) error { + return serv.handleParseEndpoint(c, locParser) + } + serv.app.Post(fmt.Sprintf("/%s/parse", parserName), parseHandler) + if slug := engineEndpointName(parser.Name()); slug != parserName { + serv.app.Post(fmt.Sprintf("/%s/parse", slug), parseHandler) } } @@ -1216,12 +1213,20 @@ func (s *Server) resolveEngines(ctx context.Context, enginesParam string) []Sear return enginesToUse } +func engineEndpointName(name string) string { + name = strings.ToLower(name) + if name == "duckduckgo" { + return "duck" + } + return name +} + func resolveEngineAlias(name string) string { - switch name { + switch strings.ToLower(name) { case "duck", "ddg": return "duckduckgo" default: - return name + return strings.ToLower(name) } } diff --git a/duckduckgo/parse_html.go b/duckduckgo/parse_html.go index 1c6b806..7daa8fa 100644 --- a/duckduckgo/parse_html.go +++ b/duckduckgo/parse_html.go @@ -20,9 +20,7 @@ func ParseHTML(r io.Reader) ([]core.SearchResult, error) { func parseDDGDocument(doc *goquery.Document) []core.SearchResult { var results []core.SearchResult - rank := 1 - adRank := 1 - absoluteRank := 1 + rank := core.NewRankState(0) resultSel := firstMatchingSelector(doc, Selectors.Results) if resultSel == "" { @@ -30,42 +28,49 @@ func parseDDGDocument(doc *goquery.Document) []core.SearchResult { } doc.Find(resultSel).Each(func(_ int, item *goquery.Selection) { - href := extractFirstAttr(item, Selectors.Link, "href") + href := ddgDocumentHref(item) if href == "" || href == "#" || strings.HasPrefix(href, "javascript:") { return } - title := extractFirstText(item, Selectors.Title) + title := firstText(item, Selectors.Title...) if title == "" { return } - desc := extractFirstText(item, Selectors.Desc) + desc := firstText(item, Selectors.Desc...) + isAd := ddgSelectionHasAdMarker(item) - isAd := duckduckgoSelectionHasAdMarker(item) - - r := core.SearchResult{ - Rank: rank, + resultRank, absoluteRank := rank.Next(isAd) + results = append(results, core.SearchResult{ + Rank: resultRank, AbsoluteRank: absoluteRank, URL: href, Title: title, Description: desc, Ad: isAd, - } - if !isAd { - rank++ - } else { - r.Rank = adRank - adRank++ - } - results = append(results, r) - absoluteRank++ + }) }) return core.AttachFeaturesToFirstResult(core.DeduplicateResults(results), extractDDGFeatures(doc)) } -func duckduckgoSelectionHasAdMarker(item *goquery.Selection) bool { +// firstText returns the normalized text of the first selector that matches a +// non-empty element. +func firstText(item *goquery.Selection, selectors ...string) string { + for _, sel := range selectors { + if tag := item.Find(sel).First(); tag.Length() > 0 { + if text := core.NormalizeWhitespace(tag.Text()); text != "" { + return text + } + } + } + return "" +} + +// ddgSelectionHasAdMarker reports whether the row self-or-descendant matches any +// DuckDuckGo ad-badge selector. +func ddgSelectionHasAdMarker(item *goquery.Selection) bool { for _, sel := range Selectors.AdBadge { if item.Is(sel) || item.Find(sel).Length() > 0 { return true @@ -74,6 +79,22 @@ func duckduckgoSelectionHasAdMarker(item *goquery.Selection) bool { return false } +// ddgDocumentHref returns the first non-empty href among the link selectors, +// trimmed. It skips a selector whose anchor has an absent/empty href and tries +// the next (the pre-refactor extractFirstAttr behavior). +func ddgDocumentHref(item *goquery.Selection) string { + for _, sel := range Selectors.Link { + tag := item.Find(sel).First() + if tag.Length() == 0 { + continue + } + if val, exists := tag.Attr("href"); exists && val != "" { + return strings.TrimSpace(val) + } + } + return "" +} + // firstMatchingSelector returns the first selector from the list that matches // at least one element in the document. func firstMatchingSelector(doc *goquery.Document, selectors []string) string { @@ -84,34 +105,3 @@ func firstMatchingSelector(doc *goquery.Document, selectors []string) string { } return "" } - -// extractFirstAttr tries each selector in order and returns the named attribute -// of the first match, or "". -func extractFirstAttr(item *goquery.Selection, selectors []string, attr string) string { - for _, sel := range selectors { - tag := item.Find(sel).First() - if tag.Length() == 0 { - continue - } - val, exists := tag.Attr(attr) - if exists && val != "" { - return strings.TrimSpace(val) - } - } - return "" -} - -// extractFirstText tries each selector in order and returns the trimmed text of -// the first match, or "". -func extractFirstText(item *goquery.Selection, selectors []string) string { - for _, sel := range selectors { - tag := item.Find(sel).First() - if tag.Length() == 0 { - continue - } - if text := strings.TrimSpace(tag.Text()); text != "" { - return text - } - } - return "" -} diff --git a/duckduckgo/search.go b/duckduckgo/search.go index 43a73c9..bdb3010 100644 --- a/duckduckgo/search.go +++ b/duckduckgo/search.go @@ -69,9 +69,7 @@ func (ddg *DuckDuckGo) isNoResults(page *rod.Page) bool { func (ddg *DuckDuckGo) parseResults(results rod.Elements, pageNum int) []core.SearchResult { searchResults := []core.SearchResult{} - organicRank := pageNum * 10 - adRank := 1 - absoluteRank := pageNum*10 + 1 + rank := core.NewRankState(pageNum) for _, r := range results { // Get URL - try multiple selectors @@ -116,15 +114,8 @@ func (ddg *DuckDuckGo) parseResults(results rod.Elements, pageNum int) []core.Se desc := core.FirstNonEmptyText(r, Selectors.Desc...) // Check if it's an ad - isAd := duckduckgoElementHasAdMarker(r) - resultRank := 0 - if isAd { - resultRank = adRank - adRank++ - } else { - organicRank++ - resultRank = organicRank - } + isAd := ddgElementHasAdMarker(r) + resultRank, absoluteRank := rank.Next(isAd) result := core.SearchResult{ Rank: resultRank, @@ -135,19 +126,17 @@ func (ddg *DuckDuckGo) parseResults(results rod.Elements, pageNum int) []core.Se Ad: isAd, } searchResults = append(searchResults, result) - absoluteRank++ } return searchResults } -func duckduckgoElementHasAdMarker(el *rod.Element) bool { +func ddgElementHasAdMarker(el *rod.Element) bool { if el == nil { return false } for _, selector := range Selectors.AdBadge { - matches, err := el.Matches(selector) - if err == nil && matches { + if matches, err := el.Matches(selector); err == nil && matches { return true } if adIndicator, err := el.Element(selector); err == nil && adIndicator != nil { diff --git a/ecosia/parse_html.go b/ecosia/parse_html.go index 3466be7..d83f9c2 100644 --- a/ecosia/parse_html.go +++ b/ecosia/parse_html.go @@ -1,6 +1,7 @@ package ecosia import ( + "fmt" "io" "strings" @@ -24,83 +25,87 @@ func parseEcosiaDocument(doc *goquery.Document) []core.SearchResult { adRank := 1 doc.Find(Selectors.Result).Each(func(_ int, item *goquery.Selection) { - res, ok := parseEcosiaItem(item, rank, false) - if !ok { - return + if res, ok := parseEcosiaSelectionRow(item, rank, false); ok { + results = append(results, res) + rank++ } - results = append(results, res) - rank++ }) doc.Find(Selectors.Ad).Each(func(_ int, item *goquery.Selection) { - res, ok := parseEcosiaItem(item, adRank, true) - if !ok { - return + if res, ok := parseEcosiaSelectionRow(item, adRank, true); ok { + results = append(results, res) + adRank++ } - results = append(results, res) - adRank++ }) - setSeparatedAdAbsoluteRanks(results, 0) + core.SetSeparatedAdAbsoluteRanks(results, 0) return core.AttachFeaturesToFirstResult(core.DeduplicateResults(results), extractEcosiaFeatures(doc)) } -func parseEcosiaItem(item *goquery.Selection, rank int, ad bool) (core.SearchResult, bool) { - linkTag := item.Find(Selectors.ResultLink).First() - if linkTag.Length() == 0 { - linkTag = item.Find("a[href]").First() - } - if linkTag.Length() == 0 { - return core.SearchResult{}, false +// parseEcosiaSelectionRow extracts a web row from a goquery selection. +func parseEcosiaSelectionRow(item *goquery.Selection, rank int, ad bool) (core.SearchResult, bool) { + link := item.Find(Selectors.ResultLink).First() + if link.Length() == 0 { + // Fall back to the first anchor when the test-id selector is absent. + link = item.Find("a[href]").First() } + href, _ := link.Attr("href") - href, exists := linkTag.Attr("href") - if !exists { - return core.SearchResult{}, false + title := selectionText(item, Selectors.Title) + if title == "" { + title = selectionText(item, "h2, h3") } + desc := selectionText(item, Selectors.Desc) + return assembleEcosiaRow(href, title, desc, rank, ad) +} + +// assembleEcosiaRow validates an already-extracted web row and builds the +// result. Shared by the rod (browser) and goquery (raw / parse) parsers. +func assembleEcosiaRow(href, title, desc string, rank int, ad bool) (core.SearchResult, bool) { href = strings.TrimSpace(href) if href == "" || strings.HasPrefix(href, "javascript:") { return core.SearchResult{}, false } + return core.SearchResult{ + Rank: rank, + URL: href, + Title: strings.TrimSpace(title), + Description: strings.TrimSpace(desc), + Ad: ad, + }, true +} - title := "" - if t := item.Find(Selectors.Title).First(); t.Length() > 0 { - title = strings.TrimSpace(t.Text()) +// assembleEcosiaImageRow validates an already-extracted image card and builds +// the result (formatting the source/dimensions description). Shared by both +// parser backends. +func assembleEcosiaImageRow(href, title, source, dims string, rank int) (core.SearchResult, bool) { + href = strings.TrimSpace(href) + if href == "" { + return core.SearchResult{}, false } - if title == "" { - if t := item.Find("h2, h3").First(); t.Length() > 0 { - title = strings.TrimSpace(t.Text()) + desc := source + if dims != "" { + if source != "" { + desc = fmt.Sprintf("%s (%s)", source, dims) + } else { + desc = dims } } - - desc := "" - if d := item.Find(Selectors.Desc).First(); d.Length() > 0 { - desc = strings.TrimSpace(d.Text()) - } - return core.SearchResult{ Rank: rank, URL: href, Title: title, Description: desc, - Ad: ad, }, true } -func setSeparatedAdAbsoluteRanks(results []core.SearchResult, start int) { - adCount := 0 - for i := range results { - if results[i].Ad { - adCount++ - results[i].AbsoluteRank = start + results[i].Rank +// selectionText returns the trimmed text of the first selector that matches, +// the goquery counterpart of the rod element's selector-fallback text lookup. +func selectionText(item *goquery.Selection, selectors ...string) string { + for _, selector := range selectors { + if tag := item.Find(selector).First(); tag.Length() > 0 { + return strings.TrimSpace(tag.Text()) } } - organicAbsoluteRank := start + adCount + 1 - for i := range results { - if results[i].Ad { - continue - } - results[i].AbsoluteRank = organicAbsoluteRank - organicAbsoluteRank++ - } + return "" } diff --git a/ecosia/search.go b/ecosia/search.go index eaa74cc..351301f 100644 --- a/ecosia/search.go +++ b/ecosia/search.go @@ -9,7 +9,6 @@ package ecosia import ( "context" "errors" - "fmt" "strings" "time" @@ -109,30 +108,10 @@ func (e *Ecosia) parseResult(elem *rod.Element, rank int, ad bool) (core.SearchR if err != nil { return core.SearchResult{}, false } - hrefStr := strings.TrimSpace(href.String()) - if hrefStr == "" || strings.HasPrefix(hrefStr, "javascript:") { - return core.SearchResult{}, false - } - title := "" - if t, err := elem.Element(Selectors.Title); err == nil { - title, _ = t.Text() - } else if t, err := elem.Element("h2, h3"); err == nil { - title, _ = t.Text() - } - - desc := "" - if d, err := elem.Element(Selectors.Desc); err == nil { - desc, _ = d.Text() - } - - return core.SearchResult{ - Rank: rank, - URL: hrefStr, - Title: strings.TrimSpace(title), - Description: strings.TrimSpace(desc), - Ad: ad, - }, true + title := core.FirstNonEmptyText(elem, Selectors.Title, "h2, h3") + desc := core.FirstNonEmptyText(elem, Selectors.Desc) + return assembleEcosiaRow(href.String(), title, desc, rank, ad) } // Search executes an Ecosia web search and returns normalized search results. @@ -223,7 +202,7 @@ func (e *Ecosia) Search(ctx context.Context, query core.Query) (results []core.S } } - setSeparatedAdAbsoluteRanks(all, query.Start) + core.SetSeparatedAdAbsoluteRanks(all, query.Start) deduped := core.DeduplicateResults(all) if query.Limit > 0 { deduped = core.LimitOrganicResults(deduped, query.Limit) @@ -243,10 +222,6 @@ func (e *Ecosia) parseImageResult(el *rod.Element, rank int) (core.SearchResult, if err != nil { return core.SearchResult{}, false } - imgURL := strings.TrimSpace(href.String()) - if imgURL == "" { - return core.SearchResult{}, false - } title := "" if img, err := link.Element("img"); err == nil { @@ -254,33 +229,19 @@ func (e *Ecosia) parseImageResult(el *rod.Element, rank int) (core.SearchResult, title = strings.TrimSpace(*alt) } } + source := elementText(el, Selectors.ImageSource) + dims := elementText(el, Selectors.ImageDims) + return assembleEcosiaImageRow(href.String(), title, source, dims, rank) +} - source := "" - if s, err := el.Element(Selectors.ImageSource); err == nil { - source, _ = s.Text() - source = strings.TrimSpace(source) +// elementText returns the trimmed text of the first descendant matching +// selector, or "" if absent. +func elementText(el *rod.Element, selector string) string { + if e, err := el.Element(selector); err == nil { + text, _ := e.Text() + return strings.TrimSpace(text) } - dims := "" - if d, err := el.Element(Selectors.ImageDims); err == nil { - dims, _ = d.Text() - dims = strings.TrimSpace(dims) - } - - desc := source - if dims != "" { - if source != "" { - desc = fmt.Sprintf("%s (%s)", source, dims) - } else { - desc = dims - } - } - - return core.SearchResult{ - Rank: rank, - URL: imgURL, - Title: title, - Description: desc, - }, true + return "" } // SearchImage executes an Ecosia image search and returns normalized image diff --git a/ecosia/search_raw.go b/ecosia/search_raw.go index 63f44df..a5a8dc2 100644 --- a/ecosia/search_raw.go +++ b/ecosia/search_raw.go @@ -57,31 +57,15 @@ func imageResultParser(response *http.Response) ([]core.SearchResult, error) { rank = 1 ) doc.Find(Selectors.ImageResult).Each(func(_ int, s *goquery.Selection) { - href, ok := s.Find(Selectors.ImageLink).Attr("href") - if !ok || strings.TrimSpace(href) == "" { - return + link := s.Find(Selectors.ImageLink) + href, _ := link.Attr("href") + title, _ := link.Find("img").Attr("alt") + source := strings.TrimSpace(s.Find(Selectors.ImageSource).Text()) + dims := strings.TrimSpace(s.Find(Selectors.ImageDims).Text()) + if res, ok := assembleEcosiaImageRow(href, strings.TrimSpace(title), source, dims, rank); ok { + results = append(results, res) + rank++ } - title, _ := s.Find(Selectors.ImageLink).Find("img").Attr("alt") - title = strings.TrimSpace(title) - var ( - source = strings.TrimSpace(s.Find(Selectors.ImageSource).Text()) - dims = strings.TrimSpace(s.Find(Selectors.ImageDims).Text()) - desc = source - ) - if dims != "" { - if source != "" { - desc = source + " (" + dims + ")" - } else { - desc = dims - } - } - results = append(results, core.SearchResult{ - Rank: rank, - URL: href, - Title: title, - Description: desc, - }) - rank++ }) return results, nil } @@ -140,7 +124,7 @@ func Search(ctx context.Context, query core.Query) (results []core.SearchResult, parsedResults[i].Rank = startRank + organicIdx organicIdx++ } - setSeparatedAdAbsoluteRanks(parsedResults, pageNum*10) + core.SetSeparatedAdAbsoluteRanks(parsedResults, pageNum*10) core.WithRequest(ctx).WithField("results_count", len(parsedResults)).Debug( fmt.Sprintf("Ecosia Raw results : %v", parsedResults), diff --git a/google/search.go b/google/search.go index bfcba29..57f4f09 100644 --- a/google/search.go +++ b/google/search.go @@ -273,9 +273,7 @@ func (gogl *Google) Search(ctx context.Context, query core.Query) (results []cor } gogl.logger.Info("Found %d total results", totalResults) - rank := query.Start - adRank := 1 - absoluteRank := query.Start + 1 + rank := core.NewRankStateAt(query.Start, query.Start+1) // When matched by the canonical organic selector (div.tF2Cxc) every element // is already an organic result, but the wrapper itself often lacks data-ved // (it sits on the outer .g/data-hveid container). Only require data-ved when @@ -331,10 +329,7 @@ func (gogl *Google) Search(ctx context.Context, query core.Query) (results []cor } else { srchRes.Description = strings.TrimSpace(text) } - srchRes.Rank = adRank - srchRes.AbsoluteRank = absoluteRank - adRank++ - absoluteRank++ + srchRes.Rank, srchRes.AbsoluteRank = rank.Next(true) searchResults = append(searchResults, srchRes) } else if isAnswerBox { @@ -454,10 +449,7 @@ func (gogl *Google) Search(ctx context.Context, query core.Query) (results []cor } srchRes.Description = desc - rank += 1 - srchRes.Rank = rank - srchRes.AbsoluteRank = absoluteRank - absoluteRank++ + srchRes.Rank, srchRes.AbsoluteRank = rank.Next(false) searchResults = append(searchResults, srchRes) continue diff --git a/google/search_raw.go b/google/search_raw.go index bd1b717..2c6834d 100644 --- a/google/search_raw.go +++ b/google/search_raw.go @@ -25,9 +25,7 @@ func ParseHTML(r io.Reader) ([]core.SearchResult, error) { func parseGoogleDocument(doc *goquery.Document) []core.SearchResult { results := []core.SearchResult{} - rank := 1 - adRank := 1 - absoluteRank := 1 + rank := core.NewRankState(0) // Prefer the canonical organic result block (div.tF2Cxc, innermost). Fall // back to the broad attribute selector only when no tF2Cxc blocks exist, so @@ -81,14 +79,7 @@ func parseGoogleDocument(doc *goquery.Document) []core.SearchResult { desc := descTag.Text() if link != "" && link != "#" { - resultRank := rank - if isAd { - resultRank = adRank - adRank++ - } else { - rank++ - } - + resultRank, absoluteRank := rank.Next(isAd) result := core.SearchResult{ Rank: resultRank, AbsoluteRank: absoluteRank, @@ -99,7 +90,6 @@ func parseGoogleDocument(doc *goquery.Document) []core.SearchResult { } results = append(results, result) - absoluteRank++ } } diff --git a/yandex/parse_html.go b/yandex/parse_html.go index 7e5189c..921ec08 100644 --- a/yandex/parse_html.go +++ b/yandex/parse_html.go @@ -21,9 +21,7 @@ func ParseHTML(r io.Reader) ([]core.SearchResult, error) { func parseYandexDocument(doc *goquery.Document) []core.SearchResult { var results []core.SearchResult - rank := 1 - adRank := 1 - absoluteRank := 1 + rank := core.NewRankState(0) doc.Find(Selectors.Results).Each(func(_ int, item *goquery.Selection) { // The neuro/AI answer renders as a serp-item li too, so it would be @@ -33,73 +31,79 @@ func parseYandexDocument(doc *goquery.Document) []core.SearchResult { return } - // Skip blocks without a result heading (filters out non-organic blocks - // that share the result-row container). - titleTag := item.Find(Selectors.Title).First() - if titleTag.Length() == 0 { - return - } - - // Prefer the canonical organic-title link, then the closest wrapping - // the title, then any in the block. - linkTag := item.Find(Selectors.LinkPrimary).First() - if linkTag.Length() == 0 { - linkTag = titleTag.Closest("a") - } - if linkTag.Length() == 0 { - linkTag = item.Find(Selectors.Link).First() - } - if linkTag.Length() == 0 { - return - } - - href, exists := linkTag.Attr("href") - if !exists { - return - } - href = strings.TrimSpace(href) - if href == "" || href == "#" || strings.HasPrefix(href, "javascript:") { + href, ok := yandexDocumentHref(item) + if !ok { return } + title := core.NormalizeWhitespace(item.Find(Selectors.Title).First().Text()) + desc := firstNonEmptyText(item, Selectors.Desc, Selectors.DescFallback) isAd := yandexSelectionHasAdMarker(item) || yandexURLLooksAd(href) - title := strings.TrimSpace(titleTag.Text()) - if title == "" { - return + if res, ok := assembleYandexRow(href, title, desc, isAd, rank); ok { + results = append(results, res) } - - desc := "" - if descTag := item.Find(Selectors.Desc).First(); descTag.Length() > 0 { - desc = strings.TrimSpace(descTag.Text()) - } - if desc == "" { - if descTag := item.Find(Selectors.DescFallback).First(); descTag.Length() > 0 { - desc = strings.TrimSpace(descTag.Text()) - } - } - - resultRank := rank - if isAd { - resultRank = adRank - adRank++ - } else { - rank++ - } - - results = append(results, core.SearchResult{ - Rank: resultRank, - AbsoluteRank: absoluteRank, - URL: href, - Title: title, - Description: desc, - Ad: isAd, - }) - absoluteRank++ }) return core.AttachFeaturesToFirstResult(core.DeduplicateResults(results), extractYandexFeatures(doc)) } +// assembleYandexRow validates an already-extracted Yandex row and assigns ranks. +// Shared by the rod (browser) and goquery (raw / parse) parsers, which differ +// only in how they pull title/href/desc out of the DOM. +func assembleYandexRow(href, title, desc string, isAd bool, rank *core.RankState) (core.SearchResult, bool) { + href = strings.TrimSpace(href) + if href == "" || href == "#" || strings.HasPrefix(href, "javascript:") { + return core.SearchResult{}, false + } + if title == "" { + return core.SearchResult{}, false + } + + resultRank, absoluteRank := rank.Next(isAd) + return core.SearchResult{ + Rank: resultRank, + AbsoluteRank: absoluteRank, + URL: href, + Title: title, + Description: desc, + Ad: isAd, + }, true +} + +// yandexDocumentHref resolves a result row's link href in the goquery path: +// the canonical organic-title link, then the wrapping the title, then any +// in the block. ok=false when no anchor with an href attribute is found. +func yandexDocumentHref(item *goquery.Selection) (string, bool) { + linkTag := item.Find(Selectors.LinkPrimary).First() + if linkTag.Length() == 0 { + linkTag = item.Find(Selectors.Title).First().Closest("a") + } + if linkTag.Length() == 0 { + linkTag = item.Find(Selectors.Link).First() + } + if linkTag.Length() == 0 { + return "", false + } + href, exists := linkTag.Attr("href") + if !exists { + return "", false + } + return href, true +} + +// firstNonEmptyText returns the normalized text of the first selector that +// matches a non-empty element, the goquery counterpart of core.FirstNonEmptyText. +func firstNonEmptyText(item *goquery.Selection, selectors ...string) string { + for _, selector := range selectors { + if tag := item.Find(selector).First(); tag.Length() > 0 { + if text := core.NormalizeWhitespace(tag.Text()); text != "" { + return text + } + } + } + return "" +} + // isYandexNeuroAnswer reports whether a serp-item is the AI/neuro answer card. func isYandexNeuroAnswer(item *goquery.Selection) bool { if name, ok := item.Attr("data-fast-name"); ok && name == "neuro_answer" { diff --git a/yandex/search.go b/yandex/search.go index 60e712d..7d18cbe 100644 --- a/yandex/search.go +++ b/yandex/search.go @@ -5,7 +5,6 @@ import ( "encoding/json" "fmt" "sort" - "strings" "time" "github.com/go-rod/rod" @@ -72,9 +71,7 @@ func (yand *Yandex) isNoResults(page *rod.Page) bool { func (yand *Yandex) parseResults(results rod.Elements, pageNum int) []core.SearchResult { searchResults := []core.SearchResult{} - organicRank := pageNum * 10 - adRank := 1 - absoluteRank := pageNum*10 + 1 + rank := core.NewRankState(pageNum) for _, r := range results { titleTag, err := r.Element(Selectors.Title) @@ -82,64 +79,62 @@ func (yand *Yandex) parseResults(results rod.Elements, pageNum int) []core.Searc yand.logger.Debug("Missing h2 title") continue } - title, err := titleTag.Text() - if err != nil || strings.TrimSpace(title) == "" { - yand.logger.Debug("Failed to extract title") + href, ok := yand.elementHref(r, titleTag) + if !ok { continue } - title = strings.TrimSpace(title) - - link, err := r.Element(Selectors.LinkPrimary) - if err != nil { - if closest := core.ClosestMatching(titleTag, Selectors.Link, 4); closest != nil { - link = closest - err = nil - } - } - if err != nil { - link, err = r.Element(Selectors.Link) - if err != nil { - yand.logger.Debug("Missing link") - continue - } - } - linkText, err := link.Property("href") - if err != nil { - yand.logger.Debug("Missing href") - continue - } - hrefStr := strings.TrimSpace(linkText.String()) - if hrefStr == "" || hrefStr == "#" || strings.HasPrefix(hrefStr, "javascript:") { - continue - } - + title := core.ElementText(titleTag) desc := core.FirstNonEmptyText(r, Selectors.Desc, Selectors.DescFallback) + isAd := yandexElementHasAdMarker(r) || yandexURLLooksAd(href) - isAd := yandexElementHasAdMarker(r) || yandexURLLooksAd(hrefStr) - resultRank := 0 - if isAd { - resultRank = adRank - adRank++ - } else { - organicRank++ - resultRank = organicRank + if res, ok := assembleYandexRow(href, title, desc, isAd, rank); ok { + searchResults = append(searchResults, res) } - - res := core.SearchResult{ - Rank: resultRank, - AbsoluteRank: absoluteRank, - URL: hrefStr, - Title: title, - Description: desc, - Ad: isAd, - } - searchResults = append(searchResults, res) - absoluteRank++ } return searchResults } +// elementHref resolves a result row's link href in the rod path: the canonical +// organic-title link, then the wrapping the title, then any in the +// block. ok=false (with a debug log) when no usable anchor is found. +func (yand *Yandex) elementHref(r, titleTag *rod.Element) (string, bool) { + link, err := r.Element(Selectors.LinkPrimary) + if err != nil { + if closest := core.ClosestMatching(titleTag, Selectors.Link, 4); closest != nil { + link, err = closest, nil + } + } + if err != nil { + link, err = r.Element(Selectors.Link) + if err != nil { + yand.logger.Debug("Missing link") + return "", false + } + } + linkText, err := link.Property("href") + if err != nil { + yand.logger.Debug("Missing href") + return "", false + } + return linkText.String(), true +} + +func yandexElementHasAdMarker(el *rod.Element) bool { + if el == nil { + return false + } + for _, selector := range Selectors.AdMarkers { + if matches, err := el.Matches(selector); err == nil && matches { + return true + } + if child, err := el.Element(selector); err == nil && child != nil { + return true + } + } + return false +} + // Yandex hydrates the results list progressively, so the first parse can be // short. After the list selector appears, re-poll briefly until we have the // requested number of organic results or the grace period elapses. @@ -175,22 +170,6 @@ func (yand *Yandex) waitForParsedResults(ctx context.Context, page *rod.Page, pa return results, nil } -func yandexElementHasAdMarker(el *rod.Element) bool { - if el == nil { - return false - } - for _, selector := range Selectors.AdMarkers { - matches, err := el.Matches(selector) - if err == nil && matches { - return true - } - if child, err := el.Element(selector); err == nil && child != nil { - return true - } - } - return false -} - func (yand *Yandex) parseImageEntities(items rod.Elements) map[string]ImageEntity { entities := make(map[string]ImageEntity) for _, item := range items {