This commit is contained in:
AIWintermuteAI
2026-07-18 09:49:38 +02:00
parent 58fdf2b7a6
commit aec0d9c42a
3 changed files with 8 additions and 12 deletions

View File

@@ -426,14 +426,14 @@ type batchExtractPayload struct {
Mode string `json:"mode"`
}
// batchExtractItem is the WebUI-compatible response item per URL.
// batchExtractItem is a response item for a single extracted URL, using
// page_content/metadata keys
type batchExtractItem struct {
PageContent string `json:"page_content"`
Metadata map[string]string `json:"metadata"`
}
func (s *Server) handleBatchExtract(c *fiber.Ctx) error {
startedAt := time.Now()
requestCtx := withRequestUsage(c.UserContext(), "extract-batch")
c.SetUserContext(requestCtx)
defer setNetworkBytesHeader(c, requestCtx)
@@ -545,6 +545,5 @@ func (s *Server) handleBatchExtract(c *fiber.Ctx) error {
}
wg.Wait()
_ = startedAt
return c.JSON(results)
}

View File

@@ -113,7 +113,7 @@ func TestValidateExtractTargetURLNormalizesBarePublicIP(t *testing.T) {
}
}
func TestBatchExtractReturnsWebUIFormat(t *testing.T) {
func TestBatchExtractSingleURL(t *testing.T) {
target := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
w.WriteHeader(http.StatusOK)
_, _ = w.Write([]byte(`<html><body><article><h1>Test Page</h1><p>This is a test page with enough content to pass the minimum runes threshold for extraction in batch mode.</p></article></body></html>`))

View File

@@ -420,8 +420,6 @@ paths:
$ref: "#/components/headers/XRequestID"
X-Cache:
$ref: "#/components/headers/XCache"
X-Fallback-Engine:
$ref: "#/components/headers/XFallbackEngine"
X-Proxy-Mode:
$ref: "#/components/headers/XProxyMode"
X-Proxy-Tag:
@@ -554,12 +552,11 @@ paths:
post:
tags: [Extract]
operationId: extractBatch
summary: Extract content from multiple URLs (WebUI-compatible)
summary: Extract content from multiple URLs
description: >
Accepts an array of URLs and returns extracted page content in the
format expected by Open WebUI's ExternalWebLoader. Each result
contains `page_content` (markdown) and `metadata` (title, source,
lang, etc.).
Accepts an array of URLs and returns extracted page content for each
one. Each result contains `page_content` (markdown) and `metadata`
(title, source, lang, etc.).
requestBody:
required: true
content:
@@ -2069,7 +2066,7 @@ components:
total:
type: integer
# ── Batch extract (WebUI) ────────────────────────────────────────
# ── Batch extract ─────────────────────────────────────────────────
BatchExtractRequest:
type: object
required: [urls]