mirror of
https://github.com/vectorize-io/hindsight.git
synced 2026-09-14 19:31:49 +08:00
ba23c6526f
One branch for the memories-store seam, so the engine's store interface and the stores that implement it move together rather than drifting apart. A store-backed extension imports the interface at module load; when those symbols live on a different branch than the deployed engine, the extension fails to import and every bank routed to it is degraded. **The seam is raised above persistence.** A store that owns its rows no longer has its writes driven by the orchestrator. `begin_retain` opens a session, the engine streams parts in, and the store decides when to commit. That choice is the point: a bulk ingest with no LLM in the loop should commit once, a long extraction should not, and neither is right in general, so the engine must not decide it. Two rules any flush policy has to honour — never memories without their bodies, and bound what an interruption loses. **One ownership flag, not several.** `store_owned` replaces the separate questions about who holds memory rows, document bodies, the retain write, the knowledge-page index, and the persistence half of a retain. No store ever answered them in a mixed combination, so each extra flag was a branch every call site kept handling for a state that does not occur. The per-bank probes (`store_owned_for`, `derives_semantic_links_internally_for`) stay, because a router genuinely does hold banks in different backends — and reading the class attribute instead of the probe is the bug they exist to prevent. **Cross-store write-group transactions are removed.** They made a store write atomic with a Postgres witness row, but mint-early / witness-late meant a crash or a sibling cancel between the two left a transaction pending with no witness. Consolidation is idempotent on retry, so the recovery this bought never paid for the state it stranded. **Delta re-ingest is fenced by the store's own compare-and-set** (`StoreWriteConflict` → `ConcurrentAppendConflict`), scoped to the chunks that actually moved, and holds no pooled connection while it runs. Alongside: retain phase timing that records seconds *and* round trips, because a phase slow per-call and one slow per-count need opposite fixes; a store's backpressure treated as backpressure rather than as a failed task; and a per-process cache for the two bank rows a retain reads on every call, invalidated by every path that writes the bank row so a caller that writes and reads back sees what it wrote.
221 lines
8.2 KiB
Python
221 lines
8.2 KiB
Python
"""
|
|
Tests for LATERAL entity fanout cap in graph expansion.
|
|
|
|
Verifies that the per-entity LIMIT in _expand_combined prevents high-fanout
|
|
entities from exploding the self-join, while still returning entity-based
|
|
graph results.
|
|
"""
|
|
|
|
import asyncio
|
|
from datetime import datetime, timezone
|
|
|
|
import pytest
|
|
|
|
# These read `result.trace["retrieval_results"]` and assert an entry per ARM (method_name ==
|
|
# "graph"). That structure is produced by the engine running the arms itself; a store that answers
|
|
# a whole recall in one hop runs them internally and reports phases ("arms", "fuse", "trim"), not
|
|
# one entry per arm -- so there is nothing here to assert against for such a store, and the arm
|
|
# behaviour is covered by the extension's own recall suite instead.
|
|
pytestmark = pytest.mark.memory_backend_incompatible
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
@pytest.mark.timeout(1200)
|
|
async def test_high_fanout_entity_returns_results(memory, request_context):
|
|
"""
|
|
A high-fanout entity (appearing in many facts) should still produce
|
|
graph retrieval results — the LATERAL cap limits rows per entity but
|
|
does not drop the entity entirely.
|
|
"""
|
|
bank_id = f"test_fanout_cap_{datetime.now(timezone.utc).timestamp()}"
|
|
|
|
try:
|
|
# Create many facts sharing one common entity ("Acme Corp") plus
|
|
# a few with a unique entity so we can query for the unique one
|
|
# and verify graph expansion finds siblings via "Acme Corp".
|
|
contents = [
|
|
# Target: unique entity "Zara" shares "Acme Corp" with the rest
|
|
{
|
|
"content": "Zara joined Acme Corp as a senior engineer last month",
|
|
"context": "hr update",
|
|
"entities": [{"text": "Zara"}, {"text": "Acme Corp"}],
|
|
},
|
|
]
|
|
# Add many facts that all share "Acme Corp" — creates a high-fanout entity
|
|
for i in range(60):
|
|
contents.append(
|
|
{
|
|
"content": f"Employee {i} completed onboarding at Acme Corp in department {i % 5}",
|
|
"context": "hr update",
|
|
"entities": [{"text": f"Employee {i}"}, {"text": "Acme Corp"}],
|
|
}
|
|
)
|
|
|
|
await memory.retain_batch_async(
|
|
bank_id=bank_id,
|
|
contents=contents,
|
|
request_context=request_context,
|
|
)
|
|
|
|
from hindsight_api.engine.memory_engine import Budget
|
|
|
|
# Query for "Zara" — semantic search finds Zara's fact as a seed,
|
|
# then graph expansion should find other Acme Corp facts via the
|
|
# shared entity, even though "Acme Corp" has 60+ mentions.
|
|
result = await memory.recall_async(
|
|
bank_id=bank_id,
|
|
query="Zara",
|
|
budget=Budget.HIGH,
|
|
max_tokens=4096,
|
|
enable_trace=True,
|
|
request_context=request_context,
|
|
_quiet=True,
|
|
)
|
|
|
|
assert result.results is not None
|
|
assert len(result.results) > 0
|
|
|
|
# Verify graph retrieval ran and found results
|
|
retrieval_results = result.trace.get("retrieval_results", [])
|
|
graph_results = [r for r in retrieval_results if r.get("method_name") == "graph"]
|
|
assert len(graph_results) > 0, "Graph retrieval should have run"
|
|
|
|
# At least one graph result should contain Acme Corp content
|
|
# (found via shared entity, not just semantic similarity)
|
|
all_texts = [r.text for r in result.results]
|
|
acme_found = any("Acme Corp" in t for t in all_texts)
|
|
assert acme_found, "Should find Acme Corp facts via entity graph expansion"
|
|
|
|
finally:
|
|
await memory.delete_bank(bank_id, request_context=request_context)
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_entity_expansion_timeout_fallback(memory, request_context):
|
|
"""
|
|
When graph_expansion_timeout is set very low, entity expansion should
|
|
time out gracefully and fall back to semantic+causal links only,
|
|
rather than failing the entire recall.
|
|
"""
|
|
bank_id = f"test_timeout_fallback_{datetime.now(timezone.utc).timestamp()}"
|
|
|
|
try:
|
|
await memory.retain_batch_async(
|
|
bank_id=bank_id,
|
|
contents=[
|
|
{
|
|
"content": "Alice works on the backend API at TechCorp",
|
|
"context": "team info",
|
|
"entities": [{"text": "Alice"}, {"text": "TechCorp"}],
|
|
},
|
|
{
|
|
"content": "Bob maintains the frontend at TechCorp",
|
|
"context": "team info",
|
|
"entities": [{"text": "Bob"}, {"text": "TechCorp"}],
|
|
},
|
|
],
|
|
request_context=request_context,
|
|
)
|
|
|
|
from hindsight_api.config import _get_raw_config
|
|
from hindsight_api.engine.memory_engine import Budget
|
|
|
|
config = _get_raw_config()
|
|
original_timeout = config.link_expansion_timeout
|
|
|
|
try:
|
|
# Set an impossibly low timeout to force the fallback path
|
|
config.link_expansion_timeout = 0.0001
|
|
|
|
result = await memory.recall_async(
|
|
bank_id=bank_id,
|
|
query="Alice",
|
|
budget=Budget.MID,
|
|
max_tokens=2048,
|
|
enable_trace=True,
|
|
request_context=request_context,
|
|
_quiet=True,
|
|
)
|
|
|
|
# Recall should succeed even when entity expansion times out
|
|
assert result.results is not None
|
|
assert len(result.results) > 0
|
|
|
|
# Alice should still be found via semantic search
|
|
result_texts = [r.text for r in result.results]
|
|
alice_found = any("Alice" in t for t in result_texts)
|
|
assert alice_found, "Should find Alice via semantic search despite graph timeout"
|
|
finally:
|
|
config.link_expansion_timeout = original_timeout
|
|
|
|
finally:
|
|
await memory.delete_bank(bank_id, request_context=request_context)
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
@pytest.mark.timeout(1200)
|
|
async def test_per_entity_limit_caps_expansion(memory, request_context):
|
|
"""
|
|
With graph_per_entity_limit set to a small value, entity expansion should
|
|
still work but return fewer results from high-fanout entities.
|
|
"""
|
|
bank_id = f"test_per_entity_limit_{datetime.now(timezone.utc).timestamp()}"
|
|
|
|
try:
|
|
# Create facts with a shared entity
|
|
contents = [
|
|
{
|
|
"content": "Lead engineer Dana oversees the Widgets project at MegaCorp",
|
|
"context": "project info",
|
|
"entities": [{"text": "Dana"}, {"text": "MegaCorp"}],
|
|
},
|
|
]
|
|
for i in range(30):
|
|
contents.append(
|
|
{
|
|
"content": f"MegaCorp hired contractor {i} for the Q4 push",
|
|
"context": "hiring info",
|
|
"entities": [{"text": f"Contractor {i}"}, {"text": "MegaCorp"}],
|
|
}
|
|
)
|
|
|
|
await memory.retain_batch_async(
|
|
bank_id=bank_id,
|
|
contents=contents,
|
|
request_context=request_context,
|
|
)
|
|
|
|
from hindsight_api.config import _get_raw_config
|
|
from hindsight_api.engine.memory_engine import Budget
|
|
|
|
config = _get_raw_config()
|
|
original_limit = config.link_expansion_per_entity_limit
|
|
|
|
try:
|
|
# Set a very small per-entity limit
|
|
config.link_expansion_per_entity_limit = 5
|
|
|
|
result = await memory.recall_async(
|
|
bank_id=bank_id,
|
|
query="Dana",
|
|
budget=Budget.HIGH,
|
|
max_tokens=4096,
|
|
enable_trace=True,
|
|
request_context=request_context,
|
|
_quiet=True,
|
|
)
|
|
|
|
# Recall should succeed with the cap
|
|
assert result.results is not None
|
|
assert len(result.results) > 0
|
|
|
|
# Graph retrieval should have run
|
|
retrieval_results = result.trace.get("retrieval_results", [])
|
|
graph_results = [r for r in retrieval_results if r.get("method_name") == "graph"]
|
|
assert len(graph_results) > 0, "Graph retrieval should have run"
|
|
finally:
|
|
config.link_expansion_per_entity_limit = original_limit
|
|
|
|
finally:
|
|
await memory.delete_bank(bank_id, request_context=request_context)
|