mirror of
https://github.com/vectorize-io/hindsight.git
synced 2026-09-14 19:31:49 +08:00
5d46f9c8c8
- Update version to 0.10.0 in all components - Regenerate OpenAPI spec and client SDKs - Python packages: hindsight-api, hindsight-dev, hindsight-all, hindsight-embed - Python client: hindsight-clients/python - TypeScript client: hindsight-clients/typescript - hindsight-all npm wrapper: hindsight-all-npm - Rust CLI: hindsight-cli - Control Plane: hindsight-control-plane - Helm chart - Create documentation version-0.10
318 lines
15 KiB
TOML
318 lines
15 KiB
TOML
[build-system]
|
|
requires = ["hatchling>=1.27"]
|
|
build-backend = "hatchling.build"
|
|
|
|
[project]
|
|
name = "hindsight-api-slim"
|
|
version = "0.10.0"
|
|
description = "Hindsight: Agent Memory That Learns"
|
|
license = "MIT"
|
|
readme = "README.md"
|
|
requires-python = ">=3.11"
|
|
dependencies = [
|
|
# 0.30.0 added Connection.get_reset_query(), the only supported way to skip
|
|
# asyncpg's release-time reset (see engine/db/postgresql.py). On 0.29 the
|
|
# override would be silently ignored, making the flag a no-op.
|
|
"asyncpg>=0.30.0",
|
|
"python-dotenv>=1.0.0",
|
|
"openai>=1.66.0", # Responses API (client.responses.create, reasoning=, text.format) for the openai-responses provider
|
|
"pydantic>=2.0.0",
|
|
"rich>=13.0.0",
|
|
"fastapi[standard]>=0.120.3",
|
|
"uvicorn>=0.38.0",
|
|
"wsproto>=1.0.0",
|
|
# Cap below 2.1: SQLAlchemy 2.1 switches the default `postgresql://` DBAPI
|
|
# from psycopg2 to psycopg (v3), which we don't ship — a bare install would
|
|
# fail migrations with "No module named 'psycopg'". Pin to the tested 2.0
|
|
# line (which keeps psycopg2 the default driver) until psycopg3 is adopted.
|
|
"sqlalchemy>=2.0.44,<2.1",
|
|
"alembic>=1.17.1",
|
|
"pgvector>=0.4.1",
|
|
"greenlet>=3.2.4,<3.4.0", # 3.4.0 lacks arm64 wheels for manylinux_2_41
|
|
"psycopg2-binary>=2.9.11",
|
|
# Tokenizer for all token counting and chunking. Replaces tiktoken: same ids,
|
|
# several times faster, a count-only API that allocates no id list, and the
|
|
# vocabularies ship in the wheel so nothing is downloaded at runtime.
|
|
# Contained to engine/token_encoding.py — see that module for the rationale.
|
|
"toktok-rs>=0.1.3", # 0.1.3 adds truncate()/batch_truncate()
|
|
"httpx>=0.27.0",
|
|
"fastmcp>=3.2.0", # SSRF/path traversal, OAuth confused deputy, command injection fixes
|
|
"python-dateutil>=2.8.0",
|
|
# Two coupled reasons for these floors — keep all six pins moving together:
|
|
# (1) opentelemetry-exporter-prometheus calls
|
|
# MetricReader.__init__(otel_component_type=…), a kwarg added in
|
|
# opentelemetry-sdk 1.41.0 (open-telemetry/opentelemetry-python#4970).
|
|
# Without matched floors, pip resolves a new exporter against an older
|
|
# sdk and metric initialisation crashes at startup with
|
|
# "MetricReader.__init__() got an unexpected keyword argument
|
|
# 'otel_component_type'".
|
|
# (2) opentelemetry-proto <1.44 caps protobuf<7.0; 1.44.0 raised it to
|
|
# protobuf<8.0, which is what lets the protobuf pin below reach 7.x.
|
|
"opentelemetry-api>=1.44.0",
|
|
"opentelemetry-sdk>=1.44.0",
|
|
"opentelemetry-instrumentation-fastapi>=0.65b0",
|
|
"opentelemetry-exporter-prometheus>=0.65b0",
|
|
"opentelemetry-exporter-otlp-proto-http>=1.44.0",
|
|
"opentelemetry-semantic-conventions>=0.65b0",
|
|
"dateparser>=1.4.2",
|
|
# Transitive via dateparser, pinned here because the floor is load-bearing:
|
|
# 2025.11.3 turns concurrent entry into dateparser's locale caches into a SIGSEGV
|
|
# inside _regex (verified -- exit 139 in seconds, 3/3; the same probe on 2026.9.3
|
|
# survives). The engine also serialises those entries now (see
|
|
# engine/query_analyzer.py _DATEPARSER_LOCK), so this is the second of two
|
|
# independent guards, not the only one.
|
|
"regex>=2026.9.3",
|
|
"google-genai>=1.72.0",
|
|
"google-auth>=2.0.0",
|
|
"anthropic>=0.40.0",
|
|
"typer>=0.9.0",
|
|
"cohere>=5.0.0",
|
|
# 1.82.7/1.82.8 had a supply chain compromise (yanked); 1.83.0+ also fixes
|
|
# GHSA-jjhc-v7c2-5hh6 / GHSA-53mr-6c8q-9789 / GHSA-pq44-5pcq-4r5g /
|
|
# GHSA-8cjq-wjmh-q42r; 1.84.0 fixes GHSA-4xpc-pv4p-pm3w.
|
|
# Floor raised to 1.93.0 for Python 3.14: litellm ships its own Rust
|
|
# extension (litellm-rust python-bridge). Releases before 1.93.0 publish no
|
|
# cp314 wheel and their sdist fails to build because PyO3 0.23.5 rejects
|
|
# any interpreter newer than 3.13. 1.93.0 adds cp314 wheels and a PyO3 that
|
|
# builds on 3.14.
|
|
"litellm>=1.93.0; sys_platform != 'darwin'",
|
|
# macOS: litellm publishes NO macOS wheels for any release >= 1.92.0 (only
|
|
# manylinux + win_amd64), so every install compiles the sdist's Rust/PyO3
|
|
# bridge — which requires a Rust toolchain most users don't have, and as of
|
|
# 1.95.0 (vendored aws-smithy crates) rustc >= 1.94.1 on top. Pin to the
|
|
# 1.91.x line, the last releases shipping pure-python py3-none-any wheels,
|
|
# so a plain `uvx hindsight-api` works on a stock Mac. The 1.93.0 floor's
|
|
# Python 3.14 rationale doesn't apply here: pure wheels install on any
|
|
# interpreter without building anything. Revisit when litellm ships macOS
|
|
# wheels (tracked upstream in BerriAI/litellm#31261).
|
|
"litellm>=1.91.3,<1.92; sys_platform == 'darwin'",
|
|
"markitdown[pdf,docx,pptx,xlsx,xls]>=0.1.4", # File to markdown conversion
|
|
"obstore>=0.4.0", # S3/GCS/Azure object storage client (Rust-backed)
|
|
"winloop>=0.1.0; sys_platform == 'win32'",
|
|
"uvloop>=0.22.1; sys_platform != 'win32'",
|
|
# Transitive dependency security fixes
|
|
"pyasn1>=0.6.3", # DoS vulnerability fix
|
|
"urllib3>=2.7.0", # Decompression-bomb safeguards bypass + sensitive header forwarding fixes
|
|
"protobuf>=7.35.1", # JSON recursion depth bypass fix (>=6.33.5); requires otel>=1.44 (proto <1.44 caps protobuf<7.0)
|
|
"pillow>=12.3.0", # Multiple HIGH image parsing vulnerabilities fixed in 12.3.0
|
|
"cryptography>=50.0.0", # GHSA-g6cj-pr64-35w5: Bleichenbacher oracle in PKCS#7 EnvelopedData decryption (supersedes the >=48.0.1 GHSA-537c-gmf6-5ccf floor). GHSA-537c-gmf6-5ccf: bundled-OpenSSL OOB read fix needs >=48.0.1. Prior <47 cap (47.0.0 SIGILL on ARM64 Docker/Podman, pyca/cryptography#14733) lifted — 47/48/49 verified importing + RSA sign/verify cleanly on linux/arm64 (Docker on Apple Silicon) and native arm64 macOS; upstream issue closed unconfirmed.
|
|
"filelock>=3.20.1", # TOCTOU race condition fix
|
|
"authlib>=1.6.9", # Account takeover/JWS header injection vulnerability fix
|
|
"pyjwt>=2.12.0", # Accepts unknown crit header extensions fix. Transitive only (via mcp) since the Supabase tenant extension moved to hindsight-extensions/supabase-tenant — no core module imports jwt.
|
|
"python-multipart>=0.0.22", # Arbitrary file write via non-default configuration fix
|
|
"aiohttp>=3.14.3", # GHSA-cq5v-8q36-5273: OOB heap read in the C response parser, plus earlier DoS fixes
|
|
"pygments>=2.20.0", # ReDoS via inefficient GUID regex fix
|
|
"claude-agent-sdk>=0.2.82",
|
|
"github-copilot-sdk>=1.0.11",
|
|
"boto3>=1.42.74",
|
|
"croniter>=2.0.0", # Cron parsing for scheduled mental model refresh
|
|
"json-repair>=0.63.2", # Structural repair of malformed LLM JSON (last-resort parse fallback); >=0.60.1 also fixes the circular-$ref unbounded-CPU DoS
|
|
"orjson>=3.11.0", # Fast JSON encode/decode on the recall path (audit rows, TEI embedding responses); 3.11 ships 3.14 wheels
|
|
"numpy>=1.26.0", # Core vector/array math for pgvector, embeddings, and link graph operations
|
|
]
|
|
|
|
[project.optional-dependencies]
|
|
local-ml = [
|
|
# Local ML models for embeddings/reranking
|
|
# 5.0 is the floor: LocalSTEmbeddings calls SentenceTransformer.encode_query()
|
|
# and .encode_document(), which only exist from 5.0 onwards. On 4.x those are
|
|
# an AttributeError at first encode (recall/retain), not at startup.
|
|
"sentence-transformers>=5.0.0",
|
|
"transformers>=5.5.0", # ReDoS fixes; 5.5.0 clears GHSA-fgcw-684q-jj6r (LightGlue RCE)
|
|
# transformers enforces tokenizers<=0.23.0 with a runtime check, but has
|
|
# shipped metadata declaring a wider range than it actually enforces. Keep
|
|
# this cap: without it an in-place upgrade can pull tokenizers 0.23.1 and
|
|
# break local embeddings/reranker startup. See issue #2055.
|
|
"tokenizers>=0.22.0,<=0.23.0",
|
|
# >=2.6.0 for the remote-code-execution CVE fix. The 2.11.0 floor is for
|
|
# ARMv8.0 CPUs (e.g. Cortex-A57): torch 2.10.0's aarch64 wheel bundles a
|
|
# mimalloc whose ELF constructor runs an inline ARMv8.1 LSE atomic
|
|
# (`ldaddal` in `_mi_options_init`), so `import torch` dies with SIGILL at
|
|
# dlopen time on any CPU without LSE. 2.11.0+ keeps LSE behind the
|
|
# runtime-dispatched outline-atomics helpers again. See issue #4142.
|
|
"torch>=2.11.0",
|
|
"einops>=0.8.2",
|
|
"flashrank>=0.2.0",
|
|
# Apple Silicon local inference — mlx publishes wheels only for
|
|
# macOS/Linux, not Windows, so gate on platform to let `uv sync
|
|
# --all-extras` resolve on win_amd64 runners.
|
|
"mlx>=0.31.0; sys_platform != 'win32'",
|
|
"mlx-lm>=0.31.1; sys_platform != 'win32'",
|
|
"safetensors>=0.6.2",
|
|
]
|
|
local-llm = [
|
|
# Built-in llama.cpp inference for fully offline operation
|
|
"llama-cpp-python[server]>=0.3.0",
|
|
"huggingface-hub>=0.20.0",
|
|
]
|
|
local-onnx = [
|
|
# In-process ONNX Runtime embeddings without an Ollama/TEI sidecar
|
|
"onnxruntime>=1.17.0",
|
|
"transformers>=5.5.0", # 5.5.0 clears GHSA-fgcw-684q-jj6r (LightGlue RCE)
|
|
"tokenizers>=0.22.0,<=0.23.0", # See issue #2055 (transformers caps tokenizers<=0.23.0)
|
|
"huggingface-hub>=0.20.0",
|
|
"numpy>=1.26.0",
|
|
]
|
|
embedded-db = [
|
|
"pg0-embedded>=0.15.0",
|
|
]
|
|
oracle = [
|
|
"oracledb>=2.5.0",
|
|
]
|
|
all = [
|
|
"hindsight-api-slim[local-ml,local-onnx,embedded-db]",
|
|
]
|
|
test = [
|
|
"pytest>=7.0.0",
|
|
"pytest-asyncio>=0.21.0",
|
|
"pytest-timeout>=2.4.0",
|
|
"pytest-xdist>=3.0.0",
|
|
"filelock>=3.20.1", # TOCTOU race condition fix
|
|
"testcontainers>=4.0.0",
|
|
]
|
|
|
|
[project.scripts]
|
|
hindsight-api = "hindsight_api.main:main"
|
|
hindsight-worker = "hindsight_api.worker.main:main"
|
|
hindsight-local-mcp = "hindsight_api.mcp_local:main"
|
|
hindsight-admin = "hindsight_api.admin.cli:main"
|
|
|
|
[tool.hatch.build.targets.wheel]
|
|
packages = ["hindsight_api"]
|
|
|
|
[tool.hatch.build.targets.wheel.sources]
|
|
"hindsight_api" = "hindsight_api"
|
|
|
|
[tool.hatch.build.targets.sdist]
|
|
include = [
|
|
"hindsight_api/**/*",
|
|
]
|
|
|
|
[tool.hatch.build]
|
|
include = [
|
|
"hindsight_api/**/*.py",
|
|
"hindsight_api/alembic/**/*",
|
|
]
|
|
|
|
[tool.pytest.ini_options]
|
|
log_cli = true
|
|
log_cli_level = "INFO"
|
|
log_cli_format = "%(asctime)s - %(levelname)s - %(name)s - %(message)s"
|
|
log_cli_date_format = "%Y-%m-%d %H:%M:%S"
|
|
addopts = "--timeout 300 -n 8 --dist loadgroup --durations=10 -v"
|
|
markers = [
|
|
"oracle: Oracle 23ai integration tests (require ORACLE_TEST_DSN env var)",
|
|
"hs_llm_mat: LLM minimum acceptance tests — run in CI matrix across multiple providers",
|
|
"hs_llm_core: Core pipeline tests that need a real LLM but only one provider",
|
|
"integration: Live external-API integration tests (require provider credentials; skipped without)",
|
|
"slow: Slow tests (minutes); not run in fast CI",
|
|
"memory_backend_incompatible: Asserts or SEEDS Postgres-internal state for memories — raw memory_units / memory_links / unit_entities / documents / chunks rows, raw memory_links counts (the graph read path dedupes bidirectional edges), or internal columns like embedding / search_vector that are not part of the public read model. A MEMORIES extension owns those rows itself and leaves the tables empty, so such a test measures the storage layout rather than the behaviour and cannot pass against one. Deselect when running against an alternative store with -m 'not memory_backend_incompatible'; it still runs, and must, on Postgres.",
|
|
]
|
|
asyncio_mode = "auto"
|
|
asyncio_default_fixture_loop_scope = "function"
|
|
log_auto_indent = true
|
|
filterwarnings = [
|
|
"ignore:The @wait_container_is_ready decorator is deprecated:DeprecationWarning",
|
|
"ignore::RuntimeWarning:asyncio",
|
|
]
|
|
|
|
[dependency-groups]
|
|
dev = [
|
|
"pytest>=9.0.0",
|
|
"pytest-asyncio>=1.3.0",
|
|
"pytest-timeout>=2.4.0",
|
|
"pytest-xdist>=3.8.0",
|
|
"pytest-rerunfailures>=15.0",
|
|
"python-dotenv>=1.2.1",
|
|
"filelock>=3.20.1", # TOCTOU race condition fix
|
|
"ruff>=0.8.0",
|
|
"ty>=0.0.1",
|
|
"testcontainers>=4.0.0",
|
|
]
|
|
|
|
[tool.ruff]
|
|
line-length = 120
|
|
target-version = "py311"
|
|
|
|
[tool.ruff.lint]
|
|
# Tests are formatted (via `ruff format`) but excluded from lint rules, which
|
|
# are too noisy for test code (unused imports/vars, import ordering).
|
|
exclude = [
|
|
"tests/**",
|
|
"**/tests/**",
|
|
]
|
|
select = [
|
|
"E", # pycodestyle errors
|
|
"W", # pycodestyle warnings
|
|
"F", # Pyflakes
|
|
"I", # isort
|
|
"B021", # flake8-bugbear: f-string used as docstring (leaves __doc__ None)
|
|
"TID251", # banned imports: sync HTTP clients and httpx (see banned-api below)
|
|
]
|
|
ignore = [
|
|
"E501", # line too long (handled by formatter)
|
|
"E402", # module import not at top of file
|
|
"F811", # redefined while unused
|
|
"F821", # undefined name (forward references in type hints)
|
|
]
|
|
|
|
[tool.ruff.lint.isort]
|
|
known-third-party = ["alembic"]
|
|
|
|
# Production HTTP is aiohttp, async only (see .claude/skills/code-review/SKILL.md).
|
|
[tool.ruff.lint.flake8-tidy-imports.banned-api]
|
|
"httpx".msg = "Use aiohttp (async) via hindsight_api.engine.aiohttp_session. httpx is allowed only to configure or classify a third-party SDK built on it, with a `# noqa: TID251` saying so."
|
|
"requests".msg = "Sync HTTP is forbidden in production code; use aiohttp via hindsight_api.engine.aiohttp_session."
|
|
"urllib.request".msg = "Sync HTTP is forbidden in production code; use aiohttp via hindsight_api.engine.aiohttp_session."
|
|
"urllib3".msg = "Sync HTTP is forbidden in production code; use aiohttp via hindsight_api.engine.aiohttp_session."
|
|
"http.client".msg = "Sync HTTP is forbidden in production code; use aiohttp via hindsight_api.engine.aiohttp_session."
|
|
|
|
[tool.ruff.lint.per-file-ignores]
|
|
# The container readiness probe runs as its own short-lived process and is
|
|
# stdlib-only by rule (tests/test_http_probe.py), so it is not on the event loop.
|
|
"hindsight_api/http_probe.py" = ["TID251"]
|
|
# Tests may use httpx (FastAPI TestClient / ASGITransport). `exclude` above already
|
|
# skips them on a directory run; this covers ruff invoked on an explicit test path.
|
|
"tests/**" = ["TID251"]
|
|
|
|
[tool.ruff.format]
|
|
quote-style = "double"
|
|
indent-style = "space"
|
|
|
|
[tool.uv]
|
|
# Use explicit index for PyTorch to prevent the pytorch index from serving
|
|
# non-pytorch packages (e.g. markupsafe) with incompatible wheels
|
|
[[tool.uv.index]]
|
|
name = "pytorch-cpu"
|
|
url = "https://download.pytorch.org/whl/cpu"
|
|
explicit = true
|
|
|
|
[tool.uv.sources]
|
|
# Route torch to the CPU-only PyTorch index; everything else uses PyPI
|
|
torch = { index = "pytorch-cpu" }
|
|
|
|
[tool.ty]
|
|
# Type checking configuration
|
|
# ty is an extremely fast Python type checker from Astral (same team as ruff/uv)
|
|
|
|
[tool.ty.environment]
|
|
python-version = "3.11"
|
|
|
|
[tool.ty.src]
|
|
exclude = [
|
|
"tests/",
|
|
"hindsight_api/alembic/",
|
|
]
|
|
|
|
[tool.ty.rules]
|
|
# Disable noisy rules while keeping important ones
|
|
invalid-argument-type = "ignore" # False positives with **kwargs patterns
|
|
invalid-return-type = "ignore" # Often intentional in async code
|
|
invalid-parameter-default = "ignore" # Optional params with None default
|
|
possibly-missing-attribute = "ignore" # Common with Optional types
|
|
invalid-raise = "ignore" # False positives with exception tracking
|
|
call-non-callable = "ignore" # False positives with Optional types
|
|
invalid-key = "ignore" # Pydantic ConfigDict not understood
|
|
invalid-method-override = "ignore" # Intentional signature differences
|
|
unresolved-reference = "ignore" # Forward references not always resolved
|