fleet-memory/hindsight-api-slim/tests/conftest.py
Nicolò Boschi ea662d062e
feat: fact_types and mental model exclusion filters for reflect (#615)
* feat: add fact_types and mental model exclusion filters to reflect and mental models

Adds three new filtering options to both the reflect endpoint and mental model creation/refresh:

- `fact_types`: restrict which fact types (world, experience, observation) are retrieved.
  Disables irrelevant agent tools entirely (no wasted tokens).
- `exclude_mental_models`: skip the search_mental_models tool altogether.
- `exclude_mental_model_ids`: exclude specific mental models by ID (merged with the
  existing self-exclusion logic during mental model refresh).

For mental models, options are persisted in the existing `trigger` JSONB column so they
are automatically applied on every refresh. The `UpdateMentalModelRequest` already
proxies `trigger`, so no extra endpoint changes are needed.

Also fixes the test fixture (`pg0_db_url` in conftest.py) to correctly resolve pg0://
URLs and run migrations before tests, which was causing all DB-dependent tests to fail
with "relation public.banks does not exist" when HINDSIGHT_API_DATABASE_URL=pg0://uuuu.

* fix: guard against disabled-tool hallucination and regenerate clients

- Add enabled_tools guard in reflect agent: if an LLM calls a tool that
  was excluded (e.g. recall when fact_types=["observation"]), return an
  error result instead of executing it
- Regenerate OpenAPI spec and all SDK clients (Go, Python, TypeScript)
  to include new fact_types / exclude_mental_models fields

* fix: add missing ReflectRequest fields in Rust CLI struct initializers

* fix: filter hallucinated tool calls before trace to prevent disabled tools appearing in results

* chore: merge main, fix lint formatting and update skills openapi.json

* feat: expose fact_types, exclude_mental_models, exclude_mental_model_ids in control plane UI

* fix: add missing trigger fields to MentalModel type in control plane api.ts

* fix: add missing trigger fields to local MentalModel interface in mental-models-view

* feat: tabbed mental model dialogs (Basic / Options tabs)

* refactor: shared FactTypeFilter component, tabbed mental model dialogs use General tab, clean up labels

* feat: pill-style toggle buttons for fact type filter (blue/emerald/amber per type)

* fix: add spacing between Fact Types label and pills, rename to Exclude all mental models
2026-03-19 17:03:41 +01:00

267 lines
9.3 KiB
Python

"""
Pytest configuration and shared fixtures.
"""
import pytest
import pytest_asyncio
import asyncio
import os
import filelock
from pathlib import Path
from dotenv import load_dotenv
from hindsight_api import MemoryEngine, LLMConfig, LocalSTEmbeddings, RequestContext
from hindsight_api.engine.cross_encoder import LocalSTCrossEncoder
from hindsight_api.engine.query_analyzer import DateparserQueryAnalyzer
from hindsight_api.engine.task_backend import SyncTaskBackend
from hindsight_api.pg0 import EmbeddedPostgres
# Default pg0 instance configuration for tests
DEFAULT_PG0_INSTANCE_NAME = "hindsight-test"
DEFAULT_PG0_PORT = 5556
# Load environment variables from .env at the start of test session
def pytest_configure(config):
"""Load environment variables before running tests."""
# Look for .env in the workspace root (two levels up from tests dir)
env_file = Path(__file__).parent.parent.parent / ".env"
if env_file.exists():
load_dotenv(env_file)
else:
print(f"Warning: {env_file} not found, tests may fail without proper configuration")
@pytest.fixture(scope="session")
def db_url():
"""
Provide a PostgreSQL connection URL for tests.
If HINDSIGHT_API_DATABASE_URL is set, use it directly.
Otherwise, return None to indicate pg0 should be used (managed by pg0_instance fixture).
"""
return os.getenv("HINDSIGHT_API_DATABASE_URL")
@pytest.fixture(scope="session")
def pg0_db_url(db_url, tmp_path_factory, worker_id):
"""
Session-scoped fixture that ensures pg0 is running, migrations are applied,
and returns the database URL.
If HINDSIGHT_API_DATABASE_URL is a plain postgresql:// URL, uses it directly.
If HINDSIGHT_API_DATABASE_URL is a pg0:// URL, resolves it to a real URL first.
Otherwise, starts pg0 once for the entire test session.
Uses filelock to ensure only one pytest-xdist worker starts pg0.
Migrations use PostgreSQL advisory locks internally, so they're safe to call
from multiple workers - only one will actually run migrations.
Note: We don't stop pg0 at the end because pytest-xdist runs workers in separate
processes that share the same pg0 instance. pg0 will persist for the next test run.
"""
from hindsight_api.pg0 import parse_pg0_url as _parse_pg0_url
# Determine pg0 instance name/port from db_url (if it's a pg0:// URL) or use defaults
if db_url and not _parse_pg0_url(db_url)[0]:
# Plain postgresql:// URL - use it directly but still run migrations
from hindsight_api.migrations import run_migrations
run_migrations(db_url)
return db_url
if db_url:
_, pg0_name, pg0_port = _parse_pg0_url(db_url)
pg0_instance_name = pg0_name or DEFAULT_PG0_INSTANCE_NAME
pg0_instance_port = pg0_port or DEFAULT_PG0_PORT
else:
pg0_instance_name = DEFAULT_PG0_INSTANCE_NAME
pg0_instance_port = DEFAULT_PG0_PORT
# Get shared temp dir for coordination between xdist workers
if worker_id == "master":
# Running without xdist (-n 0 or no -n flag)
root_tmp_dir = tmp_path_factory.getbasetemp()
else:
# Running with xdist - use parent dir shared by all workers
root_tmp_dir = tmp_path_factory.getbasetemp().parent
# Use a lock file to ensure only one worker starts pg0
lock_file = root_tmp_dir / f"pg0_setup_{pg0_instance_name}.lock"
url_file = root_tmp_dir / f"pg0_url_{pg0_instance_name}.txt"
with filelock.FileLock(str(lock_file)):
if url_file.exists():
# Another worker already started pg0
url = url_file.read_text().strip()
else:
# First worker - start pg0
pg0 = EmbeddedPostgres(name=pg0_instance_name, port=pg0_instance_port)
# Run ensure_running in a new event loop
loop = asyncio.new_event_loop()
try:
url = loop.run_until_complete(pg0.ensure_running())
finally:
loop.close()
# Save URL for other workers
url_file.write_text(url)
# Run migrations - uses PostgreSQL advisory lock internally,
# so safe to call from multiple workers (only one will actually run migrations)
from hindsight_api.migrations import run_migrations
run_migrations(url)
return url
@pytest.fixture(scope="function")
def request_context():
"""Provide a default RequestContext for tests."""
return RequestContext()
@pytest.fixture(scope="session")
def llm_config():
"""
Provide LLM configuration for tests.
This can be used by tests that need to call LLM directly without memory system.
"""
return LLMConfig.for_memory()
@pytest.fixture(scope="session")
def embeddings(tmp_path_factory, worker_id):
"""
Session-scoped embeddings fixture with filelock to prevent race conditions.
When pytest-xdist runs multiple workers in parallel, they all try to load
models from the HuggingFace cache simultaneously, which can cause race
conditions and meta tensor errors. We use a filelock to serialize model
initialization across workers.
"""
# Get shared temp dir for coordination between xdist workers
if worker_id == "master":
root_tmp_dir = tmp_path_factory.getbasetemp()
else:
root_tmp_dir = tmp_path_factory.getbasetemp().parent
lock_file = root_tmp_dir / "embeddings_init.lock"
emb = LocalSTEmbeddings()
# Serialize model initialization across workers
with filelock.FileLock(str(lock_file)):
loop = asyncio.new_event_loop()
try:
loop.run_until_complete(emb.initialize())
finally:
loop.close()
return emb
@pytest.fixture(scope="session")
def cross_encoder(tmp_path_factory, worker_id):
"""
Session-scoped cross-encoder fixture with filelock to prevent race conditions.
When pytest-xdist runs multiple workers in parallel, they all try to load
models from the HuggingFace cache simultaneously, which can cause race
conditions and meta tensor errors. We use a filelock to serialize model
initialization across workers.
"""
# Get shared temp dir for coordination between xdist workers
if worker_id == "master":
root_tmp_dir = tmp_path_factory.getbasetemp()
else:
root_tmp_dir = tmp_path_factory.getbasetemp().parent
lock_file = root_tmp_dir / "cross_encoder_init.lock"
ce = LocalSTCrossEncoder()
# Serialize model initialization across workers
with filelock.FileLock(str(lock_file)):
loop = asyncio.new_event_loop()
try:
loop.run_until_complete(ce.initialize())
finally:
loop.close()
return ce
@pytest.fixture(scope="session")
def query_analyzer():
return DateparserQueryAnalyzer()
@pytest_asyncio.fixture(scope="function")
async def memory(pg0_db_url, embeddings, cross_encoder, query_analyzer):
"""
Provide a MemoryEngine instance for each test.
Must be function-scoped because:
1. pytest-xdist runs tests in separate processes with different event loops
2. asyncpg pools are bound to the event loop that created them
3. Each test needs its own pool in its own event loop
Uses small pool sizes since tests run in parallel.
Uses pg0_db_url (a postgresql:// URL) directly, so MemoryEngine won't try to
manage pg0 lifecycle - that's handled by the session-scoped pg0_db_url fixture.
Migrations are disabled here since they're run once at session scope in pg0_db_url.
Uses SyncTaskBackend so async tasks execute immediately (no worker needed).
"""
mem = MemoryEngine(
db_url=pg0_db_url, # Direct postgresql:// URL, not pg0://
memory_llm_provider=os.getenv("HINDSIGHT_API_LLM_PROVIDER", "groq"),
memory_llm_api_key=os.getenv("HINDSIGHT_API_LLM_API_KEY"),
memory_llm_model=os.getenv("HINDSIGHT_API_LLM_MODEL", "openai/gpt-oss-120b"),
memory_llm_base_url=os.getenv("HINDSIGHT_API_LLM_BASE_URL") or None,
embeddings=embeddings,
cross_encoder=cross_encoder,
query_analyzer=query_analyzer,
pool_min_size=1,
pool_max_size=5,
run_migrations=False, # Migrations already run at session scope
task_backend=SyncTaskBackend(), # Execute tasks immediately in tests
)
await mem.initialize()
yield mem
try:
if mem._pool and not mem._pool._closing:
await mem.close()
except Exception:
pass
@pytest_asyncio.fixture(scope="function")
async def memory_no_llm_verify(pg0_db_url, embeddings, cross_encoder, query_analyzer):
"""
Provide a MemoryEngine instance that skips LLM connection verification.
This fixture is useful for tests that override the LLM configuration
after initialization (e.g., to test specific providers).
"""
mem = MemoryEngine(
db_url=pg0_db_url,
memory_llm_provider="mock", # Use mock provider as placeholder
memory_llm_api_key="",
memory_llm_model="mock",
embeddings=embeddings,
cross_encoder=cross_encoder,
query_analyzer=query_analyzer,
pool_min_size=1,
pool_max_size=5,
run_migrations=False,
task_backend=SyncTaskBackend(),
skip_llm_verification=True, # Skip verification - will be overridden by test
)
await mem.initialize()
yield mem
try:
if mem._pool and not mem._pool._closing:
await mem.close()
except Exception:
pass