* feat: introduce hindsight-api-slim and hindsight-all-slim packages Closes #552 - Move all source code from hindsight-api/ to new hindsight-api-slim/ - hindsight-api-slim has heavy ML deps (torch, sentence-transformers, transformers, einops, flashrank, mlx, mlx-lm, safetensors) and pg0-embedded as optional extras: [local-ml], [embedded-db], [all] - hindsight-api becomes a zero-code meta-package depending on hindsight-api-slim[all] for full backward compatibility - Add hindsight-all-slim meta-package: hindsight-api-slim + client + embed - hindsight-all updated to depend on hindsight-api-slim[all] - pg0.py: lazy-import pg0 with clear ImportError pointing to [embedded-db] - Dockerfile: replace sed hack with proper uv sync --extra flags - Update release.yml, test.yml, lint.sh, release.sh, CLAUDE.md and all path references throughout the repo * refactor: rename hindsight/ directory to hindsight-all/ * docs: document hindsight-api-slim and hindsight-all-slim package variants Add package variants table and extras explanation to installation.md * docs: remove emojis from installation.md, use professional tone * docs: link Docker slim variant to pip package variants section * docs: consolidate Docker image variants into single table * ci: fix working-directory paths after package restructure - Replace all hindsight-api → hindsight-api-slim in test.yml - Replace hindsight → hindsight-all in test.yml - Add --extra embedded-db to test-embed API install step * ci: add local-ml and embedded-db extras to API sync steps These extras were previously implicit in the old hindsight-api package (which bundled everything). Now that hindsight-api-slim uses optional extras, we must explicitly request local-ml and embedded-db in CI. * ci: add API install step with embedded-db to test-embed smoke test The smoke test starts hindsight-api as a daemon, which requires pg0-embedded. Add a dedicated install step for hindsight-api-slim with embedded-db extra so the daemon can start successfully. * ci: remove --no-install-project when using optional extras When --no-install-project is combined with --extra, the optional deps are not installed because extras require the project to be active. Remove --no-install-project from steps that need local-ml or embedded-db. * ci: fix ordering of uv sync steps to preserve optional extras When uv sync runs for a different workspace member, it removes optional extras installed for other members. Fix by always running extra-requiring API sync last, after other workspace member syncs. Also remove --no-install-project from embedded-db sync in test-embed, as --no-install-project prevents optional extras from being active. * ci: add local-ml extra to test-embed API install for smoke test The smoke test starts the full API server which needs sentence-transformers for local embeddings (default provider). Add local-ml extra to the install. * ci: simplify extras with --all-extras and add slim pip smoke test - Replace explicit --extra local-ml --extra embedded-db with --all-extras for cleaner, more maintainable sync steps - Add test-pip-slim job: tests hindsight-api-slim[embedded-db] without local ML models, using Cohere for embeddings/reranking (mirrors Docker slim smoke test approach) * ci: simplify slim smoke test to health check only (mirrors Docker test)
272 lines
9.1 KiB
Python
272 lines
9.1 KiB
Python
"""
|
|
Integration test for Hindsight server with context manager.
|
|
|
|
Tests the full workflow:
|
|
1. Starting server using context manager
|
|
2. Creating a memory bank
|
|
3. Storing memories (retain)
|
|
4. Recalling memories
|
|
5. Reflecting on memories
|
|
|
|
Note: These tests use embedded PostgreSQL (pg0) with a shared server instance
|
|
across all tests. Each test uses random bank_ids to avoid conflicts, allowing
|
|
safe parallel execution.
|
|
"""
|
|
|
|
import os
|
|
import uuid
|
|
import pytest
|
|
from hindsight import HindsightServer, HindsightClient
|
|
|
|
|
|
@pytest.fixture(scope="session")
|
|
def llm_config():
|
|
"""Get LLM configuration from environment (session-scoped)."""
|
|
provider = os.getenv("HINDSIGHT_LLM_PROVIDER", "groq")
|
|
api_key = os.getenv("HINDSIGHT_LLM_API_KEY", "")
|
|
model = os.getenv("HINDSIGHT_LLM_MODEL", "openai/gpt-oss-120b")
|
|
|
|
# vertexai uses GCP service account credentials (HINDSIGHT_API_LLM_VERTEXAI_*),
|
|
# not a traditional API key
|
|
providers_without_api_key = ("vertexai", "ollama")
|
|
if not api_key and provider not in providers_without_api_key:
|
|
raise Exception("LLM API key not configured. Set HINDSIGHT_LLM_API_KEY environment variable.")
|
|
|
|
return {
|
|
"llm_provider": provider,
|
|
"llm_api_key": api_key,
|
|
"llm_model": model,
|
|
}
|
|
|
|
|
|
@pytest.fixture(scope="session")
|
|
def shared_server(llm_config):
|
|
"""
|
|
Shared server instance for all tests (session-scoped).
|
|
|
|
This allows tests to run in parallel by sharing the same pg0 instance,
|
|
while using different bank_ids to avoid data conflicts.
|
|
"""
|
|
server = HindsightServer(db_url="pg0", **llm_config)
|
|
server.start()
|
|
yield server
|
|
server.stop()
|
|
|
|
|
|
@pytest.fixture
|
|
def client(shared_server):
|
|
"""Create a client connected to the shared server."""
|
|
return HindsightClient(base_url=shared_server.url)
|
|
|
|
|
|
def test_server_context_manager_basic_workflow(client):
|
|
"""
|
|
Test complete workflow using shared server.
|
|
|
|
This test:
|
|
1. Uses a shared server instance
|
|
2. Creates a memory bank with unique ID
|
|
3. Stores multiple memories
|
|
4. Recalls memories based on a query
|
|
5. Reflects (generates contextual answers) based on stored memories
|
|
"""
|
|
# Use random bank_id to allow parallel test execution
|
|
bank_id = f"test_assistant_{uuid.uuid4().hex[:8]}"
|
|
|
|
# Step 1: Create a memory bank with background information
|
|
print(f"\n1. Creating memory bank: {bank_id}")
|
|
bank_response = client.create_bank(
|
|
bank_id=bank_id,
|
|
name="Test Assistant",
|
|
mission="An AI assistant that helps with programming and data analysis tasks."
|
|
)
|
|
assert bank_response.bank_id == bank_id
|
|
|
|
# Step 2: Store some memories about user preferences
|
|
print("\n2. Storing memories...")
|
|
|
|
# Store first memory
|
|
retain_response1 = client.retain(
|
|
bank_id=bank_id,
|
|
content="User prefers Python over JavaScript for data analysis projects.",
|
|
context="User conversation about programming languages"
|
|
)
|
|
assert retain_response1.success is True
|
|
|
|
# Store second memory
|
|
retain_response2 = client.retain(
|
|
bank_id=bank_id,
|
|
content="User is working on a machine learning project using scikit-learn.",
|
|
context="Discussion about ML frameworks"
|
|
)
|
|
assert retain_response2.success is True
|
|
|
|
# Store third memory
|
|
retain_response3 = client.retain(
|
|
bank_id=bank_id,
|
|
content="User likes visualizing data with matplotlib and seaborn.",
|
|
context="Conversation about data visualization"
|
|
)
|
|
assert retain_response3.success is True
|
|
|
|
# Store batch memories
|
|
batch_response = client.retain_batch(
|
|
bank_id=bank_id,
|
|
items=[
|
|
{"content": "User is interested in neural networks and deep learning."},
|
|
{"content": "User asked about best practices for training models."},
|
|
]
|
|
)
|
|
# Check if the batch was submitted successfully (items_count shows how many were submitted)
|
|
assert batch_response.items_count >= 2
|
|
|
|
# Step 3: Recall memories based on a query
|
|
print("\n3. Recalling memories about programming preferences...")
|
|
recall_results = client.recall(
|
|
bank_id=bank_id,
|
|
query="What programming languages and tools does the user prefer?",
|
|
max_tokens=4096
|
|
)
|
|
|
|
# Verify recall results
|
|
assert isinstance(recall_results.results, list)
|
|
assert len(recall_results.results) > 0
|
|
print(f" Found {len(recall_results.results)} relevant memories")
|
|
|
|
# Check that results have expected structure
|
|
for result in recall_results.results:
|
|
print(f" - {result.text[:100]}")
|
|
|
|
# Step 4: Recall memories about machine learning
|
|
print("\n4. Recalling memories about machine learning...")
|
|
ml_recall_results = client.recall(
|
|
bank_id=bank_id,
|
|
query="machine learning and neural networks",
|
|
max_tokens=4096
|
|
)
|
|
|
|
# Verify recall results
|
|
assert isinstance(ml_recall_results.results, list)
|
|
assert len(ml_recall_results.results) > 0
|
|
print(f" Found {len(ml_recall_results.results)} ML-related memories")
|
|
for result in ml_recall_results.results[:3]: # Show first 3
|
|
print(f" - {result.text[:100]}")
|
|
|
|
# Step 5: Reflect (generate contextual answer based on memories)
|
|
print("\n5. Reflecting on query about recommendations...")
|
|
reflect_response = client.reflect(
|
|
bank_id=bank_id,
|
|
query="What tools and libraries should I recommend for this user's data analysis work?",
|
|
budget="mid"
|
|
)
|
|
|
|
# Verify reflection response
|
|
answer = reflect_response.text
|
|
assert len(answer) > 0
|
|
print(f" Answer: {answer[:200]}...")
|
|
|
|
# Verify the answer mentions relevant tools/libraries
|
|
answer_lower = answer.lower()
|
|
assert any(term in answer_lower for term in ["python", "scikit-learn", "matplotlib", "seaborn", "data"])
|
|
|
|
# Step 6: Another reflection with different context
|
|
print("\n6. Reflecting with additional context...")
|
|
reflect_with_context = client.reflect(
|
|
bank_id=bank_id,
|
|
query="Should I use TensorFlow or PyTorch?",
|
|
budget="low",
|
|
context="The user is starting a new deep learning project"
|
|
)
|
|
|
|
context_answer = reflect_with_context.text
|
|
assert len(context_answer) > 0
|
|
print(f" Context-aware answer: {context_answer[:150]}...")
|
|
|
|
|
|
def test_server_manual_start_stop(client):
|
|
"""
|
|
Test basic operations on shared server.
|
|
|
|
Verifies that basic bank operations work correctly.
|
|
"""
|
|
# Use random bank_id to allow parallel test execution
|
|
bank_id = f"test_manual_{uuid.uuid4().hex[:8]}"
|
|
|
|
# Create bank
|
|
bank_response = client.create_bank(
|
|
bank_id=bank_id,
|
|
name="Manual Test"
|
|
)
|
|
assert bank_response.bank_id == bank_id
|
|
|
|
# Store a memory
|
|
retain_response = client.retain(
|
|
bank_id=bank_id,
|
|
content="Testing manual server lifecycle."
|
|
)
|
|
assert retain_response.success is True
|
|
|
|
# Recall the memory
|
|
recall_results = client.recall(
|
|
bank_id=bank_id,
|
|
query="server testing"
|
|
)
|
|
assert len(recall_results.results) >= 0 # May or may not find results immediately
|
|
|
|
|
|
def test_server_with_client_context_manager(client):
|
|
"""
|
|
Test client context manager with shared server.
|
|
"""
|
|
# Use random bank_id to allow parallel test execution
|
|
bank_id = f"test_nested_context_{uuid.uuid4().hex[:8]}"
|
|
|
|
# Use client context manager (client fixture already provides this)
|
|
# Create bank
|
|
client.create_bank(bank_id=bank_id, name="Nested Context Test")
|
|
|
|
# Store memory
|
|
response = client.retain(
|
|
bank_id=bank_id,
|
|
content="Testing nested context managers."
|
|
)
|
|
assert response.success is True
|
|
|
|
# Verify we can recall
|
|
results = client.recall(bank_id=bank_id, query="context")
|
|
assert isinstance(results.results, list)
|
|
|
|
|
|
def test_list_banks(client, shared_server):
|
|
"""
|
|
Test listing banks to verify bank_id field mapping.
|
|
|
|
This test verifies that the list_banks endpoint correctly returns
|
|
bank_id (not agent_id) in the response.
|
|
"""
|
|
# Create a couple of banks with random IDs to allow parallel test execution
|
|
test_suffix = uuid.uuid4().hex[:8]
|
|
bank1_id = f"test_bank_1_{test_suffix}"
|
|
bank2_id = f"test_bank_2_{test_suffix}"
|
|
|
|
client.create_bank(bank_id=bank1_id, name="Test Bank 1", mission="First test bank")
|
|
client.create_bank(bank_id=bank2_id, name="Test Bank 2", mission="Second test bank")
|
|
|
|
# List all banks using the namespace API
|
|
response = client.banks.list()
|
|
|
|
# Verify response structure
|
|
assert hasattr(response, 'banks'), "Response should have 'banks' attribute"
|
|
assert len(response.banks) >= 2, f"Should have at least 2 banks, got {len(response.banks)}"
|
|
|
|
# Verify each bank has bank_id (not agent_id)
|
|
for bank in response.banks:
|
|
assert hasattr(bank, 'bank_id'), f"Bank should have 'bank_id' attribute"
|
|
assert bank.bank_id is not None, "Bank ID should not be None"
|
|
|
|
# Find our test banks
|
|
bank_ids = [b.bank_id if hasattr(b, 'bank_id') else b['bank_id'] for b in response.banks]
|
|
assert bank1_id in bank_ids, f"Should find {bank1_id} in bank list"
|
|
assert bank2_id in bank_ids, f"Should find {bank2_id} in bank list"
|
|
|
|
print(f"✓ Successfully listed {len(response.banks)} banks with correct bank_id field")
|