* feat: introduce hindsight-api-slim and hindsight-all-slim packages Closes #552 - Move all source code from hindsight-api/ to new hindsight-api-slim/ - hindsight-api-slim has heavy ML deps (torch, sentence-transformers, transformers, einops, flashrank, mlx, mlx-lm, safetensors) and pg0-embedded as optional extras: [local-ml], [embedded-db], [all] - hindsight-api becomes a zero-code meta-package depending on hindsight-api-slim[all] for full backward compatibility - Add hindsight-all-slim meta-package: hindsight-api-slim + client + embed - hindsight-all updated to depend on hindsight-api-slim[all] - pg0.py: lazy-import pg0 with clear ImportError pointing to [embedded-db] - Dockerfile: replace sed hack with proper uv sync --extra flags - Update release.yml, test.yml, lint.sh, release.sh, CLAUDE.md and all path references throughout the repo * refactor: rename hindsight/ directory to hindsight-all/ * docs: document hindsight-api-slim and hindsight-all-slim package variants Add package variants table and extras explanation to installation.md * docs: remove emojis from installation.md, use professional tone * docs: link Docker slim variant to pip package variants section * docs: consolidate Docker image variants into single table * ci: fix working-directory paths after package restructure - Replace all hindsight-api → hindsight-api-slim in test.yml - Replace hindsight → hindsight-all in test.yml - Add --extra embedded-db to test-embed API install step * ci: add local-ml and embedded-db extras to API sync steps These extras were previously implicit in the old hindsight-api package (which bundled everything). Now that hindsight-api-slim uses optional extras, we must explicitly request local-ml and embedded-db in CI. * ci: add API install step with embedded-db to test-embed smoke test The smoke test starts hindsight-api as a daemon, which requires pg0-embedded. Add a dedicated install step for hindsight-api-slim with embedded-db extra so the daemon can start successfully. * ci: remove --no-install-project when using optional extras When --no-install-project is combined with --extra, the optional deps are not installed because extras require the project to be active. Remove --no-install-project from steps that need local-ml or embedded-db. * ci: fix ordering of uv sync steps to preserve optional extras When uv sync runs for a different workspace member, it removes optional extras installed for other members. Fix by always running extra-requiring API sync last, after other workspace member syncs. Also remove --no-install-project from embedded-db sync in test-embed, as --no-install-project prevents optional extras from being active. * ci: add local-ml extra to test-embed API install for smoke test The smoke test starts the full API server which needs sentence-transformers for local embeddings (default provider). Add local-ml extra to the install. * ci: simplify extras with --all-extras and add slim pip smoke test - Replace explicit --extra local-ml --extra embedded-db with --all-extras for cleaner, more maintainable sync steps - Add test-pip-slim job: tests hindsight-api-slim[embedded-db] without local ML models, using Cohere for embeddings/reranking (mirrors Docker slim smoke test approach) * ci: simplify slim smoke test to health check only (mirrors Docker test)
263 lines
10 KiB
Python
263 lines
10 KiB
Python
"""
|
|
Real integration test for OpenAI Batch API.
|
|
|
|
This test makes REAL API calls to OpenAI and measures actual timing.
|
|
It will be slow (minutes to hours) depending on OpenAI's queue.
|
|
|
|
To run:
|
|
pytest tests/test_batch_api_integration.py -v -s
|
|
|
|
To skip in CI:
|
|
Add @pytest.mark.skip at the test level
|
|
"""
|
|
import pytest
|
|
import os
|
|
import asyncio
|
|
import logging
|
|
import time
|
|
from datetime import datetime, timezone
|
|
from dotenv import load_dotenv
|
|
from hindsight_api import RequestContext
|
|
from hindsight_api.engine.retain.fact_extraction import (
|
|
extract_facts_from_contents_batch_api,
|
|
RetainContent,
|
|
)
|
|
from hindsight_api.config import HindsightConfig
|
|
from hindsight_api.engine.llm_wrapper import LLMProvider
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
# Load .env file for API keys
|
|
load_dotenv()
|
|
|
|
|
|
@pytest.fixture
|
|
def openai_api_key():
|
|
"""Get OpenAI API key from environment."""
|
|
# Try both current and commented keys from .env
|
|
api_key = os.getenv("HINDSIGHT_API_LLM_API_KEY")
|
|
|
|
# Check if it's an OpenAI key (starts with sk-proj- or sk-)
|
|
if not api_key or not api_key.startswith("sk-"):
|
|
# Try the OpenAI-specific env var (if set separately)
|
|
api_key = os.getenv("OPENAI_API_KEY")
|
|
|
|
if not api_key or not api_key.startswith("sk-"):
|
|
pytest.skip("OpenAI API key not found in environment. Set OPENAI_API_KEY or uncomment OpenAI config in .env")
|
|
|
|
return api_key
|
|
|
|
|
|
@pytest.fixture
|
|
def real_llm_config(openai_api_key):
|
|
"""Create real LLM config for OpenAI."""
|
|
# Create config with OpenAI settings
|
|
config = HindsightConfig.from_env()
|
|
|
|
# Use LLMProvider wrapper (which creates _provider_impl internally)
|
|
llm_config = LLMProvider(
|
|
provider="openai",
|
|
api_key=openai_api_key,
|
|
base_url="https://api.openai.com/v1",
|
|
model="gpt-4o-mini", # Fast, cheap model for testing
|
|
reasoning_effort="medium", # Required parameter
|
|
)
|
|
|
|
return llm_config
|
|
|
|
|
|
@pytest.fixture
|
|
def test_contents_real():
|
|
"""Create realistic test content for fact extraction."""
|
|
return [
|
|
RetainContent(
|
|
content="""
|
|
Alice is a senior software engineer at TechCorp, where she has been working for 5 years.
|
|
She specializes in distributed systems and microservices architecture. Alice graduated
|
|
from MIT with a degree in Computer Science in 2015. She is known for writing clean,
|
|
well-documented code and mentoring junior developers.
|
|
""",
|
|
event_date=datetime(2024, 1, 15, 10, 30, tzinfo=timezone.utc),
|
|
context="team member profile",
|
|
),
|
|
RetainContent(
|
|
content="""
|
|
Bob joined TechCorp last month as a junior developer. He is learning React and Node.js
|
|
and recently completed his first feature, which was a user authentication flow. Bob
|
|
graduated from Berkeley with a degree in Computer Science in 2023. He is enthusiastic
|
|
and asks great questions during code reviews.
|
|
""",
|
|
event_date=datetime(2024, 1, 15, 10, 30, tzinfo=timezone.utc),
|
|
context="team member profile",
|
|
),
|
|
RetainContent(
|
|
content="""
|
|
The team uses Kubernetes for container orchestration and deploys to AWS. They follow
|
|
agile methodologies with two-week sprints. Code reviews are mandatory before merging
|
|
any pull request. The team meets every morning for a 15-minute standup to discuss
|
|
progress and blockers.
|
|
""",
|
|
event_date=datetime(2024, 1, 15, 10, 30, tzinfo=timezone.utc),
|
|
context="team processes",
|
|
),
|
|
]
|
|
|
|
|
|
@pytest.fixture
|
|
def integration_config():
|
|
"""Create config for integration test."""
|
|
config = HindsightConfig.from_env()
|
|
config.retain_batch_enabled = True
|
|
config.retain_batch_poll_interval_seconds = 30 # Poll every 30 seconds (reasonable for real API)
|
|
config.retain_chunk_size = 4000
|
|
config.retain_extraction_mode = "concise"
|
|
config.retain_extract_causal_links = False
|
|
return config
|
|
|
|
|
|
@pytest.mark.skip(reason="Real API test - takes minutes and costs money. Run manually with: pytest tests/test_batch_api_integration.py::test_real_openai_batch_api -v -s")
|
|
@pytest.mark.integration # Mark as integration test
|
|
@pytest.mark.slow # Mark as slow test
|
|
@pytest.mark.asyncio
|
|
async def test_real_openai_batch_api(real_llm_config, test_contents_real, integration_config, memory, request_context):
|
|
"""
|
|
REAL integration test: Submit actual batch to OpenAI and measure timing.
|
|
|
|
WARNING: This test:
|
|
- Makes real API calls to OpenAI
|
|
- Will take minutes to hours to complete
|
|
- Costs money (though very little with gpt-4o-mini)
|
|
- Requires valid OpenAI API key
|
|
|
|
To skip this test:
|
|
pytest tests/test_batch_api_integration.py --skip-integration
|
|
"""
|
|
bank_id = f"test_real_batch_{datetime.now(timezone.utc).timestamp()}"
|
|
|
|
logger.info("=" * 80)
|
|
logger.info("STARTING REAL OPENAI BATCH API INTEGRATION TEST")
|
|
logger.info("=" * 80)
|
|
logger.info(f"Test contents: {len(test_contents_real)} items")
|
|
logger.info(f"Poll interval: {integration_config.retain_batch_poll_interval_seconds}s")
|
|
logger.info(f"Model: {real_llm_config.model}")
|
|
logger.info("This may take several minutes to hours depending on OpenAI's queue...")
|
|
logger.info("=" * 80)
|
|
|
|
try:
|
|
# Ensure bank exists
|
|
await memory.get_bank_profile(bank_id, request_context=request_context)
|
|
|
|
# Get database pool and schema for crash recovery testing
|
|
pool = memory._pool
|
|
schema = request_context.tenant_id
|
|
|
|
# Track overall timing
|
|
test_start_time = time.time()
|
|
|
|
# Call REAL batch API extraction
|
|
logger.info("\n📤 Submitting batch to OpenAI...")
|
|
|
|
facts, chunks, usage = await extract_facts_from_contents_batch_api(
|
|
contents=test_contents_real,
|
|
llm_config=real_llm_config,
|
|
agent_name="test_agent",
|
|
config=integration_config,
|
|
pool=pool,
|
|
operation_id=None, # No crash recovery for this test
|
|
schema=schema,
|
|
)
|
|
|
|
test_end_time = time.time()
|
|
total_duration = test_end_time - test_start_time
|
|
|
|
# Log results
|
|
logger.info("\n" + "=" * 80)
|
|
logger.info("✅ BATCH COMPLETED SUCCESSFULLY")
|
|
logger.info("=" * 80)
|
|
logger.info(f"Total duration: {total_duration:.1f} seconds ({total_duration/60:.1f} minutes)")
|
|
logger.info(f"Facts extracted: {len(facts)}")
|
|
logger.info(f"Chunks processed: {len(chunks)}")
|
|
logger.info(f"Token usage: {usage.input_tokens} input + {usage.output_tokens} output = {usage.total_tokens} total")
|
|
logger.info(f"Estimated cost: ${(usage.input_tokens * 0.00015 / 1000 + usage.output_tokens * 0.0006 / 1000):.4f}")
|
|
logger.info("=" * 80)
|
|
|
|
# Log sample facts
|
|
logger.info("\n📋 Sample extracted facts:")
|
|
for i, fact in enumerate(facts[:5]): # Show first 5 facts
|
|
logger.info(f"\nFact {i+1}:")
|
|
logger.info(f" Type: {fact.fact_type}")
|
|
logger.info(f" Text: {fact.fact_text[:100]}...")
|
|
logger.info(f" Entities: {fact.entities}")
|
|
|
|
# Verify results
|
|
assert len(facts) > 0, "Should extract at least some facts"
|
|
assert len(chunks) == len(test_contents_real), f"Should have {len(test_contents_real)} chunks"
|
|
assert usage.total_tokens > 0, "Should have token usage"
|
|
|
|
# Verify fact structure
|
|
for fact in facts:
|
|
assert hasattr(fact, "fact_text"), "Fact should have fact_text"
|
|
assert hasattr(fact, "fact_type"), "Fact should have fact_type"
|
|
assert fact.fact_type in ["world", "experience", "opinion"], f"Invalid fact_type: {fact.fact_type}"
|
|
|
|
logger.info("\n✅ All assertions passed!")
|
|
|
|
# Write timing report to file for later analysis
|
|
report_path = "/tmp/openai_batch_api_timing_report.txt"
|
|
with open(report_path, "w") as f:
|
|
f.write(f"OpenAI Batch API Integration Test Report\n")
|
|
f.write(f"={'=' * 60}\n\n")
|
|
f.write(f"Test Date: {datetime.now(timezone.utc).isoformat()}\n")
|
|
f.write(f"Model: {real_llm_config.model}\n")
|
|
f.write(f"Contents: {len(test_contents_real)} items\n")
|
|
f.write(f"Poll Interval: {integration_config.retain_batch_poll_interval_seconds}s\n\n")
|
|
f.write(f"Results:\n")
|
|
f.write(f" Total Duration: {total_duration:.1f}s ({total_duration/60:.1f} min)\n")
|
|
f.write(f" Facts Extracted: {len(facts)}\n")
|
|
f.write(f" Chunks Processed: {len(chunks)}\n")
|
|
f.write(f" Token Usage: {usage.total_tokens} ({usage.input_tokens} in + {usage.output_tokens} out)\n")
|
|
f.write(f" Estimated Cost: ${(usage.input_tokens * 0.00015 / 1000 + usage.output_tokens * 0.0006 / 1000):.4f}\n")
|
|
|
|
logger.info(f"\n📄 Timing report written to: {report_path}")
|
|
|
|
finally:
|
|
# Cleanup
|
|
try:
|
|
await memory.delete_bank(bank_id, request_context=request_context)
|
|
logger.info(f"\n🧹 Cleaned up test bank: {bank_id}")
|
|
except Exception as e:
|
|
logger.error(f"Failed to cleanup bank: {e}")
|
|
|
|
|
|
@pytest.mark.skip(reason="Real API test - requires Groq API key. Run manually if needed.")
|
|
@pytest.mark.integration
|
|
@pytest.mark.slow
|
|
@pytest.mark.asyncio
|
|
async def test_real_batch_supports_groq(integration_config):
|
|
"""
|
|
Test that Groq also supports batch API (if configured).
|
|
|
|
Groq has the same batch API interface as OpenAI.
|
|
"""
|
|
groq_api_key = os.getenv("HINDSIGHT_API_LLM_API_KEY")
|
|
|
|
if not groq_api_key or not groq_api_key.startswith("gsk_"):
|
|
pytest.skip("Groq API key not found in environment")
|
|
|
|
llm_config = LLMProvider(
|
|
provider="groq",
|
|
api_key=groq_api_key,
|
|
base_url="https://api.groq.com/openai/v1",
|
|
model="llama-3.1-8b-instant",
|
|
reasoning_effort="medium",
|
|
)
|
|
|
|
# Check if Groq supports batch API
|
|
supports_batch = await llm_config._provider_impl.supports_batch_api()
|
|
|
|
logger.info(f"Groq batch API support: {supports_batch}")
|
|
|
|
# Groq should support batch API (same interface as OpenAI)
|
|
assert supports_batch, "Groq should support batch API"
|
|
|
|
logger.info("✅ Groq batch API support confirmed")
|