* feat: introduce hindsight-api-slim and hindsight-all-slim packages Closes #552 - Move all source code from hindsight-api/ to new hindsight-api-slim/ - hindsight-api-slim has heavy ML deps (torch, sentence-transformers, transformers, einops, flashrank, mlx, mlx-lm, safetensors) and pg0-embedded as optional extras: [local-ml], [embedded-db], [all] - hindsight-api becomes a zero-code meta-package depending on hindsight-api-slim[all] for full backward compatibility - Add hindsight-all-slim meta-package: hindsight-api-slim + client + embed - hindsight-all updated to depend on hindsight-api-slim[all] - pg0.py: lazy-import pg0 with clear ImportError pointing to [embedded-db] - Dockerfile: replace sed hack with proper uv sync --extra flags - Update release.yml, test.yml, lint.sh, release.sh, CLAUDE.md and all path references throughout the repo * refactor: rename hindsight/ directory to hindsight-all/ * docs: document hindsight-api-slim and hindsight-all-slim package variants Add package variants table and extras explanation to installation.md * docs: remove emojis from installation.md, use professional tone * docs: link Docker slim variant to pip package variants section * docs: consolidate Docker image variants into single table * ci: fix working-directory paths after package restructure - Replace all hindsight-api → hindsight-api-slim in test.yml - Replace hindsight → hindsight-all in test.yml - Add --extra embedded-db to test-embed API install step * ci: add local-ml and embedded-db extras to API sync steps These extras were previously implicit in the old hindsight-api package (which bundled everything). Now that hindsight-api-slim uses optional extras, we must explicitly request local-ml and embedded-db in CI. * ci: add API install step with embedded-db to test-embed smoke test The smoke test starts hindsight-api as a daemon, which requires pg0-embedded. Add a dedicated install step for hindsight-api-slim with embedded-db extra so the daemon can start successfully. * ci: remove --no-install-project when using optional extras When --no-install-project is combined with --extra, the optional deps are not installed because extras require the project to be active. Remove --no-install-project from steps that need local-ml or embedded-db. * ci: fix ordering of uv sync steps to preserve optional extras When uv sync runs for a different workspace member, it removes optional extras installed for other members. Fix by always running extra-requiring API sync last, after other workspace member syncs. Also remove --no-install-project from embedded-db sync in test-embed, as --no-install-project prevents optional extras from being active. * ci: add local-ml extra to test-embed API install for smoke test The smoke test starts the full API server which needs sentence-transformers for local embeddings (default provider). Add local-ml extra to the install. * ci: simplify extras with --all-extras and add slim pip smoke test - Replace explicit --extra local-ml --extra embedded-db with --all-extras for cleaner, more maintainable sync steps - Add test-pip-slim job: tests hindsight-api-slim[embedded-db] without local ML models, using Cohere for embeddings/reranking (mirrors Docker slim smoke test approach) * ci: simplify slim smoke test to health check only (mirrors Docker test)
207 lines
6.9 KiB
Python
207 lines
6.9 KiB
Python
"""
|
|
Abstract interface for LLM providers.
|
|
|
|
This module defines the interface that all LLM providers must implement,
|
|
enabling support for multiple LLM backends (OpenAI, Anthropic, Gemini, Codex, etc.)
|
|
"""
|
|
|
|
from abc import ABC, abstractmethod
|
|
from typing import Any
|
|
|
|
from .response_models import LLMToolCallResult, TokenUsage
|
|
|
|
|
|
class LLMInterface(ABC):
|
|
"""
|
|
Abstract interface for LLM providers.
|
|
|
|
All LLM provider implementations must inherit from this class and implement
|
|
the required methods.
|
|
"""
|
|
|
|
def __init__(
|
|
self,
|
|
provider: str,
|
|
api_key: str,
|
|
base_url: str,
|
|
model: str,
|
|
reasoning_effort: str = "low",
|
|
**kwargs: Any,
|
|
):
|
|
"""
|
|
Initialize LLM provider.
|
|
|
|
Args:
|
|
provider: Provider name (e.g., "openai", "codex", "anthropic", "gemini").
|
|
api_key: API key or authentication token.
|
|
base_url: Base URL for the API.
|
|
model: Model name.
|
|
reasoning_effort: Reasoning effort level for supported providers.
|
|
**kwargs: Additional provider-specific parameters.
|
|
"""
|
|
self.provider = provider.lower()
|
|
self.api_key = api_key
|
|
self.base_url = base_url
|
|
self.model = model
|
|
self.reasoning_effort = reasoning_effort
|
|
|
|
@abstractmethod
|
|
async def verify_connection(self) -> None:
|
|
"""
|
|
Verify that the LLM provider is configured correctly by making a simple test call.
|
|
|
|
Raises:
|
|
RuntimeError: If the connection test fails.
|
|
"""
|
|
pass
|
|
|
|
@abstractmethod
|
|
async def call(
|
|
self,
|
|
messages: list[dict[str, str]],
|
|
response_format: Any | None = None,
|
|
max_completion_tokens: int | None = None,
|
|
temperature: float | None = None,
|
|
scope: str = "memory",
|
|
max_retries: int = 10,
|
|
initial_backoff: float = 1.0,
|
|
max_backoff: float = 60.0,
|
|
skip_validation: bool = False,
|
|
strict_schema: bool = False,
|
|
return_usage: bool = False,
|
|
) -> Any:
|
|
"""
|
|
Make an LLM API call with retry logic.
|
|
|
|
Args:
|
|
messages: List of message dicts with 'role' and 'content'.
|
|
response_format: Optional Pydantic model for structured output.
|
|
max_completion_tokens: Maximum tokens in response.
|
|
temperature: Sampling temperature (0.0-2.0).
|
|
scope: Scope identifier for tracking.
|
|
max_retries: Maximum retry attempts.
|
|
initial_backoff: Initial backoff time in seconds.
|
|
max_backoff: Maximum backoff time in seconds.
|
|
skip_validation: Return raw JSON without Pydantic validation.
|
|
strict_schema: Use strict JSON schema enforcement (OpenAI only).
|
|
return_usage: If True, return tuple (result, TokenUsage) instead of just result.
|
|
|
|
Returns:
|
|
If return_usage=False: Parsed response if response_format is provided, otherwise text content.
|
|
If return_usage=True: Tuple of (result, TokenUsage) with token counts.
|
|
|
|
Raises:
|
|
OutputTooLongError: If output exceeds token limits.
|
|
Exception: Re-raises API errors after retries exhausted.
|
|
"""
|
|
pass
|
|
|
|
@abstractmethod
|
|
async def call_with_tools(
|
|
self,
|
|
messages: list[dict[str, Any]],
|
|
tools: list[dict[str, Any]],
|
|
max_completion_tokens: int | None = None,
|
|
temperature: float | None = None,
|
|
scope: str = "tools",
|
|
max_retries: int = 5,
|
|
initial_backoff: float = 1.0,
|
|
max_backoff: float = 30.0,
|
|
tool_choice: str | dict[str, Any] = "auto",
|
|
) -> LLMToolCallResult:
|
|
"""
|
|
Make an LLM API call with tool/function calling support.
|
|
|
|
Args:
|
|
messages: List of message dicts. Can include tool results with role='tool'.
|
|
tools: List of tool definitions in OpenAI format.
|
|
max_completion_tokens: Maximum tokens in response.
|
|
temperature: Sampling temperature (0.0-2.0).
|
|
scope: Scope identifier for tracking.
|
|
max_retries: Maximum retry attempts.
|
|
initial_backoff: Initial backoff time in seconds.
|
|
max_backoff: Maximum backoff time in seconds.
|
|
tool_choice: How to choose tools - "auto", "none", "required", or specific function.
|
|
|
|
Returns:
|
|
LLMToolCallResult with content and/or tool_calls.
|
|
"""
|
|
pass
|
|
|
|
async def supports_batch_api(self) -> bool:
|
|
"""
|
|
Check if this provider supports batch API operations.
|
|
|
|
Returns:
|
|
True if provider supports submit_batch/get_batch_status/retrieve_batch_results
|
|
"""
|
|
return False
|
|
|
|
async def submit_batch(
|
|
self,
|
|
requests: list[dict[str, Any]],
|
|
endpoint: str = "/v1/chat/completions",
|
|
completion_window: str = "24h",
|
|
) -> dict[str, Any]:
|
|
"""
|
|
Submit a batch of requests to the provider's batch API.
|
|
|
|
Args:
|
|
requests: List of request dicts in JSONL format (custom_id, method, url, body)
|
|
endpoint: API endpoint for the batch (e.g., "/v1/chat/completions")
|
|
completion_window: Completion window (e.g., "24h")
|
|
|
|
Returns:
|
|
Dict with batch metadata: {"batch_id": str, "status": str, ...}
|
|
|
|
Raises:
|
|
NotImplementedError: If provider doesn't support batch API
|
|
"""
|
|
raise NotImplementedError(f"Batch API not supported for provider: {self.provider}")
|
|
|
|
async def get_batch_status(self, batch_id: str) -> dict[str, Any]:
|
|
"""
|
|
Get the status of a batch job.
|
|
|
|
Args:
|
|
batch_id: Batch identifier returned from submit_batch
|
|
|
|
Returns:
|
|
Dict with status info: {"batch_id": str, "status": str, "completed_at": str, ...}
|
|
|
|
Raises:
|
|
NotImplementedError: If provider doesn't support batch API
|
|
"""
|
|
raise NotImplementedError(f"Batch API not supported for provider: {self.provider}")
|
|
|
|
async def retrieve_batch_results(self, batch_id: str) -> list[dict[str, Any]]:
|
|
"""
|
|
Retrieve completed batch results.
|
|
|
|
Args:
|
|
batch_id: Batch identifier returned from submit_batch
|
|
|
|
Returns:
|
|
List of result dicts (one per request, matched by custom_id)
|
|
|
|
Raises:
|
|
NotImplementedError: If provider doesn't support batch API
|
|
"""
|
|
raise NotImplementedError(f"Batch API not supported for provider: {self.provider}")
|
|
|
|
@abstractmethod
|
|
async def cleanup(self) -> None:
|
|
"""Clean up resources (close connections, etc.)."""
|
|
pass
|
|
|
|
|
|
class OutputTooLongError(Exception):
|
|
"""
|
|
Bridge exception raised when LLM output exceeds token limits.
|
|
|
|
This wraps provider-specific errors (e.g., OpenAI's LengthFinishReasonError)
|
|
to allow callers to handle output length issues without depending on
|
|
provider-specific implementations.
|
|
"""
|
|
|
|
pass
|