diff --git a/hindsight-api/hindsight_api/config.py b/hindsight-api/hindsight_api/config.py index e7cb0d81..d2334278 100644 --- a/hindsight-api/hindsight_api/config.py +++ b/hindsight-api/hindsight_api/config.py @@ -189,6 +189,14 @@ ENV_RERANKER_LITELLM_API_BASE = "HINDSIGHT_API_RERANKER_LITELLM_API_BASE" ENV_RERANKER_LITELLM_API_KEY = "HINDSIGHT_API_RERANKER_LITELLM_API_KEY" ENV_RERANKER_LITELLM_MODEL = "HINDSIGHT_API_RERANKER_LITELLM_MODEL" +# LiteLLM SDK configuration (direct API access, no proxy needed) +ENV_EMBEDDINGS_LITELLM_SDK_API_KEY = "HINDSIGHT_API_EMBEDDINGS_LITELLM_SDK_API_KEY" +ENV_EMBEDDINGS_LITELLM_SDK_MODEL = "HINDSIGHT_API_EMBEDDINGS_LITELLM_SDK_MODEL" +ENV_EMBEDDINGS_LITELLM_SDK_API_BASE = "HINDSIGHT_API_EMBEDDINGS_LITELLM_SDK_API_BASE" +ENV_RERANKER_LITELLM_SDK_API_KEY = "HINDSIGHT_API_RERANKER_LITELLM_SDK_API_KEY" +ENV_RERANKER_LITELLM_SDK_MODEL = "HINDSIGHT_API_RERANKER_LITELLM_SDK_MODEL" +ENV_RERANKER_LITELLM_SDK_API_BASE = "HINDSIGHT_API_RERANKER_LITELLM_SDK_API_BASE" + # Deprecated: Legacy shared LiteLLM config (for backward compatibility) ENV_LITELLM_API_BASE = "HINDSIGHT_API_LITELLM_API_BASE" ENV_LITELLM_API_KEY = "HINDSIGHT_API_LITELLM_API_KEY" @@ -337,6 +345,10 @@ DEFAULT_LITELLM_API_BASE = "http://localhost:4000" DEFAULT_EMBEDDINGS_LITELLM_MODEL = "text-embedding-3-small" DEFAULT_RERANKER_LITELLM_MODEL = "cohere/rerank-english-v3.0" +# LiteLLM SDK defaults +DEFAULT_EMBEDDINGS_LITELLM_SDK_MODEL = "cohere/embed-english-v3.0" +DEFAULT_RERANKER_LITELLM_SDK_MODEL = "cohere/rerank-english-v3.0" + DEFAULT_HOST = "0.0.0.0" DEFAULT_PORT = 8888 DEFAULT_BASE_PATH = "" # Empty string = root path @@ -532,6 +544,9 @@ class HindsightConfig: embeddings_litellm_api_base: str embeddings_litellm_api_key: str | None embeddings_litellm_model: str + embeddings_litellm_sdk_api_key: str | None + embeddings_litellm_sdk_model: str + embeddings_litellm_sdk_api_base: str | None # Reranker reranker_provider: str @@ -549,6 +564,9 @@ class HindsightConfig: reranker_litellm_api_base: str reranker_litellm_api_key: str | None reranker_litellm_model: str + reranker_litellm_sdk_api_key: str | None + reranker_litellm_sdk_model: str + reranker_litellm_sdk_api_base: str | None # Server host: str @@ -847,6 +865,12 @@ class HindsightConfig: or os.getenv(ENV_LITELLM_API_BASE, DEFAULT_LITELLM_API_BASE), embeddings_litellm_api_key=os.getenv(ENV_EMBEDDINGS_LITELLM_API_KEY) or os.getenv(ENV_LITELLM_API_KEY), embeddings_litellm_model=os.getenv(ENV_EMBEDDINGS_LITELLM_MODEL, DEFAULT_EMBEDDINGS_LITELLM_MODEL), + # LiteLLM SDK embeddings (direct API access) + embeddings_litellm_sdk_api_key=os.getenv(ENV_EMBEDDINGS_LITELLM_SDK_API_KEY), + embeddings_litellm_sdk_model=os.getenv( + ENV_EMBEDDINGS_LITELLM_SDK_MODEL, DEFAULT_EMBEDDINGS_LITELLM_SDK_MODEL + ), + embeddings_litellm_sdk_api_base=os.getenv(ENV_EMBEDDINGS_LITELLM_SDK_API_BASE) or None, # Reranker reranker_provider=os.getenv(ENV_RERANKER_PROVIDER, DEFAULT_RERANKER_PROVIDER), reranker_local_model=os.getenv(ENV_RERANKER_LOCAL_MODEL, DEFAULT_RERANKER_LOCAL_MODEL), @@ -876,6 +900,10 @@ class HindsightConfig: or os.getenv(ENV_LITELLM_API_BASE, DEFAULT_LITELLM_API_BASE), reranker_litellm_api_key=os.getenv(ENV_RERANKER_LITELLM_API_KEY) or os.getenv(ENV_LITELLM_API_KEY), reranker_litellm_model=os.getenv(ENV_RERANKER_LITELLM_MODEL, DEFAULT_RERANKER_LITELLM_MODEL), + # LiteLLM SDK reranker (direct API access) + reranker_litellm_sdk_api_key=os.getenv(ENV_RERANKER_LITELLM_SDK_API_KEY), + reranker_litellm_sdk_model=os.getenv(ENV_RERANKER_LITELLM_SDK_MODEL, DEFAULT_RERANKER_LITELLM_SDK_MODEL), + reranker_litellm_sdk_api_base=os.getenv(ENV_RERANKER_LITELLM_SDK_API_BASE) or None, # Server host=os.getenv(ENV_HOST, DEFAULT_HOST), port=int(os.getenv(ENV_PORT, DEFAULT_PORT)), diff --git a/hindsight-api/hindsight_api/engine/cross_encoder.py b/hindsight-api/hindsight_api/engine/cross_encoder.py index 7dbef6c9..75ad687f 100644 --- a/hindsight-api/hindsight_api/engine/cross_encoder.py +++ b/hindsight-api/hindsight_api/engine/cross_encoder.py @@ -21,6 +21,7 @@ from ..config import ( DEFAULT_RERANKER_FLASHRANK_CACHE_DIR, DEFAULT_RERANKER_FLASHRANK_MODEL, DEFAULT_RERANKER_LITELLM_MODEL, + DEFAULT_RERANKER_LITELLM_SDK_MODEL, DEFAULT_RERANKER_LOCAL_FORCE_CPU, DEFAULT_RERANKER_LOCAL_MAX_CONCURRENT, DEFAULT_RERANKER_LOCAL_MODEL, @@ -32,6 +33,7 @@ from ..config import ( ENV_RERANKER_COHERE_MODEL, ENV_RERANKER_FLASHRANK_CACHE_DIR, ENV_RERANKER_FLASHRANK_MODEL, + ENV_RERANKER_LITELLM_SDK_API_KEY, ENV_RERANKER_LOCAL_FORCE_CPU, ENV_RERANKER_LOCAL_MAX_CONCURRENT, ENV_RERANKER_LOCAL_MODEL, @@ -828,6 +830,126 @@ class LiteLLMCrossEncoder(CrossEncoderModel): return all_scores +class LiteLLMSDKCrossEncoder(CrossEncoderModel): + """ + LiteLLM SDK cross-encoder for direct API integration. + + Supports reranking via LiteLLM SDK without requiring a proxy server. + Supported providers: Cohere, DeepInfra, Together AI, HuggingFace, Jina AI, Voyage AI, AWS Bedrock. + + Example model names: + - cohere/rerank-english-v3.0 + - deepinfra/Qwen3-reranker-8B + - together_ai/Salesforce/Llama-Rank-V1 + - huggingface/BAAI/bge-reranker-v2-m3 + """ + + def __init__( + self, + api_key: str, + model: str = DEFAULT_RERANKER_LITELLM_SDK_MODEL, + api_base: str | None = None, + timeout: float = 60.0, + ): + """ + Initialize LiteLLM SDK cross-encoder client. + + Args: + api_key: API key for the reranking provider + model: Model name with provider prefix (e.g., "deepinfra/Qwen3-reranker-8B") + api_base: Custom base URL for API (optional) + timeout: Request timeout in seconds (default: 60.0) + """ + self.api_key = api_key + self.model = model + self.api_base = api_base + self.timeout = timeout + self._initialized = False + self._litellm = None # Will be set during initialization + + @property + def provider_name(self) -> str: + return "litellm-sdk" + + async def initialize(self) -> None: + """Initialize the LiteLLM SDK client.""" + if self._initialized: + return + + try: + import litellm + + self._litellm = litellm # Store reference + except ImportError: + raise ImportError("litellm is required for LiteLLMSDKCrossEncoder. Install it with: pip install litellm") + + api_base_msg = f" at {self.api_base}" if self.api_base else "" + logger.info(f"Reranker: initializing LiteLLM SDK provider with model {self.model}{api_base_msg}") + + self._initialized = True + logger.info("Reranker: LiteLLM SDK provider initialized") + + async def predict(self, pairs: list[tuple[str, str]]) -> list[float]: + """ + Score query-document pairs using the LiteLLM SDK. + + Args: + pairs: List of (query, document) tuples to score + + Returns: + List of relevance scores + """ + if not self._initialized: + raise RuntimeError("Reranker not initialized. Call initialize() first.") + + if not pairs: + return [] + + # Group pairs by query for efficient batching + # LiteLLM rerank expects one query with multiple documents + query_groups: dict[str, list[tuple[int, str]]] = {} + for idx, (query, text) in enumerate(pairs): + if query not in query_groups: + query_groups[query] = [] + query_groups[query].append((idx, text)) + + all_scores = [0.0] * len(pairs) + + for query, indexed_texts in query_groups.items(): + texts = [text for _, text in indexed_texts] + indices = [idx for idx, _ in indexed_texts] + + # Build kwargs for rerank call + rerank_kwargs = { + "model": self.model, + "query": query, + "documents": texts, + "api_key": self.api_key, + } + if self.api_base: + rerank_kwargs["api_base"] = self.api_base + + response = await self._litellm.arerank(**rerank_kwargs) + + # Map scores back to original positions + # Response format: RerankResponse with results list + # Each result is a TypedDict with "index" and "relevance_score" + if hasattr(response, "results") and response.results: + for result in response.results: + # Results are TypedDicts, use dict-style access + original_idx = result["index"] + score = result.get("relevance_score", result.get("score", 0.0)) + all_scores[indices[original_idx]] = score + elif isinstance(response, list): + # Direct list of scores (unlikely but defensive) + for i, score in enumerate(response): + all_scores[indices[i]] = score + else: + logger.warning(f"Unexpected response format from LiteLLM rerank: {type(response)}") + + return all_scores + + def create_cross_encoder_from_env() -> CrossEncoderModel: """ Create a CrossEncoderModel instance based on configuration. @@ -877,9 +999,20 @@ def create_cross_encoder_from_env() -> CrossEncoderModel: api_key=config.reranker_litellm_api_key, model=config.reranker_litellm_model, ) + elif provider == "litellm-sdk": + api_key = config.reranker_litellm_sdk_api_key + if not api_key: + raise ValueError( + f"{ENV_RERANKER_LITELLM_SDK_API_KEY} is required when {ENV_RERANKER_PROVIDER} is 'litellm-sdk'" + ) + return LiteLLMSDKCrossEncoder( + api_key=api_key, + model=config.reranker_litellm_sdk_model, + api_base=config.reranker_litellm_sdk_api_base, + ) elif provider == "rrf": return RRFPassthroughCrossEncoder() else: raise ValueError( - f"Unknown reranker provider: {provider}. Supported: 'local', 'tei', 'cohere', 'flashrank', 'litellm', 'rrf'" + f"Unknown reranker provider: {provider}. Supported: 'local', 'tei', 'cohere', 'flashrank', 'litellm', 'litellm-sdk', 'rrf'" ) diff --git a/hindsight-api/hindsight_api/engine/embeddings.py b/hindsight-api/hindsight_api/engine/embeddings.py index 97ecaa4a..d943591c 100644 --- a/hindsight-api/hindsight_api/engine/embeddings.py +++ b/hindsight-api/hindsight_api/engine/embeddings.py @@ -19,6 +19,7 @@ import httpx from ..config import ( DEFAULT_EMBEDDINGS_COHERE_MODEL, DEFAULT_EMBEDDINGS_LITELLM_MODEL, + DEFAULT_EMBEDDINGS_LITELLM_SDK_MODEL, DEFAULT_EMBEDDINGS_LOCAL_FORCE_CPU, DEFAULT_EMBEDDINGS_LOCAL_MODEL, DEFAULT_EMBEDDINGS_LOCAL_TRUST_REMOTE_CODE, @@ -26,6 +27,7 @@ from ..config import ( DEFAULT_EMBEDDINGS_PROVIDER, DEFAULT_LITELLM_API_BASE, ENV_EMBEDDINGS_COHERE_API_KEY, + ENV_EMBEDDINGS_LITELLM_SDK_API_KEY, ENV_EMBEDDINGS_LOCAL_FORCE_CPU, ENV_EMBEDDINGS_LOCAL_MODEL, ENV_EMBEDDINGS_LOCAL_TRUST_REMOTE_CODE, @@ -720,6 +722,148 @@ class LiteLLMEmbeddings(Embeddings): return all_embeddings +class LiteLLMSDKEmbeddings(Embeddings): + """ + LiteLLM SDK embeddings for direct API integration. + + Supports embeddings via LiteLLM SDK without requiring a proxy server. + Supported providers: Cohere, OpenAI, Azure OpenAI, HuggingFace, Voyage AI, Together AI, etc. + + Example model names: + - cohere/embed-english-v3.0 + - openai/text-embedding-3-small + - together_ai/togethercomputer/m2-bert-80M-8k-retrieval + - voyage/voyage-2 + """ + + def __init__( + self, + api_key: str, + model: str = DEFAULT_EMBEDDINGS_LITELLM_SDK_MODEL, + api_base: str | None = None, + batch_size: int = 100, + timeout: float = 60.0, + ): + """ + Initialize LiteLLM SDK embeddings client. + + Args: + api_key: API key for the embedding provider + model: Model name with provider prefix (e.g., "cohere/embed-english-v3.0") + api_base: Custom base URL for API (optional) + batch_size: Maximum batch size for embedding requests (default: 100) + timeout: Request timeout in seconds (default: 60.0) + """ + self.api_key = api_key + self.model = model + self.api_base = api_base + self.batch_size = batch_size + self.timeout = timeout + self._litellm = None # Will be set during initialization + self._dimension: int | None = None + + @property + def provider_name(self) -> str: + return "litellm-sdk" + + @property + def dimension(self) -> int: + if self._dimension is None: + raise RuntimeError("Embeddings not initialized. Call initialize() first.") + return self._dimension + + async def initialize(self) -> None: + """Initialize the LiteLLM SDK client and detect dimension.""" + if self._litellm is not None: + return + + try: + import litellm + + self._litellm = litellm # Store reference + except ImportError: + raise ImportError("litellm is required for LiteLLMSDKEmbeddings. Install it with: pip install litellm") + + api_base_msg = f" at {self.api_base}" if self.api_base else "" + logger.info(f"Embeddings: initializing LiteLLM SDK provider with model {self.model}{api_base_msg}") + + # Do a test embedding to detect dimension + try: + # Build kwargs for embedding call + embed_kwargs = { + "model": self.model, + "input": ["test"], + "api_key": self.api_key, + } + if self.api_base: + embed_kwargs["api_base"] = self.api_base + + # Use async embedding method (standard in litellm) + response = await self._litellm.aembedding(**embed_kwargs) + + # Extract dimension from response + if response.data and len(response.data) > 0: + self._dimension = len(response.data[0]["embedding"]) + else: + raise RuntimeError(f"Unable to detect embedding dimension for model {self.model}") + + except Exception as e: + raise RuntimeError(f"Failed to initialize LiteLLM SDK embeddings: {e}") + + logger.info(f"Embeddings: LiteLLM SDK provider initialized (model: {self.model}, dim: {self._dimension})") + + def encode(self, texts: list[str]) -> list[list[float]]: + """ + Generate embeddings using the LiteLLM SDK. + + Args: + texts: List of text strings to encode + + Returns: + List of embedding vectors (one per input text) + """ + if self._litellm is None: + raise RuntimeError("Embeddings not initialized. Call initialize() first.") + + if not texts: + return [] + + all_embeddings = [] + + # Process in batches + for i in range(0, len(texts), self.batch_size): + batch = texts[i : i + self.batch_size] + + try: + # Build kwargs for embedding call + embed_kwargs = { + "model": self.model, + "input": batch, + "api_key": self.api_key, + } + if self.api_base: + embed_kwargs["api_base"] = self.api_base + + # Use sync embedding (litellm doesn't have async in thread-safe way) + response = self._litellm.embedding(**embed_kwargs) + + # Extract embeddings from response + # Sort by index to ensure correct order + batch_embeddings = sorted(response.data, key=lambda x: x.get("index", 0)) + all_embeddings.extend([e["embedding"] for e in batch_embeddings]) + + except Exception as e: + import traceback + + logger.error( + f"Error in LiteLLM embedding for batch starting at index {i}: {e}\n" + f"Traceback: {traceback.format_exc()}" + ) + raise + + return all_embeddings + + def create_embeddings_from_env() -> Embeddings: """ Create an Embeddings instance based on configuration. @@ -771,7 +915,19 @@ def create_embeddings_from_env() -> Embeddings: api_key=config.embeddings_litellm_api_key, model=config.embeddings_litellm_model, ) + elif provider == "litellm-sdk": + api_key = config.embeddings_litellm_sdk_api_key + if not api_key: + raise ValueError( + f"{ENV_EMBEDDINGS_LITELLM_SDK_API_KEY} is required when {ENV_EMBEDDINGS_PROVIDER} is 'litellm-sdk'" + ) + return LiteLLMSDKEmbeddings( + api_key=api_key, + model=config.embeddings_litellm_sdk_model, + api_base=config.embeddings_litellm_sdk_api_base, + ) else: raise ValueError( - f"Unknown embeddings provider: {provider}. Supported: 'local', 'tei', 'openai', 'cohere', 'litellm'" + f"Unknown embeddings provider: {provider}. " + f"Supported: 'local', 'tei', 'openai', 'cohere', 'litellm', 'litellm-sdk'" ) diff --git a/hindsight-api/hindsight_api/main.py b/hindsight-api/hindsight_api/main.py index b77ca7e0..b5387164 100644 --- a/hindsight-api/hindsight_api/main.py +++ b/hindsight-api/hindsight_api/main.py @@ -208,6 +208,9 @@ def main(): embeddings_litellm_api_base=config.embeddings_litellm_api_base, embeddings_litellm_api_key=config.embeddings_litellm_api_key, embeddings_litellm_model=config.embeddings_litellm_model, + embeddings_litellm_sdk_api_key=config.embeddings_litellm_sdk_api_key, + embeddings_litellm_sdk_model=config.embeddings_litellm_sdk_model, + embeddings_litellm_sdk_api_base=config.embeddings_litellm_sdk_api_base, reranker_provider=config.reranker_provider, reranker_local_model=config.reranker_local_model, reranker_local_force_cpu=config.reranker_local_force_cpu, @@ -223,6 +226,9 @@ def main(): reranker_litellm_api_base=config.reranker_litellm_api_base, reranker_litellm_api_key=config.reranker_litellm_api_key, reranker_litellm_model=config.reranker_litellm_model, + reranker_litellm_sdk_api_key=config.reranker_litellm_sdk_api_key, + reranker_litellm_sdk_model=config.reranker_litellm_sdk_model, + reranker_litellm_sdk_api_base=config.reranker_litellm_sdk_api_base, host=args.host, port=args.port, base_path=config.base_path, diff --git a/hindsight-api/pyproject.toml b/hindsight-api/pyproject.toml index 1d2ad2d4..5613367b 100644 --- a/hindsight-api/pyproject.toml +++ b/hindsight-api/pyproject.toml @@ -42,6 +42,7 @@ dependencies = [ "typer>=0.9.0", "cohere>=5.0.0", "flashrank>=0.2.0", + "litellm>=1.0.0", # Local ML models for embeddings/reranking - can be excluded in Docker with INCLUDE_LOCAL_MODELS=false "sentence-transformers>=3.3.0", "transformers>=4.53.0", # Security fixes for ReDoS vulnerabilities diff --git a/hindsight-api/tests/test_litellm_sdk_cross_encoder.py b/hindsight-api/tests/test_litellm_sdk_cross_encoder.py new file mode 100644 index 00000000..3a2ae64a --- /dev/null +++ b/hindsight-api/tests/test_litellm_sdk_cross_encoder.py @@ -0,0 +1,392 @@ +""" +Tests for LiteLLMSDKCrossEncoder. + +Tests the LiteLLM SDK-based cross-encoder implementation for reranking. +""" + +import os +from unittest.mock import AsyncMock, MagicMock, patch + +import pytest + +from hindsight_api.engine.cross_encoder import LiteLLMSDKCrossEncoder, create_cross_encoder_from_env + + +class TestLiteLLMSDKCrossEncoder: + """Test suite for LiteLLMSDKCrossEncoder class.""" + + @pytest.mark.asyncio + async def test_initialization_success(self): + """Test successful initialization with valid config.""" + encoder = LiteLLMSDKCrossEncoder( + api_key="test_key", + model="deepinfra/Qwen3-reranker-8B", + ) + + assert encoder.provider_name == "litellm-sdk" + assert encoder.api_key == "test_key" + assert encoder.model == "deepinfra/Qwen3-reranker-8B" + assert encoder._initialized is False + + # Mock the litellm import + mock_litellm = MagicMock() + with patch.dict("sys.modules", {"litellm": mock_litellm}): + await encoder.initialize() + assert encoder._initialized is True + + @pytest.mark.asyncio + async def test_initialization_missing_package(self): + """Test initialization fails when litellm package is missing.""" + encoder = LiteLLMSDKCrossEncoder( + api_key="test_key", + model="cohere/rerank-english-v3.0", + ) + + with patch.dict("sys.modules", {"litellm": None}): + with pytest.raises(ImportError, match="litellm is required"): + await encoder.initialize() + + @pytest.mark.asyncio + async def test_initialization_idempotent(self): + """Test that calling initialize() multiple times is safe.""" + encoder = LiteLLMSDKCrossEncoder( + api_key="test_key", + model="cohere/rerank-english-v3.0", + ) + + mock_litellm = MagicMock() + with patch.dict("sys.modules", {"litellm": mock_litellm}): + await encoder.initialize() + assert encoder._initialized is True + + # Second call should be no-op + await encoder.initialize() + assert encoder._initialized is True + + @pytest.mark.asyncio + async def test_predict_single_query(self): + """Test prediction with a single query and multiple documents.""" + encoder = LiteLLMSDKCrossEncoder( + api_key="test_key", + model="deepinfra/Qwen3-reranker-8B", + ) + + # Create mock response with results as TypedDicts + mock_response = MagicMock() + mock_response.results = [ + {"index": 0, "relevance_score": 0.9}, + {"index": 1, "relevance_score": 0.7}, + {"index": 2, "relevance_score": 0.5}, + ] + + mock_litellm = MagicMock() + mock_litellm.arerank = AsyncMock(return_value=mock_response) + + with patch.dict("sys.modules", {"litellm": mock_litellm}): + await encoder.initialize() + + pairs = [ + ("What is Python?", "Python is a programming language"), + ("What is Python?", "Python is a snake"), + ("What is Python?", "Python is a British comedy group"), + ] + + scores = await encoder.predict(pairs) + + assert len(scores) == 3 + assert scores == [0.9, 0.7, 0.5] + + # Verify arerank was called correctly + mock_litellm.arerank.assert_called_once() + call_args = mock_litellm.arerank.call_args + assert call_args.kwargs["model"] == "deepinfra/Qwen3-reranker-8B" + assert call_args.kwargs["query"] == "What is Python?" + assert len(call_args.kwargs["documents"]) == 3 + assert call_args.kwargs["api_key"] == "test_key" + + @pytest.mark.asyncio + async def test_predict_multiple_queries(self): + """Test prediction with multiple different queries (grouped efficiently).""" + encoder = LiteLLMSDKCrossEncoder( + api_key="test_key", + model="cohere/rerank-english-v3.0", + ) + + # First query response + mock_response1 = MagicMock() + mock_response1.results = [ + {"index": 0, "relevance_score": 0.9}, + {"index": 1, "relevance_score": 0.7}, + ] + + # Second query response + mock_response2 = MagicMock() + mock_response2.results = [ + {"index": 0, "relevance_score": 0.8}, + ] + + mock_litellm = MagicMock() + mock_litellm.arerank = AsyncMock(side_effect=[mock_response1, mock_response2]) + + with patch.dict("sys.modules", {"litellm": mock_litellm}): + await encoder.initialize() + + pairs = [ + ("What is Python?", "Python is a programming language"), + ("What is Python?", "Python is a snake"), + ("What is Java?", "Java is a programming language"), + ] + + scores = await encoder.predict(pairs) + + assert len(scores) == 3 + assert scores[0] == 0.9 # First query, first doc + assert scores[1] == 0.7 # First query, second doc + assert scores[2] == 0.8 # Second query, first doc + + # Verify arerank was called twice (once per unique query) + assert mock_litellm.arerank.call_count == 2 + + @pytest.mark.asyncio + async def test_predict_empty_pairs(self): + """Test prediction with empty input.""" + encoder = LiteLLMSDKCrossEncoder( + api_key="test_key", + model="cohere/rerank-english-v3.0", + ) + + mock_litellm = MagicMock() + with patch.dict("sys.modules", {"litellm": mock_litellm}): + await encoder.initialize() + scores = await encoder.predict([]) + assert scores == [] + + @pytest.mark.asyncio + async def test_predict_not_initialized(self): + """Test that predict fails if encoder not initialized.""" + encoder = LiteLLMSDKCrossEncoder( + api_key="test_key", + model="cohere/rerank-english-v3.0", + ) + + pairs = [("query", "document")] + + with pytest.raises(RuntimeError, match="not initialized"): + await encoder.predict(pairs) + + @pytest.mark.asyncio + async def test_predict_error_handling(self): + """Test that errors during prediction are raised.""" + encoder = LiteLLMSDKCrossEncoder( + api_key="test_key", + model="cohere/rerank-english-v3.0", + ) + + # Mock litellm to raise an error + mock_litellm = MagicMock() + mock_litellm.arerank = AsyncMock(side_effect=Exception("API Error")) + + with patch.dict("sys.modules", {"litellm": mock_litellm}): + await encoder.initialize() + + pairs = [ + ("What is Python?", "Python is a programming language"), + ] + + # Should raise the exception + with pytest.raises(Exception, match="API Error"): + await encoder.predict(pairs) + + @pytest.mark.asyncio + async def test_custom_api_base(self): + """Test that custom API base URL is passed to rerank calls.""" + encoder = LiteLLMSDKCrossEncoder( + api_key="test_key", + model="cohere/rerank-english-v3.0", + api_base="https://custom.api.example.com", + ) + + mock_response = MagicMock() + mock_response.results = [ + {"index": 0, "relevance_score": 0.9}, + ] + + mock_litellm = MagicMock() + mock_litellm.arerank = AsyncMock(return_value=mock_response) + + with patch.dict("sys.modules", {"litellm": mock_litellm}): + await encoder.initialize() + + # Test that api_base is passed to arerank + pairs = [("query", "document")] + scores = await encoder.predict(pairs) + + assert scores == [0.9] + mock_litellm.arerank.assert_called_once() + call_args = mock_litellm.arerank.call_args + assert call_args.kwargs["api_base"] == "https://custom.api.example.com" + + @pytest.mark.asyncio + async def test_response_with_direct_score_list(self): + """Test handling of response format with direct score list.""" + encoder = LiteLLMSDKCrossEncoder( + api_key="test_key", + model="some-provider/model", + ) + + # Mock litellm to return direct list of scores + mock_litellm = MagicMock() + mock_litellm.arerank = AsyncMock(return_value=[0.9, 0.7, 0.5]) + + with patch.dict("sys.modules", {"litellm": mock_litellm}): + await encoder.initialize() + + pairs = [ + ("query", "doc1"), + ("query", "doc2"), + ("query", "doc3"), + ] + + scores = await encoder.predict(pairs) + + assert scores == [0.9, 0.7, 0.5] + + +class TestFactoryFunction: + """Test suite for create_cross_encoder_from_env factory function.""" + + @pytest.mark.asyncio + async def test_create_litellm_sdk_from_env(self): + """Test creating LiteLLM SDK cross-encoder from environment variables.""" + env_vars = { + "HINDSIGHT_API_RERANKER_PROVIDER": "litellm-sdk", + "HINDSIGHT_API_RERANKER_LITELLM_SDK_API_KEY": "test_key", + "HINDSIGHT_API_RERANKER_LITELLM_SDK_MODEL": "deepinfra/Qwen3-reranker-8B", + } + + with patch.dict(os.environ, env_vars, clear=False): + # Need to reload config to pick up env vars + from hindsight_api.config import HindsightConfig + + config = HindsightConfig.from_env() + + with patch("hindsight_api.config.get_config", return_value=config): + encoder = create_cross_encoder_from_env() + + assert isinstance(encoder, LiteLLMSDKCrossEncoder) + assert encoder.api_key == "test_key" + assert encoder.model == "deepinfra/Qwen3-reranker-8B" + + @pytest.mark.asyncio + async def test_create_litellm_sdk_missing_api_key(self): + """Test that factory raises error when API key is missing.""" + env_vars = { + "HINDSIGHT_API_RERANKER_PROVIDER": "litellm-sdk", + "HINDSIGHT_API_RERANKER_LITELLM_SDK_MODEL": "deepinfra/Qwen3-reranker-8B", + } + + with patch.dict(os.environ, env_vars, clear=False): + # Remove API key if set + if "HINDSIGHT_API_RERANKER_LITELLM_SDK_API_KEY" in os.environ: + del os.environ["HINDSIGHT_API_RERANKER_LITELLM_SDK_API_KEY"] + + from hindsight_api.config import HindsightConfig + + config = HindsightConfig.from_env() + + with patch("hindsight_api.config.get_config", return_value=config): + with pytest.raises(ValueError, match="HINDSIGHT_API_RERANKER_LITELLM_SDK_API_KEY is required"): + create_cross_encoder_from_env() + + @pytest.mark.asyncio + async def test_create_litellm_sdk_with_custom_api_base(self): + """Test creating LiteLLM SDK cross-encoder with custom API base.""" + env_vars = { + "HINDSIGHT_API_RERANKER_PROVIDER": "litellm-sdk", + "HINDSIGHT_API_RERANKER_LITELLM_SDK_API_KEY": "test_key", + "HINDSIGHT_API_RERANKER_LITELLM_SDK_MODEL": "cohere/rerank-english-v3.0", + "HINDSIGHT_API_RERANKER_LITELLM_SDK_API_BASE": "https://custom.api.example.com", + } + + with patch.dict(os.environ, env_vars, clear=False): + from hindsight_api.config import HindsightConfig + + config = HindsightConfig.from_env() + + with patch("hindsight_api.config.get_config", return_value=config): + encoder = create_cross_encoder_from_env() + + assert isinstance(encoder, LiteLLMSDKCrossEncoder) + assert encoder.api_base == "https://custom.api.example.com" + + +class TestLiteLLMSDKCohereCrossEncoder: + """Tests for LiteLLM SDK calling Cohere (runs in CI with COHERE_API_KEY).""" + + @pytest.fixture + async def litellm_cohere_cross_encoder(self): + """Create LiteLLM SDK cross-encoder instance for Cohere.""" + if not os.environ.get("COHERE_API_KEY"): + pytest.skip("Cohere API key not available (set COHERE_API_KEY)") + + encoder = LiteLLMSDKCrossEncoder( + api_key=os.environ["COHERE_API_KEY"], + model="cohere/rerank-english-v3.0", + ) + await encoder.initialize() + return encoder + + @pytest.mark.asyncio + async def test_litellm_sdk_cohere_initialization(self, litellm_cohere_cross_encoder): + """Test that LiteLLM SDK Cohere cross-encoder initializes correctly.""" + assert litellm_cohere_cross_encoder.provider_name == "litellm-sdk" + assert litellm_cohere_cross_encoder.model == "cohere/rerank-english-v3.0" + + @pytest.mark.asyncio + async def test_litellm_sdk_cohere_predict(self, litellm_cohere_cross_encoder): + """Test that LiteLLM SDK can call Cohere rerank API.""" + pairs = [ + ("What is the capital of France?", "Paris is the capital of France."), + ("What is the capital of France?", "The Eiffel Tower is in Paris."), + ("What is the capital of France?", "Python is a programming language."), + ] + scores = await litellm_cohere_cross_encoder.predict(pairs) + + assert len(scores) == 3 + assert all(isinstance(s, float) for s in scores) + # The first result should be most relevant + assert scores[0] > scores[2], "Direct answer should score higher than unrelated text" + # All scores should be in valid range + assert all(0.0 <= score <= 1.0 for score in scores) + + +class TestIntegration: + """Integration tests with real API (optional - requires API keys).""" + + @pytest.mark.skipif( + not os.environ.get("DEEPINFRA_API_KEY"), + reason="DEEPINFRA_API_KEY not set - skipping integration test", + ) + @pytest.mark.asyncio + async def test_real_deepinfra_api(self): + """Test with real DeepInfra API (requires DEEPINFRA_API_KEY env var).""" + encoder = LiteLLMSDKCrossEncoder( + api_key=os.environ["DEEPINFRA_API_KEY"], + model="deepinfra/Qwen3-reranker-8B", + ) + + await encoder.initialize() + + pairs = [ + ("What is Python?", "Python is a high-level programming language"), + ("What is Python?", "Python is a species of snake"), + ("What is Python?", "Python is unrelated text about cars"), + ] + + scores = await encoder.predict(pairs) + + # First doc should have highest score (most relevant) + assert len(scores) == 3 + assert scores[0] > scores[1] + assert scores[1] > scores[2] + assert all(0.0 <= score <= 1.0 for score in scores) diff --git a/hindsight-api/tests/test_litellm_sdk_embeddings.py b/hindsight-api/tests/test_litellm_sdk_embeddings.py new file mode 100644 index 00000000..857769c7 --- /dev/null +++ b/hindsight-api/tests/test_litellm_sdk_embeddings.py @@ -0,0 +1,387 @@ +""" +Tests for LiteLLM SDK embeddings implementation. + +These tests cover: +1. Initialization (success, missing package, missing API key, idempotent) +2. Encode (single text, multiple texts, batching, error handling) +3. Provider-specific configuration (Cohere, OpenAI, etc.) +4. Factory function (create from env, validation errors) +5. Dimension detection +""" + +import os +from unittest.mock import AsyncMock, MagicMock, patch + +import pytest + +from hindsight_api.config import ( + ENV_EMBEDDINGS_LITELLM_SDK_API_KEY, + ENV_EMBEDDINGS_LITELLM_SDK_MODEL, + ENV_EMBEDDINGS_PROVIDER, + HindsightConfig, +) +from hindsight_api.engine.embeddings import LiteLLMSDKEmbeddings, create_embeddings_from_env + + +class TestLiteLLMSDKEmbeddings: + """Unit tests for LiteLLMSDKEmbeddings with mocked litellm responses.""" + + @pytest.fixture + def mock_litellm(self): + """Mock litellm module.""" + mock = MagicMock() + + # Mock aembedding (async) for initialization + mock_response = MagicMock() + mock_response.data = [{"embedding": [0.1] * 768, "index": 0}] + mock.aembedding = AsyncMock(return_value=mock_response) + + # Mock embedding (sync) for encode + mock_sync_response = MagicMock() + mock_sync_response.data = [ + {"embedding": [0.1] * 768, "index": 0}, + {"embedding": [0.2] * 768, "index": 1}, + ] + mock.embedding = MagicMock(return_value=mock_sync_response) + + return mock + + @pytest.fixture + async def embeddings(self, mock_litellm): + """Create initialized LiteLLMSDKEmbeddings instance.""" + emb = LiteLLMSDKEmbeddings( + api_key="test_key", + model="cohere/embed-english-v3.0", + api_base=None, + batch_size=100, + timeout=60.0, + ) + # Manually set the mock (simulating successful initialization) + emb._litellm = mock_litellm + emb._dimension = 768 + return emb + + async def test_initialization_success(self, mock_litellm): + """Test successful initialization.""" + with patch("builtins.__import__", side_effect=lambda name, *args: mock_litellm if name == "litellm" else __import__(name, *args)): + emb = LiteLLMSDKEmbeddings( + api_key="test_key", + model="cohere/embed-english-v3.0", + api_base=None, + batch_size=100, + timeout=60.0, + ) + + assert emb._litellm is None + assert emb._dimension is None + + await emb.initialize() + + assert emb._litellm is not None + assert emb._dimension == 768 + + # Verify test embedding was called + mock_litellm.aembedding.assert_called_once_with( + model="cohere/embed-english-v3.0", + input=["test"], + api_key="test_key", + ) + + async def test_initialization_missing_package(self): + """Test initialization fails gracefully when litellm is not installed.""" + def mock_import(name, *args): + if name == "litellm": + raise ImportError("No module named 'litellm'") + return __import__(name, *args) + + with patch("builtins.__import__", side_effect=mock_import): + emb = LiteLLMSDKEmbeddings( + api_key="test_key", + model="cohere/embed-english-v3.0", + api_base=None, + batch_size=100, + timeout=60.0, + ) + + with pytest.raises(ImportError, match="litellm is required"): + await emb.initialize() + + async def test_initialization_idempotent(self, embeddings, mock_litellm): + """Test that calling initialize() multiple times is safe.""" + # embeddings._litellm is already set in fixture + assert embeddings._litellm is not None + + # Call again + await embeddings.initialize() + + # Should still have same litellm instance + assert embeddings._litellm is not None + + async def test_encode_single_text(self, embeddings, mock_litellm): + """Test encoding a single text.""" + # Set up mock response + mock_litellm.embedding.return_value.data = [ + {"embedding": [0.5] * 768, "index": 0}, + ] + + result = embeddings.encode(["Hello world"]) + + assert isinstance(result, list) + assert len(result) == 1 + assert len(result[0]) == 768 + assert all(isinstance(x, float) for x in result[0]) + assert all(abs(x - 0.5) < 0.001 for x in result[0]) + + # Verify call + mock_litellm.embedding.assert_called_once_with( + model="cohere/embed-english-v3.0", + input=["Hello world"], + api_key="test_key", + ) + + async def test_encode_multiple_texts(self, embeddings, mock_litellm): + """Test encoding multiple texts.""" + # Set up mock response + mock_litellm.embedding.return_value.data = [ + {"embedding": [0.1] * 768, "index": 0}, + {"embedding": [0.2] * 768, "index": 1}, + {"embedding": [0.3] * 768, "index": 2}, + ] + + texts = ["First text", "Second text", "Third text"] + result = embeddings.encode(texts) + + assert isinstance(result, list) + assert len(result) == 3 + assert len(result[0]) == 768 + assert len(result[1]) == 768 + assert len(result[2]) == 768 + assert all(abs(x - 0.1) < 0.001 for x in result[0]) + assert all(abs(x - 0.2) < 0.001 for x in result[1]) + assert all(abs(x - 0.3) < 0.001 for x in result[2]) + + async def test_encode_batching(self, embeddings, mock_litellm): + """Test that large inputs are batched correctly.""" + # Create embeddings with small batch size + emb = LiteLLMSDKEmbeddings( + api_key="test_key", + model="cohere/embed-english-v3.0", + api_base=None, + batch_size=2, # Small batch for testing + timeout=60.0, + ) + emb._litellm = mock_litellm + emb._initialized = True + emb._dimension = 768 + + # Mock responses for each batch + def mock_embedding_side_effect(model, input, **kwargs): + mock_response = MagicMock() + mock_response.data = [ + {"embedding": [float(i)] * 768, "index": i} for i in range(len(input)) + ] + return mock_response + + mock_litellm.embedding.side_effect = mock_embedding_side_effect + + # Encode 5 texts (should create 3 batches: 2, 2, 1) + texts = [f"Text {i}" for i in range(5)] + result = emb.encode(texts) + + assert isinstance(result, list) + assert len(result) == 5 + assert all(len(embedding) == 768 for embedding in result) + + # Verify batching: should be called 3 times + assert mock_litellm.embedding.call_count == 3 + + # Verify batch sizes + calls = mock_litellm.embedding.call_args_list + assert len(calls[0][1]["input"]) == 2 # First batch + assert len(calls[1][1]["input"]) == 2 # Second batch + assert len(calls[2][1]["input"]) == 1 # Third batch + + async def test_encode_empty_list(self, embeddings): + """Test encoding empty list returns empty list.""" + result = embeddings.encode([]) + + assert isinstance(result, list) + assert len(result) == 0 + + async def test_encode_before_initialization(self, mock_litellm): + """Test that encode raises error if not initialized.""" + emb = LiteLLMSDKEmbeddings( + api_key="test_key", + model="cohere/embed-english-v3.0", + api_base=None, + batch_size=100, + timeout=60.0, + ) + + with pytest.raises(RuntimeError, match="not initialized"): + emb.encode(["test"]) + + async def test_encode_error_handling(self, embeddings, mock_litellm): + """Test error handling during encoding.""" + # Make embedding raise an error + mock_litellm.embedding.side_effect = Exception("API Error") + + with pytest.raises(Exception, match="API Error"): + embeddings.encode(["test"]) + + async def test_dimension_property(self, embeddings): + """Test dimension property.""" + assert embeddings.dimension == 768 + + async def test_dimension_before_initialization(self, mock_litellm): + """Test dimension raises error if not initialized.""" + emb = LiteLLMSDKEmbeddings( + api_key="test_key", + model="cohere/embed-english-v3.0", + api_base=None, + batch_size=100, + timeout=60.0, + ) + + with pytest.raises(RuntimeError, match="not initialized"): + _ = emb.dimension + + async def test_custom_api_base(self, mock_litellm): + """Test custom API base URL is passed to embedding calls.""" + with patch("builtins.__import__", side_effect=lambda name, *args: mock_litellm if name == "litellm" else __import__(name, *args)): + emb = LiteLLMSDKEmbeddings( + api_key="test_key", + model="cohere/embed-english-v3.0", + api_base="https://custom.api.com", + batch_size=100, + timeout=60.0, + ) + + await emb.initialize() + + # Verify api_base is set + assert emb.api_base == "https://custom.api.com" + + # Verify api_base is passed to aembedding + mock_litellm.aembedding.assert_called_once() + call_args = mock_litellm.aembedding.call_args + assert call_args.kwargs["api_base"] == "https://custom.api.com" + + # Test encode also passes api_base + mock_litellm.embedding.return_value.data = [{"embedding": [0.1] * 768, "index": 0}] + emb.encode(["test"]) + + mock_litellm.embedding.assert_called_once() + call_args = mock_litellm.embedding.call_args + assert call_args.kwargs["api_base"] == "https://custom.api.com" + + +class TestLiteLLMSDKEmbeddingsFactory: + """Test the factory function for creating LiteLLM SDK embeddings.""" + + def test_create_from_env_success(self, monkeypatch): + """Test creating embeddings from environment variables.""" + # Mock get_config() to return configured HindsightConfig + mock_config = MagicMock() + mock_config.embeddings_provider = "litellm-sdk" + mock_config.embeddings_litellm_sdk_api_key = "test_key" + mock_config.embeddings_litellm_sdk_model = "cohere/embed-english-v3.0" + mock_config.embeddings_litellm_sdk_api_base = None + + with patch("hindsight_api.config.get_config", return_value=mock_config): + embeddings = create_embeddings_from_env() + + assert isinstance(embeddings, LiteLLMSDKEmbeddings) + assert embeddings.api_key == "test_key" + assert embeddings.model == "cohere/embed-english-v3.0" + + def test_create_from_env_missing_api_key(self, monkeypatch): + """Test that missing API key raises error.""" + # Mock get_config() with missing API key + mock_config = MagicMock() + mock_config.embeddings_provider = "litellm-sdk" + mock_config.embeddings_litellm_sdk_api_key = None # Missing key + mock_config.embeddings_litellm_sdk_model = "cohere/embed-english-v3.0" + + with patch("hindsight_api.config.get_config", return_value=mock_config): + with pytest.raises(ValueError, match=ENV_EMBEDDINGS_LITELLM_SDK_API_KEY): + create_embeddings_from_env() + + def test_create_from_env_with_api_base(self, monkeypatch): + """Test creating embeddings with custom API base.""" + # Mock get_config() with custom API base + mock_config = MagicMock() + mock_config.embeddings_provider = "litellm-sdk" + mock_config.embeddings_litellm_sdk_api_key = "test_key" + mock_config.embeddings_litellm_sdk_model = "cohere/embed-english-v3.0" + mock_config.embeddings_litellm_sdk_api_base = "https://custom.api.com" + + with patch("hindsight_api.config.get_config", return_value=mock_config): + embeddings = create_embeddings_from_env() + + assert isinstance(embeddings, LiteLLMSDKEmbeddings) + assert embeddings.api_base == "https://custom.api.com" + + +class TestLiteLLMSDKCohereEmbeddings: + """Integration tests calling real Cohere API (matches CI pattern).""" + + @pytest.fixture + async def litellm_cohere_embeddings(self): + """Create embeddings instance with real Cohere API key.""" + if not os.environ.get("COHERE_API_KEY"): + pytest.skip("Cohere API key not available") + + emb = LiteLLMSDKEmbeddings( + api_key=os.environ["COHERE_API_KEY"], + model="cohere/embed-english-v3.0", + api_base=None, + batch_size=100, + timeout=60.0, + ) + await emb.initialize() + return emb + + @pytest.mark.asyncio + async def test_litellm_sdk_cohere_encode(self, litellm_cohere_embeddings): + """Test real Cohere API call for embeddings.""" + texts = [ + "The quick brown fox jumps over the lazy dog", + "Machine learning is a subset of artificial intelligence", + "Python is a popular programming language", + ] + + result = litellm_cohere_embeddings.encode(texts) + + # Verify result type and shape + assert isinstance(result, list) + assert len(result) == 3 + assert all(len(embedding) > 0 for embedding in result) + assert all(isinstance(x, float) for x in result[0]) + + # Verify embeddings are not zeros (common API failure mode) + for i, embedding in enumerate(result): + assert not all(abs(x) < 0.0001 for x in embedding), f"Embedding {i} is all zeros" + + # Verify embeddings are normalized (Cohere returns normalized vectors) + for i, embedding in enumerate(result): + norm = sum(x * x for x in embedding) ** 0.5 + assert 0.9 < norm < 1.1, f"Embedding {i} norm {norm} is not close to 1.0" + + @pytest.mark.asyncio + async def test_litellm_sdk_cohere_dimension(self, litellm_cohere_embeddings): + """Test dimension detection with real Cohere API.""" + dimension = litellm_cohere_embeddings.dimension + + # Cohere embed-english-v3.0 has 1024 dimensions + assert dimension == 1024 + + @pytest.mark.asyncio + async def test_litellm_sdk_cohere_single_text(self, litellm_cohere_embeddings): + """Test encoding single text with real Cohere API.""" + result = litellm_cohere_embeddings.encode(["Hello world"]) + + assert isinstance(result, list) + assert len(result) == 1 + assert len(result[0]) == 1024 + assert not all(abs(x) < 0.0001 for x in result[0]) diff --git a/hindsight-docs/docs/developer/configuration.md b/hindsight-docs/docs/developer/configuration.md index ba9fac7c..967fae9b 100644 --- a/hindsight-docs/docs/developer/configuration.md +++ b/hindsight-docs/docs/developer/configuration.md @@ -332,7 +332,7 @@ export HINDSIGHT_API_RETAIN_LLM_MAX_BACKOFF=120.0 # Cap at 2min instead of 1m | Variable | Description | Default | |----------|-------------|---------| -| `HINDSIGHT_API_EMBEDDINGS_PROVIDER` | Provider: `local`, `tei`, `openai`, `cohere`, or `litellm` | `local` | +| `HINDSIGHT_API_EMBEDDINGS_PROVIDER` | Provider: `local`, `tei`, `openai`, `cohere`, `litellm`, or `litellm-sdk` | `local` | | `HINDSIGHT_API_EMBEDDINGS_LOCAL_MODEL` | Model for local provider | `BAAI/bge-small-en-v1.5` | | `HINDSIGHT_API_EMBEDDINGS_LOCAL_TRUST_REMOTE_CODE` | Allow loading models with custom code (security risk, disabled by default) | `false` | | `HINDSIGHT_API_EMBEDDINGS_TEI_URL` | TEI server URL | - | @@ -345,6 +345,9 @@ export HINDSIGHT_API_RETAIN_LLM_MAX_BACKOFF=120.0 # Cap at 2min instead of 1m | `HINDSIGHT_API_EMBEDDINGS_LITELLM_API_BASE` | LiteLLM proxy base URL for embeddings | `http://localhost:4000` | | `HINDSIGHT_API_EMBEDDINGS_LITELLM_API_KEY` | LiteLLM proxy API key for embeddings (optional, depends on proxy config) | - | | `HINDSIGHT_API_EMBEDDINGS_LITELLM_MODEL` | LiteLLM embedding model (use provider prefix, e.g., `cohere/embed-english-v3.0`) | `text-embedding-3-small` | +| `HINDSIGHT_API_EMBEDDINGS_LITELLM_SDK_API_KEY` | LiteLLM SDK API key for direct embedding provider access | - | +| `HINDSIGHT_API_EMBEDDINGS_LITELLM_SDK_MODEL` | LiteLLM SDK embedding model (use provider prefix, e.g., `cohere/embed-english-v3.0`) | `cohere/embed-english-v3.0` | +| `HINDSIGHT_API_EMBEDDINGS_LITELLM_SDK_API_BASE` | Custom base URL for LiteLLM SDK embeddings (optional) | - | ```bash # Local (default) - uses SentenceTransformers @@ -387,6 +390,18 @@ export HINDSIGHT_API_EMBEDDINGS_PROVIDER=litellm export HINDSIGHT_API_EMBEDDINGS_LITELLM_API_BASE=http://localhost:4000 export HINDSIGHT_API_EMBEDDINGS_LITELLM_API_KEY=your-litellm-key # optional export HINDSIGHT_API_EMBEDDINGS_LITELLM_MODEL=text-embedding-3-small # or cohere/embed-english-v3.0 + +# LiteLLM SDK - direct API access without proxy server (recommended) +export HINDSIGHT_API_EMBEDDINGS_PROVIDER=litellm-sdk +export HINDSIGHT_API_EMBEDDINGS_LITELLM_SDK_API_KEY=your-provider-api-key +export HINDSIGHT_API_EMBEDDINGS_LITELLM_SDK_MODEL=cohere/embed-english-v3.0 + +# Supported LiteLLM SDK embedding providers: +# - cohere/embed-english-v3.0 (1024 dimensions) +# - openai/text-embedding-3-small (1536 dimensions) +# - together_ai/togethercomputer/m2-bert-80M-8k-retrieval +# - huggingface/sentence-transformers/all-MiniLM-L6-v2 +# - voyage/voyage-2 ``` #### Embedding Dimensions @@ -409,7 +424,7 @@ Supported OpenAI embedding dimensions: | Variable | Description | Default | |----------|-------------|---------| -| `HINDSIGHT_API_RERANKER_PROVIDER` | Provider: `local`, `tei`, `cohere`, `flashrank`, `litellm`, or `rrf` | `local` | +| `HINDSIGHT_API_RERANKER_PROVIDER` | Provider: `local`, `tei`, `cohere`, `flashrank`, `litellm`, `litellm-sdk`, or `rrf` | `local` | | `HINDSIGHT_API_RERANKER_LOCAL_MODEL` | Model for local provider | `cross-encoder/ms-marco-MiniLM-L-6-v2` | | `HINDSIGHT_API_RERANKER_LOCAL_MAX_CONCURRENT` | Max concurrent local reranking (prevents CPU thrashing under load) | `4` | | `HINDSIGHT_API_RERANKER_LOCAL_TRUST_REMOTE_CODE` | Allow loading models with custom code (security risk, disabled by default) | `false` | @@ -421,7 +436,10 @@ Supported OpenAI embedding dimensions: | `HINDSIGHT_API_RERANKER_COHERE_BASE_URL` | Custom base URL for Cohere-compatible API (e.g., Azure-hosted) | - | | `HINDSIGHT_API_RERANKER_LITELLM_API_BASE` | LiteLLM proxy base URL for reranking | `http://localhost:4000` | | `HINDSIGHT_API_RERANKER_LITELLM_API_KEY` | LiteLLM proxy API key for reranking (optional, depends on proxy config) | - | -| `HINDSIGHT_API_RERANKER_LITELLM_MODEL` | LiteLLM rerank model (use provider prefix, e.g., `cohere/rerank-english-v3.0`) | `cohere/rerank-english-v3.0` | +| `HINDSIGHT_API_RERANKER_LITELLM_MODEL` | LiteLLM **proxy** rerank model (use provider prefix, e.g., `cohere/rerank-english-v3.0`) | `cohere/rerank-english-v3.0` | +| `HINDSIGHT_API_RERANKER_LITELLM_SDK_API_KEY` | LiteLLM **SDK** API key for direct reranking (no proxy needed) | - | +| `HINDSIGHT_API_RERANKER_LITELLM_SDK_MODEL` | LiteLLM SDK rerank model (e.g., `deepinfra/Qwen3-reranker-8B`) | `cohere/rerank-english-v3.0` | +| `HINDSIGHT_API_RERANKER_LITELLM_SDK_API_BASE` | Custom API base URL for LiteLLM SDK (optional) | - | | `HINDSIGHT_API_RERANKER_FLASHRANK_MODEL` | FlashRank model for fast CPU-based reranking | `ms-marco-MiniLM-L-12-v2` | | `HINDSIGHT_API_RERANKER_FLASHRANK_CACHE_DIR` | Cache directory for FlashRank models | System default | @@ -451,19 +469,31 @@ export HINDSIGHT_API_RERANKER_COHERE_API_KEY=your-azure-api-key export HINDSIGHT_API_RERANKER_COHERE_MODEL=rerank-english-v3.0 export HINDSIGHT_API_RERANKER_COHERE_BASE_URL=https://your-azure-cohere-endpoint.com -# LiteLLM proxy - unified gateway for multiple reranking providers +# LiteLLM proxy - unified gateway for multiple reranking providers (requires running LiteLLM proxy server) export HINDSIGHT_API_RERANKER_PROVIDER=litellm export HINDSIGHT_API_RERANKER_LITELLM_API_BASE=http://localhost:4000 export HINDSIGHT_API_RERANKER_LITELLM_API_KEY=your-litellm-key # optional export HINDSIGHT_API_RERANKER_LITELLM_MODEL=cohere/rerank-english-v3.0 # or voyage/rerank-2, together_ai/... + +# LiteLLM SDK - direct API access without proxy (recommended for simplicity) +export HINDSIGHT_API_RERANKER_PROVIDER=litellm-sdk +export HINDSIGHT_API_RERANKER_LITELLM_SDK_API_KEY=your-deepinfra-api-key +export HINDSIGHT_API_RERANKER_LITELLM_SDK_MODEL=deepinfra/Qwen3-reranker-8B # or cohere/rerank-english-v3.0, etc. ``` -LiteLLM supports multiple reranking providers via the `/rerank` endpoint: -- Cohere (`cohere/rerank-english-v3.0`, `cohere/rerank-multilingual-v3.0`) -- Together AI (`together_ai/...`) -- Voyage AI (`voyage/rerank-2`) -- Jina AI (`jina_ai/...`) -- AWS Bedrock (`bedrock/...`) +#### LiteLLM Proxy vs SDK + +- **`litellm`**: Requires running a separate LiteLLM proxy server. Good for centralized configuration, rate limiting, and caching. +- **`litellm-sdk`**: Direct API access without proxy. Simpler setup, lower latency, fewer infrastructure components. + +Both support the same providers: +- **Cohere** (`cohere/rerank-english-v3.0`, `cohere/rerank-multilingual-v3.0`) +- **DeepInfra** (`deepinfra/Qwen3-reranker-8B`, `deepinfra/bge-reranker-v2-m3`) +- **Together AI** (`together_ai/Salesforce/Llama-Rank-V1`) +- **HuggingFace** (`huggingface/BAAI/bge-reranker-v2-m3`) +- **Voyage AI** (`voyage/rerank-2`) +- **Jina AI** (`jina_ai/jina-reranker-v2`) +- **AWS Bedrock** (`bedrock/...`) ### Authentication diff --git a/uv.lock b/uv.lock index 7e286d0c..b6c0cdd6 100644 --- a/uv.lock +++ b/uv.lock @@ -1012,6 +1012,58 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/fc/dc/f7dd14213bf511690dccaa5094d436947c253b418c86c86211d1c76e6e44/fastmcp-2.14.3-py3-none-any.whl", hash = "sha256:103c6b4c6e97a9acc251c81d303f110fe4f2bdba31353df515d66272bf1b9414", size = 416220 }, ] +[[package]] +name = "fastuuid" +version = "0.14.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/c3/7d/d9daedf0f2ebcacd20d599928f8913e9d2aea1d56d2d355a93bfa2b611d7/fastuuid-0.14.0.tar.gz", hash = "sha256:178947fc2f995b38497a74172adee64fdeb8b7ec18f2a5934d037641ba265d26", size = 18232 } +wheels = [ + { url = "https://files.pythonhosted.org/packages/98/f3/12481bda4e5b6d3e698fbf525df4443cc7dce746f246b86b6fcb2fba1844/fastuuid-0.14.0-cp311-cp311-macosx_10_12_x86_64.macosx_11_0_arm64.macosx_10_12_universal2.whl", hash = "sha256:73946cb950c8caf65127d4e9a325e2b6be0442a224fd51ba3b6ac44e1912ce34", size = 516386 }, + { url = "https://files.pythonhosted.org/packages/59/19/2fc58a1446e4d72b655648eb0879b04e88ed6fa70d474efcf550f640f6ec/fastuuid-0.14.0-cp311-cp311-macosx_10_12_x86_64.whl", hash = "sha256:12ac85024637586a5b69645e7ed986f7535106ed3013640a393a03e461740cb7", size = 264569 }, + { url = "https://files.pythonhosted.org/packages/78/29/3c74756e5b02c40cfcc8b1d8b5bac4edbd532b55917a6bcc9113550e99d1/fastuuid-0.14.0-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:05a8dde1f395e0c9b4be515b7a521403d1e8349443e7641761af07c7ad1624b1", size = 254366 }, + { url = "https://files.pythonhosted.org/packages/52/96/d761da3fccfa84f0f353ce6e3eb8b7f76b3aa21fd25e1b00a19f9c80a063/fastuuid-0.14.0-cp311-cp311-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:09378a05020e3e4883dfdab438926f31fea15fd17604908f3d39cbeb22a0b4dc", size = 278978 }, + { url = "https://files.pythonhosted.org/packages/fc/c2/f84c90167cc7765cb82b3ff7808057608b21c14a38531845d933a4637307/fastuuid-0.14.0-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:bbb0c4b15d66b435d2538f3827f05e44e2baafcc003dd7d8472dc67807ab8fd8", size = 279692 }, + { url = "https://files.pythonhosted.org/packages/af/7b/4bacd03897b88c12348e7bd77943bac32ccf80ff98100598fcff74f75f2e/fastuuid-0.14.0-cp311-cp311-manylinux_2_5_i686.manylinux1_i686.whl", hash = "sha256:cd5a7f648d4365b41dbf0e38fe8da4884e57bed4e77c83598e076ac0c93995e7", size = 303384 }, + { url = "https://files.pythonhosted.org/packages/c0/a2/584f2c29641df8bd810d00c1f21d408c12e9ad0c0dafdb8b7b29e5ddf787/fastuuid-0.14.0-cp311-cp311-musllinux_1_1_aarch64.whl", hash = "sha256:c0a94245afae4d7af8c43b3159d5e3934c53f47140be0be624b96acd672ceb73", size = 460921 }, + { url = "https://files.pythonhosted.org/packages/24/68/c6b77443bb7764c760e211002c8638c0c7cce11cb584927e723215ba1398/fastuuid-0.14.0-cp311-cp311-musllinux_1_1_i686.whl", hash = "sha256:2b29e23c97e77c3a9514d70ce343571e469098ac7f5a269320a0f0b3e193ab36", size = 480575 }, + { url = "https://files.pythonhosted.org/packages/5a/87/93f553111b33f9bb83145be12868c3c475bf8ea87c107063d01377cc0e8e/fastuuid-0.14.0-cp311-cp311-musllinux_1_1_x86_64.whl", hash = "sha256:1e690d48f923c253f28151b3a6b4e335f2b06bf669c68a02665bc150b7839e94", size = 452317 }, + { url = "https://files.pythonhosted.org/packages/9e/8c/a04d486ca55b5abb7eaa65b39df8d891b7b1635b22db2163734dc273579a/fastuuid-0.14.0-cp311-cp311-win32.whl", hash = "sha256:a6f46790d59ab38c6aa0e35c681c0484b50dc0acf9e2679c005d61e019313c24", size = 154804 }, + { url = "https://files.pythonhosted.org/packages/9c/b2/2d40bf00820de94b9280366a122cbaa60090c8cf59e89ac3938cf5d75895/fastuuid-0.14.0-cp311-cp311-win_amd64.whl", hash = "sha256:e150eab56c95dc9e3fefc234a0eedb342fac433dacc273cd4d150a5b0871e1fa", size = 156099 }, + { url = "https://files.pythonhosted.org/packages/02/a2/e78fcc5df65467f0d207661b7ef86c5b7ac62eea337c0c0fcedbeee6fb13/fastuuid-0.14.0-cp312-cp312-macosx_10_12_x86_64.macosx_11_0_arm64.macosx_10_12_universal2.whl", hash = "sha256:77e94728324b63660ebf8adb27055e92d2e4611645bf12ed9d88d30486471d0a", size = 510164 }, + { url = "https://files.pythonhosted.org/packages/2b/b3/c846f933f22f581f558ee63f81f29fa924acd971ce903dab1a9b6701816e/fastuuid-0.14.0-cp312-cp312-macosx_10_12_x86_64.whl", hash = "sha256:caa1f14d2102cb8d353096bc6ef6c13b2c81f347e6ab9d6fbd48b9dea41c153d", size = 261837 }, + { url = "https://files.pythonhosted.org/packages/54/ea/682551030f8c4fa9a769d9825570ad28c0c71e30cf34020b85c1f7ee7382/fastuuid-0.14.0-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:d23ef06f9e67163be38cece704170486715b177f6baae338110983f99a72c070", size = 251370 }, + { url = "https://files.pythonhosted.org/packages/14/dd/5927f0a523d8e6a76b70968e6004966ee7df30322f5fc9b6cdfb0276646a/fastuuid-0.14.0-cp312-cp312-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:0c9ec605ace243b6dbe3bd27ebdd5d33b00d8d1d3f580b39fdd15cd96fd71796", size = 277766 }, + { url = "https://files.pythonhosted.org/packages/16/6e/c0fb547eef61293153348f12e0f75a06abb322664b34a1573a7760501336/fastuuid-0.14.0-cp312-cp312-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:808527f2407f58a76c916d6aa15d58692a4a019fdf8d4c32ac7ff303b7d7af09", size = 278105 }, + { url = "https://files.pythonhosted.org/packages/2d/b1/b9c75e03b768f61cf2e84ee193dc18601aeaf89a4684b20f2f0e9f52b62c/fastuuid-0.14.0-cp312-cp312-manylinux_2_5_i686.manylinux1_i686.whl", hash = "sha256:2fb3c0d7fef6674bbeacdd6dbd386924a7b60b26de849266d1ff6602937675c8", size = 301564 }, + { url = "https://files.pythonhosted.org/packages/fc/fa/f7395fdac07c7a54f18f801744573707321ca0cee082e638e36452355a9d/fastuuid-0.14.0-cp312-cp312-musllinux_1_1_aarch64.whl", hash = "sha256:ab3f5d36e4393e628a4df337c2c039069344db5f4b9d2a3c9cea48284f1dd741", size = 459659 }, + { url = "https://files.pythonhosted.org/packages/66/49/c9fd06a4a0b1f0f048aacb6599e7d96e5d6bc6fa680ed0d46bf111929d1b/fastuuid-0.14.0-cp312-cp312-musllinux_1_1_i686.whl", hash = "sha256:b9a0ca4f03b7e0b01425281ffd44e99d360e15c895f1907ca105854ed85e2057", size = 478430 }, + { url = "https://files.pythonhosted.org/packages/be/9c/909e8c95b494e8e140e8be6165d5fc3f61fdc46198c1554df7b3e1764471/fastuuid-0.14.0-cp312-cp312-musllinux_1_1_x86_64.whl", hash = "sha256:3acdf655684cc09e60fb7e4cf524e8f42ea760031945aa8086c7eae2eeeabeb8", size = 450894 }, + { url = "https://files.pythonhosted.org/packages/90/eb/d29d17521976e673c55ef7f210d4cdd72091a9ec6755d0fd4710d9b3c871/fastuuid-0.14.0-cp312-cp312-win32.whl", hash = "sha256:9579618be6280700ae36ac42c3efd157049fe4dd40ca49b021280481c78c3176", size = 154374 }, + { url = "https://files.pythonhosted.org/packages/cc/fc/f5c799a6ea6d877faec0472d0b27c079b47c86b1cdc577720a5386483b36/fastuuid-0.14.0-cp312-cp312-win_amd64.whl", hash = "sha256:d9e4332dc4ba054434a9594cbfaf7823b57993d7d8e7267831c3e059857cf397", size = 156550 }, + { url = "https://files.pythonhosted.org/packages/a5/83/ae12dd39b9a39b55d7f90abb8971f1a5f3c321fd72d5aa83f90dc67fe9ed/fastuuid-0.14.0-cp313-cp313-macosx_10_12_x86_64.macosx_11_0_arm64.macosx_10_12_universal2.whl", hash = "sha256:77a09cb7427e7af74c594e409f7731a0cf887221de2f698e1ca0ebf0f3139021", size = 510720 }, + { url = "https://files.pythonhosted.org/packages/53/b0/a4b03ff5d00f563cc7546b933c28cb3f2a07344b2aec5834e874f7d44143/fastuuid-0.14.0-cp313-cp313-macosx_10_12_x86_64.whl", hash = "sha256:9bd57289daf7b153bfa3e8013446aa144ce5e8c825e9e366d455155ede5ea2dc", size = 262024 }, + { url = "https://files.pythonhosted.org/packages/9c/6d/64aee0a0f6a58eeabadd582e55d0d7d70258ffdd01d093b30c53d668303b/fastuuid-0.14.0-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:ac60fc860cdf3c3f327374db87ab8e064c86566ca8c49d2e30df15eda1b0c2d5", size = 251679 }, + { url = "https://files.pythonhosted.org/packages/60/f5/a7e9cda8369e4f7919d36552db9b2ae21db7915083bc6336f1b0082c8b2e/fastuuid-0.14.0-cp313-cp313-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:ab32f74bd56565b186f036e33129da77db8be09178cd2f5206a5d4035fb2a23f", size = 277862 }, + { url = "https://files.pythonhosted.org/packages/f0/d3/8ce11827c783affffd5bd4d6378b28eb6cc6d2ddf41474006b8d62e7448e/fastuuid-0.14.0-cp313-cp313-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:33e678459cf4addaedd9936bbb038e35b3f6b2061330fd8f2f6a1d80414c0f87", size = 278278 }, + { url = "https://files.pythonhosted.org/packages/a2/51/680fb6352d0bbade04036da46264a8001f74b7484e2fd1f4da9e3db1c666/fastuuid-0.14.0-cp313-cp313-manylinux_2_5_i686.manylinux1_i686.whl", hash = "sha256:1e3cc56742f76cd25ecb98e4b82a25f978ccffba02e4bdce8aba857b6d85d87b", size = 301788 }, + { url = "https://files.pythonhosted.org/packages/fa/7c/2014b5785bd8ebdab04ec857635ebd84d5ee4950186a577db9eff0fb8ff6/fastuuid-0.14.0-cp313-cp313-musllinux_1_1_aarch64.whl", hash = "sha256:cb9a030f609194b679e1660f7e32733b7a0f332d519c5d5a6a0a580991290022", size = 459819 }, + { url = "https://files.pythonhosted.org/packages/01/d2/524d4ceeba9160e7a9bc2ea3e8f4ccf1ad78f3bde34090ca0c51f09a5e91/fastuuid-0.14.0-cp313-cp313-musllinux_1_1_i686.whl", hash = "sha256:09098762aad4f8da3a888eb9ae01c84430c907a297b97166b8abc07b640f2995", size = 478546 }, + { url = "https://files.pythonhosted.org/packages/bc/17/354d04951ce114bf4afc78e27a18cfbd6ee319ab1829c2d5fb5e94063ac6/fastuuid-0.14.0-cp313-cp313-musllinux_1_1_x86_64.whl", hash = "sha256:1383fff584fa249b16329a059c68ad45d030d5a4b70fb7c73a08d98fd53bcdab", size = 450921 }, + { url = "https://files.pythonhosted.org/packages/fb/be/d7be8670151d16d88f15bb121c5b66cdb5ea6a0c2a362d0dcf30276ade53/fastuuid-0.14.0-cp313-cp313-win32.whl", hash = "sha256:a0809f8cc5731c066c909047f9a314d5f536c871a7a22e815cc4967c110ac9ad", size = 154559 }, + { url = "https://files.pythonhosted.org/packages/22/1d/5573ef3624ceb7abf4a46073d3554e37191c868abc3aecd5289a72f9810a/fastuuid-0.14.0-cp313-cp313-win_amd64.whl", hash = "sha256:0df14e92e7ad3276327631c9e7cec09e32572ce82089c55cb1bb8df71cf394ed", size = 156539 }, + { url = "https://files.pythonhosted.org/packages/16/c9/8c7660d1fe3862e3f8acabd9be7fc9ad71eb270f1c65cce9a2b7a31329ab/fastuuid-0.14.0-cp314-cp314-macosx_10_12_x86_64.macosx_11_0_arm64.macosx_10_12_universal2.whl", hash = "sha256:b852a870a61cfc26c884af205d502881a2e59cc07076b60ab4a951cc0c94d1ad", size = 510600 }, + { url = "https://files.pythonhosted.org/packages/4c/f4/a989c82f9a90d0ad995aa957b3e572ebef163c5299823b4027986f133dfb/fastuuid-0.14.0-cp314-cp314-macosx_10_12_x86_64.whl", hash = "sha256:c7502d6f54cd08024c3ea9b3514e2d6f190feb2f46e6dbcd3747882264bb5f7b", size = 262069 }, + { url = "https://files.pythonhosted.org/packages/da/6c/a1a24f73574ac995482b1326cf7ab41301af0fabaa3e37eeb6b3df00e6e2/fastuuid-0.14.0-cp314-cp314-macosx_11_0_arm64.whl", hash = "sha256:1ca61b592120cf314cfd66e662a5b54a578c5a15b26305e1b8b618a6f22df714", size = 251543 }, + { url = "https://files.pythonhosted.org/packages/1a/20/2a9b59185ba7a6c7b37808431477c2d739fcbdabbf63e00243e37bd6bf49/fastuuid-0.14.0-cp314-cp314-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:aa75b6657ec129d0abded3bec745e6f7ab642e6dba3a5272a68247e85f5f316f", size = 277798 }, + { url = "https://files.pythonhosted.org/packages/ef/33/4105ca574f6ded0af6a797d39add041bcfb468a1255fbbe82fcb6f592da2/fastuuid-0.14.0-cp314-cp314-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:a8a0dfea3972200f72d4c7df02c8ac70bad1bb4c58d7e0ec1e6f341679073a7f", size = 278283 }, + { url = "https://files.pythonhosted.org/packages/fe/8c/fca59f8e21c4deb013f574eae05723737ddb1d2937ce87cb2a5d20992dc3/fastuuid-0.14.0-cp314-cp314-manylinux_2_5_i686.manylinux1_i686.whl", hash = "sha256:1bf539a7a95f35b419f9ad105d5a8a35036df35fdafae48fb2fd2e5f318f0d75", size = 301627 }, + { url = "https://files.pythonhosted.org/packages/cb/e2/f78c271b909c034d429218f2798ca4e89eeda7983f4257d7865976ddbb6c/fastuuid-0.14.0-cp314-cp314-musllinux_1_1_aarch64.whl", hash = "sha256:9a133bf9cc78fdbd1179cb58a59ad0100aa32d8675508150f3658814aeefeaa4", size = 459778 }, + { url = "https://files.pythonhosted.org/packages/1e/f0/5ff209d865897667a2ff3e7a572267a9ced8f7313919f6d6043aed8b1caa/fastuuid-0.14.0-cp314-cp314-musllinux_1_1_i686.whl", hash = "sha256:f54d5b36c56a2d5e1a31e73b950b28a0d83eb0c37b91d10408875a5a29494bad", size = 478605 }, + { url = "https://files.pythonhosted.org/packages/e0/c8/2ce1c78f983a2c4987ea865d9516dbdfb141a120fd3abb977ae6f02ba7ca/fastuuid-0.14.0-cp314-cp314-musllinux_1_1_x86_64.whl", hash = "sha256:ec27778c6ca3393ef662e2762dba8af13f4ec1aaa32d08d77f71f2a70ae9feb8", size = 450837 }, + { url = "https://files.pythonhosted.org/packages/df/60/dad662ec9a33b4a5fe44f60699258da64172c39bd041da2994422cdc40fe/fastuuid-0.14.0-cp314-cp314-win32.whl", hash = "sha256:e23fc6a83f112de4be0cc1990e5b127c27663ae43f866353166f87df58e73d06", size = 154532 }, + { url = "https://files.pythonhosted.org/packages/1f/f6/da4db31001e854025ffd26bc9ba0740a9cbba2c3259695f7c5834908b336/fastuuid-0.14.0-cp314-cp314-win_amd64.whl", hash = "sha256:df61342889d0f5e7a32f7284e55ef95103f2110fee433c2ae7c2c0956d76ac8a", size = 156457 }, +] + [[package]] name = "filelock" version = "3.20.3" @@ -1370,6 +1422,7 @@ dependencies = [ { name = "httpx" }, { name = "langchain-core" }, { name = "langchain-text-splitters" }, + { name = "litellm" }, { name = "openai" }, { name = "opentelemetry-api" }, { name = "opentelemetry-exporter-otlp-proto-http" }, @@ -1440,6 +1493,7 @@ requires-dist = [ { name = "httpx", specifier = ">=0.27.0" }, { name = "langchain-core", specifier = ">=1.2.5" }, { name = "langchain-text-splitters", specifier = ">=0.3.0" }, + { name = "litellm", specifier = ">=1.0.0" }, { name = "openai", specifier = ">=1.0.0" }, { name = "opentelemetry-api", specifier = ">=1.20.0" }, { name = "opentelemetry-exporter-otlp-proto-http", specifier = ">=1.20.0" }, @@ -2018,6 +2072,29 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/0f/17/4280bc381b40a642ea5efe1bab0237f03507a9d4281484c5baa1db82055a/langsmith-0.4.42-py3-none-any.whl", hash = "sha256:015b0a0c17eb1a61293e8cbb7d41778a4b37caddd267d54274ba94e4721b301b", size = 401937 }, ] +[[package]] +name = "litellm" +version = "1.81.10" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "aiohttp" }, + { name = "click" }, + { name = "fastuuid" }, + { name = "httpx" }, + { name = "importlib-metadata" }, + { name = "jinja2" }, + { name = "jsonschema" }, + { name = "openai" }, + { name = "pydantic" }, + { name = "python-dotenv" }, + { name = "tiktoken" }, + { name = "tokenizers" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/40/fc/78887158b4057835ba2c647a1bd4da650fd79142f8412c6d0bbe6d8c6081/litellm-1.81.10.tar.gz", hash = "sha256:8d769a7200888e1295592af5ce5cb0ff035832250bd0102a4ca50acf5820ca50", size = 16297572 } +wheels = [ + { url = "https://files.pythonhosted.org/packages/b1/bb/3f3cc3d79657bc9daaa1319ec3a9d75e4889fc88d07e327f0ac02cd2ac7d/litellm-1.81.10-py3-none-any.whl", hash = "sha256:9efa1cbe61ac051f6500c267b173d988ff2d511c2eecf1c8f2ee546c0870747c", size = 14457931 }, +] + [[package]] name = "lupa" version = "2.6" @@ -2624,7 +2701,7 @@ wheels = [ [[package]] name = "openai" -version = "2.7.2" +version = "2.20.0" source = { registry = "https://pypi.org/simple" } dependencies = [ { name = "anyio" }, @@ -2636,9 +2713,9 @@ dependencies = [ { name = "tqdm" }, { name = "typing-extensions" }, ] -sdist = { url = "https://files.pythonhosted.org/packages/71/e3/cec27fa28ef36c4ccea71e9e8c20be9b8539618732989a82027575aab9d4/openai-2.7.2.tar.gz", hash = "sha256:082ef61163074d8efad0035dd08934cf5e3afd37254f70fc9165dd6a8c67dcbd", size = 595732 } +sdist = { url = "https://files.pythonhosted.org/packages/6e/5a/f495777c02625bfa18212b6e3b73f1893094f2bf660976eb4bc6f43a1ca2/openai-2.20.0.tar.gz", hash = "sha256:2654a689208cd0bf1098bb9462e8d722af5cbe961e6bba54e6f19fb843d88db1", size = 642355 } wheels = [ - { url = "https://files.pythonhosted.org/packages/25/66/22cfe4b695b5fd042931b32c67d685e867bfd169ebf46036b95b57314c33/openai-2.7.2-py3-none-any.whl", hash = "sha256:116f522f4427f8a0a59b51655a356da85ce092f3ed6abeca65f03c8be6e073d9", size = 1008375 }, + { url = "https://files.pythonhosted.org/packages/b5/a0/cf4297aa51bbc21e83ef0ac018947fa06aea8f2364aad7c96cbf148590e6/openai-2.20.0-py3-none-any.whl", hash = "sha256:38d989c4b1075cd1f76abc68364059d822327cf1a932531d429795f4fc18be99", size = 1098479 }, ] [[package]]