feat: add built-in llama.cpp LLM provider for local inference (#933)
* feat: add built-in llama.cpp LLM provider for fully local inference Add `llamacpp` as a new LLM provider that manages a llama-cpp-python server subprocess. Auto-downloads Gemma 4 E2B Q4_K_M (~3.5 GB) on first use and runs inference locally via Metal/CUDA with no external services needed. - New provider: `HINDSIGHT_API_LLM_PROVIDER=llamacpp` - Singleton server shared across retain/reflect/consolidation - Configurable: model path, GPU layers, context size, grammar enforcement - User-extensible via `HINDSIGHT_API_LLAMACPP_EXTRA_ARGS` - Flash attention + prompt caching enabled by default - LLM provider cleanup on shutdown (stops subprocess) - hindsight-embed: `--ui` flag on `daemon start`, removed FORCE_CPU on macOS - Docs: configuration.md, models.mdx, providers grid updated * chore: regenerate docs skill and update lockfile for local-llm dep
This commit is contained in:
parent
3c633e5e16
commit
f74b577e02
17 changed files with 3821 additions and 3155 deletions
|
|
@ -344,6 +344,14 @@ ENV_WEBHOOK_SECRET = "HINDSIGHT_API_WEBHOOK_SECRET"
|
||||||
ENV_WEBHOOK_EVENT_TYPES = "HINDSIGHT_API_WEBHOOK_EVENT_TYPES"
|
ENV_WEBHOOK_EVENT_TYPES = "HINDSIGHT_API_WEBHOOK_EVENT_TYPES"
|
||||||
ENV_WEBHOOK_DELIVERY_POLL_INTERVAL_SECONDS = "HINDSIGHT_API_WEBHOOK_DELIVERY_POLL_INTERVAL_SECONDS"
|
ENV_WEBHOOK_DELIVERY_POLL_INTERVAL_SECONDS = "HINDSIGHT_API_WEBHOOK_DELIVERY_POLL_INTERVAL_SECONDS"
|
||||||
|
|
||||||
|
# Built-in llama.cpp configuration (for provider=llamacpp)
|
||||||
|
ENV_LLAMACPP_MODEL_PATH = "HINDSIGHT_API_LLAMACPP_MODEL_PATH"
|
||||||
|
ENV_LLAMACPP_GPU_LAYERS = "HINDSIGHT_API_LLAMACPP_GPU_LAYERS"
|
||||||
|
ENV_LLAMACPP_CONTEXT_SIZE = "HINDSIGHT_API_LLAMACPP_CONTEXT_SIZE"
|
||||||
|
ENV_LLAMACPP_CHAT_FORMAT = "HINDSIGHT_API_LLAMACPP_CHAT_FORMAT"
|
||||||
|
ENV_LLAMACPP_NO_GRAMMAR = "HINDSIGHT_API_LLAMACPP_NO_GRAMMAR"
|
||||||
|
ENV_LLAMACPP_EXTRA_ARGS = "HINDSIGHT_API_LLAMACPP_EXTRA_ARGS"
|
||||||
|
|
||||||
# Optimization flags
|
# Optimization flags
|
||||||
ENV_SKIP_LLM_VERIFICATION = "HINDSIGHT_API_SKIP_LLM_VERIFICATION"
|
ENV_SKIP_LLM_VERIFICATION = "HINDSIGHT_API_SKIP_LLM_VERIFICATION"
|
||||||
ENV_LAZY_RERANKER = "HINDSIGHT_API_LAZY_RERANKER"
|
ENV_LAZY_RERANKER = "HINDSIGHT_API_LAZY_RERANKER"
|
||||||
|
|
@ -397,6 +405,7 @@ PROVIDER_DEFAULT_MODELS = {
|
||||||
"groq": "openai/gpt-oss-120b",
|
"groq": "openai/gpt-oss-120b",
|
||||||
"minimax": "MiniMax-M2.7",
|
"minimax": "MiniMax-M2.7",
|
||||||
"ollama": "gemma3:12b",
|
"ollama": "gemma3:12b",
|
||||||
|
"llamacpp": "gemma-4-e2b-it",
|
||||||
"lmstudio": "local-model",
|
"lmstudio": "local-model",
|
||||||
"vertexai": "google/gemini-2.5-flash-lite",
|
"vertexai": "google/gemini-2.5-flash-lite",
|
||||||
"openai-codex": "gpt-5.2-codex",
|
"openai-codex": "gpt-5.2-codex",
|
||||||
|
|
@ -409,6 +418,13 @@ PROVIDER_DEFAULT_MODELS = {
|
||||||
"openrouter": "qwen/qwen3.5-9b",
|
"openrouter": "qwen/qwen3.5-9b",
|
||||||
}
|
}
|
||||||
DEFAULT_LLM_MODEL = "gpt-4o-mini" # Fallback if provider not in table
|
DEFAULT_LLM_MODEL = "gpt-4o-mini" # Fallback if provider not in table
|
||||||
|
# Built-in llama.cpp defaults
|
||||||
|
DEFAULT_LLAMACPP_GPU_LAYERS = -1 # -1 = offload all layers to GPU (Metal/CUDA)
|
||||||
|
DEFAULT_LLAMACPP_CONTEXT_SIZE = 8192
|
||||||
|
DEFAULT_LLAMACPP_CHAT_FORMAT = None # None = auto-detect from GGUF metadata
|
||||||
|
DEFAULT_LLAMACPP_NO_GRAMMAR = False # True = disable JSON grammar enforcement (faster but less reliable)
|
||||||
|
DEFAULT_LLAMACPP_EXTRA_ARGS = None # Space-separated extra CLI args for llama.cpp server
|
||||||
|
|
||||||
DEFAULT_LLM_MAX_CONCURRENT = 32
|
DEFAULT_LLM_MAX_CONCURRENT = 32
|
||||||
DEFAULT_LLM_MAX_RETRIES = 10 # Max retry attempts for LLM API calls
|
DEFAULT_LLM_MAX_RETRIES = 10 # Max retry attempts for LLM API calls
|
||||||
DEFAULT_LLM_INITIAL_BACKOFF = 1.0 # Initial backoff in seconds for retry exponential backoff
|
DEFAULT_LLM_INITIAL_BACKOFF = 1.0 # Initial backoff in seconds for retry exponential backoff
|
||||||
|
|
@ -695,6 +711,14 @@ class HindsightConfig:
|
||||||
# Gemini safety settings (None = use Gemini defaults; list of dicts with category/threshold)
|
# Gemini safety settings (None = use Gemini defaults; list of dicts with category/threshold)
|
||||||
llm_gemini_safety_settings: list | None
|
llm_gemini_safety_settings: list | None
|
||||||
|
|
||||||
|
# Built-in llama.cpp configuration (for provider=llamacpp)
|
||||||
|
llamacpp_model_path: str | None # Path to GGUF file (None = auto-download default)
|
||||||
|
llamacpp_gpu_layers: int # -1 = all layers on GPU, 0 = CPU only
|
||||||
|
llamacpp_context_size: int # Context window size
|
||||||
|
llamacpp_chat_format: str | None # Chat template format (None = auto-detect from GGUF)
|
||||||
|
llamacpp_no_grammar: bool # Disable JSON grammar enforcement (faster, less reliable)
|
||||||
|
llamacpp_extra_args: str | None # Space-separated extra CLI args for llama.cpp server
|
||||||
|
|
||||||
# Per-operation LLM configuration (None = use default LLM config)
|
# Per-operation LLM configuration (None = use default LLM config)
|
||||||
retain_llm_provider: str | None
|
retain_llm_provider: str | None
|
||||||
retain_llm_api_key: str | None
|
retain_llm_api_key: str | None
|
||||||
|
|
@ -1117,6 +1141,14 @@ class HindsightConfig:
|
||||||
or DEFAULT_LLM_VERTEXAI_SERVICE_ACCOUNT_KEY,
|
or DEFAULT_LLM_VERTEXAI_SERVICE_ACCOUNT_KEY,
|
||||||
# Gemini safety settings (JSON-encoded list of {category, threshold} dicts)
|
# Gemini safety settings (JSON-encoded list of {category, threshold} dicts)
|
||||||
llm_gemini_safety_settings=json.loads(os.getenv(ENV_LLM_GEMINI_SAFETY_SETTINGS, "null")),
|
llm_gemini_safety_settings=json.loads(os.getenv(ENV_LLM_GEMINI_SAFETY_SETTINGS, "null")),
|
||||||
|
# Built-in llama.cpp configuration
|
||||||
|
llamacpp_model_path=os.getenv(ENV_LLAMACPP_MODEL_PATH) or None,
|
||||||
|
llamacpp_gpu_layers=int(os.getenv(ENV_LLAMACPP_GPU_LAYERS, str(DEFAULT_LLAMACPP_GPU_LAYERS))),
|
||||||
|
llamacpp_context_size=int(os.getenv(ENV_LLAMACPP_CONTEXT_SIZE, str(DEFAULT_LLAMACPP_CONTEXT_SIZE))),
|
||||||
|
llamacpp_chat_format=os.getenv(ENV_LLAMACPP_CHAT_FORMAT) or DEFAULT_LLAMACPP_CHAT_FORMAT,
|
||||||
|
llamacpp_no_grammar=os.getenv(ENV_LLAMACPP_NO_GRAMMAR, str(DEFAULT_LLAMACPP_NO_GRAMMAR)).lower()
|
||||||
|
in ("true", "1"),
|
||||||
|
llamacpp_extra_args=os.getenv(ENV_LLAMACPP_EXTRA_ARGS) or DEFAULT_LLAMACPP_EXTRA_ARGS,
|
||||||
# Per-operation LLM config (None = use default)
|
# Per-operation LLM config (None = use default)
|
||||||
retain_llm_provider=os.getenv(ENV_RETAIN_LLM_PROVIDER) or None,
|
retain_llm_provider=os.getenv(ENV_RETAIN_LLM_PROVIDER) or None,
|
||||||
retain_llm_api_key=os.getenv(ENV_RETAIN_LLM_API_KEY) or None,
|
retain_llm_api_key=os.getenv(ENV_RETAIN_LLM_API_KEY) or None,
|
||||||
|
|
|
||||||
|
|
@ -122,6 +122,7 @@ _PROVIDERS_WITHOUT_API_KEY = frozenset(
|
||||||
{
|
{
|
||||||
"ollama",
|
"ollama",
|
||||||
"lmstudio",
|
"lmstudio",
|
||||||
|
"llamacpp",
|
||||||
"openai-codex",
|
"openai-codex",
|
||||||
"claude-code",
|
"claude-code",
|
||||||
"mock",
|
"mock",
|
||||||
|
|
@ -178,6 +179,7 @@ def create_llm_provider(
|
||||||
CodexLLM,
|
CodexLLM,
|
||||||
GeminiLLM,
|
GeminiLLM,
|
||||||
LiteLLMLLM,
|
LiteLLMLLM,
|
||||||
|
LlamaCppLLM,
|
||||||
MockLLM,
|
MockLLM,
|
||||||
NoneLLM,
|
NoneLLM,
|
||||||
OpenAICompatibleLLM,
|
OpenAICompatibleLLM,
|
||||||
|
|
@ -263,6 +265,24 @@ def create_llm_provider(
|
||||||
reasoning_effort=reasoning_effort,
|
reasoning_effort=reasoning_effort,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
elif provider_lower == "llamacpp":
|
||||||
|
from ..config import get_config
|
||||||
|
|
||||||
|
config = get_config()
|
||||||
|
return LlamaCppLLM(
|
||||||
|
provider=provider,
|
||||||
|
api_key=api_key,
|
||||||
|
base_url=base_url,
|
||||||
|
model=model,
|
||||||
|
reasoning_effort=reasoning_effort,
|
||||||
|
model_path=config.llamacpp_model_path,
|
||||||
|
gpu_layers=config.llamacpp_gpu_layers,
|
||||||
|
context_size=config.llamacpp_context_size,
|
||||||
|
chat_format=config.llamacpp_chat_format,
|
||||||
|
no_grammar=config.llamacpp_no_grammar,
|
||||||
|
extra_args=config.llamacpp_extra_args,
|
||||||
|
)
|
||||||
|
|
||||||
elif provider_lower in ("openai", "groq", "ollama", "lmstudio", "minimax", "volcano", "openrouter"):
|
elif provider_lower in ("openai", "groq", "ollama", "lmstudio", "minimax", "volcano", "openrouter"):
|
||||||
return OpenAICompatibleLLM(
|
return OpenAICompatibleLLM(
|
||||||
provider=provider,
|
provider=provider,
|
||||||
|
|
@ -333,6 +353,7 @@ class LLMProvider:
|
||||||
"gemini",
|
"gemini",
|
||||||
"anthropic",
|
"anthropic",
|
||||||
"lmstudio",
|
"lmstudio",
|
||||||
|
"llamacpp",
|
||||||
"vertexai",
|
"vertexai",
|
||||||
"openai-codex",
|
"openai-codex",
|
||||||
"claude-code",
|
"claude-code",
|
||||||
|
|
@ -714,8 +735,9 @@ class LLMProvider:
|
||||||
return ConfiguredLLMProvider(self, config.llm_gemini_safety_settings)
|
return ConfiguredLLMProvider(self, config.llm_gemini_safety_settings)
|
||||||
|
|
||||||
async def cleanup(self) -> None:
|
async def cleanup(self) -> None:
|
||||||
"""Clean up resources."""
|
"""Clean up resources (e.g. stop llamacpp subprocess)."""
|
||||||
pass
|
if self._provider_impl:
|
||||||
|
await self._provider_impl.cleanup()
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
def from_env(cls) -> "LLMProvider":
|
def from_env(cls) -> "LLMProvider":
|
||||||
|
|
|
||||||
|
|
@ -1923,6 +1923,18 @@ class MemoryEngine(MemoryEngineInterface):
|
||||||
|
|
||||||
self._initialized = False
|
self._initialized = False
|
||||||
|
|
||||||
|
# Clean up LLM providers (e.g. stop llamacpp subprocess)
|
||||||
|
for llm_config in (
|
||||||
|
self._llm_config,
|
||||||
|
self._retain_llm_config,
|
||||||
|
self._reflect_llm_config,
|
||||||
|
self._consolidation_llm_config,
|
||||||
|
):
|
||||||
|
try:
|
||||||
|
await llm_config.cleanup()
|
||||||
|
except Exception as e:
|
||||||
|
logger.warning(f"Error cleaning up LLM provider: {e}")
|
||||||
|
|
||||||
# Stop pg0 if we started it
|
# Stop pg0 if we started it
|
||||||
if self._pg0 is not None:
|
if self._pg0 is not None:
|
||||||
logger.info("Stopping pg0...")
|
logger.info("Stopping pg0...")
|
||||||
|
|
|
||||||
|
|
@ -9,6 +9,7 @@ from .claude_code_llm import ClaudeCodeLLM
|
||||||
from .codex_llm import CodexLLM
|
from .codex_llm import CodexLLM
|
||||||
from .gemini_llm import GeminiLLM
|
from .gemini_llm import GeminiLLM
|
||||||
from .litellm_llm import LiteLLMLLM
|
from .litellm_llm import LiteLLMLLM
|
||||||
|
from .llamacpp_llm import LlamaCppLLM
|
||||||
from .mock_llm import MockLLM
|
from .mock_llm import MockLLM
|
||||||
from .none_llm import NoneLLM
|
from .none_llm import NoneLLM
|
||||||
from .openai_compatible_llm import OpenAICompatibleLLM
|
from .openai_compatible_llm import OpenAICompatibleLLM
|
||||||
|
|
@ -18,6 +19,7 @@ __all__ = [
|
||||||
"ClaudeCodeLLM",
|
"ClaudeCodeLLM",
|
||||||
"CodexLLM",
|
"CodexLLM",
|
||||||
"GeminiLLM",
|
"GeminiLLM",
|
||||||
|
"LlamaCppLLM",
|
||||||
"LiteLLMLLM",
|
"LiteLLMLLM",
|
||||||
"MockLLM",
|
"MockLLM",
|
||||||
"NoneLLM",
|
"NoneLLM",
|
||||||
|
|
|
||||||
|
|
@ -0,0 +1,428 @@
|
||||||
|
"""
|
||||||
|
Built-in llama.cpp LLM provider for fully offline operation.
|
||||||
|
|
||||||
|
Manages a llama-cpp-python server as a subprocess, downloads GGUF models
|
||||||
|
from HuggingFace on first use, and delegates inference to the OpenAI-compatible API.
|
||||||
|
|
||||||
|
Usage:
|
||||||
|
HINDSIGHT_API_LLM_PROVIDER=llamacpp
|
||||||
|
HINDSIGHT_API_LLAMACPP_MODEL_PATH=~/.hindsight/models/gemma-4-E2B-it-Q4_K_M.gguf
|
||||||
|
HINDSIGHT_API_LLAMACPP_GPU_LAYERS=-1 # -1 = all layers on GPU
|
||||||
|
HINDSIGHT_API_LLAMACPP_CONTEXT_SIZE=8192
|
||||||
|
"""
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
import logging
|
||||||
|
import os
|
||||||
|
import signal
|
||||||
|
import socket
|
||||||
|
import subprocess
|
||||||
|
import sys
|
||||||
|
import time
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
from hindsight_api.engine.llm_interface import LLMInterface
|
||||||
|
from hindsight_api.engine.response_models import LLMToolCallResult
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
# Default GGUF model for offline mode
|
||||||
|
DEFAULT_LLAMACPP_HF_REPO = "bartowski/google_gemma-4-E2B-it-GGUF"
|
||||||
|
DEFAULT_LLAMACPP_HF_FILENAME = "google_gemma-4-E2B-it-Q4_K_M.gguf"
|
||||||
|
DEFAULT_LLAMACPP_MODEL_ALIAS = "gemma-4-e2b-it"
|
||||||
|
|
||||||
|
MODELS_DIR = Path.home() / ".hindsight" / "models"
|
||||||
|
|
||||||
|
# Singleton server instance — shared across all LlamaCppLLM instances
|
||||||
|
# (retain, reflect, consolidation each create their own LLMProvider,
|
||||||
|
# but they should all share one llama.cpp server process)
|
||||||
|
_shared_server: "LlamaCppServer | None" = None
|
||||||
|
_shared_server_lock = asyncio.Lock()
|
||||||
|
|
||||||
|
|
||||||
|
def _find_free_port() -> int:
|
||||||
|
"""Find a free TCP port on localhost."""
|
||||||
|
with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as s:
|
||||||
|
s.bind(("127.0.0.1", 0))
|
||||||
|
return s.getsockname()[1]
|
||||||
|
|
||||||
|
|
||||||
|
def _download_default_model() -> Path:
|
||||||
|
"""Download the default GGUF model from HuggingFace if not already cached.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Path to the downloaded GGUF file.
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
from huggingface_hub import hf_hub_download
|
||||||
|
except ImportError:
|
||||||
|
raise ImportError(
|
||||||
|
"huggingface-hub is required for automatic model download. "
|
||||||
|
"Install with: pip install 'hindsight-api-slim[local-llm]'"
|
||||||
|
)
|
||||||
|
|
||||||
|
MODELS_DIR.mkdir(parents=True, exist_ok=True)
|
||||||
|
target = MODELS_DIR / DEFAULT_LLAMACPP_HF_FILENAME
|
||||||
|
|
||||||
|
if target.exists():
|
||||||
|
logger.info(f"Using cached model: {target}")
|
||||||
|
return target
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
f"Downloading {DEFAULT_LLAMACPP_HF_FILENAME} from {DEFAULT_LLAMACPP_HF_REPO} (~3.5 GB, first run only)..."
|
||||||
|
)
|
||||||
|
|
||||||
|
downloaded = hf_hub_download(
|
||||||
|
repo_id=DEFAULT_LLAMACPP_HF_REPO,
|
||||||
|
filename=DEFAULT_LLAMACPP_HF_FILENAME,
|
||||||
|
local_dir=str(MODELS_DIR),
|
||||||
|
)
|
||||||
|
|
||||||
|
logger.info(f"Model downloaded: {downloaded}")
|
||||||
|
return Path(downloaded)
|
||||||
|
|
||||||
|
|
||||||
|
def _resolve_model_path(model_path: str | None) -> Path:
|
||||||
|
"""Resolve the model path, downloading the default if needed.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
model_path: Explicit path to a GGUF file, or None to use the default.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Resolved Path to the GGUF file.
|
||||||
|
"""
|
||||||
|
if model_path:
|
||||||
|
p = Path(model_path).expanduser()
|
||||||
|
if not p.exists():
|
||||||
|
raise FileNotFoundError(
|
||||||
|
f"GGUF model not found: {p}\n"
|
||||||
|
f"Set HINDSIGHT_API_LLAMACPP_MODEL_PATH to a valid .gguf file, "
|
||||||
|
f"or remove the setting to auto-download the default model."
|
||||||
|
)
|
||||||
|
return p
|
||||||
|
|
||||||
|
return _download_default_model()
|
||||||
|
|
||||||
|
|
||||||
|
class LlamaCppServer:
|
||||||
|
"""Manages a llama-cpp-python OpenAI-compatible server as a subprocess."""
|
||||||
|
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
model_path: Path,
|
||||||
|
port: int,
|
||||||
|
gpu_layers: int = -1,
|
||||||
|
context_size: int = 8192,
|
||||||
|
chat_format: str | None = None,
|
||||||
|
extra_args: str | None = None,
|
||||||
|
):
|
||||||
|
self.model_path = model_path
|
||||||
|
self.port = port
|
||||||
|
self.gpu_layers = gpu_layers
|
||||||
|
self.context_size = context_size
|
||||||
|
self.chat_format = chat_format
|
||||||
|
self.extra_args = extra_args
|
||||||
|
self._process: subprocess.Popen | None = None
|
||||||
|
|
||||||
|
@property
|
||||||
|
def base_url(self) -> str:
|
||||||
|
return f"http://127.0.0.1:{self.port}/v1"
|
||||||
|
|
||||||
|
async def start(self) -> None:
|
||||||
|
"""Start the llama.cpp server subprocess."""
|
||||||
|
cmd = [
|
||||||
|
sys.executable,
|
||||||
|
"-m",
|
||||||
|
"llama_cpp.server",
|
||||||
|
"--model",
|
||||||
|
str(self.model_path),
|
||||||
|
"--host",
|
||||||
|
"127.0.0.1",
|
||||||
|
"--port",
|
||||||
|
str(self.port),
|
||||||
|
"--n_gpu_layers",
|
||||||
|
str(self.gpu_layers),
|
||||||
|
"--n_ctx",
|
||||||
|
str(self.context_size),
|
||||||
|
"--flash_attn",
|
||||||
|
"true",
|
||||||
|
"--n_batch",
|
||||||
|
"2048",
|
||||||
|
# Prompt cache: reuse KV cache for repeated system prompts
|
||||||
|
"--cache",
|
||||||
|
"true",
|
||||||
|
]
|
||||||
|
# Only pass chat_format if explicitly set (most GGUF models have it embedded)
|
||||||
|
if self.chat_format:
|
||||||
|
cmd.extend(["--chat_format", self.chat_format])
|
||||||
|
# User-provided extra args (e.g. "--type_k 1 --type_v 1 --n_threads 8")
|
||||||
|
if self.extra_args:
|
||||||
|
cmd.extend(self.extra_args.split())
|
||||||
|
|
||||||
|
logger.info(f"Starting llama.cpp server: {' '.join(cmd)}")
|
||||||
|
|
||||||
|
# Write stderr to a log file to avoid pipe buffer deadlock
|
||||||
|
# (llama.cpp outputs a lot of model metadata on stderr during loading)
|
||||||
|
self._log_path = MODELS_DIR / "llamacpp_server.log"
|
||||||
|
self._log_file = open(self._log_path, "w")
|
||||||
|
|
||||||
|
self._process = subprocess.Popen(
|
||||||
|
cmd,
|
||||||
|
stdout=subprocess.DEVNULL,
|
||||||
|
stderr=self._log_file,
|
||||||
|
# Ensure the subprocess is killed when the parent exits
|
||||||
|
preexec_fn=os.setsid if hasattr(os, "setsid") else None,
|
||||||
|
)
|
||||||
|
|
||||||
|
# Wait for the server to be ready
|
||||||
|
await self._wait_for_ready()
|
||||||
|
|
||||||
|
async def _wait_for_ready(self, timeout: float = 120.0) -> None:
|
||||||
|
"""Wait for the llama.cpp server to accept connections."""
|
||||||
|
import httpx
|
||||||
|
|
||||||
|
start = time.monotonic()
|
||||||
|
url = f"http://127.0.0.1:{self.port}/v1/models"
|
||||||
|
last_log = start
|
||||||
|
|
||||||
|
while time.monotonic() - start < timeout:
|
||||||
|
# Check if process died
|
||||||
|
if self._process and self._process.poll() is not None:
|
||||||
|
stderr = ""
|
||||||
|
try:
|
||||||
|
stderr = self._log_path.read_text()[-2000:]
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
raise RuntimeError(f"llama.cpp server exited with code {self._process.returncode}.\nstderr: {stderr}")
|
||||||
|
|
||||||
|
try:
|
||||||
|
async with httpx.AsyncClient() as client:
|
||||||
|
resp = await client.get(url, timeout=5.0)
|
||||||
|
if resp.status_code == 200:
|
||||||
|
logger.info(f"llama.cpp server ready on port {self.port}")
|
||||||
|
return
|
||||||
|
except (httpx.ConnectError, httpx.TimeoutException, httpx.ConnectTimeout):
|
||||||
|
pass
|
||||||
|
|
||||||
|
# Log progress every 15s
|
||||||
|
now = time.monotonic()
|
||||||
|
if now - last_log > 15:
|
||||||
|
elapsed = int(now - start)
|
||||||
|
logger.info(f"Waiting for llama.cpp server to load model... ({elapsed}s)")
|
||||||
|
last_log = now
|
||||||
|
|
||||||
|
await asyncio.sleep(1.0)
|
||||||
|
|
||||||
|
# Timeout — read the log to help debug
|
||||||
|
stderr = ""
|
||||||
|
try:
|
||||||
|
stderr = self._log_path.read_text()[-2000:]
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
raise TimeoutError(
|
||||||
|
f"llama.cpp server did not become ready within {timeout}s.\n"
|
||||||
|
f"Check model compatibility and available memory.\n"
|
||||||
|
f"Server log: {stderr}"
|
||||||
|
)
|
||||||
|
|
||||||
|
async def stop(self) -> None:
|
||||||
|
"""Stop the llama.cpp server subprocess."""
|
||||||
|
if self._process is None:
|
||||||
|
return
|
||||||
|
|
||||||
|
logger.info("Stopping llama.cpp server...")
|
||||||
|
try:
|
||||||
|
# Send SIGTERM to the process group
|
||||||
|
if hasattr(os, "killpg"):
|
||||||
|
os.killpg(os.getpgid(self._process.pid), signal.SIGTERM)
|
||||||
|
else:
|
||||||
|
self._process.terminate()
|
||||||
|
|
||||||
|
# Wait up to 10s for graceful shutdown
|
||||||
|
try:
|
||||||
|
self._process.wait(timeout=10)
|
||||||
|
except subprocess.TimeoutExpired:
|
||||||
|
if hasattr(os, "killpg"):
|
||||||
|
os.killpg(os.getpgid(self._process.pid), signal.SIGKILL)
|
||||||
|
else:
|
||||||
|
self._process.kill()
|
||||||
|
self._process.wait(timeout=5)
|
||||||
|
except (ProcessLookupError, OSError):
|
||||||
|
pass # Process already exited
|
||||||
|
finally:
|
||||||
|
self._process = None
|
||||||
|
if hasattr(self, "_log_file") and self._log_file:
|
||||||
|
self._log_file.close()
|
||||||
|
self._log_file = None
|
||||||
|
logger.info("llama.cpp server stopped")
|
||||||
|
|
||||||
|
|
||||||
|
class LlamaCppLLM(LLMInterface):
|
||||||
|
"""
|
||||||
|
Built-in llama.cpp provider.
|
||||||
|
|
||||||
|
Manages a llama-cpp-python server subprocess and delegates to OpenAICompatibleLLM
|
||||||
|
for actual inference calls. Handles model downloading and server lifecycle.
|
||||||
|
"""
|
||||||
|
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
provider: str,
|
||||||
|
api_key: str,
|
||||||
|
base_url: str,
|
||||||
|
model: str,
|
||||||
|
reasoning_effort: str = "low",
|
||||||
|
model_path: str | None = None,
|
||||||
|
gpu_layers: int = -1,
|
||||||
|
context_size: int = 8192,
|
||||||
|
chat_format: str | None = None,
|
||||||
|
no_grammar: bool = False,
|
||||||
|
extra_args: str | None = None,
|
||||||
|
**kwargs: Any,
|
||||||
|
):
|
||||||
|
super().__init__(
|
||||||
|
provider=provider,
|
||||||
|
api_key=api_key or "llamacpp",
|
||||||
|
base_url=base_url or "",
|
||||||
|
model=model or DEFAULT_LLAMACPP_MODEL_ALIAS,
|
||||||
|
reasoning_effort=reasoning_effort,
|
||||||
|
)
|
||||||
|
self._model_path_str = model_path
|
||||||
|
self._gpu_layers = gpu_layers
|
||||||
|
self._context_size = context_size
|
||||||
|
self._chat_format = chat_format
|
||||||
|
self._no_grammar = no_grammar
|
||||||
|
self._extra_args = extra_args
|
||||||
|
self._server: LlamaCppServer | None = None
|
||||||
|
self._delegate: Any = None # OpenAICompatibleLLM, created after server starts
|
||||||
|
self._initialized = False
|
||||||
|
|
||||||
|
async def _ensure_initialized(self) -> None:
|
||||||
|
"""Lazy initialization: download model + start shared server on first use."""
|
||||||
|
if self._initialized:
|
||||||
|
return
|
||||||
|
|
||||||
|
global _shared_server
|
||||||
|
|
||||||
|
from .openai_compatible_llm import OpenAICompatibleLLM
|
||||||
|
|
||||||
|
async with _shared_server_lock:
|
||||||
|
if _shared_server is None:
|
||||||
|
# Resolve and potentially download the model
|
||||||
|
model_path = _resolve_model_path(self._model_path_str)
|
||||||
|
logger.info(f"Using GGUF model: {model_path}")
|
||||||
|
|
||||||
|
# Start the shared llama.cpp server
|
||||||
|
port = _find_free_port()
|
||||||
|
_shared_server = LlamaCppServer(
|
||||||
|
model_path=model_path,
|
||||||
|
port=port,
|
||||||
|
gpu_layers=self._gpu_layers,
|
||||||
|
context_size=self._context_size,
|
||||||
|
chat_format=self._chat_format,
|
||||||
|
extra_args=self._extra_args,
|
||||||
|
)
|
||||||
|
await _shared_server.start()
|
||||||
|
|
||||||
|
self._server = _shared_server
|
||||||
|
|
||||||
|
# Create the delegate that talks to the shared server's OpenAI-compatible API
|
||||||
|
if self._no_grammar:
|
||||||
|
logger.info("Grammar enforcement disabled (HINDSIGHT_API_LLAMACPP_NO_GRAMMAR=true)")
|
||||||
|
self._delegate = OpenAICompatibleLLM(
|
||||||
|
provider="llamacpp",
|
||||||
|
api_key="llamacpp",
|
||||||
|
base_url=self._server.base_url,
|
||||||
|
model=self.model,
|
||||||
|
reasoning_effort=self.reasoning_effort,
|
||||||
|
)
|
||||||
|
|
||||||
|
self._initialized = True
|
||||||
|
|
||||||
|
async def verify_connection(self) -> None:
|
||||||
|
"""Verify the llama.cpp server is running and can generate text."""
|
||||||
|
await self._ensure_initialized()
|
||||||
|
# Make a simple test call to verify the model can actually generate
|
||||||
|
await self._delegate.call(
|
||||||
|
messages=[{"role": "user", "content": "Say 'ok'"}],
|
||||||
|
max_completion_tokens=10,
|
||||||
|
max_retries=2,
|
||||||
|
initial_backoff=0.5,
|
||||||
|
max_backoff=2.0,
|
||||||
|
scope="verification",
|
||||||
|
)
|
||||||
|
logger.info("llama.cpp LLM verification passed")
|
||||||
|
|
||||||
|
async def call(
|
||||||
|
self,
|
||||||
|
messages: list[dict[str, str]],
|
||||||
|
response_format: Any | None = None,
|
||||||
|
max_completion_tokens: int | None = None,
|
||||||
|
temperature: float | None = None,
|
||||||
|
scope: str = "memory",
|
||||||
|
max_retries: int = 10,
|
||||||
|
initial_backoff: float = 1.0,
|
||||||
|
max_backoff: float = 60.0,
|
||||||
|
skip_validation: bool = False,
|
||||||
|
strict_schema: bool = False,
|
||||||
|
return_usage: bool = False,
|
||||||
|
) -> Any:
|
||||||
|
"""Delegate call to the OpenAI-compatible API."""
|
||||||
|
await self._ensure_initialized()
|
||||||
|
return await self._delegate.call(
|
||||||
|
messages=messages,
|
||||||
|
response_format=response_format,
|
||||||
|
max_completion_tokens=max_completion_tokens,
|
||||||
|
temperature=temperature,
|
||||||
|
scope=scope,
|
||||||
|
max_retries=max_retries,
|
||||||
|
initial_backoff=initial_backoff,
|
||||||
|
max_backoff=max_backoff,
|
||||||
|
skip_validation=skip_validation,
|
||||||
|
strict_schema=strict_schema,
|
||||||
|
return_usage=return_usage,
|
||||||
|
)
|
||||||
|
|
||||||
|
async def call_with_tools(
|
||||||
|
self,
|
||||||
|
messages: list[dict[str, Any]],
|
||||||
|
tools: list[dict[str, Any]],
|
||||||
|
max_completion_tokens: int | None = None,
|
||||||
|
temperature: float | None = None,
|
||||||
|
scope: str = "tools",
|
||||||
|
max_retries: int = 5,
|
||||||
|
initial_backoff: float = 1.0,
|
||||||
|
max_backoff: float = 30.0,
|
||||||
|
tool_choice: str | dict[str, Any] = "auto",
|
||||||
|
) -> LLMToolCallResult:
|
||||||
|
"""Delegate tool calls to the OpenAI-compatible API."""
|
||||||
|
await self._ensure_initialized()
|
||||||
|
return await self._delegate.call_with_tools(
|
||||||
|
messages=messages,
|
||||||
|
tools=tools,
|
||||||
|
max_completion_tokens=max_completion_tokens,
|
||||||
|
temperature=temperature,
|
||||||
|
scope=scope,
|
||||||
|
max_retries=max_retries,
|
||||||
|
initial_backoff=initial_backoff,
|
||||||
|
max_backoff=max_backoff,
|
||||||
|
tool_choice=tool_choice,
|
||||||
|
)
|
||||||
|
|
||||||
|
async def cleanup(self) -> None:
|
||||||
|
"""Stop the shared llama.cpp server."""
|
||||||
|
global _shared_server
|
||||||
|
|
||||||
|
if self._delegate:
|
||||||
|
await self._delegate.cleanup()
|
||||||
|
self._delegate = None
|
||||||
|
|
||||||
|
# Stop the shared server (only the first cleanup call actually stops it)
|
||||||
|
async with _shared_server_lock:
|
||||||
|
if _shared_server is not None:
|
||||||
|
await _shared_server.stop()
|
||||||
|
_shared_server = None
|
||||||
|
|
||||||
|
self._server = None
|
||||||
|
self._initialized = False
|
||||||
|
|
@ -100,7 +100,7 @@ class OpenAICompatibleLLM(LLMInterface):
|
||||||
super().__init__(provider, api_key, base_url, model, reasoning_effort, **kwargs)
|
super().__init__(provider, api_key, base_url, model, reasoning_effort, **kwargs)
|
||||||
|
|
||||||
# Validate provider
|
# Validate provider
|
||||||
valid_providers = ["openai", "groq", "ollama", "lmstudio", "minimax", "volcano", "openrouter"]
|
valid_providers = ["openai", "groq", "ollama", "lmstudio", "llamacpp", "minimax", "volcano", "openrouter"]
|
||||||
if self.provider not in valid_providers:
|
if self.provider not in valid_providers:
|
||||||
raise ValueError(f"OpenAICompatibleLLM only supports: {', '.join(valid_providers)}. Got: {self.provider}")
|
raise ValueError(f"OpenAICompatibleLLM only supports: {', '.join(valid_providers)}. Got: {self.provider}")
|
||||||
|
|
||||||
|
|
@ -201,8 +201,8 @@ class OpenAICompatibleLLM(LLMInterface):
|
||||||
require 'max_tokens'. Using a custom base_url with the openai provider
|
require 'max_tokens'. Using a custom base_url with the openai provider
|
||||||
signals a third-party compatible API, so fall back to 'max_tokens'.
|
signals a third-party compatible API, so fall back to 'max_tokens'.
|
||||||
"""
|
"""
|
||||||
# Native OpenAI (no custom base URL) and Groq use max_completion_tokens
|
# Native OpenAI (no custom base URL), Groq, and llamacpp use max_completion_tokens
|
||||||
if self.provider == "groq":
|
if self.provider in ("groq", "llamacpp"):
|
||||||
return "max_completion_tokens"
|
return "max_completion_tokens"
|
||||||
if self.provider == "openai" and not self.base_url:
|
if self.provider == "openai" and not self.base_url:
|
||||||
return "max_completion_tokens"
|
return "max_completion_tokens"
|
||||||
|
|
@ -337,8 +337,13 @@ class OpenAICompatibleLLM(LLMInterface):
|
||||||
first_msg = call_params["messages"][0]
|
first_msg = call_params["messages"][0]
|
||||||
if isinstance(first_msg, dict) and isinstance(first_msg.get("content"), str):
|
if isinstance(first_msg, dict) and isinstance(first_msg.get("content"), str):
|
||||||
first_msg["content"] = schema_msg + "\n\n" + first_msg["content"]
|
first_msg["content"] = schema_msg + "\n\n" + first_msg["content"]
|
||||||
if self.provider not in ("lmstudio", "ollama", "volcano"):
|
# Providers that skip json_object grammar enforcement
|
||||||
# LM Studio, Ollama and Volcano don't support json_object response format reliably
|
skip_grammar = self.provider in ("lmstudio", "ollama", "volcano")
|
||||||
|
if self.provider == "llamacpp":
|
||||||
|
from hindsight_api.config import get_config
|
||||||
|
|
||||||
|
skip_grammar = get_config().llamacpp_no_grammar
|
||||||
|
if not skip_grammar:
|
||||||
call_params["response_format"] = {"type": "json_object"}
|
call_params["response_format"] = {"type": "json_object"}
|
||||||
|
|
||||||
last_exception = None
|
last_exception = None
|
||||||
|
|
|
||||||
|
|
@ -909,6 +909,7 @@ def _build_user_message(
|
||||||
event_date: datetime | None,
|
event_date: datetime | None,
|
||||||
context: str,
|
context: str,
|
||||||
metadata: dict[str, str] | None = None,
|
metadata: dict[str, str] | None = None,
|
||||||
|
agent_name: str | None = None,
|
||||||
) -> str:
|
) -> str:
|
||||||
"""Build user message for fact extraction."""
|
"""Build user message for fact extraction."""
|
||||||
from .orchestrator import parse_datetime_flexible
|
from .orchestrator import parse_datetime_flexible
|
||||||
|
|
@ -927,11 +928,15 @@ def _build_user_message(
|
||||||
metadata_lines = "\n".join(f" {k}: {v}" for k, v in metadata.items())
|
metadata_lines = "\n".join(f" {k}: {v}" for k, v in metadata.items())
|
||||||
metadata_section = f"\nMetadata:\n{metadata_lines}"
|
metadata_section = f"\nMetadata:\n{metadata_lines}"
|
||||||
|
|
||||||
|
narrator_section = ""
|
||||||
|
if agent_name:
|
||||||
|
narrator_section = f'\nNarrator: {agent_name} (AI agent — first-person statements like "I did X" are the agent\'s own actions; classify as "assistant")'
|
||||||
|
|
||||||
return f"""Extract facts from the following text chunk.
|
return f"""Extract facts from the following text chunk.
|
||||||
|
|
||||||
Chunk: {chunk_index + 1}/{total_chunks}
|
Chunk: {chunk_index + 1}/{total_chunks}
|
||||||
Event Date: {event_date_str}
|
Event Date: {event_date_str}
|
||||||
Context: {sanitized_context}{metadata_section}
|
Context: {sanitized_context}{metadata_section}{narrator_section}
|
||||||
|
|
||||||
Text:
|
Text:
|
||||||
{sanitized_chunk}"""
|
{sanitized_chunk}"""
|
||||||
|
|
@ -995,7 +1000,7 @@ async def _extract_facts_from_chunk(
|
||||||
extract_causal_links = config.retain_extract_causal_links
|
extract_causal_links = config.retain_extract_causal_links
|
||||||
|
|
||||||
# Build user message using helper function
|
# Build user message using helper function
|
||||||
user_message = _build_user_message(chunk, chunk_index, total_chunks, event_date, context, metadata)
|
user_message = _build_user_message(chunk, chunk_index, total_chunks, event_date, context, metadata, agent_name)
|
||||||
|
|
||||||
# Retry logic for JSON validation errors
|
# Retry logic for JSON validation errors
|
||||||
# Use retain-specific overrides if set, otherwise fall back to global LLM config
|
# Use retain-specific overrides if set, otherwise fall back to global LLM config
|
||||||
|
|
@ -1632,7 +1637,13 @@ async def extract_facts_from_contents_batch_api(
|
||||||
|
|
||||||
# Build user message using helper function
|
# Build user message using helper function
|
||||||
user_message = _build_user_message(
|
user_message = _build_user_message(
|
||||||
chunk, chunk_index_in_content, len(chunks), item.event_date, item.context, item.metadata or None
|
chunk,
|
||||||
|
chunk_index_in_content,
|
||||||
|
len(chunks),
|
||||||
|
item.event_date,
|
||||||
|
item.context,
|
||||||
|
item.metadata or None,
|
||||||
|
agent_name,
|
||||||
)
|
)
|
||||||
|
|
||||||
# Build request body using helper function
|
# Build request body using helper function
|
||||||
|
|
|
||||||
|
|
@ -78,11 +78,16 @@ local-ml = [
|
||||||
"mlx-lm>=0.31.1",
|
"mlx-lm>=0.31.1",
|
||||||
"safetensors>=0.6.2",
|
"safetensors>=0.6.2",
|
||||||
]
|
]
|
||||||
|
local-llm = [
|
||||||
|
# Built-in llama.cpp inference for fully offline operation
|
||||||
|
"llama-cpp-python[server]>=0.3.0",
|
||||||
|
"huggingface-hub>=0.20.0",
|
||||||
|
]
|
||||||
embedded-db = [
|
embedded-db = [
|
||||||
"pg0-embedded>=0.11.0",
|
"pg0-embedded>=0.11.0",
|
||||||
]
|
]
|
||||||
all = [
|
all = [
|
||||||
"hindsight-api-slim[local-ml,embedded-db]",
|
"hindsight-api-slim[local-ml,local-llm,embedded-db]",
|
||||||
]
|
]
|
||||||
test = [
|
test = [
|
||||||
"pytest>=7.0.0",
|
"pytest>=7.0.0",
|
||||||
|
|
|
||||||
|
|
@ -161,7 +161,7 @@ To switch between backends:
|
||||||
|
|
||||||
| Variable | Description | Default |
|
| Variable | Description | Default |
|
||||||
|----------|-------------|---------|
|
|----------|-------------|---------|
|
||||||
| `HINDSIGHT_API_LLM_PROVIDER` | Provider: `openai`, `openai-codex`, `claude-code`, `anthropic`, `gemini`, `groq`, `minimax`, `ollama`, `lmstudio`, `vertexai`, `bedrock`, `litellm`, `volcano`, `openrouter`, `none` | `openai` |
|
| `HINDSIGHT_API_LLM_PROVIDER` | Provider: `openai`, `openai-codex`, `claude-code`, `anthropic`, `gemini`, `groq`, `minimax`, `ollama`, `lmstudio`, `llamacpp`, `vertexai`, `bedrock`, `litellm`, `volcano`, `openrouter`, `none` | `openai` |
|
||||||
| `HINDSIGHT_API_LLM_API_KEY` | API key for LLM provider | - |
|
| `HINDSIGHT_API_LLM_API_KEY` | API key for LLM provider | - |
|
||||||
| `HINDSIGHT_API_LLM_MODEL` | Model name | `gpt-5-mini` |
|
| `HINDSIGHT_API_LLM_MODEL` | Model name | `gpt-5-mini` |
|
||||||
| `HINDSIGHT_API_LLM_BASE_URL` | Custom LLM endpoint | Provider default |
|
| `HINDSIGHT_API_LLM_BASE_URL` | Custom LLM endpoint | Provider default |
|
||||||
|
|
@ -220,6 +220,12 @@ export HINDSIGHT_API_LLM_PROVIDER=lmstudio
|
||||||
export HINDSIGHT_API_LLM_BASE_URL=http://localhost:1234/v1
|
export HINDSIGHT_API_LLM_BASE_URL=http://localhost:1234/v1
|
||||||
export HINDSIGHT_API_LLM_MODEL=your-local-model
|
export HINDSIGHT_API_LLM_MODEL=your-local-model
|
||||||
|
|
||||||
|
# llama.cpp (built-in local inference, no external server needed)
|
||||||
|
export HINDSIGHT_API_LLM_PROVIDER=llamacpp
|
||||||
|
# No API key, base URL, or external server required.
|
||||||
|
# Auto-downloads Gemma 4 E2B (~3.5 GB GGUF) on first run.
|
||||||
|
# See "Built-in llama.cpp" section below for all configuration options.
|
||||||
|
|
||||||
# OpenAI-compatible endpoint
|
# OpenAI-compatible endpoint
|
||||||
export HINDSIGHT_API_LLM_PROVIDER=openai
|
export HINDSIGHT_API_LLM_PROVIDER=openai
|
||||||
export HINDSIGHT_API_LLM_BASE_URL=https://your-endpoint.com/v1
|
export HINDSIGHT_API_LLM_BASE_URL=https://your-endpoint.com/v1
|
||||||
|
|
@ -277,6 +283,36 @@ export HINDSIGHT_API_LLM_PROVIDER=none
|
||||||
For detailed setup instructions for **OpenAI Codex** (ChatGPT Plus/Pro), **Claude Code** (Claude Pro/Max), and **Vertex AI** (Google Cloud), see the [Models documentation](./models#openai-codex-setup-chatgpt-pluspro).
|
For detailed setup instructions for **OpenAI Codex** (ChatGPT Plus/Pro), **Claude Code** (Claude Pro/Max), and **Vertex AI** (Google Cloud), see the [Models documentation](./models#openai-codex-setup-chatgpt-pluspro).
|
||||||
:::
|
:::
|
||||||
|
|
||||||
|
### Built-in llama.cpp
|
||||||
|
|
||||||
|
The `llamacpp` provider runs a llama.cpp server as a managed subprocess — no external LLM server needed. On first run it auto-downloads a default GGUF model (~3.5 GB). Requires the `local-llm` extra: `pip install 'hindsight-api-slim[local-llm]'`.
|
||||||
|
|
||||||
|
| Variable | Description | Default |
|
||||||
|
|----------|-------------|---------|
|
||||||
|
| `HINDSIGHT_API_LLAMACPP_MODEL_PATH` | Path to a GGUF model file. If not set, auto-downloads `gemma-4-E2B-it-Q4_K_M` from HuggingFace. | Auto-download |
|
||||||
|
| `HINDSIGHT_API_LLAMACPP_GPU_LAYERS` | Number of layers to offload to GPU. `-1` = all layers (recommended). `0` = CPU only. | `-1` |
|
||||||
|
| `HINDSIGHT_API_LLAMACPP_CONTEXT_SIZE` | Context window size in tokens. | `8192` |
|
||||||
|
| `HINDSIGHT_API_LLAMACPP_CHAT_FORMAT` | Chat template format. `null` = auto-detect from GGUF metadata (recommended). | Auto-detect |
|
||||||
|
| `HINDSIGHT_API_LLAMACPP_NO_GRAMMAR` | Disable JSON grammar enforcement. Faster inference but less reliable JSON output. | `false` |
|
||||||
|
| `HINDSIGHT_API_LLAMACPP_EXTRA_ARGS` | Space-separated extra CLI args passed to the llama.cpp server (e.g. `--n_threads 8 --type_k 1`). | - |
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# Minimal setup (auto-downloads model, uses GPU)
|
||||||
|
export HINDSIGHT_API_LLM_PROVIDER=llamacpp
|
||||||
|
|
||||||
|
# Custom model with tuning
|
||||||
|
export HINDSIGHT_API_LLM_PROVIDER=llamacpp
|
||||||
|
export HINDSIGHT_API_LLM_MAX_CONCURRENT=2
|
||||||
|
export HINDSIGHT_API_LLAMACPP_MODEL_PATH=~/.hindsight/models/my-model.gguf
|
||||||
|
export HINDSIGHT_API_LLAMACPP_CONTEXT_SIZE=16384
|
||||||
|
export HINDSIGHT_API_LLAMACPP_NO_GRAMMAR=true # faster, less reliable JSON
|
||||||
|
export HINDSIGHT_API_LLAMACPP_EXTRA_ARGS="--n_threads 8"
|
||||||
|
```
|
||||||
|
|
||||||
|
:::note
|
||||||
|
The llama.cpp server is shared across all LLM operations (retain, reflect, consolidation). Set `HINDSIGHT_API_LLM_MAX_CONCURRENT=2` to allow retain and consolidation to run concurrently without blocking each other.
|
||||||
|
:::
|
||||||
|
|
||||||
### Per-Operation LLM Configuration
|
### Per-Operation LLM Configuration
|
||||||
|
|
||||||
Different memory operations have different requirements. **Retain** (fact extraction) benefits from models with strong structured output capabilities, while **Reflect** (reasoning/response generation) can use lighter, faster models. Configure separate LLM models for each operation to optimize for cost and performance.
|
Different memory operations have different requirements. **Retain** (fact extraction) benefits from models with strong structured output capabilities, while **Reflect** (reasoning/response generation) can use lighter, faster models. Configure separate LLM models for each operation to optimize for cost and performance.
|
||||||
|
|
|
||||||
|
|
@ -36,6 +36,12 @@ Set `HINDSIGHT_API_LLM_PROVIDER=bedrock` to use AWS Bedrock models directly. Mod
|
||||||
See [Configuration](./configuration#llm-provider) for setup examples.
|
See [Configuration](./configuration#llm-provider) for setup examples.
|
||||||
:::
|
:::
|
||||||
|
|
||||||
|
:::tip Built-in llama.cpp (fully local, no API key)
|
||||||
|
Set `HINDSIGHT_API_LLM_PROVIDER=llamacpp` to run a built-in llama.cpp server with no external dependencies. A Gemma 4 E2B GGUF model (~3.5 GB) is auto-downloaded on first run. Requires the `local-llm` extra: `pip install 'hindsight-api-slim[local-llm]'`.
|
||||||
|
|
||||||
|
See [Configuration](./configuration#built-in-llamacpp) for all options.
|
||||||
|
:::
|
||||||
|
|
||||||
:::tip LiteLLM Provider (Azure, Together AI, and more)
|
:::tip LiteLLM Provider (Azure, Together AI, and more)
|
||||||
Set `HINDSIGHT_API_LLM_PROVIDER=litellm` to use any model supported by [LiteLLM](https://docs.litellm.ai/docs/providers), including **Azure OpenAI**, **Together AI**, **Fireworks AI**, and many more. Model names use LiteLLM's provider prefix format (e.g., `azure/gpt-4o`).
|
Set `HINDSIGHT_API_LLM_PROVIDER=litellm` to use any model supported by [LiteLLM](https://docs.litellm.ai/docs/providers), including **Azure OpenAI**, **Together AI**, **Fireworks AI**, and many more. Model names use LiteLLM's provider prefix format (e.g., `azure/gpt-4o`).
|
||||||
|
|
||||||
|
|
@ -81,6 +87,7 @@ Each provider has a recommended default model that's used when `HINDSIGHT_API_LL
|
||||||
| `groq` | `openai/gpt-oss-120b` |
|
| `groq` | `openai/gpt-oss-120b` |
|
||||||
| `minimax` | `MiniMax-M2.7` |
|
| `minimax` | `MiniMax-M2.7` |
|
||||||
| `ollama` | `gemma3:12b` |
|
| `ollama` | `gemma3:12b` |
|
||||||
|
| `llamacpp` | `gemma-4-e2b-it` (auto-downloaded GGUF) |
|
||||||
| `lmstudio` | `local-model` |
|
| `lmstudio` | `local-model` |
|
||||||
| `vertexai` | `gemini-2.0-flash-001` |
|
| `vertexai` | `gemini-2.0-flash-001` |
|
||||||
| `openai-codex` | `gpt-5.2-codex` |
|
| `openai-codex` | `gpt-5.2-codex` |
|
||||||
|
|
|
||||||
|
|
@ -37,6 +37,7 @@ export function LLMProvidersGrid() {
|
||||||
{ label: 'Groq', icon: LuZap },
|
{ label: 'Groq', icon: LuZap },
|
||||||
{ label: 'Ollama', icon: SiOllama },
|
{ label: 'Ollama', icon: SiOllama },
|
||||||
{ label: 'LM Studio', icon: LuBrainCog },
|
{ label: 'LM Studio', icon: LuBrainCog },
|
||||||
|
{ label: 'llama.cpp', icon: LuTerminal },
|
||||||
{ label: 'MiniMax', icon: LuSparkles },
|
{ label: 'MiniMax', icon: LuSparkles },
|
||||||
{ label: 'Volcano Engine', icon: LuZap },
|
{ label: 'Volcano Engine', icon: LuZap },
|
||||||
{ label: 'OpenAI Compatible', icon: OpenAICompatibleIcon },
|
{ label: 'OpenAI Compatible', icon: OpenAICompatibleIcon },
|
||||||
|
|
|
||||||
|
|
@ -531,6 +531,23 @@ def do_daemon(args, config: dict, logger):
|
||||||
return 0
|
return 0
|
||||||
|
|
||||||
if daemon_client.ensure_daemon_running(config, profile):
|
if daemon_client.ensure_daemon_running(config, profile):
|
||||||
|
# Start UI if --ui flag was passed
|
||||||
|
if getattr(args, "ui", False):
|
||||||
|
from .daemon_embed_manager import DaemonEmbedManager
|
||||||
|
from .profile_manager import resolve_active_profile
|
||||||
|
|
||||||
|
# Use the same profile resolution as the daemon
|
||||||
|
resolved_profile = profile if profile is not None else resolve_active_profile()
|
||||||
|
manager = DaemonEmbedManager()
|
||||||
|
ui_started = manager.start_ui(resolved_profile, None, "0.0.0.0")
|
||||||
|
if not ui_started:
|
||||||
|
console.print(
|
||||||
|
Panel(
|
||||||
|
Text("Daemon is running but UI failed to start", style="yellow"),
|
||||||
|
title="[bold yellow]UI Warning[/bold yellow]",
|
||||||
|
border_style="yellow",
|
||||||
|
)
|
||||||
|
)
|
||||||
return 0
|
return 0
|
||||||
else:
|
else:
|
||||||
console.print(
|
console.print(
|
||||||
|
|
@ -591,7 +608,6 @@ def do_daemon(args, config: dict, logger):
|
||||||
return 1
|
return 1
|
||||||
|
|
||||||
elif args.daemon_command == "status":
|
elif args.daemon_command == "status":
|
||||||
import os
|
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
from rich.console import Console
|
from rich.console import Console
|
||||||
|
|
@ -1393,7 +1409,8 @@ def main():
|
||||||
# Parse daemon subcommand (profile already extracted globally)
|
# Parse daemon subcommand (profile already extracted globally)
|
||||||
parser = argparse.ArgumentParser(prog="hindsight-embed daemon")
|
parser = argparse.ArgumentParser(prog="hindsight-embed daemon")
|
||||||
subparsers = parser.add_subparsers(dest="daemon_command")
|
subparsers = parser.add_subparsers(dest="daemon_command")
|
||||||
subparsers.add_parser("start", help="Start the daemon")
|
start_parser = subparsers.add_parser("start", help="Start the daemon")
|
||||||
|
start_parser.add_argument("--ui", action="store_true", help="Also start the web UI after daemon is ready")
|
||||||
subparsers.add_parser("stop", help="Stop the daemon")
|
subparsers.add_parser("stop", help="Stop the daemon")
|
||||||
subparsers.add_parser("status", help="Check daemon status")
|
subparsers.add_parser("status", help="Check daemon status")
|
||||||
logs_parser = subparsers.add_parser("logs", help="View daemon logs")
|
logs_parser = subparsers.add_parser("logs", help="View daemon logs")
|
||||||
|
|
@ -1450,7 +1467,7 @@ def main():
|
||||||
|
|
||||||
# Check for LLM API key (not required for vertexai which uses GCP credentials)
|
# Check for LLM API key (not required for vertexai which uses GCP credentials)
|
||||||
llm_provider = config.get("llm_provider", "openai")
|
llm_provider = config.get("llm_provider", "openai")
|
||||||
providers_without_api_key = ("ollama", "vertexai")
|
providers_without_api_key = ("ollama", "vertexai", "llamacpp")
|
||||||
if not config["llm_api_key"] and llm_provider not in providers_without_api_key:
|
if not config["llm_api_key"] and llm_provider not in providers_without_api_key:
|
||||||
print("Error: LLM API key is required.", file=sys.stderr)
|
print("Error: LLM API key is required.", file=sys.stderr)
|
||||||
print("Run 'hindsight-embed configure' to set up.", file=sys.stderr)
|
print("Run 'hindsight-embed configure' to set up.", file=sys.stderr)
|
||||||
|
|
|
||||||
|
|
@ -56,7 +56,7 @@ def get_daemon_url(profile: str | None = None) -> str:
|
||||||
return _manager.get_url(profile)
|
return _manager.get_url(profile)
|
||||||
|
|
||||||
|
|
||||||
def ensure_daemon_running(config: dict, profile: str | None = None) -> bool:
|
def ensure_daemon_running(config: dict, profile: str | None = None, extra_args: list[str] | None = None) -> bool:
|
||||||
"""
|
"""
|
||||||
Ensure daemon is running, starting it if needed.
|
Ensure daemon is running, starting it if needed.
|
||||||
|
|
||||||
|
|
@ -64,6 +64,7 @@ def ensure_daemon_running(config: dict, profile: str | None = None) -> bool:
|
||||||
config: Configuration dict with LLM settings (accepts both simple keys
|
config: Configuration dict with LLM settings (accepts both simple keys
|
||||||
like "llm_api_key" and env var format like "HINDSIGHT_API_LLM_API_KEY").
|
like "llm_api_key" and env var format like "HINDSIGHT_API_LLM_API_KEY").
|
||||||
profile: Profile name (None = resolve from priority).
|
profile: Profile name (None = resolve from priority).
|
||||||
|
extra_args: Extra CLI arguments to pass to hindsight-api (e.g. ["--offline"]).
|
||||||
|
|
||||||
Returns:
|
Returns:
|
||||||
True if daemon is running.
|
True if daemon is running.
|
||||||
|
|
@ -71,7 +72,7 @@ def ensure_daemon_running(config: dict, profile: str | None = None) -> bool:
|
||||||
if profile is None:
|
if profile is None:
|
||||||
profile = resolve_active_profile()
|
profile = resolve_active_profile()
|
||||||
|
|
||||||
return _manager.ensure_running(config, profile)
|
return _manager.ensure_running(config, profile, extra_args=extra_args)
|
||||||
|
|
||||||
|
|
||||||
def stop_daemon(profile: str | None = None) -> bool:
|
def stop_daemon(profile: str | None = None) -> bool:
|
||||||
|
|
|
||||||
|
|
@ -197,7 +197,7 @@ class DaemonEmbedManager(EmbedManager):
|
||||||
logger.warning(f"Old daemon (PID {pid}) did not stop in time")
|
logger.warning(f"Old daemon (PID {pid}) did not stop in time")
|
||||||
return False
|
return False
|
||||||
|
|
||||||
def _start_daemon(self, config: dict, profile: str) -> bool:
|
def _start_daemon(self, config: dict, profile: str, extra_args: list[str] | None = None) -> bool:
|
||||||
"""Start the daemon in background."""
|
"""Start the daemon in background."""
|
||||||
paths = self._profile_manager.resolve_profile_paths(profile)
|
paths = self._profile_manager.resolve_profile_paths(profile)
|
||||||
profile_label = f"profile '{profile}'" if profile else "default profile"
|
profile_label = f"profile '{profile}'" if profile else "default profile"
|
||||||
|
|
@ -252,15 +252,6 @@ class DaemonEmbedManager(EmbedManager):
|
||||||
if "HINDSIGHT_EMBED_DAEMON_IDLE_TIMEOUT" not in env:
|
if "HINDSIGHT_EMBED_DAEMON_IDLE_TIMEOUT" not in env:
|
||||||
env["HINDSIGHT_EMBED_DAEMON_IDLE_TIMEOUT"] = str(DEFAULT_DAEMON_IDLE_TIMEOUT)
|
env["HINDSIGHT_EMBED_DAEMON_IDLE_TIMEOUT"] = str(DEFAULT_DAEMON_IDLE_TIMEOUT)
|
||||||
|
|
||||||
# On macOS, force CPU for embeddings/reranker to avoid MPS issues
|
|
||||||
import platform
|
|
||||||
|
|
||||||
if platform.system() == "Darwin":
|
|
||||||
if "HINDSIGHT_API_EMBEDDINGS_LOCAL_FORCE_CPU" not in env:
|
|
||||||
env["HINDSIGHT_API_EMBEDDINGS_LOCAL_FORCE_CPU"] = "1"
|
|
||||||
if "HINDSIGHT_API_RERANKER_LOCAL_FORCE_CPU" not in env:
|
|
||||||
env["HINDSIGHT_API_RERANKER_LOCAL_FORCE_CPU"] = "1"
|
|
||||||
|
|
||||||
# Get idle timeout from env
|
# Get idle timeout from env
|
||||||
idle_timeout = int(env.get("HINDSIGHT_EMBED_DAEMON_IDLE_TIMEOUT", str(DEFAULT_DAEMON_IDLE_TIMEOUT)))
|
idle_timeout = int(env.get("HINDSIGHT_EMBED_DAEMON_IDLE_TIMEOUT", str(DEFAULT_DAEMON_IDLE_TIMEOUT)))
|
||||||
|
|
||||||
|
|
@ -276,6 +267,8 @@ class DaemonEmbedManager(EmbedManager):
|
||||||
"--port",
|
"--port",
|
||||||
str(port),
|
str(port),
|
||||||
]
|
]
|
||||||
|
if extra_args:
|
||||||
|
cmd.extend(extra_args)
|
||||||
|
|
||||||
try:
|
try:
|
||||||
# Start daemon
|
# Start daemon
|
||||||
|
|
@ -620,13 +613,14 @@ class DaemonEmbedManager(EmbedManager):
|
||||||
|
|
||||||
return not self.is_ui_running(profile, ui_port)
|
return not self.is_ui_running(profile, ui_port)
|
||||||
|
|
||||||
def ensure_running(self, config: dict, profile: str) -> bool:
|
def ensure_running(self, config: dict, profile: str, extra_args: list[str] | None = None) -> bool:
|
||||||
"""
|
"""
|
||||||
Ensure daemon is running, starting it if needed.
|
Ensure daemon is running, starting it if needed.
|
||||||
|
|
||||||
Args:
|
Args:
|
||||||
config: Environment configuration dict (HINDSIGHT_API_* vars)
|
config: Environment configuration dict (HINDSIGHT_API_* vars)
|
||||||
profile: Profile name for isolation
|
profile: Profile name for isolation
|
||||||
|
extra_args: Extra CLI arguments to pass to hindsight-api (e.g. ["--offline"])
|
||||||
|
|
||||||
Returns:
|
Returns:
|
||||||
True if daemon is running (started or already running), False on failure
|
True if daemon is running (started or already running), False on failure
|
||||||
|
|
@ -637,7 +631,7 @@ class DaemonEmbedManager(EmbedManager):
|
||||||
paths = self._profile_manager.resolve_profile_paths(profile)
|
paths = self._profile_manager.resolve_profile_paths(profile)
|
||||||
self._register_profile(profile, paths.port, config)
|
self._register_profile(profile, paths.port, config)
|
||||||
return True
|
return True
|
||||||
return self._start_daemon(config, profile)
|
return self._start_daemon(config, profile, extra_args=extra_args)
|
||||||
|
|
||||||
def stop(self, profile: str) -> bool:
|
def stop(self, profile: str) -> bool:
|
||||||
"""
|
"""
|
||||||
|
|
|
||||||
|
|
@ -161,7 +161,7 @@ To switch between backends:
|
||||||
|
|
||||||
| Variable | Description | Default |
|
| Variable | Description | Default |
|
||||||
|----------|-------------|---------|
|
|----------|-------------|---------|
|
||||||
| `HINDSIGHT_API_LLM_PROVIDER` | Provider: `openai`, `openai-codex`, `claude-code`, `anthropic`, `gemini`, `groq`, `minimax`, `ollama`, `lmstudio`, `vertexai`, `bedrock`, `litellm`, `volcano`, `openrouter`, `none` | `openai` |
|
| `HINDSIGHT_API_LLM_PROVIDER` | Provider: `openai`, `openai-codex`, `claude-code`, `anthropic`, `gemini`, `groq`, `minimax`, `ollama`, `lmstudio`, `llamacpp`, `vertexai`, `bedrock`, `litellm`, `volcano`, `openrouter`, `none` | `openai` |
|
||||||
| `HINDSIGHT_API_LLM_API_KEY` | API key for LLM provider | - |
|
| `HINDSIGHT_API_LLM_API_KEY` | API key for LLM provider | - |
|
||||||
| `HINDSIGHT_API_LLM_MODEL` | Model name | `gpt-5-mini` |
|
| `HINDSIGHT_API_LLM_MODEL` | Model name | `gpt-5-mini` |
|
||||||
| `HINDSIGHT_API_LLM_BASE_URL` | Custom LLM endpoint | Provider default |
|
| `HINDSIGHT_API_LLM_BASE_URL` | Custom LLM endpoint | Provider default |
|
||||||
|
|
@ -220,6 +220,12 @@ export HINDSIGHT_API_LLM_PROVIDER=lmstudio
|
||||||
export HINDSIGHT_API_LLM_BASE_URL=http://localhost:1234/v1
|
export HINDSIGHT_API_LLM_BASE_URL=http://localhost:1234/v1
|
||||||
export HINDSIGHT_API_LLM_MODEL=your-local-model
|
export HINDSIGHT_API_LLM_MODEL=your-local-model
|
||||||
|
|
||||||
|
# llama.cpp (built-in local inference, no external server needed)
|
||||||
|
export HINDSIGHT_API_LLM_PROVIDER=llamacpp
|
||||||
|
# No API key, base URL, or external server required.
|
||||||
|
# Auto-downloads Gemma 4 E2B (~3.5 GB GGUF) on first run.
|
||||||
|
# See "Built-in llama.cpp" section below for all configuration options.
|
||||||
|
|
||||||
# OpenAI-compatible endpoint
|
# OpenAI-compatible endpoint
|
||||||
export HINDSIGHT_API_LLM_PROVIDER=openai
|
export HINDSIGHT_API_LLM_PROVIDER=openai
|
||||||
export HINDSIGHT_API_LLM_BASE_URL=https://your-endpoint.com/v1
|
export HINDSIGHT_API_LLM_BASE_URL=https://your-endpoint.com/v1
|
||||||
|
|
@ -277,6 +283,36 @@ export HINDSIGHT_API_LLM_PROVIDER=none
|
||||||
For detailed setup instructions for **OpenAI Codex** (ChatGPT Plus/Pro), **Claude Code** (Claude Pro/Max), and **Vertex AI** (Google Cloud), see the [Models documentation](./models#openai-codex-setup-chatgpt-pluspro).
|
For detailed setup instructions for **OpenAI Codex** (ChatGPT Plus/Pro), **Claude Code** (Claude Pro/Max), and **Vertex AI** (Google Cloud), see the [Models documentation](./models#openai-codex-setup-chatgpt-pluspro).
|
||||||
:::
|
:::
|
||||||
|
|
||||||
|
### Built-in llama.cpp
|
||||||
|
|
||||||
|
The `llamacpp` provider runs a llama.cpp server as a managed subprocess — no external LLM server needed. On first run it auto-downloads a default GGUF model (~3.5 GB). Requires the `local-llm` extra: `pip install 'hindsight-api-slim[local-llm]'`.
|
||||||
|
|
||||||
|
| Variable | Description | Default |
|
||||||
|
|----------|-------------|---------|
|
||||||
|
| `HINDSIGHT_API_LLAMACPP_MODEL_PATH` | Path to a GGUF model file. If not set, auto-downloads `gemma-4-E2B-it-Q4_K_M` from HuggingFace. | Auto-download |
|
||||||
|
| `HINDSIGHT_API_LLAMACPP_GPU_LAYERS` | Number of layers to offload to GPU. `-1` = all layers (recommended). `0` = CPU only. | `-1` |
|
||||||
|
| `HINDSIGHT_API_LLAMACPP_CONTEXT_SIZE` | Context window size in tokens. | `8192` |
|
||||||
|
| `HINDSIGHT_API_LLAMACPP_CHAT_FORMAT` | Chat template format. `null` = auto-detect from GGUF metadata (recommended). | Auto-detect |
|
||||||
|
| `HINDSIGHT_API_LLAMACPP_NO_GRAMMAR` | Disable JSON grammar enforcement. Faster inference but less reliable JSON output. | `false` |
|
||||||
|
| `HINDSIGHT_API_LLAMACPP_EXTRA_ARGS` | Space-separated extra CLI args passed to the llama.cpp server (e.g. `--n_threads 8 --type_k 1`). | - |
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# Minimal setup (auto-downloads model, uses GPU)
|
||||||
|
export HINDSIGHT_API_LLM_PROVIDER=llamacpp
|
||||||
|
|
||||||
|
# Custom model with tuning
|
||||||
|
export HINDSIGHT_API_LLM_PROVIDER=llamacpp
|
||||||
|
export HINDSIGHT_API_LLM_MAX_CONCURRENT=2
|
||||||
|
export HINDSIGHT_API_LLAMACPP_MODEL_PATH=~/.hindsight/models/my-model.gguf
|
||||||
|
export HINDSIGHT_API_LLAMACPP_CONTEXT_SIZE=16384
|
||||||
|
export HINDSIGHT_API_LLAMACPP_NO_GRAMMAR=true # faster, less reliable JSON
|
||||||
|
export HINDSIGHT_API_LLAMACPP_EXTRA_ARGS="--n_threads 8"
|
||||||
|
```
|
||||||
|
|
||||||
|
:::note
|
||||||
|
The llama.cpp server is shared across all LLM operations (retain, reflect, consolidation). Set `HINDSIGHT_API_LLM_MAX_CONCURRENT=2` to allow retain and consolidation to run concurrently without blocking each other.
|
||||||
|
:::
|
||||||
|
|
||||||
### Per-Operation LLM Configuration
|
### Per-Operation LLM Configuration
|
||||||
|
|
||||||
Different memory operations have different requirements. **Retain** (fact extraction) benefits from models with strong structured output capabilities, while **Reflect** (reasoning/response generation) can use lighter, faster models. Configure separate LLM models for each operation to optimize for cost and performance.
|
Different memory operations have different requirements. **Retain** (fact extraction) benefits from models with strong structured output capabilities, while **Reflect** (reasoning/response generation) can use lighter, faster models. Configure separate LLM models for each operation to optimize for cost and performance.
|
||||||
|
|
|
||||||
|
|
@ -33,6 +33,11 @@ See [Configuration](./configuration#llm-provider) for setup examples.
|
||||||
Set `HINDSIGHT_API_LLM_PROVIDER=bedrock` to use AWS Bedrock models directly. Model names use Bedrock model IDs (e.g., `us.amazon.nova-2-lite-v1:0`). No API key is required — authentication uses AWS credentials (`AWS_ACCESS_KEY_ID`, `AWS_SECRET_ACCESS_KEY`, `AWS_REGION_NAME`) or IAM roles.
|
Set `HINDSIGHT_API_LLM_PROVIDER=bedrock` to use AWS Bedrock models directly. Model names use Bedrock model IDs (e.g., `us.amazon.nova-2-lite-v1:0`). No API key is required — authentication uses AWS credentials (`AWS_ACCESS_KEY_ID`, `AWS_SECRET_ACCESS_KEY`, `AWS_REGION_NAME`) or IAM roles.
|
||||||
|
|
||||||
See [Configuration](./configuration#llm-provider) for setup examples.
|
See [Configuration](./configuration#llm-provider) for setup examples.
|
||||||
|
> **💡 Built-in llama.cpp (fully local, no API key)**
|
||||||
|
>
|
||||||
|
Set `HINDSIGHT_API_LLM_PROVIDER=llamacpp` to run a built-in llama.cpp server with no external dependencies. A Gemma 4 E2B GGUF model (~3.5 GB) is auto-downloaded on first run. Requires the `local-llm` extra: `pip install 'hindsight-api-slim[local-llm]'`.
|
||||||
|
|
||||||
|
See [Configuration](./configuration#built-in-llamacpp) for all options.
|
||||||
> **💡 LiteLLM Provider (Azure, Together AI, and more)**
|
> **💡 LiteLLM Provider (Azure, Together AI, and more)**
|
||||||
>
|
>
|
||||||
Set `HINDSIGHT_API_LLM_PROVIDER=litellm` to use any model supported by [LiteLLM](https://docs.litellm.ai/docs/providers), including **Azure OpenAI**, **Together AI**, **Fireworks AI**, and many more. Model names use LiteLLM's provider prefix format (e.g., `azure/gpt-4o`).
|
Set `HINDSIGHT_API_LLM_PROVIDER=litellm` to use any model supported by [LiteLLM](https://docs.litellm.ai/docs/providers), including **Azure OpenAI**, **Together AI**, **Fireworks AI**, and many more. Model names use LiteLLM's provider prefix format (e.g., `azure/gpt-4o`).
|
||||||
|
|
@ -77,6 +82,7 @@ Each provider has a recommended default model that's used when `HINDSIGHT_API_LL
|
||||||
| `groq` | `openai/gpt-oss-120b` |
|
| `groq` | `openai/gpt-oss-120b` |
|
||||||
| `minimax` | `MiniMax-M2.7` |
|
| `minimax` | `MiniMax-M2.7` |
|
||||||
| `ollama` | `gemma3:12b` |
|
| `ollama` | `gemma3:12b` |
|
||||||
|
| `llamacpp` | `gemma-4-e2b-it` (auto-downloaded GGUF) |
|
||||||
| `lmstudio` | `local-model` |
|
| `lmstudio` | `local-model` |
|
||||||
| `vertexai` | `gemini-2.0-flash-001` |
|
| `vertexai` | `gemini-2.0-flash-001` |
|
||||||
| `openai-codex` | `gpt-5.2-codex` |
|
| `openai-codex` | `gpt-5.2-codex` |
|
||||||
|
|
|
||||||
Loading…
Reference in a new issue