* feat: introduce hindsight-api-slim and hindsight-all-slim packages Closes #552 - Move all source code from hindsight-api/ to new hindsight-api-slim/ - hindsight-api-slim has heavy ML deps (torch, sentence-transformers, transformers, einops, flashrank, mlx, mlx-lm, safetensors) and pg0-embedded as optional extras: [local-ml], [embedded-db], [all] - hindsight-api becomes a zero-code meta-package depending on hindsight-api-slim[all] for full backward compatibility - Add hindsight-all-slim meta-package: hindsight-api-slim + client + embed - hindsight-all updated to depend on hindsight-api-slim[all] - pg0.py: lazy-import pg0 with clear ImportError pointing to [embedded-db] - Dockerfile: replace sed hack with proper uv sync --extra flags - Update release.yml, test.yml, lint.sh, release.sh, CLAUDE.md and all path references throughout the repo * refactor: rename hindsight/ directory to hindsight-all/ * docs: document hindsight-api-slim and hindsight-all-slim package variants Add package variants table and extras explanation to installation.md * docs: remove emojis from installation.md, use professional tone * docs: link Docker slim variant to pip package variants section * docs: consolidate Docker image variants into single table * ci: fix working-directory paths after package restructure - Replace all hindsight-api → hindsight-api-slim in test.yml - Replace hindsight → hindsight-all in test.yml - Add --extra embedded-db to test-embed API install step * ci: add local-ml and embedded-db extras to API sync steps These extras were previously implicit in the old hindsight-api package (which bundled everything). Now that hindsight-api-slim uses optional extras, we must explicitly request local-ml and embedded-db in CI. * ci: add API install step with embedded-db to test-embed smoke test The smoke test starts hindsight-api as a daemon, which requires pg0-embedded. Add a dedicated install step for hindsight-api-slim with embedded-db extra so the daemon can start successfully. * ci: remove --no-install-project when using optional extras When --no-install-project is combined with --extra, the optional deps are not installed because extras require the project to be active. Remove --no-install-project from steps that need local-ml or embedded-db. * ci: fix ordering of uv sync steps to preserve optional extras When uv sync runs for a different workspace member, it removes optional extras installed for other members. Fix by always running extra-requiring API sync last, after other workspace member syncs. Also remove --no-install-project from embedded-db sync in test-embed, as --no-install-project prevents optional extras from being active. * ci: add local-ml extra to test-embed API install for smoke test The smoke test starts the full API server which needs sentence-transformers for local embeddings (default provider). Add local-ml extra to the install. * ci: simplify extras with --all-extras and add slim pip smoke test - Replace explicit --extra local-ml --extra embedded-db with --all-extras for cleaner, more maintainable sync steps - Add test-pip-slim job: tests hindsight-api-slim[embedded-db] without local ML models, using Cohere for embeddings/reranking (mirrors Docker slim smoke test approach) * ci: simplify slim smoke test to health check only (mirrors Docker test)
153 lines
4.6 KiB
Python
153 lines
4.6 KiB
Python
"""PostgreSQL BYTEA-based file storage (default, zero-config)."""
|
|
|
|
import logging
|
|
from collections.abc import Callable
|
|
from typing import TYPE_CHECKING
|
|
|
|
if TYPE_CHECKING:
|
|
import asyncpg
|
|
|
|
from .base import FileStorage
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
|
|
def fq_table(table: str, schema: str | None = None) -> str:
|
|
"""Get fully-qualified table name with optional schema prefix."""
|
|
if schema:
|
|
return f'"{schema}".{table}'
|
|
return table
|
|
|
|
|
|
class PostgreSQLFileStorage(FileStorage):
|
|
"""
|
|
PostgreSQL BYTEA-based file storage.
|
|
|
|
Stores files directly in PostgreSQL using BYTEA columns.
|
|
This is the default storage backend - zero configuration required!
|
|
|
|
Pros:
|
|
- Works out of the box (no external dependencies)
|
|
- Transactional consistency with database
|
|
- Simple backups (included in pg_dump)
|
|
- Good performance for <10MB files
|
|
|
|
Cons:
|
|
- Database bloat for large/many files
|
|
- Not ideal for distributed deployments
|
|
- Higher cost than object storage at scale
|
|
|
|
For production/scale, consider S3FileStorage instead.
|
|
"""
|
|
|
|
def __init__(
|
|
self,
|
|
pool_getter: Callable[[], "asyncpg.Pool"],
|
|
schema: str | None = None,
|
|
schema_getter: Callable[[], str] | None = None,
|
|
):
|
|
"""
|
|
Initialize PostgreSQL file storage.
|
|
|
|
Args:
|
|
pool_getter: Function that returns asyncpg connection pool
|
|
schema: Static database schema (fallback for single-tenant / tests)
|
|
schema_getter: Callable returning current schema at query time (for multi-tenant)
|
|
"""
|
|
self._pool_getter = pool_getter
|
|
self._static_schema = schema
|
|
self._schema_getter = schema_getter
|
|
|
|
@property
|
|
def _schema(self) -> str | None:
|
|
"""Resolve schema dynamically per-request when schema_getter is provided."""
|
|
if self._schema_getter:
|
|
return self._schema_getter()
|
|
return self._static_schema
|
|
|
|
async def store(
|
|
self,
|
|
file_data: bytes,
|
|
key: str,
|
|
metadata: dict[str, str] | None = None,
|
|
) -> str:
|
|
"""Store file in PostgreSQL."""
|
|
pool = self._pool_getter()
|
|
|
|
async with pool.acquire() as conn:
|
|
await conn.execute(
|
|
f"""
|
|
INSERT INTO {fq_table("file_storage", self._schema)}
|
|
(storage_key, data)
|
|
VALUES ($1, $2)
|
|
ON CONFLICT (storage_key) DO UPDATE SET
|
|
data = EXCLUDED.data
|
|
""",
|
|
key,
|
|
file_data,
|
|
)
|
|
|
|
logger.debug(f"Stored file {key} ({len(file_data)} bytes) in PostgreSQL")
|
|
return key
|
|
|
|
async def retrieve(self, key: str) -> bytes:
|
|
"""Retrieve file from PostgreSQL."""
|
|
pool = self._pool_getter()
|
|
|
|
async with pool.acquire() as conn:
|
|
row = await conn.fetchrow(
|
|
f"""
|
|
SELECT data FROM {fq_table("file_storage", self._schema)}
|
|
WHERE storage_key = $1
|
|
""",
|
|
key,
|
|
)
|
|
|
|
if not row:
|
|
raise FileNotFoundError(f"File not found: {key}")
|
|
|
|
return bytes(row["data"])
|
|
|
|
async def delete(self, key: str) -> None:
|
|
"""Delete file from PostgreSQL."""
|
|
pool = self._pool_getter()
|
|
|
|
async with pool.acquire() as conn:
|
|
result = await conn.execute(
|
|
f"""
|
|
DELETE FROM {fq_table("file_storage", self._schema)}
|
|
WHERE storage_key = $1
|
|
""",
|
|
key,
|
|
)
|
|
|
|
# Check if anything was deleted
|
|
if result == "DELETE 0":
|
|
logger.warning(f"Attempted to delete non-existent file: {key}")
|
|
|
|
async def exists(self, key: str) -> bool:
|
|
"""Check if file exists in PostgreSQL."""
|
|
pool = self._pool_getter()
|
|
|
|
async with pool.acquire() as conn:
|
|
row = await conn.fetchrow(
|
|
f"""
|
|
SELECT 1 FROM {fq_table("file_storage", self._schema)}
|
|
WHERE storage_key = $1
|
|
""",
|
|
key,
|
|
)
|
|
|
|
return row is not None
|
|
|
|
async def get_download_url(self, key: str, expires_in: int = 3600) -> str:
|
|
"""
|
|
Get download URL for PostgreSQL-stored file.
|
|
|
|
Returns an API endpoint path (not a pre-signed URL since the file
|
|
is stored in the database). The expires_in parameter is ignored
|
|
for PostgreSQL storage.
|
|
"""
|
|
# Return API path for download endpoint
|
|
# (expires_in ignored for database storage - auth handled at API level)
|
|
return f"/v1/default/files/download/{key}"
|