* feat: introduce hindsight-api-slim and hindsight-all-slim packages Closes #552 - Move all source code from hindsight-api/ to new hindsight-api-slim/ - hindsight-api-slim has heavy ML deps (torch, sentence-transformers, transformers, einops, flashrank, mlx, mlx-lm, safetensors) and pg0-embedded as optional extras: [local-ml], [embedded-db], [all] - hindsight-api becomes a zero-code meta-package depending on hindsight-api-slim[all] for full backward compatibility - Add hindsight-all-slim meta-package: hindsight-api-slim + client + embed - hindsight-all updated to depend on hindsight-api-slim[all] - pg0.py: lazy-import pg0 with clear ImportError pointing to [embedded-db] - Dockerfile: replace sed hack with proper uv sync --extra flags - Update release.yml, test.yml, lint.sh, release.sh, CLAUDE.md and all path references throughout the repo * refactor: rename hindsight/ directory to hindsight-all/ * docs: document hindsight-api-slim and hindsight-all-slim package variants Add package variants table and extras explanation to installation.md * docs: remove emojis from installation.md, use professional tone * docs: link Docker slim variant to pip package variants section * docs: consolidate Docker image variants into single table * ci: fix working-directory paths after package restructure - Replace all hindsight-api → hindsight-api-slim in test.yml - Replace hindsight → hindsight-all in test.yml - Add --extra embedded-db to test-embed API install step * ci: add local-ml and embedded-db extras to API sync steps These extras were previously implicit in the old hindsight-api package (which bundled everything). Now that hindsight-api-slim uses optional extras, we must explicitly request local-ml and embedded-db in CI. * ci: add API install step with embedded-db to test-embed smoke test The smoke test starts hindsight-api as a daemon, which requires pg0-embedded. Add a dedicated install step for hindsight-api-slim with embedded-db extra so the daemon can start successfully. * ci: remove --no-install-project when using optional extras When --no-install-project is combined with --extra, the optional deps are not installed because extras require the project to be active. Remove --no-install-project from steps that need local-ml or embedded-db. * ci: fix ordering of uv sync steps to preserve optional extras When uv sync runs for a different workspace member, it removes optional extras installed for other members. Fix by always running extra-requiring API sync last, after other workspace member syncs. Also remove --no-install-project from embedded-db sync in test-embed, as --no-install-project prevents optional extras from being active. * ci: add local-ml extra to test-embed API install for smoke test The smoke test starts the full API server which needs sentence-transformers for local embeddings (default provider). Add local-ml extra to the install. * ci: simplify extras with --all-extras and add slim pip smoke test - Replace explicit --extra local-ml --extra embedded-db with --all-extras for cleaner, more maintainable sync steps - Add test-pip-slim job: tests hindsight-api-slim[embedded-db] without local ML models, using Cohere for embeddings/reranking (mirrors Docker slim smoke test approach) * ci: simplify slim smoke test to health check only (mirrors Docker test)
113 lines
4.1 KiB
Python
113 lines
4.1 KiB
Python
"""
|
|
Helper functions for hybrid search (semantic + BM25 + graph).
|
|
"""
|
|
|
|
from typing import Any
|
|
|
|
from .types import MergedCandidate, RetrievalResult
|
|
|
|
|
|
def reciprocal_rank_fusion(result_lists: list[list[RetrievalResult]], k: int = 60) -> list[MergedCandidate]:
|
|
"""
|
|
Merge multiple ranked result lists using Reciprocal Rank Fusion.
|
|
|
|
RRF formula: score(d) = sum_over_lists(1 / (k + rank(d)))
|
|
|
|
Args:
|
|
result_lists: List of result lists, each containing RetrievalResult objects
|
|
k: Constant for RRF formula (default: 60)
|
|
|
|
Returns:
|
|
Merged list of MergedCandidate objects, sorted by RRF score
|
|
|
|
Example:
|
|
semantic_results = [RetrievalResult(...), RetrievalResult(...), ...]
|
|
bm25_results = [RetrievalResult(...), RetrievalResult(...), ...]
|
|
graph_results = [RetrievalResult(...), RetrievalResult(...), ...]
|
|
|
|
merged = reciprocal_rank_fusion([semantic_results, bm25_results, graph_results])
|
|
# Returns: [MergedCandidate(...), MergedCandidate(...), ...]
|
|
"""
|
|
# Track scores from each list
|
|
rrf_scores = {}
|
|
source_ranks = {} # Track rank from each source for each doc_id
|
|
all_retrievals = {} # Store the actual RetrievalResult (use first occurrence)
|
|
|
|
source_names = ["semantic", "bm25", "graph", "temporal"]
|
|
|
|
for source_idx, results in enumerate(result_lists):
|
|
source_name = source_names[source_idx] if source_idx < len(source_names) else f"source_{source_idx}"
|
|
|
|
for rank, retrieval in enumerate(results, start=1):
|
|
# Type check to catch tuple issues
|
|
if isinstance(retrieval, tuple):
|
|
raise TypeError(
|
|
f"Expected RetrievalResult but got tuple in {source_name} results at rank {rank}. "
|
|
f"Tuple value: {retrieval[:2] if len(retrieval) >= 2 else retrieval}. "
|
|
f"This suggests the retrieval function returned tuples instead of RetrievalResult objects."
|
|
)
|
|
if not isinstance(retrieval, RetrievalResult):
|
|
raise TypeError(
|
|
f"Expected RetrievalResult but got {type(retrieval).__name__} in {source_name} results at rank {rank}"
|
|
)
|
|
doc_id = retrieval.id
|
|
|
|
# Store retrieval result (use first occurrence)
|
|
if doc_id not in all_retrievals:
|
|
all_retrievals[doc_id] = retrieval
|
|
|
|
# Calculate RRF score contribution
|
|
if doc_id not in rrf_scores:
|
|
rrf_scores[doc_id] = 0.0
|
|
source_ranks[doc_id] = {}
|
|
|
|
rrf_scores[doc_id] += 1.0 / (k + rank)
|
|
source_ranks[doc_id][f"{source_name}_rank"] = rank
|
|
|
|
# Combine into final results with metadata
|
|
merged_results = []
|
|
for rrf_rank, (doc_id, rrf_score) in enumerate(
|
|
sorted(rrf_scores.items(), key=lambda x: x[1], reverse=True), start=1
|
|
):
|
|
merged_candidate = MergedCandidate(
|
|
retrieval=all_retrievals[doc_id], rrf_score=rrf_score, rrf_rank=rrf_rank, source_ranks=source_ranks[doc_id]
|
|
)
|
|
merged_results.append(merged_candidate)
|
|
|
|
return merged_results
|
|
|
|
|
|
def normalize_scores_on_deltas(results: list[dict[str, Any]], score_keys: list[str]) -> list[dict[str, Any]]:
|
|
"""
|
|
Normalize scores based on deltas (min-max normalization within result set).
|
|
|
|
This ensures all scores are in [0, 1] range based on the spread in THIS result set.
|
|
|
|
Args:
|
|
results: List of result dicts
|
|
score_keys: Keys to normalize (e.g., ["recency", "frequency"])
|
|
|
|
Returns:
|
|
Results with normalized scores added as "{key}_normalized"
|
|
"""
|
|
for key in score_keys:
|
|
values = [r.get(key, 0.0) for r in results if key in r]
|
|
|
|
if not values:
|
|
continue
|
|
|
|
min_val = min(values)
|
|
max_val = max(values)
|
|
delta = max_val - min_val
|
|
|
|
if delta > 0:
|
|
for r in results:
|
|
if key in r:
|
|
r[f"{key}_normalized"] = (r[key] - min_val) / delta
|
|
else:
|
|
# All values are the same, set to 0.5
|
|
for r in results:
|
|
if key in r:
|
|
r[f"{key}_normalized"] = 0.5
|
|
|
|
return results
|