feat: include doc metadata in fact extraction (#424)
This commit is contained in:
parent
9f0c031df7
commit
87219b731d
5 changed files with 117 additions and 5 deletions
|
|
@ -733,7 +733,14 @@ def _build_extraction_prompt_and_schema(config) -> tuple[str, type]:
|
||||||
return prompt, response_schema
|
return prompt, response_schema
|
||||||
|
|
||||||
|
|
||||||
def _build_user_message(chunk: str, chunk_index: int, total_chunks: int, event_date: datetime, context: str) -> str:
|
def _build_user_message(
|
||||||
|
chunk: str,
|
||||||
|
chunk_index: int,
|
||||||
|
total_chunks: int,
|
||||||
|
event_date: datetime,
|
||||||
|
context: str,
|
||||||
|
metadata: dict[str, str] | None = None,
|
||||||
|
) -> str:
|
||||||
"""Build user message for fact extraction."""
|
"""Build user message for fact extraction."""
|
||||||
from .orchestrator import parse_datetime_flexible
|
from .orchestrator import parse_datetime_flexible
|
||||||
|
|
||||||
|
|
@ -742,11 +749,16 @@ def _build_user_message(chunk: str, chunk_index: int, total_chunks: int, event_d
|
||||||
event_date = parse_datetime_flexible(event_date)
|
event_date = parse_datetime_flexible(event_date)
|
||||||
event_date_formatted = event_date.strftime("%A, %B %d, %Y")
|
event_date_formatted = event_date.strftime("%A, %B %d, %Y")
|
||||||
|
|
||||||
|
metadata_section = ""
|
||||||
|
if metadata:
|
||||||
|
metadata_lines = "\n".join(f" {k}: {v}" for k, v in metadata.items())
|
||||||
|
metadata_section = f"\nMetadata:\n{metadata_lines}"
|
||||||
|
|
||||||
return f"""Extract facts from the following text chunk.
|
return f"""Extract facts from the following text chunk.
|
||||||
|
|
||||||
Chunk: {chunk_index + 1}/{total_chunks}
|
Chunk: {chunk_index + 1}/{total_chunks}
|
||||||
Event Date: {event_date_formatted} ({event_date.isoformat()})
|
Event Date: {event_date_formatted} ({event_date.isoformat()})
|
||||||
Context: {sanitized_context}
|
Context: {sanitized_context}{metadata_section}
|
||||||
|
|
||||||
Text:
|
Text:
|
||||||
{sanitized_chunk}"""
|
{sanitized_chunk}"""
|
||||||
|
|
@ -788,6 +800,7 @@ async def _extract_facts_from_chunk(
|
||||||
llm_config: "LLMConfig",
|
llm_config: "LLMConfig",
|
||||||
config,
|
config,
|
||||||
agent_name: str = None,
|
agent_name: str = None,
|
||||||
|
metadata: dict[str, str] | None = None,
|
||||||
) -> tuple[list[dict[str, str]], TokenUsage]:
|
) -> tuple[list[dict[str, str]], TokenUsage]:
|
||||||
"""
|
"""
|
||||||
Extract facts from a single chunk (internal helper for parallel processing).
|
Extract facts from a single chunk (internal helper for parallel processing).
|
||||||
|
|
@ -809,7 +822,7 @@ async def _extract_facts_from_chunk(
|
||||||
extract_causal_links = config.retain_extract_causal_links
|
extract_causal_links = config.retain_extract_causal_links
|
||||||
|
|
||||||
# Build user message using helper function
|
# Build user message using helper function
|
||||||
user_message = _build_user_message(chunk, chunk_index, total_chunks, event_date, context)
|
user_message = _build_user_message(chunk, chunk_index, total_chunks, event_date, context, metadata)
|
||||||
|
|
||||||
# Retry logic for JSON validation errors
|
# Retry logic for JSON validation errors
|
||||||
max_retries = 2
|
max_retries = 2
|
||||||
|
|
@ -1089,6 +1102,7 @@ async def _extract_facts_with_auto_split(
|
||||||
llm_config: LLMConfig,
|
llm_config: LLMConfig,
|
||||||
config,
|
config,
|
||||||
agent_name: str = None,
|
agent_name: str = None,
|
||||||
|
metadata: dict[str, str] | None = None,
|
||||||
) -> tuple[list[dict[str, str]], TokenUsage]:
|
) -> tuple[list[dict[str, str]], TokenUsage]:
|
||||||
"""
|
"""
|
||||||
Extract facts from a chunk with automatic splitting if output exceeds token limits.
|
Extract facts from a chunk with automatic splitting if output exceeds token limits.
|
||||||
|
|
@ -1105,6 +1119,7 @@ async def _extract_facts_with_auto_split(
|
||||||
llm_config: LLM configuration to use
|
llm_config: LLM configuration to use
|
||||||
config: Resolved HindsightConfig for this bank
|
config: Resolved HindsightConfig for this bank
|
||||||
agent_name: Optional agent name (memory owner)
|
agent_name: Optional agent name (memory owner)
|
||||||
|
metadata: Optional document metadata key-value pairs
|
||||||
|
|
||||||
Returns:
|
Returns:
|
||||||
Tuple of (facts list, token usage) extracted from the chunk (possibly from sub-chunks)
|
Tuple of (facts list, token usage) extracted from the chunk (possibly from sub-chunks)
|
||||||
|
|
@ -1124,6 +1139,7 @@ async def _extract_facts_with_auto_split(
|
||||||
llm_config=llm_config,
|
llm_config=llm_config,
|
||||||
config=config,
|
config=config,
|
||||||
agent_name=agent_name,
|
agent_name=agent_name,
|
||||||
|
metadata=metadata,
|
||||||
)
|
)
|
||||||
except OutputTooLongError:
|
except OutputTooLongError:
|
||||||
# Output exceeded token limits - split the chunk in half and retry
|
# Output exceeded token limits - split the chunk in half and retry
|
||||||
|
|
@ -1169,6 +1185,7 @@ async def _extract_facts_with_auto_split(
|
||||||
llm_config=llm_config,
|
llm_config=llm_config,
|
||||||
config=config,
|
config=config,
|
||||||
agent_name=agent_name,
|
agent_name=agent_name,
|
||||||
|
metadata=metadata,
|
||||||
),
|
),
|
||||||
_extract_facts_with_auto_split(
|
_extract_facts_with_auto_split(
|
||||||
chunk=second_half,
|
chunk=second_half,
|
||||||
|
|
@ -1179,6 +1196,7 @@ async def _extract_facts_with_auto_split(
|
||||||
llm_config=llm_config,
|
llm_config=llm_config,
|
||||||
config=config,
|
config=config,
|
||||||
agent_name=agent_name,
|
agent_name=agent_name,
|
||||||
|
metadata=metadata,
|
||||||
),
|
),
|
||||||
]
|
]
|
||||||
|
|
||||||
|
|
@ -1203,6 +1221,7 @@ async def extract_facts_from_text(
|
||||||
agent_name: str,
|
agent_name: str,
|
||||||
config,
|
config,
|
||||||
context: str = "",
|
context: str = "",
|
||||||
|
metadata: dict[str, str] | None = None,
|
||||||
) -> tuple[list[Fact], list[tuple[str, int]], TokenUsage]:
|
) -> tuple[list[Fact], list[tuple[str, int]], TokenUsage]:
|
||||||
"""
|
"""
|
||||||
Extract semantic facts from conversational or narrative text using LLM.
|
Extract semantic facts from conversational or narrative text using LLM.
|
||||||
|
|
@ -1220,6 +1239,7 @@ async def extract_facts_from_text(
|
||||||
agent_name: Agent name (memory owner)
|
agent_name: Agent name (memory owner)
|
||||||
config: Resolved HindsightConfig for this bank
|
config: Resolved HindsightConfig for this bank
|
||||||
context: Context about the conversation/document
|
context: Context about the conversation/document
|
||||||
|
metadata: Optional document metadata key-value pairs
|
||||||
|
|
||||||
Returns:
|
Returns:
|
||||||
Tuple of (facts, chunks, usage) where:
|
Tuple of (facts, chunks, usage) where:
|
||||||
|
|
@ -1247,6 +1267,7 @@ async def extract_facts_from_text(
|
||||||
llm_config=llm_config,
|
llm_config=llm_config,
|
||||||
config=config,
|
config=config,
|
||||||
agent_name=agent_name,
|
agent_name=agent_name,
|
||||||
|
metadata=metadata,
|
||||||
)
|
)
|
||||||
for i, chunk in enumerate(chunks)
|
for i, chunk in enumerate(chunks)
|
||||||
]
|
]
|
||||||
|
|
@ -1356,7 +1377,7 @@ async def extract_facts_from_contents_batch_api(
|
||||||
|
|
||||||
# Build user message using helper function
|
# Build user message using helper function
|
||||||
user_message = _build_user_message(
|
user_message = _build_user_message(
|
||||||
chunk, chunk_index_in_content, len(chunks), item.event_date, item.context
|
chunk, chunk_index_in_content, len(chunks), item.event_date, item.context, item.metadata or None
|
||||||
)
|
)
|
||||||
|
|
||||||
# Build request body using helper function
|
# Build request body using helper function
|
||||||
|
|
@ -1736,6 +1757,7 @@ async def extract_facts_from_contents(
|
||||||
llm_config=llm_config,
|
llm_config=llm_config,
|
||||||
agent_name=agent_name,
|
agent_name=agent_name,
|
||||||
config=config,
|
config=config,
|
||||||
|
metadata=item.metadata or None,
|
||||||
)
|
)
|
||||||
fact_extraction_tasks.append(task)
|
fact_extraction_tasks.append(task)
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -231,6 +231,8 @@ def main():
|
||||||
reranker_litellm_sdk_api_key=config.reranker_litellm_sdk_api_key,
|
reranker_litellm_sdk_api_key=config.reranker_litellm_sdk_api_key,
|
||||||
reranker_litellm_sdk_model=config.reranker_litellm_sdk_model,
|
reranker_litellm_sdk_model=config.reranker_litellm_sdk_model,
|
||||||
reranker_litellm_sdk_api_base=config.reranker_litellm_sdk_api_base,
|
reranker_litellm_sdk_api_base=config.reranker_litellm_sdk_api_base,
|
||||||
|
reranker_zeroentropy_api_key=config.reranker_zeroentropy_api_key,
|
||||||
|
reranker_zeroentropy_model=config.reranker_zeroentropy_model,
|
||||||
host=args.host,
|
host=args.host,
|
||||||
port=args.port,
|
port=args.port,
|
||||||
base_path=config.base_path,
|
base_path=config.base_path,
|
||||||
|
|
|
||||||
61
hindsight-api/tests/test_fact_extraction_metadata.py
Normal file
61
hindsight-api/tests/test_fact_extraction_metadata.py
Normal file
|
|
@ -0,0 +1,61 @@
|
||||||
|
"""
|
||||||
|
Unit tests for metadata inclusion in fact extraction LLM prompt.
|
||||||
|
"""
|
||||||
|
from datetime import datetime
|
||||||
|
|
||||||
|
from hindsight_api.engine.retain.fact_extraction import _build_user_message
|
||||||
|
|
||||||
|
|
||||||
|
def test_build_user_message_includes_metadata():
|
||||||
|
"""Metadata key-value pairs should appear in the user message."""
|
||||||
|
event_date = datetime(2024, 6, 15, 12, 0, 0)
|
||||||
|
metadata = {"title": "Q2 Planning Doc", "source": "confluence", "author": "Alice"}
|
||||||
|
|
||||||
|
msg = _build_user_message(
|
||||||
|
chunk="Some content.",
|
||||||
|
chunk_index=0,
|
||||||
|
total_chunks=1,
|
||||||
|
event_date=event_date,
|
||||||
|
context="planning meeting",
|
||||||
|
metadata=metadata,
|
||||||
|
)
|
||||||
|
|
||||||
|
assert "title" in msg
|
||||||
|
assert "Q2 Planning Doc" in msg
|
||||||
|
assert "source" in msg
|
||||||
|
assert "confluence" in msg
|
||||||
|
assert "author" in msg
|
||||||
|
assert "Alice" in msg
|
||||||
|
|
||||||
|
|
||||||
|
def test_build_user_message_no_metadata():
|
||||||
|
"""When metadata is empty, the message should still be valid and not include a metadata section."""
|
||||||
|
event_date = datetime(2024, 6, 15, 12, 0, 0)
|
||||||
|
|
||||||
|
msg = _build_user_message(
|
||||||
|
chunk="Some content.",
|
||||||
|
chunk_index=0,
|
||||||
|
total_chunks=1,
|
||||||
|
event_date=event_date,
|
||||||
|
context="planning meeting",
|
||||||
|
metadata={},
|
||||||
|
)
|
||||||
|
|
||||||
|
assert "Some content." in msg
|
||||||
|
assert "Metadata:" not in msg
|
||||||
|
|
||||||
|
|
||||||
|
def test_build_user_message_without_metadata_arg():
|
||||||
|
"""Calling without metadata (default) should behave the same as empty metadata."""
|
||||||
|
event_date = datetime(2024, 6, 15, 12, 0, 0)
|
||||||
|
|
||||||
|
msg = _build_user_message(
|
||||||
|
chunk="Some content.",
|
||||||
|
chunk_index=0,
|
||||||
|
total_chunks=1,
|
||||||
|
event_date=event_date,
|
||||||
|
context="none",
|
||||||
|
)
|
||||||
|
|
||||||
|
assert "Some content." in msg
|
||||||
|
assert "Metadata:" not in msg
|
||||||
|
|
@ -90,7 +90,7 @@ Providing context consistently is one of the highest-leverage things you can do
|
||||||
|
|
||||||
### metadata
|
### metadata
|
||||||
|
|
||||||
Arbitrary key-value string pairs attached to every fact extracted from this item. For example: `{"source": "slack", "channel": "engineering", "thread_id": "T123"}`. The LLM never sees this field — it is passed through as-is and stored on each memory unit. During recall, every returned memory includes its metadata, which lets you do client-side filtering or static enrichment without extra lookups — for example, linking a memory back to its source URL, thread ID, or any application-specific identifier.
|
Arbitrary key-value string pairs that provide context about this item. For example: `{"source": "slack", "channel": "engineering", "thread_id": "T123"}`. Metadata is included in the fact extraction prompt, so the LLM can use it as additional context when extracting facts — for instance, knowing the document title or source can improve accuracy. It is also stored on each memory unit and returned with every recalled memory, letting you do client-side filtering or static enrichment without extra lookups — for example, linking a memory back to its source URL, thread ID, or any application-specific identifier.
|
||||||
|
|
||||||
### document_id
|
### document_id
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -176,6 +176,33 @@ See [Performance](/developer/performance) for tuning options.
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
### Does Hindsight support metadata filtering?
|
||||||
|
|
||||||
|
Yes — through **Tags**. Tags are string labels attached to memories at retain time and used as a visibility filter at recall/reflect time. Only memories tagged with a matching value are returned.
|
||||||
|
|
||||||
|
```python
|
||||||
|
# Tag memories at retain time
|
||||||
|
client.retain(bank_id="my-bank", items=[{
|
||||||
|
"content": "...",
|
||||||
|
"tags": ["user:alice"],
|
||||||
|
}])
|
||||||
|
|
||||||
|
# Filter by tag at recall time
|
||||||
|
client.recall(bank_id="my-bank", query="...", tags=["user:alice"])
|
||||||
|
```
|
||||||
|
|
||||||
|
See [Tags](/developer/api/retain#tags-and-document_tags) for full details including document-level tagging.
|
||||||
|
|
||||||
|
**What about document `metadata`?**
|
||||||
|
|
||||||
|
Document metadata (the `metadata` key-value pairs on a retain item) serves a different purpose. It is:
|
||||||
|
- **Included in the fact extraction prompt**, so the LLM can use it as additional context when extracting facts — for example, knowing the document title or source can improve accuracy.
|
||||||
|
- **Returned with every recalled memory** as-is, so your application can link memories back to source systems (e.g. a URL, thread ID, or ticket number) without extra lookups.
|
||||||
|
|
||||||
|
Metadata is not a filter — use tags when you need recall to be scoped to a subset of documents.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
## Still have questions?
|
## Still have questions?
|
||||||
|
|
||||||
Join our [Slack community](https://join.slack.com/t/hindsight-space/shared_invite/zt-3nhbm4w29-LeSJ5Ixi6j8PdiYOCPlOgg) or report issues on [GitHub](https://github.com/vectorize-io/hindsight/issues).
|
Join our [Slack community](https://join.slack.com/t/hindsight-space/shared_invite/zt-3nhbm4w29-LeSJ5Ixi6j8PdiYOCPlOgg) or report issues on [GitHub](https://github.com/vectorize-io/hindsight/issues).
|
||||||
|
|
|
||||||
Loading…
Reference in a new issue