From c65c6a9dc042885268527e878baeb758801cbb30 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Nicol=C3=B2=20Boschi?= Date: Thu, 8 Jan 2026 12:14:41 +0100 Subject: [PATCH] feat: support for multilingual content (#124) * feat: support for multilingual content * feat: support for multilingual content --- .../engine/retain/fact_extraction.py | 3 + .../engine/search/think_utils.py | 2 +- hindsight-api/tests/test_multilingual.py | 286 ++++++++++++++++++ hindsight-docs/docs/developer/multilingual.md | 191 ++++++++++++ hindsight-docs/sidebars.ts | 5 + 5 files changed, 486 insertions(+), 1 deletion(-) create mode 100644 hindsight-api/tests/test_multilingual.py create mode 100644 hindsight-docs/docs/developer/multilingual.md diff --git a/hindsight-api/hindsight_api/engine/retain/fact_extraction.py b/hindsight-api/hindsight_api/engine/retain/fact_extraction.py index b51c52a5..af72f264 100644 --- a/hindsight-api/hindsight_api/engine/retain/fact_extraction.py +++ b/hindsight-api/hindsight_api/engine/retain/fact_extraction.py @@ -415,6 +415,9 @@ async def _extract_facts_from_chunk( prompt = f"""Extract facts from text into structured format with FOUR required dimensions - BE EXTREMELY DETAILED. +LANGUAGE REQUIREMENT: Detect the language of the input text. All extracted facts, entity names, descriptions, +and other output MUST be in the SAME language as the input. Do not translate to English if the input is in another language. + {fact_types_instruction} diff --git a/hindsight-api/hindsight_api/engine/search/think_utils.py b/hindsight-api/hindsight_api/engine/search/think_utils.py index f8722605..6980ec93 100644 --- a/hindsight-api/hindsight_api/engine/search/think_utils.py +++ b/hindsight-api/hindsight_api/engine/search/think_utils.py @@ -172,7 +172,7 @@ def get_system_message(disposition: DispositionTraits) -> str: " ".join(instructions) if instructions else "Balance your disposition traits when interpreting information." ) - return f"You are a person with your own thoughts, experiences, opinions, and disposition. When asked a question, respond naturally like a human would - share what you genuinely think based on what you know and have experienced. {disposition_instruction} Be direct, express your views confidently, and use 'I think', 'I believe', 'in my view', etc. Respond in plain text without markdown formatting." + return f"You are a person with your own thoughts, experiences, opinions, and disposition. When asked a question, respond naturally like a human would - share what you genuinely think based on what you know and have experienced. {disposition_instruction} Be direct, express your views confidently, and use 'I think', 'I believe', 'in my view', etc. Respond in plain text without markdown formatting. IMPORTANT: Detect the language of the question and respond in the SAME language. Do not translate to English if the question is in another language." async def extract_opinions_from_text(llm_config, text: str, query: str) -> list[Opinion]: diff --git a/hindsight-api/tests/test_multilingual.py b/hindsight-api/tests/test_multilingual.py new file mode 100644 index 00000000..fb3b42f8 --- /dev/null +++ b/hindsight-api/tests/test_multilingual.py @@ -0,0 +1,286 @@ +""" +Test multilingual support for retain and reflect operations. + +Tests that the system correctly handles non-English input and produces +output in the same language as the input. +""" + +import pytest +import logging +from datetime import datetime, timezone +from hindsight_api.engine.memory_engine import Budget +from hindsight_api import RequestContext + +logger = logging.getLogger(__name__) + + +@pytest.mark.asyncio +async def test_retain_chinese_content(memory, request_context): + """ + Test that retain correctly extracts facts from Chinese content + and keeps the output in Chinese. + + This test verifies: + 1. Facts are extracted from Chinese text + 2. The extracted facts contain Chinese characters + 3. Entity names are preserved in Chinese + """ + bank_id = f"test_chinese_retain_{datetime.now(timezone.utc).timestamp()}" + + try: + # Chinese content about a person and their activities + chinese_content = """ + 张伟是一位资深软件工程师,在腾讯工作了五年。他专门研究分布式系统, + 并领导了公司微服务架构的开发。他以编写干净、文档完善的代码而闻名。 + + 李明上个月加入团队担任初级开发人员。他正在学习React和Node.js。 + 李明很有热情,在代码审查中提出很好的问题。他最近完成了他的第一个功能, + 这是一个用户认证流程。 + + 团队使用Kubernetes进行容器编排,并部署到阿里云。他们遵循敏捷方法论, + 采用两周冲刺周期。合并前必须进行代码审查。 + """ + + # Retain the Chinese content + unit_ids = await memory.retain_async( + bank_id=bank_id, + content=chinese_content, + context="团队概述", # Chinese context + event_date=datetime(2024, 1, 15, tzinfo=timezone.utc), + request_context=request_context, + ) + + logger.info(f"Retained {len(unit_ids)} facts from Chinese content") + assert len(unit_ids) > 0, "Should have extracted and stored facts from Chinese content" + + # Recall the facts with a Chinese query + result = await memory.recall_async( + bank_id=bank_id, + query="告诉我关于张伟的信息", # "Tell me about Zhang Wei" + budget=Budget.MID, + max_tokens=1000, + fact_type=["world"], + request_context=request_context, + ) + + logger.info(f"Recalled {len(result.results)} facts") + assert len(result.results) > 0, "Should recall facts about Zhang Wei" + + # Verify that the facts contain Chinese characters + # At least one fact should mention 张伟 (Zhang Wei) or related Chinese content + chinese_facts_found = 0 + for fact in result.results: + logger.info(f"Fact: {fact.text[:100]}...") + # Check for common Chinese characters or the name + if any( + char in fact.text + for char in ["张", "伟", "腾讯", "软件", "工程师", "分布式", "系统", "代码"] + ): + chinese_facts_found += 1 + + logger.info(f"Found {chinese_facts_found} facts with Chinese content") + assert chinese_facts_found > 0, ( + f"Expected facts to contain Chinese characters, but none found. " + f"Facts: {[f.text for f in result.results]}" + ) + + logger.info("Chinese retain test passed - facts preserved in Chinese") + + finally: + await memory.delete_bank(bank_id, request_context=request_context) + + +@pytest.mark.asyncio +async def test_reflect_chinese_content(memory, request_context): + """ + Test that reflect correctly generates responses in Chinese + when given Chinese facts and a Chinese query. + + This test verifies: + 1. Reflection produces a response in Chinese + 2. The response references the Chinese facts + 3. Opinions are formed and expressed in Chinese + """ + bank_id = f"test_chinese_reflect_{datetime.now(timezone.utc).timestamp()}" + + try: + # Store some Chinese facts to give context for opinion formation + await memory.retain_async( + bank_id=bank_id, + content="张伟是一位优秀的软件工程师,完成了五个重大项目。他总是按时交付,代码整洁有良好的文档。", + context="绩效评估", # "Performance review" + event_date=datetime(2024, 1, 15, tzinfo=timezone.utc), + request_context=request_context, + ) + + await memory.retain_async( + bank_id=bank_id, + content="李明最近加入团队。他错过了第一个截止日期,代码有很多bug。", + context="绩效评估", + event_date=datetime(2024, 2, 1, tzinfo=timezone.utc), + request_context=request_context, + ) + + # Reflect with a Chinese query + query = "谁是更可靠的工程师?" # "Who is a more reliable engineer?" + result = await memory.reflect_async( + bank_id=bank_id, + query=query, + budget=Budget.LOW, + request_context=request_context, + ) + + logger.info(f"Reflection answer: {result.text}") + + # Verify we got an answer + assert result.text, "Reflection should return an answer" + + # Check that the response contains Chinese characters + # The response should be in Chinese, not English + chinese_chars_found = sum(1 for char in result.text if "\u4e00" <= char <= "\u9fff") + total_chars = len(result.text.replace(" ", "").replace("\n", "")) + + logger.info(f"Chinese characters: {chinese_chars_found}, Total characters: {total_chars}") + + # At least 30% of characters should be Chinese (allowing for numbers, punctuation) + chinese_ratio = chinese_chars_found / max(total_chars, 1) + assert chinese_ratio > 0.3, ( + f"Expected response to be in Chinese (>30% Chinese characters), " + f"but only {chinese_ratio:.1%} are Chinese. Response: {result.text}" + ) + + # Check that Chinese names are mentioned + assert "张伟" in result.text or "李明" in result.text, ( + f"Expected response to mention Chinese names 张伟 or 李明. Response: {result.text}" + ) + + logger.info("Chinese reflect test passed - response generated in Chinese") + + finally: + await memory.delete_bank(bank_id, request_context=request_context) + + +@pytest.mark.asyncio +async def test_retain_japanese_content(memory, request_context): + """ + Test that retain correctly handles Japanese content. + + This test verifies multilingual support extends beyond Chinese + to other non-Latin languages. + """ + bank_id = f"test_japanese_retain_{datetime.now(timezone.utc).timestamp()}" + + try: + # Japanese content about a developer + japanese_content = """ + 田中さんはソフトウェアエンジニアで、東京のスタートアップで働いています。 + 彼女はPythonとTypeScriptが得意で、毎日コードレビューをしています。 + 先週、新しいAPIを完成させました。 + """ + + unit_ids = await memory.retain_async( + bank_id=bank_id, + content=japanese_content, + context="チームプロフィール", # "Team profile" + event_date=datetime(2024, 1, 15, tzinfo=timezone.utc), + request_context=request_context, + ) + + logger.info(f"Retained {len(unit_ids)} facts from Japanese content") + assert len(unit_ids) > 0, "Should have extracted facts from Japanese content" + + # Recall with Japanese query + result = await memory.recall_async( + bank_id=bank_id, + query="田中さんについて教えてください", # "Tell me about Tanaka-san" + budget=Budget.MID, + max_tokens=1000, + fact_type=["world"], + request_context=request_context, + ) + + assert len(result.results) > 0, "Should recall facts about Tanaka" + + # Check for Japanese content in facts + japanese_facts_found = 0 + for fact in result.results: + logger.info(f"Fact: {fact.text[:100]}...") + # Check for Japanese characters (hiragana, katakana, or kanji) + if any( + ("\u3040" <= char <= "\u309f") # Hiragana + or ("\u30a0" <= char <= "\u30ff") # Katakana + or ("\u4e00" <= char <= "\u9fff") # Kanji + for char in fact.text + ): + japanese_facts_found += 1 + + assert japanese_facts_found > 0, ( + f"Expected facts to contain Japanese characters. " + f"Facts: {[f.text for f in result.results]}" + ) + + logger.info("Japanese retain test passed - facts preserved in Japanese") + + finally: + await memory.delete_bank(bank_id, request_context=request_context) + + +@pytest.mark.asyncio +async def test_mixed_language_entities(memory, request_context): + """ + Test that entity extraction works correctly with mixed language content. + + Some entities (like company names) might be in English while the + description is in Chinese. + """ + bank_id = f"test_mixed_lang_{datetime.now(timezone.utc).timestamp()}" + + try: + # Mixed language content - Chinese with English company names + mixed_content = """ + 王芳在Google北京办公室工作,她是一名高级产品经理。 + 之前她在Microsoft和Amazon工作过。 + 她负责管理YouTube在中国市场的推广策略。 + """ + + unit_ids = await memory.retain_async( + bank_id=bank_id, + content=mixed_content, + context="员工资料", + event_date=datetime(2024, 1, 15, tzinfo=timezone.utc), + request_context=request_context, + ) + + assert len(unit_ids) > 0, "Should extract facts from mixed language content" + + # Recall and check entities + result = await memory.recall_async( + bank_id=bank_id, + query="王芳在哪里工作?", # "Where does Wang Fang work?" + budget=Budget.MID, + max_tokens=1000, + fact_type=["world"], + include_entities=True, + request_context=request_context, + ) + + assert len(result.results) > 0, "Should recall facts about Wang Fang" + + # Check that both Chinese and English entities are preserved + all_text = " ".join(f.text for f in result.results) + logger.info(f"Combined facts: {all_text}") + + # Should contain Chinese name and/or English company names + has_chinese_name = "王芳" in all_text + has_english_company = any( + company in all_text for company in ["Google", "Microsoft", "Amazon", "YouTube"] + ) + + assert has_chinese_name or has_english_company, ( + f"Expected mixed language entities. Facts: {all_text}" + ) + + logger.info("Mixed language entity test passed") + + finally: + await memory.delete_bank(bank_id, request_context=request_context) diff --git a/hindsight-docs/docs/developer/multilingual.md b/hindsight-docs/docs/developer/multilingual.md new file mode 100644 index 00000000..5f6c63cf --- /dev/null +++ b/hindsight-docs/docs/developer/multilingual.md @@ -0,0 +1,191 @@ +--- +sidebar_position: 5 +--- + +# Multilingual Support + +Hindsight automatically detects the language of your input and responds in the same language. This means facts, entities, and reflections are preserved in their original language without translation to English. + +## How It Works + +```mermaid +graph LR + A[Chinese Input] --> B[Language Detection] + B --> C[Extract Facts in Chinese] + C --> D[Chinese Entities] + D --> E[Chinese Response] +``` + +When you retain content or reflect on a query, Hindsight: + +1. **Detects the input language** automatically from the content +2. **Extracts facts in the original language** - preserving nuance and meaning +3. **Stores entities in their native script** - 张伟 stays 张伟, not "Zhang Wei" +4. **Responds in the same language** - queries in Chinese get Chinese answers + +--- + +## Retain with Non-English Content + +When you retain content in any language, Hindsight extracts and stores facts in that same language. + +### Example: Chinese Content + +```python +from hindsight import Hindsight + +hindsight = Hindsight() + +# Retain Chinese content +hindsight.retain( + bank_id="user-123", + content=""" + 张伟是一位资深软件工程师,在腾讯工作了五年。 + 他专门研究分布式系统,并领导了公司微服务架构的开发。 + """, + context="团队概述" +) + +# Query in Chinese - get Chinese results +results = hindsight.recall( + bank_id="user-123", + query="告诉我关于张伟的信息" +) + +# Facts are returned in Chinese: +# - 张伟是一位资深软件工程师,在腾讯工作了五年 +# - 张伟专门研究分布式系统,并领导了公司微服务架构的开发 +``` + +### Example: Japanese Content + +```python +hindsight.retain( + bank_id="user-123", + content=""" + 田中さんはソフトウェアエンジニアで、東京のスタートアップで働いています。 + 彼女はPythonとTypeScriptが得意で、毎日コードレビューをしています。 + """, + context="チームプロフィール" +) + +# Query in Japanese +results = hindsight.recall( + bank_id="user-123", + query="田中さんについて教えてください" +) +``` + +--- + +## Reflect with Non-English Queries + +The `reflect` operation also respects the input language, generating thoughtful responses in the same language as the query. + +### Example: Chinese Reflection + +```python +# Store facts about team members (in Chinese) +hindsight.retain( + bank_id="team-eval", + content="张伟是一位优秀的软件工程师,完成了五个重大项目。他总是按时交付,代码整洁有良好的文档。", + context="绩效评估" +) + +hindsight.retain( + bank_id="team-eval", + content="李明最近加入团队。他错过了第一个截止日期,代码有很多bug。", + context="绩效评估" +) + +# Reflect in Chinese +result = hindsight.reflect( + bank_id="team-eval", + query="谁是更可靠的工程师?" +) + +# Response is in Chinese: +# "我认为张伟更可靠。张伟完成了五个重大项目,按时交付,代码质量高..." +``` + +--- + +## Mixed Language Content + +Hindsight handles mixed-language content gracefully, preserving both languages where appropriate. + +### Example: Chinese Text with English Company Names + +```python +hindsight.retain( + bank_id="user-123", + content=""" + 王芳在Google北京办公室工作,她是一名高级产品经理。 + 之前她在Microsoft和Amazon工作过。 + 她负责管理YouTube在中国市场的推广策略。 + """, + context="员工资料" +) + +# Facts preserve both languages: +# - 王芳在Google北京办公室工作,担任高级产品经理 +# - 王芳曾在Microsoft和Amazon工作过 +# - 王芳负责管理YouTube在中国市场的推广策略 +``` + +--- + +## Supported Languages + +Hindsight supports any language that your configured LLM can understand. This typically includes: + +| Language | Script | Example | +|----------|--------|---------| +| Chinese (Simplified) | 简体中文 | 张伟是软件工程师 | +| Chinese (Traditional) | 繁體中文 | 張偉是軟體工程師 | +| Japanese | 日本語 | 田中さんはエンジニアです | +| Korean | 한국어 | 김철수는 개발자입니다 | +| Arabic | العربية | أحمد مهندس برمجيات | +| Russian | Русский | Иван - разработчик | +| Spanish | Español | María es ingeniera | +| French | Français | Pierre est développeur | +| German | Deutsch | Hans ist Entwickler | +| And many more... | | | + +The actual language support depends on your LLM provider's capabilities. + +--- + +## Best Practices + +### 1. Keep Content in One Language Per Retain Call +While mixed content works, keeping each `retain` call in a single language produces more consistent results. + +### 2. Query in the Same Language as Your Content +For best results, query using the same language as your stored content. Cross-language queries (e.g., English query for Chinese content) may work but results can vary. + +### 3. Consider Embedding Model Language Support +The default embedding model (`BAAI/bge-small-en-v1.5`) is English-optimized. For better multilingual semantic search, consider using a multilingual embedding model: + +```bash +# In your .env file +HINDSIGHT_API_EMBEDDINGS_LOCAL_MODEL=BAAI/bge-m3 +``` + +The `bge-m3` model supports 100+ languages with better cross-lingual retrieval. + +--- + +## Technical Details + +Multilingual support is implemented through LLM prompt instructions rather than external language detection libraries. This approach: + +- **Requires no additional dependencies** +- **Works with any LLM** that supports multiple languages +- **Handles edge cases** like mixed-language content naturally +- **Preserves semantic meaning** better than rule-based translation + +The LLM is instructed to: +1. Detect the input language +2. Extract all facts, entities, and descriptions in that same language +3. Never translate to English unless the input is in English diff --git a/hindsight-docs/sidebars.ts b/hindsight-docs/sidebars.ts index a392a314..392cf358 100644 --- a/hindsight-docs/sidebars.ts +++ b/hindsight-docs/sidebars.ts @@ -27,6 +27,11 @@ const sidebars: SidebarsConfig = { id: 'developer/reflect', label: 'Reflect', }, + { + type: 'doc', + id: 'developer/multilingual', + label: 'Multilingual', + }, { type: 'doc', id: 'developer/performance',