* feat: introduce hindsight-api-slim and hindsight-all-slim packages Closes #552 - Move all source code from hindsight-api/ to new hindsight-api-slim/ - hindsight-api-slim has heavy ML deps (torch, sentence-transformers, transformers, einops, flashrank, mlx, mlx-lm, safetensors) and pg0-embedded as optional extras: [local-ml], [embedded-db], [all] - hindsight-api becomes a zero-code meta-package depending on hindsight-api-slim[all] for full backward compatibility - Add hindsight-all-slim meta-package: hindsight-api-slim + client + embed - hindsight-all updated to depend on hindsight-api-slim[all] - pg0.py: lazy-import pg0 with clear ImportError pointing to [embedded-db] - Dockerfile: replace sed hack with proper uv sync --extra flags - Update release.yml, test.yml, lint.sh, release.sh, CLAUDE.md and all path references throughout the repo * refactor: rename hindsight/ directory to hindsight-all/ * docs: document hindsight-api-slim and hindsight-all-slim package variants Add package variants table and extras explanation to installation.md * docs: remove emojis from installation.md, use professional tone * docs: link Docker slim variant to pip package variants section * docs: consolidate Docker image variants into single table * ci: fix working-directory paths after package restructure - Replace all hindsight-api → hindsight-api-slim in test.yml - Replace hindsight → hindsight-all in test.yml - Add --extra embedded-db to test-embed API install step * ci: add local-ml and embedded-db extras to API sync steps These extras were previously implicit in the old hindsight-api package (which bundled everything). Now that hindsight-api-slim uses optional extras, we must explicitly request local-ml and embedded-db in CI. * ci: add API install step with embedded-db to test-embed smoke test The smoke test starts hindsight-api as a daemon, which requires pg0-embedded. Add a dedicated install step for hindsight-api-slim with embedded-db extra so the daemon can start successfully. * ci: remove --no-install-project when using optional extras When --no-install-project is combined with --extra, the optional deps are not installed because extras require the project to be active. Remove --no-install-project from steps that need local-ml or embedded-db. * ci: fix ordering of uv sync steps to preserve optional extras When uv sync runs for a different workspace member, it removes optional extras installed for other members. Fix by always running extra-requiring API sync last, after other workspace member syncs. Also remove --no-install-project from embedded-db sync in test-embed, as --no-install-project prevents optional extras from being active. * ci: add local-ml extra to test-embed API install for smoke test The smoke test starts the full API server which needs sentence-transformers for local embeddings (default provider). Add local-ml extra to the install. * ci: simplify extras with --all-extras and add slim pip smoke test - Replace explicit --extra local-ml --extra embedded-db with --all-extras for cleaner, more maintainable sync steps - Add test-pip-slim job: tests hindsight-api-slim[embedded-db] without local ML models, using Cohere for embeddings/reranking (mirrors Docker slim smoke test approach) * ci: simplify slim smoke test to health check only (mirrors Docker test)
321 lines
12 KiB
Python
321 lines
12 KiB
Python
"""
|
|
Reproduce issue #520: Reflect fails with LM Studio due to unsupported tool_choice format.
|
|
|
|
The reflect agent forces tool selection via named tool_choice dicts on the first few iterations:
|
|
{"type": "function", "function": {"name": "search_mental_models"}}
|
|
|
|
LM Studio (and Ollama) reject this format with HTTP 400:
|
|
"Tool choice of type 'function' is not supported. Use 'auto', 'none', or 'required'."
|
|
|
|
The fix should convert named tool_choice to "required" and filter the tools list
|
|
to only the requested tool for providers that don't support named tool_choice.
|
|
"""
|
|
|
|
import json
|
|
from unittest.mock import AsyncMock, MagicMock, patch
|
|
|
|
import pytest
|
|
from openai import APIStatusError
|
|
|
|
from hindsight_api.engine.providers.openai_compatible_llm import OpenAICompatibleLLM
|
|
|
|
# Reflect agent tools (subset matching what agent.py uses)
|
|
REFLECT_TOOLS = [
|
|
{
|
|
"type": "function",
|
|
"function": {
|
|
"name": "search_mental_models",
|
|
"description": "Search consolidated mental models",
|
|
"parameters": {
|
|
"type": "object",
|
|
"properties": {"query": {"type": "string"}},
|
|
"required": ["query"],
|
|
},
|
|
},
|
|
},
|
|
{
|
|
"type": "function",
|
|
"function": {
|
|
"name": "search_observations",
|
|
"description": "Search raw observations",
|
|
"parameters": {
|
|
"type": "object",
|
|
"properties": {"query": {"type": "string"}},
|
|
"required": ["query"],
|
|
},
|
|
},
|
|
},
|
|
{
|
|
"type": "function",
|
|
"function": {
|
|
"name": "recall",
|
|
"description": "Recall semantic memories",
|
|
"parameters": {
|
|
"type": "object",
|
|
"properties": {"query": {"type": "string"}},
|
|
"required": ["query"],
|
|
},
|
|
},
|
|
},
|
|
{
|
|
"type": "function",
|
|
"function": {
|
|
"name": "done",
|
|
"description": "Finish and return the answer",
|
|
"parameters": {
|
|
"type": "object",
|
|
"properties": {"answer": {"type": "string"}},
|
|
"required": ["answer"],
|
|
},
|
|
},
|
|
},
|
|
]
|
|
|
|
|
|
def _make_lmstudio_llm() -> OpenAICompatibleLLM:
|
|
return OpenAICompatibleLLM(
|
|
provider="lmstudio",
|
|
api_key="local",
|
|
base_url="http://localhost:1234/v1",
|
|
model="openai/gpt-oss-20b",
|
|
)
|
|
|
|
|
|
def _lmstudio_400_error(msg: str = "Tool choice of type 'function' is not supported. Use 'auto', 'none', or 'required'.") -> APIStatusError:
|
|
"""Simulate the HTTP 400 LM Studio returns for unsupported tool_choice format."""
|
|
mock_response = MagicMock()
|
|
mock_response.status_code = 400
|
|
mock_response.headers = {}
|
|
return APIStatusError(
|
|
message=msg,
|
|
response=mock_response,
|
|
body={"error": {"message": msg, "type": "invalid_request_error"}},
|
|
)
|
|
|
|
|
|
def _make_tool_call_response(tool_name: str, arguments: dict) -> MagicMock:
|
|
"""Build a mock successful tool call response from the LLM API."""
|
|
mock_tc = MagicMock()
|
|
mock_tc.id = "call_abc123"
|
|
mock_tc.function.name = tool_name
|
|
mock_tc.function.arguments = json.dumps(arguments)
|
|
|
|
mock_response = MagicMock()
|
|
mock_response.usage.prompt_tokens = 120
|
|
mock_response.usage.completion_tokens = 40
|
|
mock_response.usage.total_tokens = 160
|
|
mock_response.choices[0].finish_reason = "tool_calls"
|
|
mock_response.choices[0].message.content = None
|
|
mock_response.choices[0].message.tool_calls = [mock_tc]
|
|
return mock_response
|
|
|
|
|
|
class TestLMStudioNamedToolChoiceBug:
|
|
"""
|
|
Reproduces issue #520.
|
|
|
|
The reflect agent (agent.py lines 546-555) sets tool_choice to a named dict
|
|
on the first iterations to force sequential retrieval:
|
|
|
|
iteration=0, has_mental_models=True → {"type": "function", "function": {"name": "search_mental_models"}}
|
|
iteration=0, has_mental_models=False → {"type": "function", "function": {"name": "search_observations"}}
|
|
iteration=1, has_mental_models=True → {"type": "function", "function": {"name": "search_observations"}}
|
|
iteration=1 or (2 with models) → {"type": "function", "function": {"name": "recall"}}
|
|
|
|
LM Studio rejects these dict formats with HTTP 400.
|
|
"""
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_lmstudio_named_tool_choice_no_longer_causes_400(self):
|
|
"""
|
|
Regression test for issue #520: named tool_choice dict is converted to
|
|
"required" + filtered tools before the API call, so LM Studio never
|
|
sees the unsupported format and the 400 error no longer occurs.
|
|
"""
|
|
llm = _make_lmstudio_llm()
|
|
named_tool_choice = {"type": "function", "function": {"name": "search_mental_models"}}
|
|
success_response = _make_tool_call_response("search_mental_models", {"query": "user name"})
|
|
|
|
with patch.object(llm._client.chat.completions, "create", new_callable=AsyncMock) as mock_create:
|
|
mock_create.return_value = success_response
|
|
|
|
# Should succeed — no 400 because the dict is converted before sending
|
|
result = await llm.call_with_tools(
|
|
messages=[{"role": "user", "content": "What is the user's name?"}],
|
|
tools=REFLECT_TOOLS,
|
|
tool_choice=named_tool_choice,
|
|
max_retries=0,
|
|
)
|
|
|
|
assert len(result.tool_calls) == 1
|
|
assert result.tool_calls[0].name == "search_mental_models"
|
|
|
|
sent_kwargs = mock_create.call_args.kwargs
|
|
assert sent_kwargs["tool_choice"] == "required"
|
|
assert len(sent_kwargs["tools"]) == 1
|
|
assert sent_kwargs["tools"][0]["function"]["name"] == "search_mental_models"
|
|
|
|
@pytest.mark.asyncio
|
|
@pytest.mark.parametrize(
|
|
"forced_tool_name",
|
|
["search_mental_models", "search_observations", "recall"],
|
|
)
|
|
async def test_all_reflect_forced_tools_fail_on_lmstudio(self, forced_tool_name: str):
|
|
"""
|
|
Each named tool_choice the reflect agent uses on iterations 0-2 triggers
|
|
the same 400 error on LM Studio.
|
|
"""
|
|
llm = _make_lmstudio_llm()
|
|
named_tool_choice = {"type": "function", "function": {"name": forced_tool_name}}
|
|
|
|
with patch.object(llm._client.chat.completions, "create", new_callable=AsyncMock) as mock_create:
|
|
mock_create.side_effect = _lmstudio_400_error()
|
|
|
|
with pytest.raises(APIStatusError) as exc_info:
|
|
await llm.call_with_tools(
|
|
messages=[{"role": "user", "content": "Test query"}],
|
|
tools=REFLECT_TOOLS,
|
|
tool_choice=named_tool_choice,
|
|
max_retries=0,
|
|
)
|
|
|
|
assert exc_info.value.status_code == 400
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_lmstudio_string_tool_choice_works_fine(self):
|
|
"""
|
|
String tool_choice values ("auto", "none", "required") ARE supported by LM Studio.
|
|
Only the dict format {"type": "function", "function": {"name": "..."}} fails.
|
|
This test confirms the control case works.
|
|
"""
|
|
llm = _make_lmstudio_llm()
|
|
success_response = _make_tool_call_response("search_mental_models", {"query": "user name"})
|
|
|
|
with patch.object(llm._client.chat.completions, "create", new_callable=AsyncMock) as mock_create:
|
|
mock_create.return_value = success_response
|
|
|
|
result = await llm.call_with_tools(
|
|
messages=[{"role": "user", "content": "What is the user's name?"}],
|
|
tools=REFLECT_TOOLS,
|
|
tool_choice="required", # string form — LM Studio accepts this
|
|
max_retries=0,
|
|
)
|
|
|
|
assert len(result.tool_calls) == 1
|
|
assert result.tool_calls[0].name == "search_mental_models"
|
|
|
|
# Confirm "required" was sent, not a dict
|
|
sent_kwargs = mock_create.call_args.kwargs
|
|
assert sent_kwargs["tool_choice"] == "required"
|
|
|
|
|
|
class TestExpectedFixBehavior:
|
|
"""
|
|
Tests that document the EXPECTED behavior after the fix is applied.
|
|
|
|
For lmstudio (and ollama) providers, when tool_choice is a named dict:
|
|
{"type": "function", "function": {"name": "search_mental_models"}}
|
|
|
|
The fix should:
|
|
1. Convert tool_choice to "required"
|
|
2. Filter tools to only the requested tool
|
|
|
|
These tests currently FAIL (because the fix is not yet implemented).
|
|
After the fix is applied, they should PASS.
|
|
"""
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_fix_converts_named_tool_choice_to_required(self):
|
|
"""
|
|
After fix: named tool_choice dict is converted to "required" for lmstudio.
|
|
The API receives tool_choice="required" instead of the unsupported dict.
|
|
"""
|
|
llm = _make_lmstudio_llm()
|
|
named_tool_choice = {"type": "function", "function": {"name": "search_mental_models"}}
|
|
success_response = _make_tool_call_response("search_mental_models", {"query": "user name"})
|
|
|
|
with patch.object(llm._client.chat.completions, "create", new_callable=AsyncMock) as mock_create:
|
|
mock_create.return_value = success_response
|
|
|
|
result = await llm.call_with_tools(
|
|
messages=[{"role": "user", "content": "What is the user's name?"}],
|
|
tools=REFLECT_TOOLS,
|
|
tool_choice=named_tool_choice,
|
|
max_retries=0,
|
|
)
|
|
|
|
assert len(result.tool_calls) == 1
|
|
assert result.tool_calls[0].name == "search_mental_models"
|
|
|
|
sent_kwargs = mock_create.call_args.kwargs
|
|
# Fix: dict was converted to "required"
|
|
assert sent_kwargs["tool_choice"] == "required", (
|
|
f"Expected tool_choice='required', got {sent_kwargs['tool_choice']!r}"
|
|
)
|
|
# Fix: tools filtered to just the requested one
|
|
assert len(sent_kwargs["tools"]) == 1
|
|
assert sent_kwargs["tools"][0]["function"]["name"] == "search_mental_models"
|
|
|
|
@pytest.mark.asyncio
|
|
@pytest.mark.parametrize(
|
|
"forced_tool_name",
|
|
["search_mental_models", "search_observations", "recall"],
|
|
)
|
|
async def test_fix_filters_tools_to_requested_tool(self, forced_tool_name: str):
|
|
"""
|
|
After fix: tools list is filtered to only the forced tool so the model
|
|
can only call that one tool (equivalent to the named tool_choice behavior).
|
|
"""
|
|
llm = _make_lmstudio_llm()
|
|
named_tool_choice = {"type": "function", "function": {"name": forced_tool_name}}
|
|
success_response = _make_tool_call_response(forced_tool_name, {"query": "test"})
|
|
|
|
with patch.object(llm._client.chat.completions, "create", new_callable=AsyncMock) as mock_create:
|
|
mock_create.return_value = success_response
|
|
|
|
await llm.call_with_tools(
|
|
messages=[{"role": "user", "content": "Test query"}],
|
|
tools=REFLECT_TOOLS,
|
|
tool_choice=named_tool_choice,
|
|
max_retries=0,
|
|
)
|
|
|
|
sent_kwargs = mock_create.call_args.kwargs
|
|
assert sent_kwargs["tool_choice"] == "required"
|
|
assert len(sent_kwargs["tools"]) == 1
|
|
assert sent_kwargs["tools"][0]["function"]["name"] == forced_tool_name
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_fix_also_applies_to_openai_provider(self):
|
|
"""
|
|
The fix is generalized: all providers convert named tool_choice to
|
|
"required" + filtered tools. OpenAI natively supports the dict format
|
|
too, so the behaviour is semantically identical either way.
|
|
"""
|
|
from hindsight_api.engine.providers.openai_compatible_llm import OpenAICompatibleLLM
|
|
|
|
openai_llm = OpenAICompatibleLLM(
|
|
provider="openai",
|
|
api_key="sk-test",
|
|
base_url="",
|
|
model="gpt-4o-mini",
|
|
)
|
|
|
|
named_tool_choice = {"type": "function", "function": {"name": "search_mental_models"}}
|
|
success_response = _make_tool_call_response("search_mental_models", {"query": "test"})
|
|
|
|
with patch.object(openai_llm._client.chat.completions, "create", new_callable=AsyncMock) as mock_create:
|
|
mock_create.return_value = success_response
|
|
|
|
await openai_llm.call_with_tools(
|
|
messages=[{"role": "user", "content": "Test"}],
|
|
tools=REFLECT_TOOLS,
|
|
tool_choice=named_tool_choice,
|
|
max_retries=0,
|
|
)
|
|
|
|
sent_kwargs = mock_create.call_args.kwargs
|
|
# Generalized fix applies to OpenAI too
|
|
assert sent_kwargs["tool_choice"] == "required"
|
|
assert len(sent_kwargs["tools"]) == 1
|
|
assert sent_kwargs["tools"][0]["function"]["name"] == "search_mental_models"
|