* Convert codex tool_choice test to pytest style Follow-up to #734: replace unittest.TestCase + manual sys.path manipulation with idiomatic pytest + @pytest.mark.asyncio, matching the rest of the test suite. * Fix test_hierarchical_fields_categorization for new configurable fields Update expected count from 20 to 21 and add assertions for fields added by recent PRs: retain_default_strategy, retain_strategies, max_observations_per_scope, reflect_source_facts_max_tokens, llm_gemini_safety_settings, mcp_enabled_tools. * Add LlamaIndex doc to v0.4 versioned docs and sidebars The LlamaIndex integration doc was added to docs/ (next version) in #672 but not to versioned_docs/version-0.4/, causing a broken link on the /integrations page which resolves to the latest version. * Regenerate docs skill references Run generate-docs-skill.sh to pick up new integration pages (codex, llamaindex) and updated configuration docs. * Add Codex integration doc to v0.4 versioned docs and sidebar Same issue as LlamaIndex: doc was added to docs/ (next) but not versioned_docs/version-0.4/, causing broken link on /integrations.
88 lines
3.1 KiB
Python
88 lines
3.1 KiB
Python
"""
|
|
Regression tests for Codex provider tool_choice normalization.
|
|
|
|
The reflect agent forces tool selection via named tool_choice dicts on early iterations:
|
|
{"type": "function", "function": {"name": "recall"}}
|
|
|
|
The Codex Responses API expects the function name at the top level instead:
|
|
{"type": "function", "name": "recall"}
|
|
|
|
Without normalization, Codex rejects the request with:
|
|
400 Unknown parameter: 'tool_choice.function'
|
|
"""
|
|
|
|
from unittest.mock import AsyncMock, MagicMock, patch
|
|
|
|
import pytest
|
|
|
|
from hindsight_api.engine.providers.codex_llm import CodexLLM
|
|
|
|
TOOLS = [
|
|
{
|
|
"type": "function",
|
|
"function": {
|
|
"name": "recall",
|
|
"description": "Recall semantic memories",
|
|
"parameters": {
|
|
"type": "object",
|
|
"properties": {"query": {"type": "string"}},
|
|
"required": ["query"],
|
|
},
|
|
},
|
|
}
|
|
]
|
|
|
|
|
|
def build_llm() -> CodexLLM:
|
|
with patch.object(CodexLLM, "_load_codex_auth", return_value=("token", "account")):
|
|
return CodexLLM(
|
|
provider="openai-codex",
|
|
api_key="ignored",
|
|
base_url="https://chatgpt.com/backend-api",
|
|
model="gpt-5.4-mini",
|
|
)
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_codex_normalizes_legacy_named_tool_choice_shape():
|
|
llm = build_llm()
|
|
response = MagicMock()
|
|
response.status_code = 200
|
|
response.raise_for_status.return_value = None
|
|
with patch.object(llm._client, "post", new_callable=AsyncMock) as mock_post:
|
|
mock_post.return_value = response
|
|
with patch.object(llm, "_parse_sse_tool_stream", new_callable=AsyncMock) as mock_parse:
|
|
mock_parse.return_value = (None, [])
|
|
await llm.call_with_tools(
|
|
messages=[{"role": "user", "content": "recall the memory"}],
|
|
tools=TOOLS,
|
|
tool_choice={"type": "function", "function": {"name": "recall"}},
|
|
max_retries=0,
|
|
)
|
|
sent_payload = mock_post.call_args.kwargs["json"]
|
|
|
|
assert sent_payload["tool_choice"] == {"type": "function", "name": "recall"}
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_codex_forced_tool_choice_still_yields_tool_calls():
|
|
llm = build_llm()
|
|
response = MagicMock()
|
|
response.status_code = 200
|
|
response.raise_for_status.return_value = None
|
|
tool_call = {"id": "call-1", "name": "recall", "arguments": {"query": "memory"}}
|
|
with patch.object(llm._client, "post", new_callable=AsyncMock) as mock_post:
|
|
mock_post.return_value = response
|
|
with patch.object(llm, "_parse_sse_tool_stream", new_callable=AsyncMock) as mock_parse:
|
|
mock_parse.return_value = (None, [tool_call])
|
|
result = await llm.call_with_tools(
|
|
messages=[{"role": "user", "content": "recall the memory"}],
|
|
tools=TOOLS,
|
|
tool_choice={"type": "function", "function": {"name": "recall"}},
|
|
max_retries=0,
|
|
)
|
|
sent_payload = mock_post.call_args.kwargs["json"]
|
|
|
|
assert len(result.tool_calls) == 1
|
|
assert result.tool_calls[0].name == "recall"
|
|
assert sent_payload["tool_choice"] == {"type": "function", "name": "recall"}
|