fix(billing): mark reflect's internal recall calls as internal (#972)
Reflect's tool functions (tool_search_observations, tool_recall) call recall_async with the user's original request_context, which has internal=False. The usage metering extension sees these as user-facing recall operations and bills them separately — double-charging the customer for recalls that are already included in the reflect operation cost. Fix: wrap request_context with dataclasses.replace(internal=True) before passing to recall_async. This matches the pattern used by consolidation, which already creates an internal RequestContext for its sub-operations. The internal flag causes the metering extension to: - Record the usage as "internal_recall" (tracked but not billed) - Skip credit deduction entirely Observed impact: a single reflect call was generating 2 extra billed recall entries (one from tool_search_observations, one from tool_recall), inflating the customer's recall token count by ~26 tokens per reflect.
This commit is contained in:
parent
aad07a141b
commit
d38ecdb9ec
1 changed files with 9 additions and 2 deletions
|
|
@ -9,6 +9,7 @@ Implements hierarchical retrieval:
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
import uuid
|
import uuid
|
||||||
|
from dataclasses import replace
|
||||||
from datetime import datetime, timezone
|
from datetime import datetime, timezone
|
||||||
from typing import TYPE_CHECKING, Any
|
from typing import TYPE_CHECKING, Any
|
||||||
|
|
||||||
|
|
@ -162,13 +163,18 @@ async def tool_search_observations(
|
||||||
if include_source_facts and source_facts_max_tokens > 0:
|
if include_source_facts and source_facts_max_tokens > 0:
|
||||||
recall_kwargs["max_source_facts_tokens"] = source_facts_max_tokens
|
recall_kwargs["max_source_facts_tokens"] = source_facts_max_tokens
|
||||||
|
|
||||||
|
# Use an internal request context so this recall is not billed as a
|
||||||
|
# user-facing operation. The reflect caller is already billed for the
|
||||||
|
# overall reflect operation; double-billing the sub-recalls would
|
||||||
|
# overcharge the customer.
|
||||||
|
internal_ctx = replace(request_context, internal=True)
|
||||||
result = await memory_engine.recall_async(
|
result = await memory_engine.recall_async(
|
||||||
bank_id=bank_id,
|
bank_id=bank_id,
|
||||||
query=query,
|
query=query,
|
||||||
fact_type=["observation"],
|
fact_type=["observation"],
|
||||||
max_tokens=max_tokens,
|
max_tokens=max_tokens,
|
||||||
enable_trace=False,
|
enable_trace=False,
|
||||||
request_context=request_context,
|
request_context=internal_ctx,
|
||||||
tags=tags,
|
tags=tags,
|
||||||
tags_match=tags_match,
|
tags_match=tags_match,
|
||||||
tag_groups=tag_groups,
|
tag_groups=tag_groups,
|
||||||
|
|
@ -233,13 +239,14 @@ async def tool_recall(
|
||||||
# Only world/experience are valid for raw recall (observation is handled by search_observations)
|
# Only world/experience are valid for raw recall (observation is handled by search_observations)
|
||||||
recall_fact_type = [ft for ft in (fact_types or ["experience", "world"]) if ft in ("world", "experience")]
|
recall_fact_type = [ft for ft in (fact_types or ["experience", "world"]) if ft in ("world", "experience")]
|
||||||
include_chunks = True
|
include_chunks = True
|
||||||
|
internal_ctx = replace(request_context, internal=True)
|
||||||
result = await memory_engine.recall_async(
|
result = await memory_engine.recall_async(
|
||||||
bank_id=bank_id,
|
bank_id=bank_id,
|
||||||
query=query,
|
query=query,
|
||||||
fact_type=recall_fact_type,
|
fact_type=recall_fact_type,
|
||||||
max_tokens=max_tokens,
|
max_tokens=max_tokens,
|
||||||
enable_trace=False,
|
enable_trace=False,
|
||||||
request_context=request_context,
|
request_context=internal_ctx,
|
||||||
tags=tags,
|
tags=tags,
|
||||||
tags_match=tags_match,
|
tags_match=tags_match,
|
||||||
tag_groups=tag_groups,
|
tag_groups=tag_groups,
|
||||||
|
|
|
||||||
Loading…
Reference in a new issue