mirror of
https://github.com/HKUDS/nanobot.git
synced 2026-09-05 10:41:58 +03:00
fix(provider): stabilize Codex prompt cache routing (#5540)
This commit is contained in:
@@ -495,6 +495,7 @@ class AgentRunner:
|
||||
model=spec.runtime.model,
|
||||
messages=messages,
|
||||
state=spec.provider_state,
|
||||
session_id=spec.session_key,
|
||||
)
|
||||
governance_config = ContextGovernanceConfig(
|
||||
provider=spec.runtime.provider,
|
||||
|
||||
@@ -252,10 +252,13 @@ class ProviderCallContext:
|
||||
The regular ``chat`` contract stays provider-agnostic. Responses-capable
|
||||
providers consume this context through the opt-in ``chat_with_context``
|
||||
hooks, while every other provider inherits the context-free delegation.
|
||||
``session_id`` gives providers a stable conversation-scoped routing key
|
||||
without exposing that identity in the public message transcript.
|
||||
"""
|
||||
|
||||
conversation_state: ProviderConversationState | None = field(default=None, repr=False)
|
||||
context_window_tokens: int | None = None
|
||||
session_id: str | None = field(default=None, repr=False)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
@@ -1640,6 +1643,7 @@ class LLMProvider(ABC):
|
||||
context_window_tokens=(
|
||||
provider_context.context_window_tokens
|
||||
),
|
||||
session_id=provider_context.session_id,
|
||||
)
|
||||
if stripped is not None or stripped_context is not None:
|
||||
logger.warning(
|
||||
|
||||
@@ -42,9 +42,11 @@ class ProviderConversationStateController:
|
||||
model: str | None,
|
||||
messages: list[dict[str, Any]],
|
||||
state: ProviderConversationState | None = None,
|
||||
session_id: str | None = None,
|
||||
) -> None:
|
||||
self._provider = provider
|
||||
self._model = model
|
||||
self._session_id = session_id
|
||||
self._state = (
|
||||
state
|
||||
if state is not None
|
||||
@@ -60,9 +62,12 @@ class ProviderConversationStateController:
|
||||
context_window_tokens: int | None,
|
||||
) -> ProviderCallContext | None:
|
||||
"""Return typed provider context for a request that does not resume state."""
|
||||
if context_window_tokens is None:
|
||||
if context_window_tokens is None and self._session_id is None:
|
||||
return None
|
||||
return ProviderCallContext(context_window_tokens=context_window_tokens)
|
||||
return ProviderCallContext(
|
||||
context_window_tokens=context_window_tokens,
|
||||
session_id=self._session_id,
|
||||
)
|
||||
|
||||
def prepare_request(
|
||||
self,
|
||||
@@ -112,6 +117,7 @@ class ProviderConversationStateController:
|
||||
if independent_context is not None
|
||||
else None
|
||||
),
|
||||
session_id=self._session_id,
|
||||
)
|
||||
|
||||
def observe_response(
|
||||
|
||||
@@ -186,6 +186,7 @@ class FallbackProvider(LLMProvider):
|
||||
return ProviderCallContext(
|
||||
conversation_state=provider_context.conversation_state,
|
||||
context_window_tokens=context_window_tokens,
|
||||
session_id=provider_context.session_id,
|
||||
)
|
||||
|
||||
def _primary_available(self) -> bool:
|
||||
@@ -541,6 +542,7 @@ class FallbackProvider(LLMProvider):
|
||||
fallback_kwargs["provider_context"] = ProviderCallContext(
|
||||
conversation_state=state,
|
||||
context_window_tokens=context_window_tokens,
|
||||
session_id=provider_context.session_id,
|
||||
)
|
||||
if fallback.reasoning_effort is None:
|
||||
fallback_kwargs.pop("reasoning_effort", None)
|
||||
|
||||
@@ -103,6 +103,7 @@ class OpenAICodexProvider(LLMProvider):
|
||||
provider=self._responses_state_provider(),
|
||||
model=_strip_model_prefix(model),
|
||||
)
|
||||
session_id = provider_context.session_id if provider_context is not None else None
|
||||
|
||||
body: dict[str, Any] = {
|
||||
"model": _strip_model_prefix(model),
|
||||
@@ -111,10 +112,11 @@ class OpenAICodexProvider(LLMProvider):
|
||||
"instructions": system_prompt,
|
||||
"input": input_items,
|
||||
"text": {"verbosity": "medium"},
|
||||
"prompt_cache_key": _prompt_cache_key(messages[:2]),
|
||||
"tool_choice": tool_choice or "auto",
|
||||
"parallel_tool_calls": True,
|
||||
}
|
||||
if session_id:
|
||||
body["prompt_cache_key"] = _prompt_cache_key(session_id)
|
||||
body["include"] = ["reasoning.encrypted_content"]
|
||||
reasoning_options = _build_reasoning_options(reasoning_effort)
|
||||
if replayed and "gpt-5.6" in _strip_model_prefix(model).lower():
|
||||
@@ -496,9 +498,8 @@ async def _request_codex(
|
||||
return result
|
||||
|
||||
|
||||
def _prompt_cache_key(messages: list[dict[str, Any]]) -> str:
|
||||
raw = json.dumps(messages, ensure_ascii=True, sort_keys=True)
|
||||
return hashlib.sha256(raw.encode("utf-8")).hexdigest()
|
||||
def _prompt_cache_key(session_id: str) -> str:
|
||||
return hashlib.sha256(session_id.encode("utf-8")).hexdigest()
|
||||
|
||||
|
||||
def _friendly_error(status_code: int, raw: str) -> str:
|
||||
|
||||
Reference in New Issue
Block a user