mirror of
https://github.com/HKUDS/nanobot.git
synced 2026-08-13 07:39:15 +03:00
feat(providers): support DeepSeek V4 Pro Responses
This commit is contained in:
@@ -360,7 +360,7 @@ request, while other tools such as `web_fetch` remain available.
|
|||||||
<details>
|
<details>
|
||||||
<summary><b>DeepSeek native web search</b></summary>
|
<summary><b>DeepSeek native web search</b></summary>
|
||||||
|
|
||||||
DeepSeek V4 Flash uses DeepSeek's native Responses API. Its provider-hosted web search is
|
DeepSeek V4 Flash and Pro use DeepSeek's native Responses API. Their provider-hosted web search is
|
||||||
enabled by default because it does not require a separate paid add-on. Turn it off from the
|
enabled by default because it does not require a separate paid add-on. Turn it off from the
|
||||||
WebUI provider settings, or with:
|
WebUI provider settings, or with:
|
||||||
|
|
||||||
@@ -377,9 +377,9 @@ WebUI provider settings, or with:
|
|||||||
}
|
}
|
||||||
```
|
```
|
||||||
|
|
||||||
The switch applies to `deepseek-v4-flash`; DeepSeek models that remain on Chat Completions
|
The switch applies to `deepseek-v4-flash` and `deepseek-v4-pro`; DeepSeek models that remain on
|
||||||
cannot use this Responses tool. Native search calls appear in the WebUI activity stream, and
|
Chat Completions cannot use this Responses tool. Native search calls appear in the WebUI activity
|
||||||
their opaque output items are preserved for multi-turn Responses state replay.
|
stream, and their opaque output items are preserved for multi-turn Responses state replay.
|
||||||
|
|
||||||
</details>
|
</details>
|
||||||
|
|
||||||
@@ -391,7 +391,7 @@ Providers that use the Responses API can keep reasoning context across a
|
|||||||
conversation, which helps with multi-step tasks. Supported providers can also
|
conversation, which helps with multi-step tasks. Supported providers can also
|
||||||
compact long conversations automatically.
|
compact long conversations automatically.
|
||||||
|
|
||||||
nanobot preserves Responses conversation state automatically for OpenAI Responses, OpenAI Codex, Azure OpenAI, DeepSeek V4 Flash, and compatible GitHub Copilot models.
|
nanobot preserves Responses conversation state automatically for OpenAI Responses, OpenAI Codex, Azure OpenAI, DeepSeek V4, and compatible GitHub Copilot models.
|
||||||
Native compaction is also automatic when the provider supports it. The
|
Native compaction is also automatic when the provider supports it. The
|
||||||
threshold is derived from the active model's context window and reserved output
|
threshold is derived from the active model's context window and reserved output
|
||||||
headroom; no provider configuration is required.
|
headroom; no provider configuration is required.
|
||||||
|
|||||||
+1
-1
@@ -287,7 +287,7 @@ Arbitrary custom provider names are OpenAI-compatible only; they do not use the
|
|||||||
|
|
||||||
`providers.openai.apiType` may be set when you need to force a specific OpenAI API surface. Other providers reject `apiType`; leave it unset outside `providers.openai`. Replace the model with a model ID available to your OpenAI account. Direct OpenAI Responses, OpenAI Codex, Azure OpenAI Responses, and eligible GitHub Copilot models share [opaque Responses state retention](./configuration.md#responses-state-and-compaction); native compaction is enabled only where the backend supports it. The WebUI exposes provider-native switches for OpenAI web search, Codex Fast mode, DeepSeek web search, and Grok X Search. These switches write the corresponding raw provider request fields under `extraBody`.
|
`providers.openai.apiType` may be set when you need to force a specific OpenAI API surface. Other providers reject `apiType`; leave it unset outside `providers.openai`. Replace the model with a model ID available to your OpenAI account. Direct OpenAI Responses, OpenAI Codex, Azure OpenAI Responses, and eligible GitHub Copilot models share [opaque Responses state retention](./configuration.md#responses-state-and-compaction); native compaction is enabled only where the backend supports it. The WebUI exposes provider-native switches for OpenAI web search, Codex Fast mode, DeepSeek web search, and Grok X Search. These switches write the corresponding raw provider request fields under `extraBody`.
|
||||||
|
|
||||||
DeepSeek is the model-level exception in the OpenAI-compatible provider: `deepseek-v4-flash` automatically uses DeepSeek's native Responses API, while `deepseek-v4-pro` remains on Chat Completions. Its native `web_search` tool is enabled by default and shows its lifecycle in WebUI chat activity; set `providers.deepseek.extraBody.tools` to `[]` to disable it.
|
DeepSeek is the model-level exception in the OpenAI-compatible provider: `deepseek-v4-flash` and `deepseek-v4-pro` automatically use DeepSeek's native Responses API. Its native `web_search` tool is enabled by default and shows its lifecycle in WebUI chat activity; set `providers.deepseek.extraBody.tools` to `[]` to disable it.
|
||||||
|
|
||||||
### Custom OpenAI-Compatible Endpoint
|
### Custom OpenAI-Compatible Endpoint
|
||||||
|
|
||||||
|
|||||||
@@ -1158,7 +1158,8 @@ class OpenAICompatProvider(LLMProvider):
|
|||||||
self._sanitize_empty_content(sanitized_state.pending_messages)
|
self._sanitize_empty_content(sanitized_state.pending_messages)
|
||||||
)
|
)
|
||||||
)
|
)
|
||||||
preserve_reasoning = bool(self._spec and self._spec.name == "deepseek")
|
is_deepseek = bool(self._spec and self._spec.name == "deepseek")
|
||||||
|
preserve_reasoning = is_deepseek
|
||||||
instructions, input_items, replayed = prepare_responses_input(
|
instructions, input_items, replayed = prepare_responses_input(
|
||||||
sanitized_messages,
|
sanitized_messages,
|
||||||
state=sanitized_state,
|
state=sanitized_state,
|
||||||
@@ -1194,7 +1195,7 @@ class OpenAICompatProvider(LLMProvider):
|
|||||||
|
|
||||||
if not self._supports_temperature(model_name, reasoning_effort) and not preserve_reasoning:
|
if not self._supports_temperature(model_name, reasoning_effort) and not preserve_reasoning:
|
||||||
body["include"] = ["reasoning.encrypted_content"]
|
body["include"] = ["reasoning.encrypted_content"]
|
||||||
if reasoning_effort and reasoning_effort.lower() != "none":
|
if reasoning_effort and (reasoning_effort.lower() != "none" or is_deepseek):
|
||||||
body["reasoning"] = {"effort": reasoning_effort}
|
body["reasoning"] = {"effort": reasoning_effort}
|
||||||
if replayed and "gpt-5.6" in model_name.lower():
|
if replayed and "gpt-5.6" in model_name.lower():
|
||||||
body.setdefault("reasoning", {})["context"] = "all_turns"
|
body.setdefault("reasoning", {})["context"] = "all_turns"
|
||||||
|
|||||||
@@ -112,8 +112,7 @@ class ProviderSpec:
|
|||||||
implicit_reasoning_models: tuple[str, ...] = ()
|
implicit_reasoning_models: tuple[str, ...] = ()
|
||||||
|
|
||||||
# Models that expose the OpenAI Responses wire format. This is model-level
|
# Models that expose the OpenAI Responses wire format. This is model-level
|
||||||
# because providers may add Responses support incrementally (DeepSeek V4
|
# because providers may add Responses support incrementally.
|
||||||
# Flash is supported before V4 Pro).
|
|
||||||
responses_models: tuple[str, ...] = ()
|
responses_models: tuple[str, ...] = ()
|
||||||
|
|
||||||
# Provider-hosted Responses tools sent unless extraBody.tools explicitly
|
# Provider-hosted Responses tools sent unless extraBody.tools explicitly
|
||||||
@@ -482,7 +481,7 @@ PROVIDERS: tuple[ProviderSpec, ...] = (
|
|||||||
backend="openai_compat",
|
backend="openai_compat",
|
||||||
default_api_base="https://api.deepseek.com",
|
default_api_base="https://api.deepseek.com",
|
||||||
thinking_style="thinking_type",
|
thinking_style="thinking_type",
|
||||||
responses_models=("deepseek-v4-flash",),
|
responses_models=("deepseek-v4-flash", "deepseek-v4-pro"),
|
||||||
responses_default_tools=("web_search",),
|
responses_default_tools=("web_search",),
|
||||||
),
|
),
|
||||||
# Gemini: Google's OpenAI-compatible endpoint
|
# Gemini: Google's OpenAI-compatible endpoint
|
||||||
|
|||||||
@@ -299,8 +299,8 @@ def _fake_chat_stream_legacy_function_call_chunks():
|
|||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
async def test_openai_compat_stream_forwards_reasoning_deltas_deepseek_style() -> None:
|
async def test_openai_compat_chat_stream_forwards_reasoning_deltas_deepseek_style() -> None:
|
||||||
"""Regression: DeepSeek-V4 / reasoner expose ``delta.reasoning_content`` during streaming."""
|
"""DeepSeek Chat Completions exposes ``delta.reasoning_content`` while streaming."""
|
||||||
mock_chat = AsyncMock(return_value=_fake_chat_stream_reasoning_chunks())
|
mock_chat = AsyncMock(return_value=_fake_chat_stream_reasoning_chunks())
|
||||||
spec = find_by_name("deepseek")
|
spec = find_by_name("deepseek")
|
||||||
thinking: list[str] = []
|
thinking: list[str] = []
|
||||||
@@ -321,6 +321,7 @@ async def test_openai_compat_stream_forwards_reasoning_deltas_deepseek_style() -
|
|||||||
default_model="deepseek-v4-pro",
|
default_model="deepseek-v4-pro",
|
||||||
spec=spec,
|
spec=spec,
|
||||||
)
|
)
|
||||||
|
provider._api_type = "chat_completions"
|
||||||
result = await provider.chat_stream(
|
result = await provider.chat_stream(
|
||||||
messages=[{"role": "user", "content": "hi"}],
|
messages=[{"role": "user", "content": "hi"}],
|
||||||
model="deepseek-v4-pro",
|
model="deepseek-v4-pro",
|
||||||
@@ -336,6 +337,37 @@ async def test_openai_compat_stream_forwards_reasoning_deltas_deepseek_style() -
|
|||||||
mock_chat.assert_awaited_once()
|
mock_chat.assert_awaited_once()
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
async def test_deepseek_v4_pro_uses_responses_api() -> None:
|
||||||
|
mock_chat = AsyncMock(return_value=_fake_chat_response())
|
||||||
|
mock_responses = AsyncMock(return_value=_fake_responses_response("from responses"))
|
||||||
|
|
||||||
|
with patch("nanobot.providers.openai_compat_provider.AsyncOpenAI") as mock_client_class:
|
||||||
|
client_instance = mock_client_class.return_value
|
||||||
|
client_instance.chat.completions.create = mock_chat
|
||||||
|
client_instance.responses.create = mock_responses
|
||||||
|
|
||||||
|
provider = OpenAICompatProvider(
|
||||||
|
api_key="sk-test",
|
||||||
|
default_model="deepseek-v4-pro",
|
||||||
|
spec=find_by_name("deepseek"),
|
||||||
|
)
|
||||||
|
result = await provider.chat(
|
||||||
|
messages=[{"role": "user", "content": "hello"}],
|
||||||
|
model="deepseek-v4-pro",
|
||||||
|
reasoning_effort="none",
|
||||||
|
)
|
||||||
|
|
||||||
|
assert result.content == "from responses"
|
||||||
|
mock_responses.assert_awaited_once()
|
||||||
|
mock_chat.assert_not_awaited()
|
||||||
|
call_kwargs = mock_responses.call_args.kwargs
|
||||||
|
assert call_kwargs["model"] == "deepseek-v4-pro"
|
||||||
|
assert call_kwargs["reasoning"] == {"effort": "none"}
|
||||||
|
assert call_kwargs["tools"] == [{"type": "web_search"}]
|
||||||
|
assert "include" not in call_kwargs
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
@pytest.mark.parametrize(
|
@pytest.mark.parametrize(
|
||||||
("provider_name", "model"),
|
("provider_name", "model"),
|
||||||
|
|||||||
@@ -11,6 +11,7 @@ from nanobot.providers.openai_compat_provider import (
|
|||||||
OpenAICompatProvider,
|
OpenAICompatProvider,
|
||||||
)
|
)
|
||||||
from nanobot.providers.openai_responses.state import build_responses_state
|
from nanobot.providers.openai_responses.state import build_responses_state
|
||||||
|
from nanobot.providers.registry import find_by_name
|
||||||
|
|
||||||
|
|
||||||
@pytest.fixture()
|
@pytest.fixture()
|
||||||
@@ -30,30 +31,22 @@ def test_responses_api_available_by_default(provider):
|
|||||||
assert provider._should_use_responses_api("gpt-5", None) is True
|
assert provider._should_use_responses_api("gpt-5", None) is True
|
||||||
|
|
||||||
|
|
||||||
def test_deepseek_v4_flash_uses_responses_by_model(provider):
|
@pytest.mark.parametrize("model", ["deepseek-v4-flash", "deepseek-v4-pro"])
|
||||||
provider._spec = type("Spec", (), {
|
def test_deepseek_v4_models_use_responses_by_model(provider, model):
|
||||||
"name": "deepseek",
|
provider._spec = find_by_name("deepseek")
|
||||||
"responses_models": ("deepseek-v4-flash",),
|
|
||||||
"strip_model_prefix": False,
|
|
||||||
"strip_model_prefixes": (),
|
|
||||||
})()
|
|
||||||
provider._effective_base = "https://api.deepseek.com"
|
provider._effective_base = "https://api.deepseek.com"
|
||||||
provider.default_model = "deepseek-v4-flash"
|
provider.default_model = model
|
||||||
|
|
||||||
assert provider._should_use_responses_api("deepseek-v4-flash", None) is True
|
assert provider._should_use_responses_api(model, None) is True
|
||||||
assert provider._should_use_responses_api("deepseek-v4-pro", None) is False
|
assert provider._should_use_responses_api("deepseek-chat", None) is False
|
||||||
|
|
||||||
|
|
||||||
def test_deepseek_v4_flash_matches_provider_prefixed_model(provider):
|
@pytest.mark.parametrize("model", ["deepseek-v4-flash", "deepseek-v4-pro"])
|
||||||
provider._spec = type("Spec", (), {
|
def test_deepseek_v4_models_match_provider_prefixed_model(provider, model):
|
||||||
"name": "deepseek",
|
provider._spec = find_by_name("deepseek")
|
||||||
"responses_models": ("deepseek-v4-flash",),
|
|
||||||
"strip_model_prefix": False,
|
|
||||||
"strip_model_prefixes": (),
|
|
||||||
})()
|
|
||||||
provider._effective_base = "https://api.deepseek.com"
|
provider._effective_base = "https://api.deepseek.com"
|
||||||
|
|
||||||
assert provider._should_use_responses_api("deepseek/deepseek-v4-flash", None) is True
|
assert provider._should_use_responses_api(f"deepseek/{model}", None) is True
|
||||||
|
|
||||||
|
|
||||||
def test_direct_openai_enables_server_compaction(provider):
|
def test_direct_openai_enables_server_compaction(provider):
|
||||||
|
|||||||
Reference in New Issue
Block a user