Merge pull request #26 from cgycorey/feat/llm-reasoning-effort

feat(llm): forward configured reasoning_effort to native LLM calls
This commit is contained in:
LZH-YS1998
2026-08-06 15:31:38 +08:00
4 changed files with 154 additions and 0 deletions
+2
View File
@@ -7,6 +7,8 @@ llm:
temperature: 1
max_tokens: 32768
# reasoning_effort: "high" # optional; use a provider-supported level
# such as low, medium, or high
routing: {}
fallback: {}
+1
View File
@@ -273,6 +273,7 @@ class LLMConfig(BaseModel):
fallback: dict[str, Any] = Field(default_factory=dict)
temperature: float = 0.3
max_tokens: int = 32768
reasoning_effort: str | None = None
# Total input context window (tokens) for the active model. Set this when
# the model is not mapped in litellm (e.g. proxy/self-hosted models like
# doubao/minimax/glm), so the context-usage ring and compaction thresholds
+4
View File
@@ -569,6 +569,8 @@ class LLMProvider:
"max_tokens": max_tok,
**kwargs,
}
if self.config.reasoning_effort and "reasoning_effort" not in call_kwargs:
call_kwargs["reasoning_effort"] = self.config.reasoning_effort
if self._api_base:
call_kwargs["api_base"] = self._api_base
if self._api_key:
@@ -715,6 +717,8 @@ class LLMProvider:
"stream": True,
**kwargs,
}
if self.config.reasoning_effort and "reasoning_effort" not in call_kwargs:
call_kwargs["reasoning_effort"] = self.config.reasoning_effort
if self._api_base:
call_kwargs["api_base"] = self._api_base
if self._api_key:
+147
View File
@@ -0,0 +1,147 @@
from __future__ import annotations
import unittest
from types import SimpleNamespace
from unittest.mock import AsyncMock, patch
from opc.core.config import LLMConfig
from opc.llm.provider import LLMProvider
def _completion_response() -> SimpleNamespace:
return SimpleNamespace(
choices=[
SimpleNamespace(
message=SimpleNamespace(content="ok", tool_calls=[]),
finish_reason="stop",
)
],
usage=SimpleNamespace(prompt_tokens=1, completion_tokens=1),
)
async def _completion_stream():
yield SimpleNamespace(
usage=None,
choices=[
SimpleNamespace(
delta=SimpleNamespace(
content="ok",
reasoning=None,
reasoning_content=None,
thinking=None,
tool_calls=[],
),
finish_reason="stop",
)
],
)
class TestLLMProviderReasoningEffort(unittest.IsolatedAsyncioTestCase):
def test_config_retains_reasoning_effort(self) -> None:
config = LLMConfig.model_validate({
"default_model": "openai/gpt-5.6-luna",
"reasoning_effort": "max",
})
assert config.reasoning_effort == "max"
async def test_chat_forwards_configured_reasoning_effort(self) -> None:
provider = LLMProvider(LLMConfig(
default_model="openai/gpt-5.6-luna",
reasoning_effort="max",
))
with (
patch("opc.llm.provider._clamp_max_tokens", return_value=128),
patch(
"opc.llm.provider.litellm.acompletion",
new=AsyncMock(return_value=_completion_response()),
) as completion,
):
await provider.chat([{"role": "user", "content": "hello"}])
assert completion.await_args.kwargs["reasoning_effort"] == "max"
async def test_chat_allows_per_call_reasoning_effort_override(self) -> None:
provider = LLMProvider(LLMConfig(
default_model="openai/gpt-5.6-luna",
reasoning_effort="high",
))
with (
patch("opc.llm.provider._clamp_max_tokens", return_value=128),
patch(
"opc.llm.provider.litellm.acompletion",
new=AsyncMock(return_value=_completion_response()),
) as completion,
):
await provider.chat(
[{"role": "user", "content": "hello"}],
reasoning_effort="low",
)
assert completion.await_args.kwargs["reasoning_effort"] == "low"
async def test_chat_stream_forwards_configured_reasoning_effort(self) -> None:
provider = LLMProvider(LLMConfig(
default_model="openai/gpt-5.6-luna",
reasoning_effort="max",
))
with (
patch("opc.llm.provider._clamp_max_tokens", return_value=128),
patch(
"opc.llm.provider.litellm.acompletion",
new=AsyncMock(return_value=_completion_stream()),
) as completion,
):
events = [
event
async for event in provider.chat_stream([
{"role": "user", "content": "hello"},
])
]
assert events
assert completion.await_args.kwargs["reasoning_effort"] == "max"
assert completion.await_args.kwargs["stream"] is True
async def test_chat_stream_allows_per_call_reasoning_effort_override(self) -> None:
provider = LLMProvider(LLMConfig(
default_model="openai/gpt-5.6-luna",
reasoning_effort="high",
))
with (
patch("opc.llm.provider._clamp_max_tokens", return_value=128),
patch(
"opc.llm.provider.litellm.acompletion",
new=AsyncMock(return_value=_completion_stream()),
) as completion,
):
events = [
event
async for event in provider.chat_stream(
[{"role": "user", "content": "hello"}],
reasoning_effort="low",
)
]
assert events
assert completion.await_args.kwargs["reasoning_effort"] == "low"
async def test_unset_reasoning_effort_is_not_added_to_requests(self) -> None:
provider = LLMProvider(LLMConfig(default_model="openai/gpt-5.6-luna"))
with (
patch("opc.llm.provider._clamp_max_tokens", return_value=128),
patch(
"opc.llm.provider.litellm.acompletion",
new=AsyncMock(return_value=_completion_response()),
) as completion,
):
await provider.chat([{"role": "user", "content": "hello"}])
assert "reasoning_effort" not in completion.await_args.kwargs