diff --git a/README.md b/README.md index e6e7167..28f7969 100644 --- a/README.md +++ b/README.md @@ -554,24 +554,25 @@ Run `opc init` once from the repo root. It creates `.opc/`, copies the template After `opc init`, edit `.opc/config/llm_config.yaml` in the repo-local OPC home. If you set `OPC_HOME`, edit `$OPC_HOME/config/llm_config.yaml` instead. -The template leaves secrets empty. Configure either a literal key or an env var: +The template leaves secrets empty. Write your key directly into the file: ```yaml llm: default_model: "openai/gpt-5.4" api_base: "https://openrouter.ai/api/v1" - api_key: "" - api_key_env: "OPENROUTER_API_KEY" + api_key: "sk-or-v1-..." # your OpenRouter (or other provider) API key + + max_tokens: 32768 # max output tokens per request; lower it if your + # model's output cap is smaller + # context_window: 128000 # total input window. Usually auto-detected via + # litellm; unmapped models fall back to 128000. + # Uncomment and set only when the fallback is + # wrong for your model. ``` -Then: +Then verify with `opc status`. -```bash -export OPENROUTER_API_KEY="..." -opc status -``` - -You can also use provider-specific env vars such as `OPENAI_API_KEY` or `ANTHROPIC_API_KEY` if your model/provider configuration expects them. +If you prefer not to store the key in the file, leave `api_key` empty and set `api_key_env` to the name of an environment variable that holds it (e.g. `api_key_env: "OPENROUTER_API_KEY"`). ### External Agents diff --git a/config/llm_config.yaml b/config/llm_config.yaml index 47a0287..880fc3b 100644 --- a/config/llm_config.yaml +++ b/config/llm_config.yaml @@ -11,11 +11,11 @@ llm: fallback: {} # Optional: total input context window, used as the denominator for the - # context-usage ring and to trigger auto-summarization. You normally do NOT - # need this — for models litellm can map (most OpenAI / Anthropic models) the - # window is resolved automatically. Only set it when your model is served - # through a proxy / self-hosted endpoint that litellm cannot map, so the ring - # has no denominator. Scalar applies to the default model: + # context-usage ring and to trigger auto-summarization. Auto-detected via + # litellm for most OpenAI / Anthropic models; models litellm cannot map + # (e.g. deepseek/doubao/glm or proxy / self-hosted endpoints) fall back to + # 128000. Set this only when the fallback is wrong for your model. + # Scalar applies to the default model: # context_window: 200000 # Or per-model (wins over the scalar); keys are matched by model name: # context_window_overrides: diff --git a/opc/core/config.py b/opc/core/config.py index a1ca202..4ea120d 100644 --- a/opc/core/config.py +++ b/opc/core/config.py @@ -272,12 +272,13 @@ class LLMConfig(BaseModel): routing: dict[str, str] = Field(default_factory=dict) fallback: dict[str, Any] = Field(default_factory=dict) temperature: float = 0.3 - max_tokens: int = 8192 + max_tokens: int = 32768 # Total input context window (tokens) for the active model. Set this when # the model is not mapped in litellm (e.g. proxy/self-hosted models like # doubao/minimax/glm), so the context-usage ring and compaction thresholds - # have a real denominator. 0 = auto-detect via litellm. Optional per-model - # overrides keyed by model name take precedence over the scalar value. + # have a real denominator. 0 = auto-detect via litellm; unmapped models + # fall back to 128000. Optional per-model overrides keyed by model name + # take precedence over the scalar value. context_window: int = 0 context_window_overrides: dict[str, int] = Field(default_factory=dict) diff --git a/opc/llm/provider.py b/opc/llm/provider.py index 49d3b1b..3a490b7 100644 --- a/opc/llm/provider.py +++ b/opc/llm/provider.py @@ -82,6 +82,12 @@ def _normalized_model_name(model: str) -> str: return model.strip().lower() +# Used when neither user config nor litellm can supply a window. Conservative +# enough for modern models so compaction still has a real denominator. +_CONTEXT_WINDOW_FALLBACK = 128_000 +_context_window_fallback_warned: set[str] = set() + + _CONTEXT_WINDOW_OVERRIDES: tuple[tuple[str, int], ...] = ( ("gpt-5.4-pro", 1_050_000), ("gpt-5.4-mini", 400_000), @@ -269,11 +275,27 @@ class LLMProvider: if override is not None: return override try: - limit = litellm.get_max_tokens(resolved_model) - return int(limit) if limit else None + # max_input_tokens is the context window; litellm.get_max_tokens() + # returns the "max_tokens" map entry, which for many models (e.g. + # deepseek) is the OUTPUT cap and would wildly under-report here. + info = litellm.get_model_info(resolved_model) + limit = info.get("max_input_tokens") or info.get("max_tokens") + if limit: + return int(limit) + reason = "model is not mapped in litellm" except Exception as e: - logger.warning(f"Unable to resolve context window for model={resolved_model}: {e}") - return None + reason = str(e) + if resolved_model not in _context_window_fallback_warned: + _context_window_fallback_warned.add(resolved_model) + logger.warning( + "Unable to resolve context window for model={} ({}); assuming {} tokens. " + "Set llm.context_window or llm.context_window_overrides in llm_config.yaml " + "to use the model's real window.", + resolved_model, + reason, + _CONTEXT_WINDOW_FALLBACK, + ) + return _CONTEXT_WINDOW_FALLBACK def count_input_tokens( self, diff --git a/tests/test_llm_provider_context_window.py b/tests/test_llm_provider_context_window.py index ca5ec82..82264e4 100644 --- a/tests/test_llm_provider_context_window.py +++ b/tests/test_llm_provider_context_window.py @@ -35,7 +35,7 @@ class TestLLMProviderContextWindow(unittest.TestCase): def test_gpt_5_4_override_applies_on_official_openai_base(self) -> None: provider = LLMProvider(LLMConfig(default_model="openai/gpt-5.4")) - with patch("opc.llm.provider.litellm.get_max_tokens", return_value=128000): + with patch("opc.llm.provider.litellm.get_model_info", return_value={"max_input_tokens": 128000}): self.assertEqual(provider.get_context_window(), 1_050_000) def test_gpt_5_4_override_does_not_apply_on_proxy_base(self) -> None: @@ -44,7 +44,7 @@ class TestLLMProviderContextWindow(unittest.TestCase): api_base="https://openrouter.ai/api/v1", )) - with patch("opc.llm.provider.litellm.get_max_tokens", return_value=128000): + with patch("opc.llm.provider.litellm.get_model_info", return_value={"max_input_tokens": 128000}): self.assertEqual(provider.get_context_window(), 128000) def test_poe_claude_sonnet_4_5_model_uses_local_context_window(self) -> None: @@ -53,9 +53,9 @@ class TestLLMProviderContextWindow(unittest.TestCase): api_base="https://api.poe.com/v1", )) - with patch("opc.llm.provider.litellm.get_max_tokens") as get_max_tokens: + with patch("opc.llm.provider.litellm.get_model_info") as get_model_info: self.assertEqual(provider.get_context_window(), 64_000) - get_max_tokens.assert_not_called() + get_model_info.assert_not_called() def test_poe_openai_compatible_legacy_prefix_uses_same_context_window(self) -> None: provider = LLMProvider(LLMConfig( @@ -63,16 +63,27 @@ class TestLLMProviderContextWindow(unittest.TestCase): api_base="https://api.poe.com/v1", )) - with patch("opc.llm.provider.litellm.get_max_tokens") as get_max_tokens: + with patch("opc.llm.provider.litellm.get_model_info") as get_model_info: self.assertEqual(provider.get_context_window(), 64_000) - get_max_tokens.assert_not_called() + get_model_info.assert_not_called() def test_non_overridden_model_still_uses_litellm(self) -> None: provider = LLMProvider(LLMConfig(default_model="openai/gpt-4o")) - with patch("opc.llm.provider.litellm.get_max_tokens", return_value=128000): + with patch("opc.llm.provider.litellm.get_model_info", return_value={"max_input_tokens": 128000}): self.assertEqual(provider.get_context_window(), 128000) + def test_context_window_uses_max_input_tokens_not_output_cap(self) -> None: + """deepseek-style entries: max_tokens is the OUTPUT cap (8192), the + context window is max_input_tokens (1M). The window must not be 8192.""" + provider = LLMProvider(LLMConfig(default_model="deepseek/deepseek-v4-pro")) + + with patch( + "opc.llm.provider.litellm.get_model_info", + return_value={"max_input_tokens": 1_000_000, "max_tokens": 8192, "max_output_tokens": 8192}, + ): + self.assertEqual(provider.get_context_window(), 1_000_000) + def test_config_scalar_override_supplies_window_for_unmapped_model(self) -> None: """Unmapped proxy models (doubao/minimax/…) get a real window from config.""" provider = LLMProvider(LLMConfig( @@ -81,19 +92,28 @@ class TestLLMProviderContextWindow(unittest.TestCase): context_window=256000, )) - with patch("opc.llm.provider.litellm.get_max_tokens", return_value=None) as get_max_tokens: + with patch("opc.llm.provider.litellm.get_model_info", return_value={}) as get_model_info: self.assertEqual(provider.get_context_window(), 256000) - get_max_tokens.assert_not_called() + get_model_info.assert_not_called() - def test_unmapped_model_without_override_returns_none(self) -> None: - """No override + litellm can't map → None (unchanged fallback).""" + def test_unmapped_model_without_override_falls_back_to_default(self) -> None: + """No override + litellm can't map → 128000 fallback, not None.""" provider = LLMProvider(LLMConfig( default_model="openai/doubao-seed-2.0-pro", api_base="https://ark.cn-beijing.volces.com/api/coding/v3", )) - with patch("opc.llm.provider.litellm.get_max_tokens", return_value=None): - self.assertIsNone(provider.get_context_window()) + with patch("opc.llm.provider.litellm.get_model_info", return_value={}): + self.assertEqual(provider.get_context_window(), 128000) + + def test_unmapped_model_litellm_error_falls_back_to_default(self) -> None: + provider = LLMProvider(LLMConfig(default_model="deepseek/deepseek-v4-pro")) + + with patch( + "opc.llm.provider.litellm.get_model_info", + side_effect=Exception("Model deepseek-v4-pro isn't mapped yet."), + ): + self.assertEqual(provider.get_context_window(), 128000) def test_config_per_model_override_takes_precedence(self) -> None: provider = LLMProvider(LLMConfig( @@ -102,15 +122,15 @@ class TestLLMProviderContextWindow(unittest.TestCase): context_window_overrides={"doubao-seed-2.0-pro": 262144}, )) - with patch("opc.llm.provider.litellm.get_max_tokens", return_value=None): + with patch("opc.llm.provider.litellm.get_model_info", return_value={}): self.assertEqual(provider.get_context_window(), 262144) def test_config_override_wins_over_litellm_for_mapped_model(self) -> None: provider = LLMProvider(LLMConfig(default_model="openai/gpt-4o", context_window=50000)) - with patch("opc.llm.provider.litellm.get_max_tokens", return_value=128000) as get_max_tokens: + with patch("opc.llm.provider.litellm.get_model_info", return_value={"max_input_tokens": 128000}) as get_model_info: self.assertEqual(provider.get_context_window(), 50000) - get_max_tokens.assert_not_called() + get_model_info.assert_not_called() if __name__ == "__main__":