Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
475 changes: 475 additions & 0 deletions .github/workflows/version-bump-prs.yml

Large diffs are not rendered by default.

Original file line number Diff line number Diff line change
Expand Up @@ -55,6 +55,7 @@
from openhands.sdk.io import FileStore, LocalFileStore
from openhands.sdk.llm import LLM, Message, TextContent, content_to_str
from openhands.sdk.llm.auth.openai import create_subscription_llm_from_config
from openhands.sdk.llm.exceptions import LLMAuthenticationError
from openhands.sdk.llm.llm import LLMCallContext
from openhands.sdk.llm.llm_profile_store import LLMProfileStore
from openhands.sdk.llm.llm_registry import LLMRegistry
Expand Down Expand Up @@ -2040,6 +2041,21 @@ def run(self) -> None:
)
)
break
except LLMAuthenticationError as e:
with self._state:
self._state.execution_status = ConversationExecutionStatus.ERROR
self._on_event(
ConversationErrorEvent(
source="environment",
code="LLMAuthenticationError",
detail=(
"Your LLM API key appears to be invalid or has expired."
),
)
)
raise ConversationRunError(
self._state.id, e, persistence_dir=self._state.persistence_dir
) from e
except Exception as e:
with self._state:
self._state.execution_status = ConversationExecutionStatus.ERROR
Expand Down Expand Up @@ -2540,6 +2556,21 @@ async def arun(self) -> None:

self._state.execution_status = ConversationExecutionStatus.PAUSED
self._on_event(InterruptEvent())
except LLMAuthenticationError as e:
with self._state:
self._state.execution_status = ConversationExecutionStatus.ERROR
self._on_event(
ConversationErrorEvent(
source="environment",
code="LLMAuthenticationError",
detail=(
"Your LLM API key appears to be invalid or has expired."
),
)
)
raise ConversationRunError(
self._state.id, e, persistence_dir=self._state.persistence_dir
) from e
except Exception as e:
with self._state:
updated_agent_state = dict(self._state.agent_state)
Expand Down
7 changes: 4 additions & 3 deletions openhands-sdk/openhands/sdk/hooks/executor.py
Original file line number Diff line number Diff line change
Expand Up @@ -566,9 +566,10 @@ def execute(
try:
output_data = json.loads(result.stdout)
if isinstance(output_data, dict):
# Parse decision
if "decision" in output_data:
decision_str = output_data["decision"].lower()
# Non-string values represent no decision.
decision_value = output_data.get("decision")
if isinstance(decision_value, str):
decision_str = decision_value.lower()
if decision_str == "allow":
hook_result.decision = HookDecision.ALLOW
elif decision_str == "deny":
Expand Down
2 changes: 1 addition & 1 deletion openhands-sdk/openhands/sdk/llm/utils/AGENTS.md
Original file line number Diff line number Diff line change
Expand Up @@ -6,7 +6,7 @@ See the [SDK AGENTS.md](../../AGENTS.md) for package-wide policies.

These lists are a curated set of models that work well, not a catalog of everything a provider offers. Keep them short.

- For each model line, keep only the **two latest versions** (for example `gpt-6` and `gpt-5.6`; `claude-opus-5` and `claude-opus-4-8`).
- For each model line, keep only the **two latest versions** (for example `gpt-6` and `gpt-5.6`; `claude-opus-5-5` and `claude-opus-5`).
- Variants of a kept version (`-pro`, `-mini`, `-codex`, `-flash`, dated aliases) stay with that version. Unversioned "current" aliases (`deepseek-chat`, `kimi-for-coding`) stay.
- When you add a new version, remove the oldest version in the same line, in every list where it appears (the provider list and `VERIFIED_OPENHANDS_MODELS`).
- Every entry in `VERIFIED_OPENHANDS_MODELS` must also appear in a provider list, unless it is OpenHands-only; `tests/sdk/llm/test_model_list.py` checks this.
Expand Down
1 change: 0 additions & 1 deletion openhands-sdk/openhands/sdk/llm/utils/model_features.py
Original file line number Diff line number Diff line change
Expand Up @@ -109,7 +109,6 @@ def _normalized_supported_openai_params(model: str | None) -> frozenset[str]:


REASONING_EFFORT_MODEL_OVERRIDES = {
"gpt-5.2-codex": "gpt-5.2-codex",
"kimi-k3": "moonshot/kimi-k3",
}

Expand Down
10 changes: 7 additions & 3 deletions openhands-sdk/openhands/sdk/llm/utils/verified_models.py
Original file line number Diff line number Diff line change
Expand Up @@ -11,6 +11,8 @@

# GPT: gpt-6 and gpt-5.6. Codex: gpt-5.3-codex and gpt-5.2-codex.
VERIFIED_OPENAI_MODELS = [
"gpt-6-sol",
"gpt-6-luna",
"gpt-6-astra",
"gpt-5.6",
"gpt-5.6-sol",
Expand All @@ -20,10 +22,10 @@
"gpt-5.2-codex",
]

# Opus: 5 and 4.8. Sonnet: 5 and 4.6. Haiku: 4.5. Fable: 5.1 and 5.
# Opus: 5.5 and 5. Sonnet: 5 and 4.6. Haiku: 4.5. Fable: 5.1 and 5.
VERIFIED_ANTHROPIC_MODELS = [
"claude-opus-5-5",
"claude-opus-5",
"claude-opus-4-8",
"claude-sonnet-5",
"claude-sonnet-4-6",
"claude-haiku-4-5-20251001",
Expand Down Expand Up @@ -104,12 +106,14 @@
# What the ``openhands/`` provider serves. Same rule; every entry must also be in
# a provider list above, except OpenHands-only models.
VERIFIED_OPENHANDS_MODELS = [
"claude-opus-5-5",
"claude-opus-5",
"claude-opus-4-8",
"claude-sonnet-5",
"claude-sonnet-4-6",
"claude-fable-5-1",
"claude-fable-5",
"gpt-6-sol",
"gpt-6-luna",
"gpt-6-astra",
"gpt-5.6",
"gpt-5.3-codex",
Expand Down
118 changes: 118 additions & 0 deletions tests/sdk/conversation/local/test_auth_error_friendly_message.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,118 @@
"""Tests that LLMAuthenticationError surfaces a user-friendly message.

Regression tests for:
https://github.com/OpenHands/software-agent-sdk/issues/3411

When a user has an invalid/expired API key the raw litellm error (e.g. the
full AnthropicException JSON) must NOT appear in the ConversationErrorEvent
detail that is sent to the UI. Instead, a clear, actionable message should
be emitted, while the ConversationRunError is still raised so server logs
remain unaffected.
"""

import asyncio
import tempfile

import pytest

from openhands.sdk.agent import Agent
from openhands.sdk.conversation import Conversation, LocalConversation
from openhands.sdk.conversation.exceptions import ConversationRunError
from openhands.sdk.event.conversation_error import ConversationErrorEvent
from openhands.sdk.llm import Message, TextContent
from openhands.sdk.llm.exceptions import LLMAuthenticationError
from openhands.sdk.testing import TestLLM


_RAW_LITELLM_ERROR = (
"litellm.AuthenticationError: AnthropicException - "
'{"type":"error","error":{"type":"authentication_error",'
'"message":"invalid x-api-key"},'
'"request_id":"req_011CbTfF4jtKVAB95FSH6ESb"}'
)
_FRIENDLY_SUBSTRING = "invalid or has expired"


def _make_auth_failing_conversation(tmpdir: str) -> LocalConversation:
llm = TestLLM.from_messages([LLMAuthenticationError(_RAW_LITELLM_ERROR)])
agent = Agent(llm=llm, tools=[])
conv = Conversation(agent=agent, persistence_dir=tmpdir, workspace=tmpdir)
assert isinstance(conv, LocalConversation)
conv.send_message(Message(role="user", content=[TextContent(text="hello")]))
return conv


# ---------------------------------------------------------------------------
# Sync path (run)
# ---------------------------------------------------------------------------


def test_auth_error_run_raises_conversation_run_error():
"""ConversationRunError is still raised so server logs are unaffected."""
with tempfile.TemporaryDirectory() as tmpdir:
conv = _make_auth_failing_conversation(tmpdir)
with pytest.raises(ConversationRunError) as exc_info:
conv.run()
assert isinstance(exc_info.value.__cause__, LLMAuthenticationError)


def test_auth_error_run_emits_friendly_detail():
"""ConversationErrorEvent.detail is user-readable, not the raw litellm string."""
with tempfile.TemporaryDirectory() as tmpdir:
conv = _make_auth_failing_conversation(tmpdir)
with pytest.raises(ConversationRunError):
conv.run()

error_events = [
e for e in conv.state.events if isinstance(e, ConversationErrorEvent)
]
assert error_events, "Expected at least one ConversationErrorEvent"

auth_error_event = next(
(e for e in error_events if e.code == "LLMAuthenticationError"), None
)
assert auth_error_event is not None, (
"Expected a ConversationErrorEvent with code='LLMAuthenticationError'"
)
assert _FRIENDLY_SUBSTRING in auth_error_event.detail, (
f"Expected friendly message in detail, got: {auth_error_event.detail!r}"
)
assert _RAW_LITELLM_ERROR not in auth_error_event.detail, (
"Raw litellm error string must not appear in the UI-facing detail"
)


# ---------------------------------------------------------------------------
# Async path (arun)
# ---------------------------------------------------------------------------


def test_auth_error_arun_raises_conversation_run_error():
"""Async path: ConversationRunError is still raised."""
with tempfile.TemporaryDirectory() as tmpdir:
conv = _make_auth_failing_conversation(tmpdir)
with pytest.raises(ConversationRunError) as exc_info:
asyncio.run(conv.arun())
assert isinstance(exc_info.value.__cause__, LLMAuthenticationError)


def test_auth_error_arun_emits_friendly_detail():
"""Async path: ConversationErrorEvent.detail is user-readable."""
with tempfile.TemporaryDirectory() as tmpdir:
conv = _make_auth_failing_conversation(tmpdir)
with pytest.raises(ConversationRunError):
asyncio.run(conv.arun())

error_events = [
e for e in conv.state.events if isinstance(e, ConversationErrorEvent)
]
assert error_events, "Expected at least one ConversationErrorEvent"

auth_error_event = next(
(e for e in error_events if e.code == "LLMAuthenticationError"), None
)
assert auth_error_event is not None, (
"Expected a ConversationErrorEvent with code='LLMAuthenticationError'"
)
assert _FRIENDLY_SUBSTRING in auth_error_event.detail
assert _RAW_LITELLM_ERROR not in auth_error_event.detail
30 changes: 30 additions & 0 deletions tests/sdk/hooks/test_executor.py
Original file line number Diff line number Diff line change
Expand Up @@ -56,6 +56,36 @@ def test_execute_receives_json_stdin(self, executor, sample_event, tmp_path):
assert output_data["event_type"] == "PreToolUse"
assert output_data["tool_name"] == "BashTool"

def test_null_decision_leaves_a_successful_hook_successful(
self, executor, sample_event
):
hook = HookDefinition(command="""echo '{"decision": null}'""")

result = executor.execute(hook, sample_event)

assert result.success
assert result.exit_code == 0
assert result.decision is None
assert not result.error

def test_non_string_decision_is_ignored_rather_than_fatal(
self, executor, sample_event
):
"""Any non-string decision means no decision, not a failed hook."""
hook = HookDefinition(command="""echo '{"decision": 1}'""")

result = executor.execute(hook, sample_event)

assert result.success
assert result.decision is None

def test_string_decisions_still_parse(self, executor, sample_event):
hook = HookDefinition(command="""echo '{"decision": "allow"}'""")

result = executor.execute(hook, sample_event)

assert result.decision == HookDecision.ALLOW

def test_execute_blocking_exit_code(self, executor, sample_event):
"""Test that exit code 2 blocks the operation."""
hook = HookDefinition(command=python_command("import sys; sys.exit(2)"))
Expand Down
8 changes: 4 additions & 4 deletions tests/sdk/llm/test_llm_profile_store.py
Original file line number Diff line number Diff line change
Expand Up @@ -154,15 +154,15 @@ def test_load_migrates_legacy_openhands_proxy_profile(
json.dumps(
{
"schema_version": 1,
"model": "litellm_proxy/claude-opus-4-8",
"model": "litellm_proxy/claude-opus-5",
"base_url": "https://llm-proxy.app.all-hands.dev/",
}
)
)

loaded = profile_store.load("legacy")

assert loaded.model == "openhands/claude-opus-4-8"
assert loaded.model == "openhands/claude-opus-5"
assert loaded.base_url is None


Expand All @@ -174,7 +174,7 @@ def test_list_summaries_migrates_legacy_openhands_proxy_profile(
json.dumps(
{
"schema_version": 1,
"model": "litellm_proxy/claude-opus-4-8",
"model": "litellm_proxy/claude-opus-5",
"base_url": "https://llm-proxy.app.all-hands.dev/",
}
)
Expand All @@ -185,7 +185,7 @@ def test_list_summaries_migrates_legacy_openhands_proxy_profile(
assert summaries == [
{
"name": "legacy",
"model": "openhands/claude-opus-4-8",
"model": "openhands/claude-opus-5",
"base_url": None,
"provider_connection_id": None,
"provider_connection_broken": False,
Expand Down
11 changes: 8 additions & 3 deletions tests/sdk/llm/test_model_list.py
Original file line number Diff line number Diff line change
Expand Up @@ -157,7 +157,10 @@ def test_verified_lists_keep_two_latest_versions_per_line():
"""
expectations = {
"openai": ({"gpt-6-astra", "gpt-5.6"}, {"gpt-5.5", "gpt-5.4", "gpt-4o", "o3"}),
"anthropic": ({"claude-opus-5", "claude-opus-4-8"}, {"claude-opus-4-7"}),
"anthropic": (
{"claude-opus-5-5", "claude-opus-5"},
{"claude-opus-4-8", "claude-opus-4-7"},
),
"mistral": (
{"devstral-2512", "devstral-medium-2512"},
{"devstral-medium-2507"},
Expand All @@ -175,8 +178,10 @@ def test_verified_lists_keep_two_latest_versions_per_line():
models = set(VERIFIED_MODELS[provider])
assert present <= models, f"{provider}: missing {present - models}"
assert not (absent & models), f"{provider}: stale {absent & models}"
assert {"gpt-6-astra", "gpt-5.6", "claude-opus-5"} <= set(VERIFIED_OPENHANDS_MODELS)
assert not {"gpt-5.5", "claude-opus-4-7", "minimax-m2.5"} & set(
assert {"gpt-6-astra", "gpt-5.6", "claude-opus-5-5", "claude-opus-5"} <= set(
VERIFIED_OPENHANDS_MODELS
)
assert not {"gpt-5.5", "claude-opus-4-8", "claude-opus-4-7", "minimax-m2.5"} & set(
VERIFIED_OPENHANDS_MODELS
)

Expand Down
4 changes: 2 additions & 2 deletions tests/sdk/test_settings.py
Original file line number Diff line number Diff line change
Expand Up @@ -523,15 +523,15 @@ def test_validate_agent_settings_migrates_legacy_openhands_proxy_llm() -> None:
"schema_version": 3,
"agent_kind": "openhands",
"llm": {
"model": "litellm_proxy/claude-opus-4-8",
"model": "litellm_proxy/claude-opus-5",
"base_url": "https://llm-proxy.app.all-hands.dev/",
},
}
)

assert isinstance(settings, OpenHandsAgentSettings)
assert settings.schema_version == AGENT_SETTINGS_SCHEMA_VERSION
assert settings.llm.model == "openhands/claude-opus-4-8"
assert settings.llm.model == "openhands/claude-opus-5"
assert settings.llm.base_url is None


Expand Down
Loading