fixed changes based on code review

2026-06-17 15:25:17 +02:00 · 2025-12-02 06:04:52 -08:00 · 2025-12-02 06:04:52 -08:00 · 8fbdddca47
commit 8fbdddca47
parent a47fd17815
5 changed files with 197 additions and 33 deletions
--- a/tests/e2e/test_openai_responses_api_client.py
+++ b/tests/e2e/test_openai_responses_api_client.py
@ -325,3 +325,154 @@ def test_openai_responses_api_streaming_with_tools_upstream_chat_completions():
    assert (
        full_text or tool_calls
    ), "Expected streamed text or tool call argument deltas from Responses tools stream"
+
+
+def test_openai_responses_api_non_streaming_upstream_anthropic():
+    """Send a v1/responses request using the grok alias to verify translation/routing"""
+    base_url = LLM_GATEWAY_ENDPOINT.replace("/v1/chat/completions", "")
+    client = openai.OpenAI(api_key="test-key", base_url=f"{base_url}/v1")
+
+    resp = client.responses.create(
+        model="claude-sonnet-4-20250514", input="Hello, translate this via grok alias"
+    )
+
+    # Print the response content - handle both responses format and chat completions format
+    print(f"\n{'='*80}")
+    print(f"Model: {resp.model}")
+    print(f"Output: {resp.output_text}")
+    print(f"{'='*80}\n")
+
+    assert resp is not None
+    assert resp.id is not None
+
+
+def test_openai_responses_api_with_streaming_upstream_anthropic():
+    """Build a v1/responses API streaming request (pass-through) and ensure gateway accepts it"""
+    base_url = LLM_GATEWAY_ENDPOINT.replace("/v1/chat/completions", "")
+    client = openai.OpenAI(api_key="test-key", base_url=f"{base_url}/v1")
+
+    # Simple streaming responses API request using a direct model (pass-through)
+    stream = client.responses.create(
+        model="claude-sonnet-4-20250514",
+        input="Write a short haiku about coding",
+        stream=True,
+    )
+
+    # Collect streamed content using the official Responses API streaming shape
+    text_chunks = []
+    final_message = None
+
+    for event in stream:
+        # The Python SDK surfaces a high-level Responses streaming interface.
+        # We rely on its typed helpers instead of digging into model_extra.
+        if getattr(event, "type", None) == "response.output_text.delta" and getattr(
+            event, "delta", None
+        ):
+            # Each delta contains a text fragment
+            text_chunks.append(event.delta)
+
+        # Track the final response message if provided by the SDK
+        if getattr(event, "type", None) == "response.completed" and getattr(
+            event, "response", None
+        ):
+            final_message = event.response
+
+    full_content = "".join(text_chunks)
+
+    # Print the streaming response
+    print(f"\n{'='*80}")
+    print(
+        f"Model: {getattr(final_message, 'model', 'unknown') if final_message else 'unknown'}"
+    )
+    print(f"Streamed Output: {full_content}")
+    print(f"{'='*80}\n")
+
+    assert len(text_chunks) > 0, "Should have received streaming text deltas"
+    assert len(full_content) > 0, "Should have received content"
+
+
+def test_openai_responses_api_non_streaming_with_tools_upstream_anthropic():
+    """Responses API with tools routed to grok via alias"""
+    base_url = LLM_GATEWAY_ENDPOINT.replace("/v1/chat/completions", "")
+    client = openai.OpenAI(api_key="test-key", base_url=f"{base_url}/v1")
+
+    tools = [
+        {
+            "type": "function",
+            "name": "echo_tool",
+            "description": "Echo back the provided input: hello_world",
+            "parameters": {
+                "type": "object",
+                "properties": {"text": {"type": "string"}},
+                "required": ["text"],
+            },
+        }
+    ]
+
+    resp = client.responses.create(
+        model="claude-sonnet-4-20250514",
+        input="Call the echo tool",
+        tools=tools,
+    )
+
+    assert resp.id is not None
+
+    print(f"\n{'='*80}")
+    print(f"Model: {resp.model}")
+    print(f"Output: {resp.output_text}")
+    print(f"{'='*80}\n")
+
+
+def test_openai_responses_api_streaming_with_tools_upstream_anthropic():
+    """Responses API with a function/tool definition (streaming, pass-through)"""
+    base_url = LLM_GATEWAY_ENDPOINT.replace("/v1/chat/completions", "")
+    client = openai.OpenAI(api_key="test-key", base_url=f"{base_url}/v1", max_retries=0)
+
+    tools = [
+        {
+            "type": "function",
+            "name": "echo_tool",
+            "description": "Echo back the provided input: hello_world",
+            "parameters": {
+                "type": "object",
+                "properties": {"text": {"type": "string"}},
+                "required": ["text"],
+            },
+        }
+    ]
+
+    stream = client.responses.create(
+        model="claude-sonnet-4-20250514",
+        input="Call the echo tool",
+        tools=tools,
+        stream=True,
+    )
+
+    text_chunks = []
+    tool_calls = []
+
+    for event in stream:
+        etype = getattr(event, "type", None)
+
+        # Collect streamed text output
+        if etype == "response.output_text.delta" and getattr(event, "delta", None):
+            text_chunks.append(event.delta)
+
+        # Collect streamed tool call arguments
+        if etype == "response.function_call_arguments.delta" and getattr(
+            event, "delta", None
+        ):
+            tool_calls.append(event.delta)
+
+    full_text = "".join(text_chunks)
+
+    print(f"\n{'='*80}")
+    print("Responses tools streaming test")
+    print(f"Streamed text: {full_text}")
+    print(f"Tool call argument chunks: {len(tool_calls)}")
+    print(f"{'='*80}\n")
+
+    # We expect either streamed text output or streamed tool-call arguments
+    assert (
+        full_text or tool_calls
+    ), "Expected streamed text or tool call argument deltas from Responses tools stream"