diff --git a/src/ucode/agents/claude.py b/src/ucode/agents/claude.py index a4b1435..f64b536 100644 --- a/src/ucode/agents/claude.py +++ b/src/ucode/agents/claude.py @@ -16,6 +16,9 @@ from typing import cast from ucode.agent_updates import available_npm_package_update +from ucode.anthropic_gateway_proxy import ( + start_proxy as start_anthropic_gateway_proxy, +) from ucode.config_io import ( APP_DIR, ToolSpec, @@ -24,12 +27,16 @@ read_json_safe, write_json_file, ) +from ucode.constants import LOOPBACK_HOST from ucode.databricks import ( build_auth_shell_command, build_tool_base_url, get_databricks_token, ) -from ucode.gateway_proxy import AI_GATEWAY_TOKEN_HEADER, AUTHORIZATION_HEADER, start_proxy +from ucode.gateway_proxy import ( + AI_GATEWAY_TOKEN_HEADER, + AUTHORIZATION_HEADER, +) from ucode.launcher import exec_or_spawn from ucode.managed_files import OS, current_os, write_managed_file from ucode.smart_routing.claude_hooks import ( @@ -214,10 +221,10 @@ def relayed_proxy_base_url(state: dict) -> str: port = state.get("relayed_proxy_port") if not isinstance(port, int): with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as sock: - sock.bind(("127.0.0.1", 0)) + sock.bind((LOOPBACK_HOST, 0)) port = sock.getsockname()[1] state["relayed_proxy_port"] = port - return f"http://127.0.0.1:{port}" + return f"http://{LOOPBACK_HOST}:{port}" def _web_search_mcp_entry(workspace: str, search_model: str, profile: str | None = None) -> dict: @@ -966,7 +973,7 @@ def _rewrite_relayed_port(state: dict, port: int) -> None: settings = read_json_safe(CLAUDE_SETTINGS_PATH) env = settings.get("env") if isinstance(env, dict): - env["ANTHROPIC_BASE_URL"] = f"http://127.0.0.1:{port}" + env["ANTHROPIC_BASE_URL"] = f"http://{LOOPBACK_HOST}:{port}" write_json_file(CLAUDE_SETTINGS_PATH, settings) @@ -999,7 +1006,7 @@ def _launch_relayed(state: dict, binary: str, tool_args: list[str]) -> None: if not isinstance(port, int): raise RuntimeError("Relayed proxy port was not configured; re-run `ucode claude`.") - server, cache, client = start_proxy( + server, cache, client = start_anthropic_gateway_proxy( workspace, state.get("profile"), port, @@ -1031,7 +1038,7 @@ def _launch_relayed(state: dict, binary: str, tool_args: list[str]) -> None: def _launch_gateway(state: dict, binary: str, tool_args: list[str]) -> None: workspace = state["workspace"] - server, cache, client = start_proxy( + server, cache, client = start_anthropic_gateway_proxy( workspace, state.get("profile"), 0, @@ -1041,7 +1048,7 @@ def _launch_gateway(state: dict, binary: str, tool_args: list[str]) -> None: token = cache.token os.environ["OAUTH_TOKEN"] = token os.environ["ANTHROPIC_AUTH_TOKEN"] = token - os.environ["ANTHROPIC_BASE_URL"] = f"http://127.0.0.1:{server.server_address[1]}" + os.environ["ANTHROPIC_BASE_URL"] = f"http://{LOOPBACK_HOST}:{server.server_address[1]}" os.environ["CLAUDE_CODE_USE_GATEWAY"] = "1" server_thread = threading.Thread(target=server.serve_forever, daemon=True) diff --git a/src/ucode/anthropic_gateway_proxy.py b/src/ucode/anthropic_gateway_proxy.py new file mode 100644 index 0000000..39798f5 --- /dev/null +++ b/src/ucode/anthropic_gateway_proxy.py @@ -0,0 +1,518 @@ +"""Loopback refresh proxy for Claude gateway requests. + +A relayed Model Provider Service authenticates the caller's own Anthropic +subscription OAuth (which Claude Code owns in the `Authorization` header) and +carries a Databricks credential in the `X-Databricks-AI-Gateway-Token` swap +header. Native gateway discovery instead carries the Databricks credential in +`Authorization`. The proxy refreshes the applicable header, streams inference +responses verbatim, and rewrites model discovery responses when needed. + +Security invariants (mirroring `databricks.py` token handling): + - Binds 127.0.0.1 only; never exposed off-host. + - Never logs header values or bodies. The Databricks token lives in memory, + refreshed off the request path; the Anthropic OAuth in `Authorization` is + passed through untouched in relayed mode and never logged. +""" + +from __future__ import annotations + +import base64 +import binascii +import json +import os +import sys +import threading +import time +import uuid +from collections.abc import Iterable +from http import HTTPStatus +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +from typing import cast +from urllib.parse import parse_qsl, urlencode, urlsplit, urlunsplit + +import httpx + +from ucode.constants import LOOPBACK_HOST +from ucode.databricks import get_databricks_token + +# Header we overwrite with the freshly-minted Databricks credential. Any +# client-supplied value is replaced, so a stale settings.json value can't leak. +AI_GATEWAY_TOKEN_HEADER = "X-Databricks-AI-Gateway-Token" +AUTHORIZATION_HEADER = "Authorization" +# Hop-by-hop headers must not be forwarded across the proxy. +_HOP_BY_HOP = frozenset( + h.lower() + for h in ( + "connection", + "keep-alive", + "proxy-authenticate", + "proxy-authorization", + "te", + "trailers", + "transfer-encoding", + "upgrade", + "host", + "content-length", + ) +) +# Per-operation upstream timeouts. `read` is generous because model turns stream +# over a single response and Anthropic emits SSE pings, so inter-chunk gaps stay +# small; `connect`/`pool` fail fast when the gateway is unreachable. +_UPSTREAM_TIMEOUT = httpx.Timeout(connect=10.0, read=600.0, write=600.0, pool=10.0) +# Refresh once the token has less than this many seconds of life left. Databricks +# access tokens live ~1h; a 10-min buffer leaves ample headroom for a retry. +_REFRESH_BUFFER_S = 600 +# How often the background thread re-checks freshness. Cheap: it only shells out +# to the CLI when actually within the buffer, otherwise it's a bare clock compare. +_REFRESHER_POLL_S = 120 +# Assumed lifetime when a token carries no decodable `exp` (defensive fallback). +_DEFAULT_TTL_S = 3600 +# Opt-in transport diagnostics for intermittent streaming failures. Events only +# contain locally-generated request ids, timings, status codes, byte counts, +# and exception class names — never headers, bodies, or credentials. +_DIAGNOSTICS_ENV = "UCODE_RELAYED_PROXY_DIAGNOSTICS" +_DIAGNOSTICS_TRUE = frozenset({"1", "true", "yes", "on"}) + + +def _diagnostics_enabled() -> bool: + return os.environ.get(_DIAGNOSTICS_ENV, "").strip().lower() in _DIAGNOSTICS_TRUE + + +def _diagnostic_log(event: str, **fields: object) -> None: + if not _diagnostics_enabled(): + return + payload = {"event": event, **fields} + sys.stderr.write( + f"[ucode-relay] {json.dumps(payload, sort_keys=True, separators=(',', ':'))}\n" + ) + sys.stderr.flush() + + +def _jwt_exp(token: str) -> float | None: + """Best-effort `exp` (epoch seconds) from a JWT access token, else None.""" + try: + payload = token.split(".")[1] + payload += "=" * (-len(payload) % 4) # restore base64 padding + return float(json.loads(base64.urlsafe_b64decode(payload))["exp"]) + except (IndexError, ValueError, KeyError, binascii.Error, json.JSONDecodeError): + return None + + +def _log_refresh_failure(exc: BaseException) -> None: + """Surface (never silently swallow) a refresh failure, without leaking any + token or header value.""" + sys.stderr.write( + f"[ucode] Databricks token refresh failed: {exc}. If the session stalls, " + "run `databricks auth login` for your workspace profile.\n" + ) + + +class _TokenCache: + """Holds the current Databricks token and its expiry, refreshing lazily as it + nears expiry. + + A background thread refreshes proactively so the request path rarely blocks, + but the request path also refreshes on demand — which is what carries the + token across events the timer can't (laptop sleep suspends the monotonic + clock, so a fixed interval silently stops advancing). All refreshes are + single-flighted through ``_refresh_lock`` so a burst of requests at the expiry + boundary triggers exactly one CLI call, not a thundering herd on the shared + token cache.""" + + def __init__( + self, + workspace: str, + profile: str | None, + *, + force_refresh_near_expiry: bool = False, + ) -> None: + self._workspace = workspace + self._profile = profile + self._force_refresh_near_expiry = force_refresh_near_expiry + self._state_lock = threading.Lock() # guards _token / _expiry (brief) + self._refresh_lock = threading.Lock() # single-flights the CLI refresh + self._stop = threading.Event() + self._token = "" + self._expiry = 0.0 + # Preserve the existing non-forced relayed-auth fetch. Gateway discovery + # opts into a forced fetch so its static client token starts with a full TTL. + self._refresh(force=force_refresh_near_expiry) + + def _refresh(self, *, force: bool) -> None: + """Mint a token and record its expiry.""" + token = get_databricks_token(self._workspace, self._profile, force_refresh=force) + expiry = _jwt_exp(token) or (time.time() + _DEFAULT_TTL_S) + with self._state_lock: + self._token = token + self._expiry = expiry + + def _fresh_enough(self) -> bool: + with self._state_lock: + return bool(self._token) and time.time() < self._expiry - _REFRESH_BUFFER_S + + def _ensure_fresh(self) -> None: + if self._fresh_enough(): + return + with self._refresh_lock: + if self._fresh_enough(): # another thread refreshed while we waited + return + try: + self._refresh(force=self._force_refresh_near_expiry) + except RuntimeError as exc: + # Keep serving the current token; a request that then 401s triggers + # a forced refresh + retry (see _ProxyHandler._handle). + _log_refresh_failure(exc) + + @property + def token(self) -> str: + self._ensure_fresh() + with self._state_lock: + return self._token + + def refresh(self) -> None: + """Force a fresh mint now (used by the retry-on-401 path).""" + with self._refresh_lock: + self._refresh(force=True) + + def run_refresher(self) -> None: + while not self._stop.wait(_REFRESHER_POLL_S): + try: + self._ensure_fresh() + except Exception as exc: # noqa: BLE001 - a stray error must NOT kill the thread + # If this thread dies, nothing refreshes and the session lapses at + # the ~1h mark until restart. Log and keep looping instead. + _log_refresh_failure(exc) + + def stop(self) -> None: + self._stop.set() + + +def _forwarded_request_headers( + handler: BaseHTTPRequestHandler, + token: str, + token_header: str = AI_GATEWAY_TOKEN_HEADER, +) -> dict[str, str]: + strip_on_forward = _HOP_BY_HOP | {token_header.lower()} + headers = { + key: value for key, value in handler.headers.items() if key.lower() not in strip_on_forward + } + headers[token_header] = f"Bearer {token}" + return headers + + +class _ProxyHandler(BaseHTTPRequestHandler): + # Set by the server factory. + cache: _TokenCache + client: httpx.Client + token_header = AI_GATEWAY_TOKEN_HEADER + + def log_message(self, format: str, *args: object) -> None: + return + + def _safe_send_error(self, code: int, message: str) -> None: + # The client (Claude Code) may already have disconnected, in which case + # reporting the error writes to a dead socket and raises again; swallow it. + try: + self.send_error(code, message) + except OSError: + pass + + def _transform_request(self, body: bytes | None) -> tuple[str, bytes | None]: + return self.path.lstrip("/"), body + + def _response_chunks(self, resp: httpx.Response) -> tuple[Iterable[bytes], frozenset[str]]: + return resp.iter_raw(), frozenset() + + def _handle(self) -> None: + diagnostic_id = uuid.uuid4().hex[:12] + started = time.monotonic() + length = int(self.headers.get("Content-Length", 0) or 0) + body = self.rfile.read(length) if length else None + url, body = self._transform_request(body) + _diagnostic_log( + "request_start", + request_id=diagnostic_id, + method=self.command, + path=self.path.split("?", 1)[0], + ) + try: + # First attempt with the current token. + headers = _forwarded_request_headers(self, self.cache.token, self.token_header) + with self.client.stream(self.command, url, headers=headers, content=body) as resp: + _diagnostic_log( + "upstream_headers", + request_id=diagnostic_id, + attempt=1, + status=resp.status_code, + elapsed_ms=round((time.monotonic() - started) * 1000), + ) + if resp.status_code not in (401, 403): + self._relay_response(resp, diagnostic_id=diagnostic_id, started=started) + return + # Auth rejected. Drain the (small) error body so the pooled + # connection can be reused, then fall through to one retry. + resp.read() + # A relayed 401/403 may be a stale Databricks swap token rather than a + # bad Anthropic OAuth — the two are indistinguishable from the status + # alone. Force-refresh the Databricks token and retry once. If it was the + # Anthropic layer, the retry still 401s and we relay it verbatim, so a + # genuine re-auth is triggered; a stale-Databricks 401 self-heals here + # instead of surfacing to Claude Code as a spurious Anthropic prompt. + try: + self.cache.refresh() + except RuntimeError as exc: + # Refresh failed: the Databricks OAuth session is dead (not just the + # access token) and can't be re-minted non-interactively. Surface the + # `databricks auth login` hint rather than silently relaying a bare 401, + # which otherwise reads as an Anthropic `/login` prompt and sends the + # user to the wrong re-auth. Still retry + relay with the existing token. + _log_refresh_failure(exc) + headers = _forwarded_request_headers(self, self.cache.token, self.token_header) + with self.client.stream(self.command, url, headers=headers, content=body) as resp: + _diagnostic_log( + "upstream_headers", + request_id=diagnostic_id, + attempt=2, + status=resp.status_code, + elapsed_ms=round((time.monotonic() - started) * 1000), + ) + self._relay_response(resp, diagnostic_id=diagnostic_id, started=started) + except (BrokenPipeError, ConnectionResetError): + # Client closed before/while we relayed headers — routine on cancel. + _diagnostic_log( + "client_disconnect", + request_id=diagnostic_id, + phase="request", + elapsed_ms=round((time.monotonic() - started) * 1000), + ) + return + except httpx.HTTPError as exc: + # Upstream failed before any bytes reached the client; a 502 is still + # sendable. (An HTTP *status* like 429 is not an error here — httpx + # only raises for transport failures — so real gateway errors are + # relayed verbatim by `_relay_response`.) + _diagnostic_log( + "upstream_request_error", + request_id=diagnostic_id, + error_type=type(exc).__name__, + elapsed_ms=round((time.monotonic() - started) * 1000), + ) + self._safe_send_error(502, "gateway proxy upstream error") + + # Streaming passthrough: forward chunks as they arrive so SSE token streaming + # is not buffered (buffering would add full-response latency to first token). + # `iter_raw` preserves any Content-Encoding verbatim (we relay that header), + # so the proxy stays byte-transparent. + def _relay_response( + self, + resp: httpx.Response, + *, + diagnostic_id: str | None = None, + started: float | None = None, + ) -> None: + started = started if started is not None else time.monotonic() + chunks = 0 + bytes_relayed = 0 + first_byte_ms: int | None = None + try: + # The upstream request has completed through response headers before + # this hook selects raw streaming or a buffered response body. + response_chunks, dropped_headers = self._response_chunks(resp) + self.send_response(resp.status_code) + for key, value in resp.headers.items(): + header_name = key.lower() + if header_name not in _HOP_BY_HOP and header_name not in dropped_headers: + self.send_header(key, value) + self.end_headers() + # Do not pass a fixed chunk size here. httpx accumulates bytes until + # that size is reached, which can hide small SSE heartbeat frames + # from Claude Code for minutes during a slow artifact/tool call. + # With ``chunk_size=None`` (the default), raw upstream chunks are + # yielded as they arrive and pings keep the downstream connection + # alive even before the model produces a large content block. + for chunk in response_chunks: + if chunk: + if first_byte_ms is None: + first_byte_ms = round((time.monotonic() - started) * 1000) + self.wfile.write(chunk) + self.wfile.flush() + chunks += 1 + bytes_relayed += len(chunk) + _diagnostic_log( + "response_complete", + request_id=diagnostic_id, + status=resp.status_code, + chunks=chunks, + bytes=bytes_relayed, + first_byte_ms=first_byte_ms, + elapsed_ms=round((time.monotonic() - started) * 1000), + ) + except (BrokenPipeError, ConnectionResetError): + # Client (Claude Code) closed the connection mid-response — routine on + # cancelled turns / SSE teardown. Nothing left to relay to, so stop + # quietly rather than crashing the handler thread. + _diagnostic_log( + "client_disconnect", + request_id=diagnostic_id, + phase="response", + chunks=chunks, + bytes=bytes_relayed, + elapsed_ms=round((time.monotonic() - started) * 1000), + ) + return + except httpx.HTTPError as exc: + # Upstream dropped mid-stream. Headers (and status) may already be + # sent, so we can't reliably signal a fresh error — stop and let the + # client see a truncated stream rather than corrupt the framing. + _diagnostic_log( + "upstream_stream_error", + request_id=diagnostic_id, + error_type=type(exc).__name__, + status=resp.status_code, + chunks=chunks, + bytes=bytes_relayed, + elapsed_ms=round((time.monotonic() - started) * 1000), + ) + return + + # Forward every method: this is a transparent pass-through, so routing any + # `do_` lookup to `_handle` lets the gateway reject unsupported methods. + def __getattr__(self, name: str): + if name.startswith("do_"): + return self._handle + raise AttributeError(name) + + +_MODEL_ALIAS_PREFIX = "anthropic-aigw-" +_ANTHROPIC_MODELS_PATH = "/v1/models" +_ANTHROPIC_MESSAGES_PATH = "/v1/messages" + + +class _AnthropicModelAliases: + """Maps Claude-compatible discovery IDs back to their gateway model IDs.""" + + def __init__(self) -> None: + self._original_by_alias: dict[str, str] = {} + self._lock = threading.Lock() + + def prefix_model_ids(self, body: bytes) -> bytes: + try: + payload = json.loads(body) + models = payload["data"] + if not isinstance(models, list): + return body + except (UnicodeDecodeError, json.JSONDecodeError, KeyError, TypeError): + return body + + aliases: dict[str, str] = {} + for model in models: + if not isinstance(model, dict) or not isinstance(model.get("id"), str): + continue + model_id = model["id"] + lowered = model_id.lower() + if "claude" in lowered or "anthropic" in lowered: + continue + alias = f"{_MODEL_ALIAS_PREFIX}{model_id}" + model["id"] = alias + aliases[alias] = model_id + + with self._lock: + self._original_by_alias.update(aliases) + + for cursor in ("first_id", "last_id"): + model_id = payload.get(cursor) + alias = f"{_MODEL_ALIAS_PREFIX}{model_id}" + if alias in aliases: + payload[cursor] = alias + return json.dumps(payload, separators=(",", ":")).encode() + + def original_id(self, model_id: str) -> str: + with self._lock: + return self._original_by_alias.get(model_id, model_id) + + def rewrite_path(self, path: str) -> str: + parsed = urlsplit(path) + if parsed.path != _ANTHROPIC_MODELS_PATH: + return path + query = [ + (key, self.original_id(value) if key == "after_id" else value) + for key, value in parse_qsl(parsed.query, keep_blank_values=True) + ] + return urlunsplit( + (parsed.scheme, parsed.netloc, parsed.path, urlencode(query), parsed.fragment) + ) + + def rewrite_body(self, path: str, body: bytes | None) -> bytes | None: + if urlsplit(path).path != _ANTHROPIC_MESSAGES_PATH or body is None: + return body + try: + payload = json.loads(body) + model_id = payload.get("model") + if not isinstance(model_id, str): + return body + except (UnicodeDecodeError, json.JSONDecodeError, AttributeError): + return body + original_id = self.original_id(model_id) + if original_id == model_id: + return body + payload["model"] = original_id + return json.dumps(payload, separators=(",", ":")).encode() + + +class _AnthropicGatewayHandler(_ProxyHandler): + anthropic_model_aliases: _AnthropicModelAliases + + def _transform_request(self, body: bytes | None) -> tuple[str, bytes | None]: + body = self.anthropic_model_aliases.rewrite_body(self.path, body) + url = self.anthropic_model_aliases.rewrite_path(self.path).lstrip("/") + return url, body + + def _response_chunks(self, resp: httpx.Response) -> tuple[Iterable[bytes], frozenset[str]]: + should_prefix_model_ids = ( + self.command == "GET" + and urlsplit(self.path).path == _ANTHROPIC_MODELS_PATH + and HTTPStatus.OK <= resp.status_code < HTTPStatus.MULTIPLE_CHOICES + ) + if not should_prefix_model_ids: + return super()._response_chunks(resp) + body = self.anthropic_model_aliases.prefix_model_ids(resp.read()) + # resp.read() decodes compression; rewritten JSON is uncompressed. + return (body,), frozenset({"content-encoding"}) + + +def start_proxy( + workspace: str, + profile: str | None, + port: int, + token_header: str, + force_refresh_near_expiry: bool, +) -> tuple[ThreadingHTTPServer, _TokenCache, httpx.Client]: + """Start the Anthropic loopback proxy and token refresher.""" + upstream_base = f"{workspace.rstrip('/')}/ai-gateway/anthropic/" + cache = _TokenCache( + workspace, + profile, + force_refresh_near_expiry=force_refresh_near_expiry, + ) + client = httpx.Client(base_url=upstream_base, timeout=_UPSTREAM_TIMEOUT, follow_redirects=False) + handler = cast( + type[BaseHTTPRequestHandler], + type( + "BoundProxyHandler", + (_AnthropicGatewayHandler,), + { + "cache": cache, + "client": client, + "token_header": token_header, + "anthropic_model_aliases": _AnthropicModelAliases(), + }, + ), + ) + try: + server = ThreadingHTTPServer((LOOPBACK_HOST, port), handler) + except OSError: + server = ThreadingHTTPServer((LOOPBACK_HOST, 0), handler) + + refresher = threading.Thread(target=cache.run_refresher, daemon=True) + refresher.start() + return server, cache, client diff --git a/src/ucode/constants.py b/src/ucode/constants.py new file mode 100644 index 0000000..3e9ff67 --- /dev/null +++ b/src/ucode/constants.py @@ -0,0 +1,3 @@ +"""Shared UCode constants.""" + +LOOPBACK_HOST = "127.0.0.1" diff --git a/tests/test_agent_claude.py b/tests/test_agent_claude.py index 6b49782..e66ed27 100644 --- a/tests/test_agent_claude.py +++ b/tests/test_agent_claude.py @@ -666,7 +666,7 @@ def test_default_launch_keeps_existing_auth_path(self, monkeypatch): assert os.environ["OAUTH_TOKEN"] == "token" assert calls == [["claude", "--settings", str(claude.CLAUDE_SETTINGS_PATH), "--debug"]] - def test_runs_through_refresh_proxy(self, monkeypatch): + def test_gateway_discovery_uses_anthropic_proxy(self, monkeypatch): calls: list[tuple] = [] class Server: @@ -713,7 +713,7 @@ def start_proxy(workspace, profile, port, token_header, force_refresh_near_expir monkeypatch.delenv("ANTHROPIC_AUTH_TOKEN", raising=False) monkeypatch.delenv("ANTHROPIC_BASE_URL", raising=False) monkeypatch.delenv("CLAUDE_CODE_USE_GATEWAY", raising=False) - monkeypatch.setattr(claude, "start_proxy", start_proxy) + monkeypatch.setattr(claude, "start_anthropic_gateway_proxy", start_proxy) monkeypatch.setattr(claude.subprocess, "Popen", Process) with pytest.raises(SystemExit) as exc: diff --git a/tests/test_anthropic_gateway_proxy.py b/tests/test_anthropic_gateway_proxy.py new file mode 100644 index 0000000..6a2f7e9 --- /dev/null +++ b/tests/test_anthropic_gateway_proxy.py @@ -0,0 +1,249 @@ +"""Tests for Anthropic model discovery transformations.""" + +from __future__ import annotations + +import io +import json + +from ucode import anthropic_gateway_proxy + + +class _FakeResponse: + def __init__(self, status_code: int, headers: dict[str, str], body: bytes): + self.status_code = status_code + self.headers = headers + self._body = body + self.read_calls = 0 + self.iter_raw_calls = 0 + + def read(self): + self.read_calls += 1 + return self._body + + def iter_raw(self): + self.iter_raw_calls += 1 + yield self._body + + def __enter__(self): + return self + + def __exit__(self, *_args): + return False + + +class _FakeClient: + def __init__(self, response): + self.response = response + self.request = None + + def stream(self, method, url, headers, content): + self.request = (method, url, headers, content) + return self.response + + +class _FakeCache: + token = "databricks-token" + + def refresh(self): + return None + + +class _Collect(io.RawIOBase): + def __init__(self): + self.data = bytearray() + + def write(self, body): # type: ignore[override] + self.data += bytes(body) + return len(body) + + def flush(self): + return None + + +def _handler(wfile, path="/v1/models", command="GET"): + handler = object.__new__(anthropic_gateway_proxy._AnthropicGatewayHandler) + handler.wfile = wfile + handler.request_version = "HTTP/1.1" + handler.requestline = f"{command} {path} HTTP/1.1" + handler.command = command + handler.path = path + handler._headers_buffer = [] + handler.anthropic_model_aliases = anthropic_gateway_proxy._AnthropicModelAliases() + return handler + + +class TestAnthropicModelAliases: + def test_prefixes_custom_model_ids_without_changing_display_name(self): + aliases = anthropic_gateway_proxy._AnthropicModelAliases() + body = json.dumps( + { + "data": [ + {"id": "catalog.schema.custom", "display_name": "Custom model"}, + {"id": "system.ai.claude-sonnet"}, + {"id": "claude-sonnet"}, + {"id": "catalog.schema.anthropic-provider"}, + {"id": "anthropic-provider"}, + ], + "first_id": "catalog.schema.custom", + "last_id": "catalog.schema.anthropic-provider", + } + ).encode() + + payload = json.loads(aliases.prefix_model_ids(body)) + + assert payload == { + "data": [ + { + "id": "anthropic-aigw-catalog.schema.custom", + "display_name": "Custom model", + }, + {"id": "system.ai.claude-sonnet"}, + {"id": "claude-sonnet"}, + {"id": "catalog.schema.anthropic-provider"}, + {"id": "anthropic-provider"}, + ], + "first_id": "anthropic-aigw-catalog.schema.custom", + "last_id": "catalog.schema.anthropic-provider", + } + + def test_rewrites_known_alias_in_messages_body(self): + aliases = anthropic_gateway_proxy._AnthropicModelAliases() + aliases.prefix_model_ids(b'{"data":[{"id":"catalog.schema.custom"}]}') + + body = aliases.rewrite_body( + "/v1/messages", + b'{"model":"anthropic-aigw-catalog.schema.custom","messages":[]}', + ) + + assert json.loads(body) == {"model": "catalog.schema.custom", "messages": []} + + def test_rewrites_known_alias_in_pagination_cursor(self): + aliases = anthropic_gateway_proxy._AnthropicModelAliases() + aliases.prefix_model_ids(b'{"data":[{"id":"catalog.schema.custom"}]}') + + assert ( + aliases.rewrite_path( + "/v1/models?limit=1000&after_id=anthropic-aigw-catalog.schema.custom" + ) + == "/v1/models?limit=1000&after_id=catalog.schema.custom" + ) + + def test_ignores_non_anthropic_models_path(self): + aliases = anthropic_gateway_proxy._AnthropicModelAliases() + path = "/codex/v1/models?after_id=anthropic-aigw-catalog.schema.custom" + + assert aliases.rewrite_path(path) == path + + def test_does_not_strip_unknown_prefixed_id(self): + aliases = anthropic_gateway_proxy._AnthropicModelAliases() + unknown = "anthropic-aigw-legitimate-upstream-id" + + assert aliases.rewrite_path(f"/v1/models?after_id={unknown}") == ( + f"/v1/models?after_id={unknown}" + ) + assert ( + aliases.rewrite_body("/v1/messages", json.dumps({"model": unknown}).encode()) + == json.dumps({"model": unknown}).encode() + ) + + def test_leaves_malformed_discovery_response_unchanged(self): + aliases = anthropic_gateway_proxy._AnthropicModelAliases() + assert aliases.prefix_model_ids(b"not-json") == b"not-json" + + +class TestAnthropicGatewayHandler: + def test_inherits_relayed_auth_and_prefixes_models(self): + out = _Collect() + handler = _handler(out) + handler.headers = {"Authorization": "Bearer subscription-token"} + handler.rfile = io.BytesIO() + handler.cache = _FakeCache() + handler.client = _FakeClient(_FakeResponse(200, {}, b'{"data":[{"id":"custom-model"}]}')) + + handler._handle() + + method, url, headers, body = handler.client.request + assert (method, url, body) == ("GET", "v1/models", None) + assert headers["Authorization"] == "Bearer subscription-token" + assert headers["X-Databricks-AI-Gateway-Token"] == "Bearer databricks-token" + assert b"anthropic-aigw-custom-model" in bytes(out.data) + + def test_prefixes_successful_model_response_and_drops_content_encoding(self): + out = _Collect() + handler = _handler(out) + response = _FakeResponse( + 200, + {"Content-Encoding": "gzip"}, + b'{"data":[{"id":"custom-model"}]}', + ) + + handler._relay_response(response) + + assert b"Content-Encoding" not in bytes(out.data) + assert b"anthropic-aigw-custom-model" in bytes(out.data) + + def test_keeps_content_encoding_for_unchanged_error(self): + out = _Collect() + handler = _handler(out) + response = _FakeResponse(400, {"Content-Encoding": "gzip"}, b"compressed-error") + + handler._relay_response(response) + + assert b"Content-Encoding: gzip" in bytes(out.data) + assert b"compressed-error" in bytes(out.data) + assert response.read_calls == 0 + assert response.iter_raw_calls == 1 + + def test_streams_relayed_inference_response_without_buffering(self): + out = _Collect() + handler = _handler(out, path="/v1/messages", command="POST") + handler.headers = {"Authorization": "Bearer subscription-token", "Content-Length": "2"} + handler.rfile = io.BytesIO(b"{}") + handler.cache = _FakeCache() + response = _FakeResponse(200, {"Content-Type": "text/event-stream"}, b"data: event\n\n") + handler.client = _FakeClient(response) + + handler._handle() + + _method, _url, headers, _body = handler.client.request + assert headers["Authorization"] == "Bearer subscription-token" + assert headers["X-Databricks-AI-Gateway-Token"] == "Bearer databricks-token" + assert response.read_calls == 0 + assert response.iter_raw_calls == 1 + assert b"data: event\n\n" in bytes(out.data) + + def test_strips_known_alias_from_message_request(self): + handler = _handler(_Collect(), path="/v1/messages", command="POST") + handler.anthropic_model_aliases.prefix_model_ids( + b'{"data":[{"id":"catalog.schema.custom"}]}' + ) + + url, body = handler._transform_request(b'{"model":"anthropic-aigw-catalog.schema.custom"}') + + assert url == "v1/messages" + assert json.loads(body) == {"model": "catalog.schema.custom"} + + +def test_start_proxy_uses_discovery_handler(monkeypatch): + class _StubCache: + def run_refresher(self): + return None + + cache = _StubCache() + monkeypatch.setattr(anthropic_gateway_proxy, "_TokenCache", lambda *_args, **_kwargs: cache) + + server, actual_cache, client = anthropic_gateway_proxy.start_proxy( + "https://workspace.example.com", "profile", 0, "header", False + ) + try: + handler = server.RequestHandlerClass + assert issubclass(handler, anthropic_gateway_proxy._AnthropicGatewayHandler) + assert handler.cache is cache + assert isinstance( + handler.anthropic_model_aliases, + anthropic_gateway_proxy._AnthropicModelAliases, + ) + assert actual_cache is cache + finally: + server.server_close() + client.close()