The compact pane is prompt-only because the feed above shows the context. An error is the exception: the feed cannot show it. A tool_result only reaches us with the next request, which a turn that died never sends, and an `API Error` row is Claude Code's own text, never in the stream at all. So the frame walks further up to take the report in, and the scheduled ctrl-l wipe is cancelled while one is framed — Ink redraws only the live frame, so wiping would delete the error 400ms after it appeared. Detection is text shape, not colour: the tool error renders palette 211 and the API error 220, both theme values. ERROR_SCAN is the only bound, which is what makes it self-limiting — the error drifts out of the window as the next turn prints output and the pane shrinks back on its own.
356 lines
18 KiB
Python
356 lines
18 KiB
Python
#!/usr/bin/env python3
|
|
"""Offline stand-in for api.anthropic.com — zero API usage.
|
|
|
|
Point the proxy at it (`CT_UPSTREAM=http://127.0.0.1:9911`) and every request a
|
|
real `claude` child makes is answered locally with a canned SSE stream. That
|
|
lets us make Claude Code render its client-side tool UIs (AskUserQuestion,
|
|
ExitPlanMode, TodoWrite/Task*) on demand, so the pane's frame detector in
|
|
`src/term.rs` can be developed against what Ink actually draws.
|
|
|
|
Scenario is picked per turn from `CT_FAKE_SCENARIO`
|
|
(ask | plan | todo | taskupdate | agent | websearch | ansi | toolerror | apierror | text).
|
|
|
|
`websearch` also answers the *nested* request Claude Code makes to run WebSearch:
|
|
that call declares Anthropic's server-side `web_search` tool, so it is replied to
|
|
with `server_tool_use` + `web_search_tool_result` + `citations_delta` — the block
|
|
types only a hosted tool produces.
|
|
Each incoming request is logged to `dev/fake_upstream.log` (declared tool names
|
|
+ the trailing user text) so we can see what CC sends.
|
|
"""
|
|
|
|
import json
|
|
import os
|
|
import sys
|
|
import time
|
|
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
|
|
|
LOG = os.path.join(os.path.dirname(os.path.abspath(__file__)), "fake_upstream.log")
|
|
MODEL = "claude-opus-4-5-20251101"
|
|
|
|
|
|
def log(msg):
|
|
with open(LOG, "a") as f:
|
|
f.write(msg + "\n")
|
|
|
|
|
|
def read_scenario():
|
|
"""Scenario for the *next* turn: `dev/.fake_scenario` wins over the env, so
|
|
you can switch scenarios without restarting the server."""
|
|
path = os.path.join(os.path.dirname(os.path.abspath(__file__)), ".fake_scenario")
|
|
try:
|
|
with open(path) as f:
|
|
return f.read().strip()
|
|
except OSError:
|
|
return os.environ.get("CT_FAKE_SCENARIO", "ask")
|
|
|
|
|
|
def sse(event, data):
|
|
return f"event: {event}\ndata: {json.dumps(data)}\n\n".encode()
|
|
|
|
|
|
def stream_text(text):
|
|
"""A plain assistant text turn."""
|
|
yield sse("message_start", {"type": "message_start", "message": {
|
|
"id": "msg_fake", "type": "message", "role": "assistant", "model": MODEL,
|
|
"content": [], "stop_reason": None, "stop_sequence": None,
|
|
"usage": {"input_tokens": 12, "output_tokens": 1}}})
|
|
yield sse("content_block_start", {"type": "content_block_start", "index": 0,
|
|
"content_block": {"type": "text", "text": ""}})
|
|
for chunk in text.split(" "):
|
|
yield sse("content_block_delta", {"type": "content_block_delta", "index": 0,
|
|
"delta": {"type": "text_delta", "text": chunk + " "}})
|
|
yield sse("content_block_stop", {"type": "content_block_stop", "index": 0})
|
|
yield sse("message_delta", {"type": "message_delta",
|
|
"delta": {"stop_reason": "end_turn", "stop_sequence": None},
|
|
"usage": {"output_tokens": 9}})
|
|
yield sse("message_stop", {"type": "message_stop"})
|
|
|
|
|
|
_TOOL_SEQ = 0
|
|
|
|
|
|
def stream_tool(name, tool_input, lead="Working on it."):
|
|
"""A turn that calls one client-side tool."""
|
|
yield from stream_tools([(name, tool_input)], lead)
|
|
|
|
|
|
def stream_tools(calls, lead="Working on it."):
|
|
"""A turn that calls several client-side tools in one message."""
|
|
yield sse("message_start", {"type": "message_start", "message": {
|
|
"id": "msg_fake", "type": "message", "role": "assistant", "model": MODEL,
|
|
"content": [], "stop_reason": None, "stop_sequence": None,
|
|
"usage": {"input_tokens": 12, "output_tokens": 1}}})
|
|
yield sse("content_block_start", {"type": "content_block_start", "index": 0,
|
|
"content_block": {"type": "text", "text": ""}})
|
|
yield sse("content_block_delta", {"type": "content_block_delta", "index": 0,
|
|
"delta": {"type": "text_delta", "text": lead}})
|
|
yield sse("content_block_stop", {"type": "content_block_stop", "index": 0})
|
|
for n, (name, tool_input) in enumerate(calls, start=1):
|
|
# Ids must be unique across the whole session, exactly as the real API
|
|
# guarantees: Claude Code resends the full history every request, so a
|
|
# reused id makes an *old* tool_result re-attach to the newest call.
|
|
global _TOOL_SEQ
|
|
_TOOL_SEQ += 1
|
|
yield sse("content_block_start", {"type": "content_block_start", "index": n,
|
|
"content_block": {"type": "tool_use", "id": f"toolu_fake{_TOOL_SEQ}",
|
|
"name": name, "input": {}}})
|
|
blob = json.dumps(tool_input)
|
|
for i in range(0, len(blob), 40):
|
|
yield sse("content_block_delta", {"type": "content_block_delta", "index": n,
|
|
"delta": {"type": "input_json_delta", "partial_json": blob[i:i + 40]}})
|
|
yield sse("content_block_stop", {"type": "content_block_stop", "index": n})
|
|
yield sse("message_delta", {"type": "message_delta",
|
|
"delta": {"stop_reason": "tool_use", "stop_sequence": None},
|
|
"usage": {"output_tokens": 40}})
|
|
yield sse("message_stop", {"type": "message_stop"})
|
|
|
|
|
|
def stream_server_websearch():
|
|
"""What a *hosted* web_search turn looks like: `server_tool_use`, then a
|
|
complete `web_search_tool_result` block (no deltas — the whole payload
|
|
rides in `content_block_start`), then cited text. Claude Code issues this
|
|
nested request itself when it runs the client-side `WebSearch` tool."""
|
|
yield sse("message_start", {"type": "message_start", "message": {
|
|
"id": "msg_ws", "type": "message", "role": "assistant", "model": MODEL,
|
|
"content": [], "stop_reason": None, "stop_sequence": None,
|
|
"usage": {"input_tokens": 50, "output_tokens": 1}}})
|
|
yield sse("content_block_start", {"type": "content_block_start", "index": 0,
|
|
"content_block": {"type": "server_tool_use", "id": "srvtoolu_fake1",
|
|
"name": "web_search", "input": {}}})
|
|
blob = json.dumps({"query": "ratatui scrollbar thumb"})
|
|
yield sse("content_block_delta", {"type": "content_block_delta", "index": 0,
|
|
"delta": {"type": "input_json_delta", "partial_json": blob}})
|
|
yield sse("content_block_stop", {"type": "content_block_stop", "index": 0})
|
|
yield sse("content_block_start", {"type": "content_block_start", "index": 1,
|
|
"content_block": {
|
|
"type": "web_search_tool_result", "tool_use_id": "srvtoolu_fake1",
|
|
"content": [
|
|
{"type": "web_search_result", "title": "Ratatui Scrollbar docs",
|
|
"url": "https://ratatui.rs/widgets/scrollbar", "page_age": "2 days"},
|
|
{"type": "web_search_result", "title": "Scrollbar example",
|
|
"url": "https://ratatui.rs/examples/scrollbar", "page_age": None},
|
|
]}})
|
|
yield sse("content_block_stop", {"type": "content_block_stop", "index": 1})
|
|
yield sse("content_block_start", {"type": "content_block_start", "index": 2,
|
|
"content_block": {"type": "text", "text": ""}})
|
|
for chunk in "Ratatui renders the thumb through its Scrollbar widget. ".split(" "):
|
|
yield sse("content_block_delta", {"type": "content_block_delta", "index": 2,
|
|
"delta": {"type": "text_delta", "text": chunk + " "}})
|
|
yield sse("content_block_delta", {"type": "content_block_delta", "index": 2,
|
|
"delta": {"type": "citations_delta", "citation": {
|
|
"type": "web_search_result_location",
|
|
"url": "https://ratatui.rs/widgets/scrollbar",
|
|
"title": "Ratatui Scrollbar docs",
|
|
"cited_text": "Scrollbar renders a thumb over the track."}}})
|
|
yield sse("content_block_stop", {"type": "content_block_stop", "index": 2})
|
|
yield sse("message_delta", {"type": "message_delta",
|
|
"delta": {"stop_reason": "end_turn", "stop_sequence": None},
|
|
"usage": {"output_tokens": 30, "server_tool_use": {"web_search_requests": 1}}})
|
|
yield sse("message_stop", {"type": "message_stop"})
|
|
|
|
|
|
# A command whose *output* carries real SGR codes, so the feed's ANSI handling
|
|
# is exercised by a genuine tool_result rather than a hand-written fixture.
|
|
ANSI_INPUT = {
|
|
"command": (
|
|
"printf '\\033[1mbold heading\\033[22m\\n'; "
|
|
"printf '\\033[31m- removed line\\033[0m\\n'; "
|
|
"printf '\\033[32m+ added line\\033[0m\\n'; "
|
|
"printf '\\033[38;5;208m256-colour orange\\033[0m\\n'"
|
|
),
|
|
"description": "Print coloured output",
|
|
}
|
|
|
|
|
|
# A tool call that *fails* on the child's side, so Claude Code prints its own
|
|
# red error row (`⎿ Error: …`) above the input box. `Read` is auto-approved, so
|
|
# this needs no permission prompt — the failure comes from the missing file.
|
|
TOOL_ERROR_INPUT = {"file_path": "/nonexistent/claude-cloak/does-not-exist.txt"}
|
|
|
|
|
|
ASK_INPUT = {"questions": [{
|
|
"question": "The compact pane currently crops the top of this prompt. Which framing "
|
|
"should the pane use when an interactive question is on screen, given that "
|
|
"the question text itself can wrap over several rows?",
|
|
"header": "Framing",
|
|
"multiSelect": False,
|
|
"options": [
|
|
{"label": "Measure the box", "description": "Frame from the question box's own top border down to the input box, so nothing is cut."},
|
|
{"label": "Fixed 75% height", "description": "Always give the pane three quarters of the screen while a question waits, however tall it really is."},
|
|
{"label": "Estimate from JSON", "description": "Keep guessing the row count from the tool input, as today, and accept that wrapped text breaks the guess."},
|
|
{"label": "Fullscreen the pane", "description": "Take the whole screen for as long as a question is waiting for an answer."},
|
|
]}, {
|
|
"question": "And what should happen when the prompt is taller than the pane can ever be?",
|
|
"header": "Overflow",
|
|
"multiSelect": False,
|
|
"options": [
|
|
{"label": "Keep the selection", "description": "Slide the window so the highlighted option stays visible."},
|
|
{"label": "Crop the tail", "description": "Always top-anchor and let the last options fall off."},
|
|
]}]}
|
|
|
|
PLAN_INPUT = {"plan": "## Plan\n\n1. Size the PTY to the whole screen.\n2. Measure the "
|
|
"rendered box instead of guessing rows.\n3. Extend the frame over an "
|
|
"active task list.\n\nThis is a long enough plan to need several rows."}
|
|
|
|
TODO_INPUT = {"todos": [
|
|
{"content": "Capture ground truth screens", "status": "completed", "activeForm": "Capturing ground truth screens"},
|
|
{"content": "Fix interactive pane sizing", "status": "in_progress", "activeForm": "Fixing interactive pane sizing"},
|
|
{"content": "Expand pane while a task list is active", "status": "pending", "activeForm": "Expanding pane for task lists"},
|
|
{"content": "Add regression tests", "status": "pending", "activeForm": "Adding regression tests"},
|
|
{"content": "Update CLAUDE.md", "status": "pending", "activeForm": "Updating CLAUDE.md"},
|
|
]}
|
|
|
|
|
|
TASK_INPUTS = [
|
|
{"subject": "Capture ground truth screens", "description": "Drive a real claude child offline.",
|
|
"activeForm": "Capturing ground truth screens"},
|
|
{"subject": "Fix interactive pane sizing", "description": "Measure the box instead of guessing rows.",
|
|
"activeForm": "Fixing interactive pane sizing"},
|
|
{"subject": "Expand pane while a task list is active", "description": "Show task status near the prompt.",
|
|
"activeForm": "Expanding pane for task lists"},
|
|
{"subject": "Add regression tests", "description": "Pin the framing math.",
|
|
"activeForm": "Adding regression tests"},
|
|
]
|
|
|
|
|
|
# Two parallel subagents with the *same* type and byte-identical prompts: the
|
|
# worst case for correlation (only Claude Code's `x-claude-code-agent-id` tells
|
|
# them apart), and what makes the subagent rows worth having.
|
|
AGENT_INPUTS = [
|
|
{"subagent_type": "Explore", "description": "find the retry helper",
|
|
"prompt": "Locate the retry helper and report back.", "run_in_background": False},
|
|
{"subagent_type": "Explore", "description": "find the retry helper",
|
|
"prompt": "Locate the retry helper and report back.", "run_in_background": False},
|
|
{"subagent_type": "oracle", "description": "review the lane design",
|
|
"prompt": "Review the lane design and report back.", "run_in_background": True},
|
|
]
|
|
|
|
|
|
class Handler(BaseHTTPRequestHandler):
|
|
protocol_version = "HTTP/1.1"
|
|
|
|
def log_message(self, *a): # keep stdout clean
|
|
pass
|
|
|
|
def do_POST(self):
|
|
n = int(self.headers.get("content-length", 0))
|
|
raw = self.rfile.read(n) if n else b"{}"
|
|
try:
|
|
body = json.loads(raw)
|
|
except Exception:
|
|
body = {}
|
|
raw_tools = body.get("tools", []) or []
|
|
tools = [t.get("name") for t in raw_tools]
|
|
# A hosted tool carries a `type` and no `input_schema`; that is the
|
|
# nested WebSearch call, not a turn start.
|
|
hosted = bool(raw_tools) and all(
|
|
t.get("input_schema") is None and t.get("type") not in (None, "custom")
|
|
for t in raw_tools
|
|
)
|
|
msgs = body.get("messages", []) or []
|
|
tail = json.dumps(msgs[-1])[:300] if msgs else ""
|
|
# Only the immediate reply to *our* canned tool call ends the turn with
|
|
# text. A user message can carry both the tool_result and fresh user
|
|
# text (a new prompt typed after declining), and that is a new turn —
|
|
# so require the tool_result to be the *whole* last message.
|
|
last = msgs[-1] if msgs else {}
|
|
blocks = last.get("content") if isinstance(last.get("content"), list) else []
|
|
has_result = bool(blocks) and all(
|
|
b.get("type") == "tool_result" for b in blocks if isinstance(b, dict)
|
|
) and "toolu_fake" in json.dumps(blocks)
|
|
# `model` + `anthropic-beta` show how a `--model opus[1m]` pick reaches
|
|
# the wire (the 1M context window is a beta header, not a model id).
|
|
log(
|
|
f"[{time.strftime('%H:%M:%S')}] {self.path} model={body.get('model')} "
|
|
f"betas={self.headers.get('anthropic-beta', '')} tools={tools} tail={tail}"
|
|
)
|
|
|
|
if "count_tokens" in self.path:
|
|
out = json.dumps({"input_tokens": 100}).encode()
|
|
self.send_response(200)
|
|
self.send_header("content-type", "application/json")
|
|
self.send_header("content-length", str(len(out)))
|
|
self.end_headers()
|
|
self.wfile.write(out)
|
|
return
|
|
|
|
scenario = read_scenario()
|
|
# `apierror`: a non-retryable upstream failure, so Claude Code prints
|
|
# its own red `API Error: 400 …` row where the turn would have been.
|
|
# Only the *turn* request fails — the side/title calls still answer, so
|
|
# the pane behaves normally around the error.
|
|
if scenario == "apierror" and tools and not hosted and not has_result:
|
|
out = json.dumps({"type": "error", "error": {
|
|
"type": "invalid_request_error",
|
|
"message": "fake upstream rejected this turn on purpose"}}).encode()
|
|
self.send_response(400)
|
|
self.send_header("content-type", "application/json")
|
|
self.send_header("content-length", str(len(out)))
|
|
self.end_headers()
|
|
self.wfile.write(out)
|
|
return
|
|
|
|
# The nested hosted-tool request answers itself, whatever the scenario.
|
|
if hosted:
|
|
gen = stream_server_websearch()
|
|
# A request with no tools is CC's side/title call — answer with text.
|
|
elif not tools or has_result:
|
|
gen = stream_text("Done. Ask me anything else.")
|
|
elif scenario == "ask":
|
|
gen = stream_tool("AskUserQuestion", ASK_INPUT, "Let me check how you want this framed.")
|
|
elif scenario == "plan":
|
|
gen = stream_tool("ExitPlanMode", PLAN_INPUT, "Here is the plan.")
|
|
elif scenario == "todo":
|
|
# Which task tool exists depends on the child's agent: a plain CC
|
|
# session has TodoWrite, an agent like `minimal` has the newer
|
|
# TaskCreate/TaskUpdate set instead.
|
|
if "TodoWrite" in tools:
|
|
gen = stream_tool("TodoWrite", TODO_INPUT, "Setting up the task list.")
|
|
else:
|
|
gen = stream_tools(
|
|
[("TaskCreate", t) for t in TASK_INPUTS], "Setting up the task list."
|
|
)
|
|
elif scenario == "agent":
|
|
# Spawn subagents. A real child then issues their requests itself,
|
|
# each carrying `x-claude-code-agent-id`.
|
|
gen = stream_tools(
|
|
[("Agent", a) for a in AGENT_INPUTS], "Delegating this."
|
|
)
|
|
elif scenario == "websearch":
|
|
gen = stream_tool("WebSearch", {"query": "ratatui scrollbar thumb"},
|
|
"Let me search for that.")
|
|
elif scenario == "ansi":
|
|
gen = stream_tool("Bash", ANSI_INPUT, "Printing coloured output.")
|
|
elif scenario == "basherror":
|
|
gen = stream_tool("Bash", {"command": "echo boom >&2; exit 3", "description": "Fail"}, "Running that.")
|
|
elif scenario == "toolerror":
|
|
gen = stream_tool("Read", TOOL_ERROR_INPUT, "Reading that file.")
|
|
elif scenario == "taskupdate":
|
|
gen = stream_tool("TaskUpdate", {"taskId": "1", "status": "in_progress"},
|
|
"Starting the first task.")
|
|
else:
|
|
gen = stream_text("Hello from the fake upstream.")
|
|
|
|
self.send_response(200)
|
|
self.send_header("content-type", "text/event-stream")
|
|
self.send_header("cache-control", "no-cache")
|
|
self.send_header("transfer-encoding", "chunked")
|
|
self.end_headers()
|
|
for part in gen:
|
|
self.wfile.write(b"%x\r\n" % len(part) + part + b"\r\n")
|
|
self.wfile.flush()
|
|
time.sleep(0.02)
|
|
self.wfile.write(b"0\r\n\r\n")
|
|
self.wfile.flush()
|
|
|
|
def do_GET(self):
|
|
self.send_response(404)
|
|
self.send_header("content-length", "0")
|
|
self.end_headers()
|
|
|
|
|
|
if __name__ == "__main__":
|
|
port = int(sys.argv[1]) if len(sys.argv) > 1 else 9911
|
|
print(f"fake upstream on 127.0.0.1:{port} scenario={read_scenario()}")
|
|
ThreadingHTTPServer(("127.0.0.1", port), Handler).serve_forever()
|