Files
claude-cloak/dev/fake_upstream.py
Jonas H fc9c05ce97 fake upstream
+1
2026-08-17 07:47:35 +02:00

221 lines
10 KiB
Python

#!/usr/bin/env python3
"""Offline stand-in for api.anthropic.com — zero API usage.
Point the proxy at it (`CT_UPSTREAM=http://127.0.0.1:9911`) and every request a
real `claude` child makes is answered locally with a canned SSE stream. That
lets us make Claude Code render its client-side tool UIs (AskUserQuestion,
ExitPlanMode, TodoWrite/Task*) on demand, so the pane's frame detector in
`src/term.rs` can be developed against what Ink actually draws.
Scenario is picked per turn from `CT_FAKE_SCENARIO` (ask | plan | todo | text).
Each incoming request is logged to `dev/fake_upstream.log` (declared tool names
+ the trailing user text) so we can see what CC sends.
"""
import json
import os
import sys
import time
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
LOG = os.path.join(os.path.dirname(os.path.abspath(__file__)), "fake_upstream.log")
MODEL = "claude-opus-4-5-20251101"
def log(msg):
with open(LOG, "a") as f:
f.write(msg + "\n")
def read_scenario():
"""Scenario for the *next* turn: `dev/.fake_scenario` wins over the env, so
you can switch scenarios without restarting the server."""
path = os.path.join(os.path.dirname(os.path.abspath(__file__)), ".fake_scenario")
try:
with open(path) as f:
return f.read().strip()
except OSError:
return os.environ.get("CT_FAKE_SCENARIO", "ask")
def sse(event, data):
return f"event: {event}\ndata: {json.dumps(data)}\n\n".encode()
def stream_text(text):
"""A plain assistant text turn."""
yield sse("message_start", {"type": "message_start", "message": {
"id": "msg_fake", "type": "message", "role": "assistant", "model": MODEL,
"content": [], "stop_reason": None, "stop_sequence": None,
"usage": {"input_tokens": 12, "output_tokens": 1}}})
yield sse("content_block_start", {"type": "content_block_start", "index": 0,
"content_block": {"type": "text", "text": ""}})
for chunk in text.split(" "):
yield sse("content_block_delta", {"type": "content_block_delta", "index": 0,
"delta": {"type": "text_delta", "text": chunk + " "}})
yield sse("content_block_stop", {"type": "content_block_stop", "index": 0})
yield sse("message_delta", {"type": "message_delta",
"delta": {"stop_reason": "end_turn", "stop_sequence": None},
"usage": {"output_tokens": 9}})
yield sse("message_stop", {"type": "message_stop"})
def stream_tool(name, tool_input, lead="Working on it."):
"""A turn that calls one client-side tool."""
yield from stream_tools([(name, tool_input)], lead)
def stream_tools(calls, lead="Working on it."):
"""A turn that calls several client-side tools in one message."""
yield sse("message_start", {"type": "message_start", "message": {
"id": "msg_fake", "type": "message", "role": "assistant", "model": MODEL,
"content": [], "stop_reason": None, "stop_sequence": None,
"usage": {"input_tokens": 12, "output_tokens": 1}}})
yield sse("content_block_start", {"type": "content_block_start", "index": 0,
"content_block": {"type": "text", "text": ""}})
yield sse("content_block_delta", {"type": "content_block_delta", "index": 0,
"delta": {"type": "text_delta", "text": lead}})
yield sse("content_block_stop", {"type": "content_block_stop", "index": 0})
for n, (name, tool_input) in enumerate(calls, start=1):
yield sse("content_block_start", {"type": "content_block_start", "index": n,
"content_block": {"type": "tool_use", "id": f"toolu_fake{n}",
"name": name, "input": {}}})
blob = json.dumps(tool_input)
for i in range(0, len(blob), 40):
yield sse("content_block_delta", {"type": "content_block_delta", "index": n,
"delta": {"type": "input_json_delta", "partial_json": blob[i:i + 40]}})
yield sse("content_block_stop", {"type": "content_block_stop", "index": n})
yield sse("message_delta", {"type": "message_delta",
"delta": {"stop_reason": "tool_use", "stop_sequence": None},
"usage": {"output_tokens": 40}})
yield sse("message_stop", {"type": "message_stop"})
ASK_INPUT = {"questions": [{
"question": "The compact pane currently crops the top of this prompt. Which framing "
"should the pane use when an interactive question is on screen, given that "
"the question text itself can wrap over several rows?",
"header": "Framing",
"multiSelect": False,
"options": [
{"label": "Measure the box", "description": "Frame from the question box's own top border down to the input box, so nothing is cut."},
{"label": "Fixed 75% height", "description": "Always give the pane three quarters of the screen while a question waits, however tall it really is."},
{"label": "Estimate from JSON", "description": "Keep guessing the row count from the tool input, as today, and accept that wrapped text breaks the guess."},
{"label": "Fullscreen the pane", "description": "Take the whole screen for as long as a question is waiting for an answer."},
]}, {
"question": "And what should happen when the prompt is taller than the pane can ever be?",
"header": "Overflow",
"multiSelect": False,
"options": [
{"label": "Keep the selection", "description": "Slide the window so the highlighted option stays visible."},
{"label": "Crop the tail", "description": "Always top-anchor and let the last options fall off."},
]}]}
PLAN_INPUT = {"plan": "## Plan\n\n1. Size the PTY to the whole screen.\n2. Measure the "
"rendered box instead of guessing rows.\n3. Extend the frame over an "
"active task list.\n\nThis is a long enough plan to need several rows."}
TODO_INPUT = {"todos": [
{"content": "Capture ground truth screens", "status": "completed", "activeForm": "Capturing ground truth screens"},
{"content": "Fix interactive pane sizing", "status": "in_progress", "activeForm": "Fixing interactive pane sizing"},
{"content": "Expand pane while a task list is active", "status": "pending", "activeForm": "Expanding pane for task lists"},
{"content": "Add regression tests", "status": "pending", "activeForm": "Adding regression tests"},
{"content": "Update CLAUDE.md", "status": "pending", "activeForm": "Updating CLAUDE.md"},
]}
TASK_INPUTS = [
{"subject": "Capture ground truth screens", "description": "Drive a real claude child offline.",
"activeForm": "Capturing ground truth screens"},
{"subject": "Fix interactive pane sizing", "description": "Measure the box instead of guessing rows.",
"activeForm": "Fixing interactive pane sizing"},
{"subject": "Expand pane while a task list is active", "description": "Show task status near the prompt.",
"activeForm": "Expanding pane for task lists"},
{"subject": "Add regression tests", "description": "Pin the framing math.",
"activeForm": "Adding regression tests"},
]
class Handler(BaseHTTPRequestHandler):
protocol_version = "HTTP/1.1"
def log_message(self, *a): # keep stdout clean
pass
def do_POST(self):
n = int(self.headers.get("content-length", 0))
raw = self.rfile.read(n) if n else b"{}"
try:
body = json.loads(raw)
except Exception:
body = {}
tools = [t.get("name") for t in body.get("tools", []) or []]
msgs = body.get("messages", []) or []
tail = json.dumps(msgs[-1])[:300] if msgs else ""
# Only the immediate reply to *our* canned tool call ends the turn with
# text. A user message can carry both the tool_result and fresh user
# text (a new prompt typed after declining), and that is a new turn —
# so require the tool_result to be the *whole* last message.
last = msgs[-1] if msgs else {}
blocks = last.get("content") if isinstance(last.get("content"), list) else []
has_result = bool(blocks) and all(
b.get("type") == "tool_result" for b in blocks if isinstance(b, dict)
) and "toolu_fake" in json.dumps(blocks)
log(f"[{time.strftime('%H:%M:%S')}] {self.path} tools={tools} tail={tail}")
if "count_tokens" in self.path:
out = json.dumps({"input_tokens": 100}).encode()
self.send_response(200)
self.send_header("content-type", "application/json")
self.send_header("content-length", str(len(out)))
self.end_headers()
self.wfile.write(out)
return
scenario = read_scenario()
# A request with no tools is CC's side/title call — answer with text.
if not tools or has_result:
gen = stream_text("Done. Ask me anything else.")
elif scenario == "ask":
gen = stream_tool("AskUserQuestion", ASK_INPUT, "Let me check how you want this framed.")
elif scenario == "plan":
gen = stream_tool("ExitPlanMode", PLAN_INPUT, "Here is the plan.")
elif scenario == "todo":
# Which task tool exists depends on the child's agent: a plain CC
# session has TodoWrite, an agent like `minimal` has the newer
# TaskCreate/TaskUpdate set instead.
if "TodoWrite" in tools:
gen = stream_tool("TodoWrite", TODO_INPUT, "Setting up the task list.")
else:
gen = stream_tools(
[("TaskCreate", t) for t in TASK_INPUTS], "Setting up the task list."
)
elif scenario == "taskupdate":
gen = stream_tool("TaskUpdate", {"taskId": "1", "status": "in_progress"},
"Starting the first task.")
else:
gen = stream_text("Hello from the fake upstream.")
self.send_response(200)
self.send_header("content-type", "text/event-stream")
self.send_header("cache-control", "no-cache")
self.send_header("transfer-encoding", "chunked")
self.end_headers()
for part in gen:
self.wfile.write(b"%x\r\n" % len(part) + part + b"\r\n")
self.wfile.flush()
time.sleep(0.02)
self.wfile.write(b"0\r\n\r\n")
self.wfile.flush()
def do_GET(self):
self.send_response(404)
self.send_header("content-length", "0")
self.end_headers()
if __name__ == "__main__":
port = int(sys.argv[1]) if len(sys.argv) > 1 else 9911
print(f"fake upstream on 127.0.0.1:{port} scenario={os.environ.get('CT_FAKE_SCENARIO', 'ask')}")
ThreadingHTTPServer(("127.0.0.1", port), Handler).serve_forever()