Add M2a plan: thirteen tasks, tests, fake server and recordings

The tasks build the inference path: emsha-backed SHA-256, inferproxy,
config, a hand-written HTTP and SSE client, request building, delta
assembly, the chat state machine, the thinking cap, the slot gate with
retry, the startup self-test and on-device verification.

Everything the tasks copy in was checked against a private reference
implementation: the gate passes after each task in order, the timing
tests pass repeatedly under CPU load, and the reference passes the
self-test and all four device checks on straylight. Expected results
for the recorded streams were derived by a separate script.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
2026-09-17 13:34:11 -07:00
co-authored by Claude Fable 5.1
parent a49db39b54
commit 76ccc251cd
56 changed files with 6367 additions and 3 deletions
+101
View File
@@ -0,0 +1,101 @@
#!/usr/bin/env python3
"""THROWAWAY: records raw HTTP responses from llama-server for the M2a fixtures. Stdlib only.
Talks plain HTTP/1.1 over TCP with `Connection: close`, exactly as loopd's client will, and saves
the response bytes untouched (status line, headers, chunked body). Prompts are synthetic.
"""
import json, os, socket, sys, threading, time
HOST, PORT = os.environ.get("LLAMA_HOST", "straylight"), int(os.environ.get("LLAMA_PORT", "11434"))
MODEL = "ornith-1.5-35b-a3b"
OUT = os.path.join(os.path.dirname(os.path.abspath(__file__)), "out", "m2a")
COMMON = {"model": MODEL, "id_slot": 0, "cache_prompt": True, "stream": True, "return_progress": True,
"timings_per_token": True, "reasoning_control": True, "temperature": 0.6, "top_p": 0.95, "top_k": 20}
TOOL = {"type": "function", "function": {"name": "read_file", "description": "Read a text file and return its contents.",
"parameters": {"type": "object", "properties": {"path": {"type": "string", "description": "Absolute path"}},
"required": ["path"]}}}
SYSTEM = {"role": "system", "content": "You are Boxmaker, a careful personal agent."}
def exchange(method, path, body=None, on_bytes=None):
payload = b"" if body is None else json.dumps(body, separators=(",", ":")).encode()
head = "%s %s HTTP/1.1\r\nHost: %s\r\nConnection: close\r\nAccept: */*\r\n" % (method, path, HOST)
if body is not None:
head += "Content-Type: application/json\r\nContent-Length: %d\r\n" % len(payload)
s = socket.create_connection((HOST, PORT), timeout=600)
s.sendall(head.encode() + b"\r\n" + payload)
got = b""
while True:
b = s.recv(65536)
if not b:
break
got += b
if on_bytes:
on_bytes(got)
s.close()
return payload, got
def save(name, request_body, response):
with open(os.path.join(OUT, name + ".http"), "wb") as f:
f.write(response)
if request_body:
with open(os.path.join(OUT, name + ".request.json"), "wb") as f:
f.write(request_body + b"\n")
print("%-22s %7d bytes %s" % (name, len(response), response.split(b"\r\n", 1)[0].decode()))
def chat(name, messages, tools=None, thinking=False, max_tokens=256, on_bytes=None, **extra):
body = {**COMMON, "max_tokens": max_tokens, "messages": messages,
"chat_template_kwargs": {"enable_thinking": thinking}, **extra}
if tools:
body["tools"] = tools
req, resp = exchange("POST", "/v1/chat/completions", body, on_bytes)
save(name, req, resp)
return resp
def main():
os.makedirs(OUT, exist_ok=True)
nonce = "%08x" % int(time.time())
user = lambda text: {"role": "user", "content": text + " (run " + nonce + ")"}
chat("plain", [SYSTEM, user("Reply with exactly: The box is made.")], max_tokens=32)
chat("thinking", [SYSTEM, user("What is 17 * 23? Think briefly, then answer in one short sentence.")],
thinking=True, max_tokens=1024)
chat("tool_call", [SYSTEM, user("Read /etc/hostname and tell me what it says.")], tools=[TOOL], max_tokens=256)
filler = " ".join("lattice cork brass feather shard coil drift vault ledger salt".split() * 700)
chat("progress", [SYSTEM, user(filler + "\nReply with exactly: ok")], max_tokens=8)
# A forced end to reasoning: post the control call once ~60 reasoning chunks have arrived.
fired = {"done": False}
def maybe_cap(got):
if fired["done"] or got.count(b'"reasoning_content"') < 60:
return
fired["done"] = True
cid = json.loads(got.split(b"data: ", 2)[1].split(b"\n", 1)[0])["id"]
def post():
req, resp = exchange("POST", "/v1/chat/completions/control", {"id": cid, "action": "reasoning_end", "model": MODEL})
save("control", req, resp)
threading.Thread(target=post).start()
chat("capped", [SYSTEM, user("Think at great length about how many ways 8 rooks fit on a chessboard without "
"attacking each other and with none on the main diagonal. Then answer in one sentence.")],
thinking=True, max_tokens=2048, on_bytes=maybe_cap)
time.sleep(1)
# Two turns for the cache check: turn 2 replays turn 1's assistant message exactly.
t1 = [SYSTEM, user("Name one colour. One word.")]
r1 = chat("turn1", t1, max_tokens=16)
content = "".join(json.loads(l[6:])["choices"][0]["delta"].get("content") or ""
for l in r1.decode("utf-8", "replace").split("\n")
if l.startswith("data: {") and json.loads(l[6:]).get("choices"))
chat("turn2", t1 + [{"role": "assistant", "content": content}, user("Name another. One word.")], max_tokens=16)
for name, path in (("props", "/props?model=" + MODEL), ("slots", "/slots?model=" + MODEL)):
_, resp = exchange("GET", path); save(name, None, resp)
req, resp = exchange("POST", "/tokenize", {"model": MODEL, "content": "The box is made."}); save("tokenize", req, resp)
req, resp = exchange("POST", "/v1/chat/completions", {"model": MODEL, "messages": "not a list"}); save("bad_request", req, resp)
req, resp = exchange("POST", "/v1/chat/completions", {**COMMON, "model": "no-such-model", "messages": [user("hi")]}); save("unknown_model", req, resp)
if __name__ == "__main__":
main()