Files
alkcall/fuzz/gen_fuzz_seeds.py
T
glm-5.3-flash a1f257757b fix(channels): duplicate adopt/open must not destroy the live channel; fuzz targets 3-5
Found by the manager_routing fuzz target (docs/research/fuzzing.md \u00a77.8):

- ChannelManager::open_channel / adopt_channel used
  HashMap::insert(...).is_some() as a collision check \u2014 insert REPLACES
  the existing entry, so a duplicate adopt/open installed the new state,
  dropped the live channel's demux_sender (spurious EOF to its readers,
  subsequently routed chunks lost) and still returned
  Err(ChannelExists). Fixed with contains_key pre-check; map untouched
  on collision. Regression tests: adopt_channel_duplicate_id_leaves_
  live_channel_intact, open_channel_duplicate_id_leaves_live_channel_
  intact.

New fuzz targets (\u00a77.4 step 3):
- manager_routing: Arbitrary op sequences over ChannelManager; exact
  counter models (parked/dropped must equal the manager's monotonic
  counters), parked-bytes bound per \u00a76.2-1, clear_all ledger-vs-map
  semantics, drainer-byte reconciliation (lossless routing)
- envelope_semantic: constructors -> serde -> write_frame/read_frame
  structural round-trip; event-type constants; call.error parse-back
- spec_parse: OpRegisterRequest::from_json -> rebuild -> registry
  registration (attacker schemas compile at register, CF-003)
- fuzz/shared/src/arbitrary_value.rs: bounded Arbitrary for
  serde_json::Value; 20 spec_parse + 4 manager_routing seeds
- corpus replay for the new targets in fuzz/shared tests

Verification: cargo test 684 passed (682 + 2 regression); clippy
-D warnings clean (main + fuzz/shared); fmt clean (main + fuzz);
cargo fuzz build clean; 20 s smoke on all three new targets clean
(manager_routing 79k, envelope_semantic 141k, spec_parse 517k runs);
crash input replays clean post-fix
2026-09-27 23:19:18 +00:00

262 lines
8.6 KiB
Python

#!/usr/bin/env python3
"""Regenerate the committed seed corpora for alkcall's fuzz targets.
Writes into fuzz/corpus/<target>/. Deterministic: fixed inputs only, no
randomness. Run from the repo root:
python3 fuzz/gen_fuzz_seeds.py
"""
import json
import os
import struct
MAX_FRAME_SIZE = 64 * 1024 * 1024
MAX_CHUNK_LEN = 16 * 1024 * 1024
ENVELOPE_EVENT_TYPES = [
"call.requested",
"call.responded",
"call.completed",
"call.aborted",
"call.error",
"call.published",
"totally.unknown.event",
"",
]
SAMPLE_PAYLOADS = [
{},
None,
0,
"",
"payload string",
{"operationId": "/fs/readFile", "input": {"path": "/etc/hosts"}},
{"output": {"ok": True}},
{"input": [1, 2, 3]},
{"code": "NOT_FOUND", "message": "missing", "retryable": False},
{"nested": {"deep": {"deeper": [1, {"a": None}]}}},
{"output": "x" * 4096},
]
def seed_name(target, i):
return os.path.join("fuzz", "corpus", target, f"seed-{i:03d}")
def write_seed(target, i, data):
path = seed_name(target, i)
os.makedirs(os.path.dirname(path), exist_ok=True)
with open(path, "wb") as f:
f.write(data)
def envelope_seeds():
i = 0
for event_type in ENVELOPE_EVENT_TYPES:
for payload in SAMPLE_PAYLOADS[:6]:
body = json.dumps(
{"type": event_type, "id": "req-1", "payload": payload},
separators=(",", ":"),
).encode()
write_seed("envelope_frame", i, struct.pack(">I", len(body)) + body)
i += 1
# Truncations at every prefix length of a valid frame.
body = json.dumps(
{
"type": "call.requested",
"id": "req-1",
"payload": {"operationId": "/fs/readFile", "input": {"path": "/etc/hosts"}},
},
separators=(",", ":"),
).encode()
frame = struct.pack(">I", len(body)) + body
for cut in range(len(frame)):
write_seed("envelope_frame", i, frame[:cut])
i += 1
# Length prefix edge cases.
for name, length in [
("zero", 0),
("max", MAX_FRAME_SIZE),
("max-plus-1", MAX_FRAME_SIZE + 1),
("u32-max", 0xFFFFFFFF),
("huge-but-under-max", MAX_FRAME_SIZE - 1),
]:
write_seed("envelope_frame", i, struct.pack(">I", length))
i += 1
# A length prefix claiming a small body but truncated at each offset.
for claimed in (1, 2, 4, 16):
for have in range(0, claimed):
write_seed(
"envelope_frame", i, struct.pack(">I", claimed) + b'{"a":1}'[:have]
)
i += 1
# Invalid JSON bodies: bad UTF-8, wrong top-level type, missing fields,
# wrong field types, JSON fragments.
invalid_bodies = [
b'{"type":"call.requested","id":"\xff\xfe","payload":null}',
b'{"type":"call.requested","id":"\xc3","payload":null}',
b"[1,2,3]",
b'"just a string"',
b"null",
b"42",
b'{"id":"req-1","payload":null}',
b'{"type":"call.requested","payload":null}',
b'{"type":"call.requested","id":"req-1"}',
b'{"type":123,"id":"req-1","payload":null}',
b'{"type":"call.requested","id":99,"payload":null}',
b'{"type":"call.requested","id":"req-1","payload":',
b'{"type":"call.requested","id":"req-1","payload":undefined}',
b'{"type":"call.requested","id":"req-1","payload":NaN}',
b"{",
b"}",
b'{"type":"call.requested","id":"req-1","payload":{}}extra',
]
for body in invalid_bodies:
write_seed("envelope_frame", i, struct.pack(">I", len(body)) + body)
i += 1
# Deep-ish nesting inside the payload (well under serde_json's 128-depth
# recursion limit and libFuzzer's -max_len).
depth = 64
nested = ""
for _ in range(depth):
nested += '{"a":'
nested += "1"
for _ in range(depth):
nested += "}"
for depth in (2, 16, 64, 100, 127):
nested = ""
for _ in range(depth):
nested += '{"a":'
nested += "1"
for _ in range(depth):
nested += "}"
body = json.dumps(
{"type": "call.requested", "id": "deep", "payload": json.loads(nested)},
separators=(",", ":"),
).encode()
write_seed("envelope_frame", i, struct.pack(">I", len(body)) + body)
i += 1
# A structurally valid envelope with oversized claimed length and
# empty stream (allocation-bound probe).
write_seed("envelope_frame", i, struct.pack(">I", MAX_FRAME_SIZE) + b"{")
i += 1
def chunk_header_seeds():
i = 0
def header(channel_id, length):
return struct.pack(">II", channel_id, length)
edge_lengths = [
0,
1,
64,
MAX_CHUNK_LEN,
MAX_CHUNK_LEN + 1,
0xFFFFFFFF,
]
edge_ids = [0, 1, 2**31, 0xFFFFFFFF]
for channel_id in edge_ids:
for length in edge_lengths:
write_seed("chunk_header", i, header(channel_id, length))
i += 1
# Truncations at every prefix length.
full = header(0x01020304, 0x05060708)
for cut in range(8):
write_seed("chunk_header", i, full[:cut])
i += 1
# Oversized buffers (8 bytes is the minimum; longer inputs are legal,
# parse_header must ignore the rest).
write_seed("chunk_header", i, header(7, 3) + b"payload-bytes")
i += 1
def manager_routing_seeds():
"""Op-sequence seeds. These are arbitrary-encoded `ManagerSequence`
inputs, so they must be produced by the same encoder libFuzzer uses
(arbitrary 1.x byte format). Rather than hand-encoding the
byte format, generate them by running the shared crate's seed
helper: a tiny Rust tool would be a dependency; instead commit a
small set of raw-byte patterns that decode into useful sequences
(generated by the decode tool during development, deterministic).
The pattern bytes below were produced once by encoding with
`arbitrary` and are stable for arbitrary 1.4.x."""
# Generated via: Unstructured from the bytes below, ManagerSequence
# decode, verified with the mr-decode tool. See docs/research/
# fuzzing.md §7.7 for the generator note.
patterns = {
"single-open": "6400000000000000",
"adopt-then-route": "6c00000002000000",
"route-spray": "6c00000001000000",
"clear-all": "14000000",
}
i = 0
for name in sorted(patterns):
data = bytes.fromhex(patterns[name])
write_seed("manager_routing", i, data)
i += 1
def spec_parse_seeds():
"""JSON byte patterns for the op/register rebuild path: valid spec,
missing fields, wrong types, pathological schemas (deep nesting,
huge strings), and non-JSON bytes."""
bodies = [
b'{"spec":{"name":"a/b","op_type":"Query"},"replace":true}',
b'{"spec":{"name":"a/b","op_type":"Mutation"}}',
b'{"spec":{"name":"a/b","op_type":"Sub"}}',
b'{"spec":{"name":"a/b","op_type":"Pub","publish_schema":{"type":"object"}}}',
b'{"spec":{"name":"channels/tty/sub","op_type":"Sub","channel_open":true}}',
b'{"spec":{"name":"channels/tunnel/direct","op_type":"Sub","channel_open":true,"channel_open_alpn":"alk/tunnel"}}',
b'{"spec":{"name":"x","op_type":"Query","visibility":"internal","input_schema":{"type":"object","required":["a"],"properties":{"a":{"type":"string"}}}}}',
b'{"spec":{"name":"x","op_type":"Query","error_schemas":[{"code":"E1","description":"d","schema":{"type":"string"}}]}}',
b'{"spec":{"name":"x","op_type":"Bogus"}}',
b'{"spec":{"op_type":"Query"}}',
b'{"replace":true}',
b'{}',
b'{"spec":null}',
b'{"spec":"not-an-object"}',
b'{"spec":{"name":123,"op_type":"Query"}}',
b'{"spec":{"name":"","op_type":""}}',
b'{"spec":{"name":"x","op_type":"Query","access_control":{"required_scopes":["a","b"]}}}',
b'{"spec":{"name":"x","op_type":"Query","input_schema":{"$ref":"#/definitions/Recursive"}}}',
]
i = 0
for body in bodies:
write_seed("spec_parse", i, body)
i += 1
# Deep-nesting schema (jsonschema compile probe)
nested = b'{"spec":{"name":"deep","op_type":"Query","input_schema":' + (b'{"a":' * 100) + b'1' + (b'}' * 100) + b'}}'
write_seed("spec_parse", i, nested)
i += 1
# Non-JSON
write_seed("spec_parse", i, b"\xff\xfe not json")
i += 1
def main():
envelope_seeds()
chunk_header_seeds()
spec_parse_seeds()
manager_routing_seeds()
for target in ("chunk_header", "envelope_frame", "spec_parse", "manager_routing"):
d = os.path.join("fuzz", "corpus", target)
n = len(os.listdir(d))
print(f"{target}: {n} seeds")
if __name__ == "__main__":
main()