Found by the manager_routing fuzz target (docs/research/fuzzing.md \u00a77.8): - ChannelManager::open_channel / adopt_channel used HashMap::insert(...).is_some() as a collision check \u2014 insert REPLACES the existing entry, so a duplicate adopt/open installed the new state, dropped the live channel's demux_sender (spurious EOF to its readers, subsequently routed chunks lost) and still returned Err(ChannelExists). Fixed with contains_key pre-check; map untouched on collision. Regression tests: adopt_channel_duplicate_id_leaves_ live_channel_intact, open_channel_duplicate_id_leaves_live_channel_ intact. New fuzz targets (\u00a77.4 step 3): - manager_routing: Arbitrary op sequences over ChannelManager; exact counter models (parked/dropped must equal the manager's monotonic counters), parked-bytes bound per \u00a76.2-1, clear_all ledger-vs-map semantics, drainer-byte reconciliation (lossless routing) - envelope_semantic: constructors -> serde -> write_frame/read_frame structural round-trip; event-type constants; call.error parse-back - spec_parse: OpRegisterRequest::from_json -> rebuild -> registry registration (attacker schemas compile at register, CF-003) - fuzz/shared/src/arbitrary_value.rs: bounded Arbitrary for serde_json::Value; 20 spec_parse + 4 manager_routing seeds - corpus replay for the new targets in fuzz/shared tests Verification: cargo test 684 passed (682 + 2 regression); clippy -D warnings clean (main + fuzz/shared); fmt clean (main + fuzz); cargo fuzz build clean; 20 s smoke on all three new targets clean (manager_routing 79k, envelope_semantic 141k, spec_parse 517k runs); crash input replays clean post-fix
262 lines
8.6 KiB
Python
262 lines
8.6 KiB
Python
#!/usr/bin/env python3
|
|
"""Regenerate the committed seed corpora for alkcall's fuzz targets.
|
|
|
|
Writes into fuzz/corpus/<target>/. Deterministic: fixed inputs only, no
|
|
randomness. Run from the repo root:
|
|
|
|
python3 fuzz/gen_fuzz_seeds.py
|
|
"""
|
|
|
|
import json
|
|
import os
|
|
import struct
|
|
|
|
MAX_FRAME_SIZE = 64 * 1024 * 1024
|
|
MAX_CHUNK_LEN = 16 * 1024 * 1024
|
|
|
|
ENVELOPE_EVENT_TYPES = [
|
|
"call.requested",
|
|
"call.responded",
|
|
"call.completed",
|
|
"call.aborted",
|
|
"call.error",
|
|
"call.published",
|
|
"totally.unknown.event",
|
|
"",
|
|
]
|
|
|
|
SAMPLE_PAYLOADS = [
|
|
{},
|
|
None,
|
|
0,
|
|
"",
|
|
"payload string",
|
|
{"operationId": "/fs/readFile", "input": {"path": "/etc/hosts"}},
|
|
{"output": {"ok": True}},
|
|
{"input": [1, 2, 3]},
|
|
{"code": "NOT_FOUND", "message": "missing", "retryable": False},
|
|
{"nested": {"deep": {"deeper": [1, {"a": None}]}}},
|
|
{"output": "x" * 4096},
|
|
]
|
|
|
|
|
|
def seed_name(target, i):
|
|
return os.path.join("fuzz", "corpus", target, f"seed-{i:03d}")
|
|
|
|
|
|
def write_seed(target, i, data):
|
|
path = seed_name(target, i)
|
|
os.makedirs(os.path.dirname(path), exist_ok=True)
|
|
with open(path, "wb") as f:
|
|
f.write(data)
|
|
|
|
|
|
def envelope_seeds():
|
|
i = 0
|
|
|
|
for event_type in ENVELOPE_EVENT_TYPES:
|
|
for payload in SAMPLE_PAYLOADS[:6]:
|
|
body = json.dumps(
|
|
{"type": event_type, "id": "req-1", "payload": payload},
|
|
separators=(",", ":"),
|
|
).encode()
|
|
write_seed("envelope_frame", i, struct.pack(">I", len(body)) + body)
|
|
i += 1
|
|
|
|
# Truncations at every prefix length of a valid frame.
|
|
body = json.dumps(
|
|
{
|
|
"type": "call.requested",
|
|
"id": "req-1",
|
|
"payload": {"operationId": "/fs/readFile", "input": {"path": "/etc/hosts"}},
|
|
},
|
|
separators=(",", ":"),
|
|
).encode()
|
|
frame = struct.pack(">I", len(body)) + body
|
|
for cut in range(len(frame)):
|
|
write_seed("envelope_frame", i, frame[:cut])
|
|
i += 1
|
|
|
|
# Length prefix edge cases.
|
|
for name, length in [
|
|
("zero", 0),
|
|
("max", MAX_FRAME_SIZE),
|
|
("max-plus-1", MAX_FRAME_SIZE + 1),
|
|
("u32-max", 0xFFFFFFFF),
|
|
("huge-but-under-max", MAX_FRAME_SIZE - 1),
|
|
]:
|
|
write_seed("envelope_frame", i, struct.pack(">I", length))
|
|
i += 1
|
|
|
|
# A length prefix claiming a small body but truncated at each offset.
|
|
for claimed in (1, 2, 4, 16):
|
|
for have in range(0, claimed):
|
|
write_seed(
|
|
"envelope_frame", i, struct.pack(">I", claimed) + b'{"a":1}'[:have]
|
|
)
|
|
i += 1
|
|
|
|
# Invalid JSON bodies: bad UTF-8, wrong top-level type, missing fields,
|
|
# wrong field types, JSON fragments.
|
|
invalid_bodies = [
|
|
b'{"type":"call.requested","id":"\xff\xfe","payload":null}',
|
|
b'{"type":"call.requested","id":"\xc3","payload":null}',
|
|
b"[1,2,3]",
|
|
b'"just a string"',
|
|
b"null",
|
|
b"42",
|
|
b'{"id":"req-1","payload":null}',
|
|
b'{"type":"call.requested","payload":null}',
|
|
b'{"type":"call.requested","id":"req-1"}',
|
|
b'{"type":123,"id":"req-1","payload":null}',
|
|
b'{"type":"call.requested","id":99,"payload":null}',
|
|
b'{"type":"call.requested","id":"req-1","payload":',
|
|
b'{"type":"call.requested","id":"req-1","payload":undefined}',
|
|
b'{"type":"call.requested","id":"req-1","payload":NaN}',
|
|
b"{",
|
|
b"}",
|
|
b'{"type":"call.requested","id":"req-1","payload":{}}extra',
|
|
]
|
|
for body in invalid_bodies:
|
|
write_seed("envelope_frame", i, struct.pack(">I", len(body)) + body)
|
|
i += 1
|
|
|
|
# Deep-ish nesting inside the payload (well under serde_json's 128-depth
|
|
# recursion limit and libFuzzer's -max_len).
|
|
depth = 64
|
|
nested = ""
|
|
for _ in range(depth):
|
|
nested += '{"a":'
|
|
nested += "1"
|
|
for _ in range(depth):
|
|
nested += "}"
|
|
for depth in (2, 16, 64, 100, 127):
|
|
nested = ""
|
|
for _ in range(depth):
|
|
nested += '{"a":'
|
|
nested += "1"
|
|
for _ in range(depth):
|
|
nested += "}"
|
|
body = json.dumps(
|
|
{"type": "call.requested", "id": "deep", "payload": json.loads(nested)},
|
|
separators=(",", ":"),
|
|
).encode()
|
|
write_seed("envelope_frame", i, struct.pack(">I", len(body)) + body)
|
|
i += 1
|
|
|
|
# A structurally valid envelope with oversized claimed length and
|
|
# empty stream (allocation-bound probe).
|
|
write_seed("envelope_frame", i, struct.pack(">I", MAX_FRAME_SIZE) + b"{")
|
|
i += 1
|
|
|
|
|
|
def chunk_header_seeds():
|
|
i = 0
|
|
|
|
def header(channel_id, length):
|
|
return struct.pack(">II", channel_id, length)
|
|
|
|
edge_lengths = [
|
|
0,
|
|
1,
|
|
64,
|
|
MAX_CHUNK_LEN,
|
|
MAX_CHUNK_LEN + 1,
|
|
0xFFFFFFFF,
|
|
]
|
|
edge_ids = [0, 1, 2**31, 0xFFFFFFFF]
|
|
|
|
for channel_id in edge_ids:
|
|
for length in edge_lengths:
|
|
write_seed("chunk_header", i, header(channel_id, length))
|
|
i += 1
|
|
|
|
# Truncations at every prefix length.
|
|
full = header(0x01020304, 0x05060708)
|
|
for cut in range(8):
|
|
write_seed("chunk_header", i, full[:cut])
|
|
i += 1
|
|
|
|
# Oversized buffers (8 bytes is the minimum; longer inputs are legal,
|
|
# parse_header must ignore the rest).
|
|
write_seed("chunk_header", i, header(7, 3) + b"payload-bytes")
|
|
i += 1
|
|
|
|
|
|
def manager_routing_seeds():
|
|
"""Op-sequence seeds. These are arbitrary-encoded `ManagerSequence`
|
|
inputs, so they must be produced by the same encoder libFuzzer uses
|
|
(arbitrary 1.x byte format). Rather than hand-encoding the
|
|
byte format, generate them by running the shared crate's seed
|
|
helper: a tiny Rust tool would be a dependency; instead commit a
|
|
small set of raw-byte patterns that decode into useful sequences
|
|
(generated by the decode tool during development, deterministic).
|
|
The pattern bytes below were produced once by encoding with
|
|
`arbitrary` and are stable for arbitrary 1.4.x."""
|
|
|
|
# Generated via: Unstructured from the bytes below, ManagerSequence
|
|
# decode, verified with the mr-decode tool. See docs/research/
|
|
# fuzzing.md §7.7 for the generator note.
|
|
patterns = {
|
|
"single-open": "6400000000000000",
|
|
"adopt-then-route": "6c00000002000000",
|
|
"route-spray": "6c00000001000000",
|
|
"clear-all": "14000000",
|
|
}
|
|
i = 0
|
|
for name in sorted(patterns):
|
|
data = bytes.fromhex(patterns[name])
|
|
write_seed("manager_routing", i, data)
|
|
i += 1
|
|
|
|
|
|
def spec_parse_seeds():
|
|
"""JSON byte patterns for the op/register rebuild path: valid spec,
|
|
missing fields, wrong types, pathological schemas (deep nesting,
|
|
huge strings), and non-JSON bytes."""
|
|
bodies = [
|
|
b'{"spec":{"name":"a/b","op_type":"Query"},"replace":true}',
|
|
b'{"spec":{"name":"a/b","op_type":"Mutation"}}',
|
|
b'{"spec":{"name":"a/b","op_type":"Sub"}}',
|
|
b'{"spec":{"name":"a/b","op_type":"Pub","publish_schema":{"type":"object"}}}',
|
|
b'{"spec":{"name":"channels/tty/sub","op_type":"Sub","channel_open":true}}',
|
|
b'{"spec":{"name":"channels/tunnel/direct","op_type":"Sub","channel_open":true,"channel_open_alpn":"alk/tunnel"}}',
|
|
b'{"spec":{"name":"x","op_type":"Query","visibility":"internal","input_schema":{"type":"object","required":["a"],"properties":{"a":{"type":"string"}}}}}',
|
|
b'{"spec":{"name":"x","op_type":"Query","error_schemas":[{"code":"E1","description":"d","schema":{"type":"string"}}]}}',
|
|
b'{"spec":{"name":"x","op_type":"Bogus"}}',
|
|
b'{"spec":{"op_type":"Query"}}',
|
|
b'{"replace":true}',
|
|
b'{}',
|
|
b'{"spec":null}',
|
|
b'{"spec":"not-an-object"}',
|
|
b'{"spec":{"name":123,"op_type":"Query"}}',
|
|
b'{"spec":{"name":"","op_type":""}}',
|
|
b'{"spec":{"name":"x","op_type":"Query","access_control":{"required_scopes":["a","b"]}}}',
|
|
b'{"spec":{"name":"x","op_type":"Query","input_schema":{"$ref":"#/definitions/Recursive"}}}',
|
|
]
|
|
i = 0
|
|
for body in bodies:
|
|
write_seed("spec_parse", i, body)
|
|
i += 1
|
|
# Deep-nesting schema (jsonschema compile probe)
|
|
nested = b'{"spec":{"name":"deep","op_type":"Query","input_schema":' + (b'{"a":' * 100) + b'1' + (b'}' * 100) + b'}}'
|
|
write_seed("spec_parse", i, nested)
|
|
i += 1
|
|
# Non-JSON
|
|
write_seed("spec_parse", i, b"\xff\xfe not json")
|
|
i += 1
|
|
|
|
|
|
def main():
|
|
envelope_seeds()
|
|
chunk_header_seeds()
|
|
spec_parse_seeds()
|
|
manager_routing_seeds()
|
|
for target in ("chunk_header", "envelope_frame", "spec_parse", "manager_routing"):
|
|
d = os.path.join("fuzz", "corpus", target)
|
|
n = len(os.listdir(d))
|
|
print(f"{target}: {n} seeds")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main() |