Files
StarForth/scripts/extract_doe_region.py
T
Robert Allan JamesandClaude Sonnet 5 9aebbb224b
Build / build-amd64-iso (push) Canceled after 0s
Build / build-aarch64-iso (push) Canceled after 0s
Build / build-riscv64-img (push) Canceled after 0s
Fix extract_doe_region.py SLOT_FMT drift; regenerate report with real data
FABRIC-3.md §XXXV.15. §XXXV.12 added an explicit _align_pad field to
doe_trial_slot_t to fix a struct-alignment bug, but never updated the
Python extractor's SLOT_FMT/SLOT_CRC_SPAN to match -- every field after
isa was read from the wrong byte offset, so slot_crc read wrong bytes
entirely and every legitimate record (including the real, live
campaign's own first two clean trials) was flagged as CRC-corrupt.
Fixed; SLOT_SIZE assertion tightened from <= to == to catch this drift
immediately next time.

Regenerated multiuser_doe_report.pdf against the real campaign's actual
first two trials: run=0 (cfg=0, nw=2, uniform, fail=0) and run=1 (cfg=5,
nw=8, heterogeneous, fail=0), both fleet_conserved=1 at exactly Q48.16
1.0. Real report, real data, first time this has happened for this
campaign. disk/artemis.img itself and the live campaign's log directory
are deliberately left uncommitted -- still being written by the running
campaign.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BWpNjdwPtFLuVLaAq44L9K
2026-09-17 23:28:50 -04:00

257 lines
10 KiB
Python
Executable File

#!/usr/bin/env python3
"""
extract_doe_region.py -- spool the DOE-PERSIST-TRIAL binary ring off
disk/artemis.img into a CSV, for the multiuser/multitasking DoE campaign
report (FABRIC-3.md §XXXV.10).
Reads the SAME devblock_from_top-addressed ring doe_region.c writes
(include/starkernel/doe_region.h has the authoritative layout: control
header at devblock_from_top=98, slot devblocks growable from 99, 64-byte
doe_trial_slot_t records, 64 slots/devblock). Two access modes:
--img PATH Read the image file directly (default). No privilege
needed, works on a plain regular file. This is what
"reading the drive" reduces to when the image is
just a file, which is the normal QEMU-dev case.
--loop PATH Attach PATH via losetup first (needs sudo), read the
resulting /dev/loopN node, then detach it again on
exit -- the literal "mount the virtual drive on a
loop device" step for when this runs against a real
block device or a copy pulled off real hardware. The
ring is raw binary at a fixed byte offset, not a
filesystem -- there is nothing to `mount` in the
POSIX-filesystem sense, so this attaches the loop
device and reads it exactly like a regular file.
Either mode is read-only and never modifies the image. Safe to run
against a live campaign's artemis.img (the ring's control header CRC
means a torn read mid-write is simply rejected as "ring not yet
initialized" by the same discipline doe_region.c itself uses -- see
doe_region_ctrl_read()); worst case is missing the very latest record
until the next run of this script.
CRC verification (§XXXV.12): both the control header and each individual
trial slot carry a CRC-64/XZ checksum (same polynomial and algorithm as
src/block_subsystem.c's compute_crc64() -- reimplemented here in Python,
not guessed at; verified bit-for-bit against a live kernel-written
record). A bad header CRC is treated as "ring not yet initialized" (same
as a bad magic/version). A bad slot CRC means that record is skipped and
reported, not silently included as if it were real data -- since
DOE_SLOTS_PER_DEVBLOCK=64 slots share one devblock via a read-modify-
write, a crash mid-write to any slot can corrupt its siblings too.
"""
import argparse
import csv
import struct
import subprocess
import sys
from contextlib import contextmanager
from pathlib import Path
DEVBLOCK_SIZE = 4096
# Mirrors include/starkernel/doe_region.h and include/starkernel/log_region.h
ARTEMIS_SIG_DEVBLOCK_FROM_TOP = 64
LOG_REGION_DEVBLOCK_FROM_TOP_BASE = ARTEMIS_SIG_DEVBLOCK_FROM_TOP + 1 # 65
LOG_REGION_MAX_DEVBLOCKS = 32
DOE_REGION_DEVBLOCK_FROM_TOP_BASE = (
LOG_REGION_DEVBLOCK_FROM_TOP_BASE + 1 + LOG_REGION_MAX_DEVBLOCKS
) # 98
DOE_SLOT_SIZE = 64
DOE_SLOTS_PER_DEVBLOCK = DEVBLOCK_SIZE // DOE_SLOT_SIZE # 64
DOE_REGION_MAGIC = 0x44454F44 # 'DOED'
DOE_REGION_VERSION_0 = 0
# doe_region_ctrl_t: magic(u64) devblocks(u32) head_slot(u32) tail_slot(u32)
# record_count(u32) hdr_crc(u64), little-endian, rest is padding to 4096.
CTRL_FMT = "<QIIIIQ"
CTRL_SIZE = struct.calcsize(CTRL_FMT)
CTRL_CRC_SPAN = struct.calcsize("<QIIII") # everything before hdr_crc
# doe_trial_slot_t: fleet_k_q48(u64) switch_count_cumulative(u64)
# run_id(u32) cfg(u32) rep(u32) nw(u32) mode(u32) fail_count(u32)
# fleet_conserved(u32) isa(8s) _align_pad(u32, explicit -- see
# doe_region.h's own comment on why: brings slot_crc to an 8-aligned
# offset without relying on implicit compiler padding) slot_crc(u64).
SLOT_FMT = "<QQIIIIIII8sIQ"
SLOT_SIZE = struct.calcsize(SLOT_FMT)
SLOT_CRC_SPAN = struct.calcsize("<QQIIIIIII8sI") # everything before slot_crc
assert SLOT_SIZE == DOE_SLOT_SIZE
# CRC-64/XZ, same polynomial/algorithm as compute_crc64() in
# src/block_subsystem.c (crc64_init()'s poly=0x42F0E1EBA9EA3693, table-
# driven, init/final XOR 0xFFFFFFFFFFFFFFFF) -- reimplemented here so
# host-side verification genuinely matches the kernel side, not guessed.
_CRC64_POLY = 0x42F0E1EBA9EA3693
_CRC64_TABLE = None
def _crc64_table():
global _CRC64_TABLE
if _CRC64_TABLE is None:
table = []
for i in range(256):
crc = i
for _ in range(8):
crc = (crc >> 1) ^ _CRC64_POLY if (crc & 1) else (crc >> 1)
table.append(crc & 0xFFFFFFFFFFFFFFFF)
_CRC64_TABLE = table
return _CRC64_TABLE
def compute_crc64(data: bytes) -> int:
table = _crc64_table()
crc = 0xFFFFFFFFFFFFFFFF
for byte in data:
idx = (crc ^ byte) & 0xFF
crc = table[idx] ^ (crc >> 8)
return crc ^ 0xFFFFFFFFFFFFFFFF
def dft_to_offset(devblock_from_top: int, total_devblocks: int) -> int:
idx = total_devblocks - 1 - devblock_from_top
if idx < 0:
raise ValueError(
f"devblock_from_top={devblock_from_top} exceeds total_devblocks={total_devblocks}"
)
return idx * DEVBLOCK_SIZE
@contextmanager
def loop_attach(img_path: Path):
"""Attach img_path via losetup (read-only), yield the /dev/loopN path,
detach on exit. Requires sudo -- the literal 'mount on a loop device'
step; nothing here is a filesystem mount."""
result = subprocess.run(
["sudo", "losetup", "--find", "--show", "--read-only", str(img_path)],
capture_output=True, text=True, check=True,
)
loop_dev = result.stdout.strip()
print(f" attached {img_path} -> {loop_dev}", file=sys.stderr)
try:
yield loop_dev
finally:
subprocess.run(["sudo", "losetup", "--detach", loop_dev], check=False)
print(f" detached {loop_dev}", file=sys.stderr)
def read_ring(dev_path: str):
"""Returns a list of dicts, one per live trial record, in ring order
(oldest first). Empty list if the ring has never been written."""
size = Path(dev_path).stat().st_size if not dev_path.startswith("/dev/") else None
if size is None:
with open(dev_path, "rb") as f:
f.seek(0, 2)
size = f.tell()
total_devblocks = size // DEVBLOCK_SIZE
if total_devblocks <= DOE_REGION_DEVBLOCK_FROM_TOP_BASE:
raise ValueError(
f"image too small ({size} bytes, {total_devblocks} devblocks) "
f"for the DOE region fence (needs > {DOE_REGION_DEVBLOCK_FROM_TOP_BASE})"
)
ctrl_off = dft_to_offset(DOE_REGION_DEVBLOCK_FROM_TOP_BASE, total_devblocks)
with open(dev_path, "rb") as f:
f.seek(ctrl_off)
ctrl_raw = f.read(DEVBLOCK_SIZE)
magic, devblocks, head_slot, tail_slot, record_count, hdr_crc = struct.unpack_from(
CTRL_FMT, ctrl_raw, 0
)
magic_lo = magic & 0xFFFFFFFF
version = (magic >> 32) & 0xFF
if magic_lo != DOE_REGION_MAGIC or version != DOE_REGION_VERSION_0:
print(" DOE region not yet initialized (no trials persisted yet)", file=sys.stderr)
return []
want_hdr_crc = compute_crc64(ctrl_raw[:CTRL_CRC_SPAN])
if want_hdr_crc != hdr_crc:
print(" DOE region control header CRC mismatch -- treating as "
"'not yet initialized' (same discipline doe_region_ctrl_read() "
"uses on the kernel side)", file=sys.stderr)
return []
total_slots = devblocks * DOE_SLOTS_PER_DEVBLOCK
records = []
corrupt = 0
with open(dev_path, "rb") as f:
idx = head_slot
for _ in range(record_count):
slot_devblock_dft = DOE_REGION_DEVBLOCK_FROM_TOP_BASE + 1 + (idx // DOE_SLOTS_PER_DEVBLOCK)
slot_off_in_devblock = (idx % DOE_SLOTS_PER_DEVBLOCK) * DOE_SLOT_SIZE
blk_off = dft_to_offset(slot_devblock_dft, total_devblocks)
f.seek(blk_off + slot_off_in_devblock)
raw = f.read(SLOT_SIZE)
(fleet_k_q48, switch_count, run_id, cfg, rep, nw, mode,
fail_count, fleet_conserved, isa_raw, _align_pad, slot_crc) = struct.unpack(SLOT_FMT, raw)
want_slot_crc = compute_crc64(raw[:SLOT_CRC_SPAN])
if want_slot_crc != slot_crc:
corrupt += 1
print(f" slot {idx} CRC mismatch (devblock-sharing torn write?) -- skipped", file=sys.stderr)
idx = (idx + 1) % total_slots
continue
isa = isa_raw.split(b"\x00", 1)[0].decode("ascii", errors="replace")
records.append({
"isa": isa,
"run_id": run_id,
"cfg": cfg,
"rep": rep,
"nw": nw,
"mode": mode,
"fail_count": fail_count,
"fleet_k_q48": fleet_k_q48,
"fleet_conserved": fleet_conserved,
"switch_count_cumulative": switch_count,
})
idx = (idx + 1) % total_slots
print(f" ring: devblocks={devblocks} head={head_slot} tail={tail_slot} "
f"record_count={record_count} -> {len(records)} valid records read"
+ (f", {corrupt} corrupt slot(s) skipped" if corrupt else ""),
file=sys.stderr)
return records
def main():
ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
ap.add_argument("--img", default="disk/artemis.img", help="Path to the disk image (default: disk/artemis.img)")
ap.add_argument("--loop", action="store_true", help="Attach --img via losetup first (needs sudo)")
ap.add_argument("-o", "--out", default=None, help="Output CSV path (default: stdout)")
args = ap.parse_args()
img_path = Path(args.img)
if not img_path.exists():
print(f"extract_doe_region: image not found: {img_path}", file=sys.stderr)
sys.exit(1)
if args.loop:
with loop_attach(img_path) as loop_dev:
records = read_ring(loop_dev)
else:
records = read_ring(str(img_path))
fieldnames = ["isa", "run_id", "cfg", "rep", "nw", "mode", "fail_count",
"fleet_k_q48", "fleet_conserved", "switch_count_cumulative"]
out_f = open(args.out, "w", newline="") if args.out else sys.stdout
try:
writer = csv.DictWriter(out_f, fieldnames=fieldnames)
writer.writeheader()
for rec in records:
writer.writerow(rec)
finally:
if args.out:
out_f.close()
print(f" wrote {len(records)} rows -> {args.out}", file=sys.stderr)
if __name__ == "__main__":
main()