ProCreations/repro-provably-data-driven-lagrangian-relaxation-for-mixed-integer-linear-programming
0
1#!/usr/bin/env python32"""Normalize paired native-run artifacts without changing model arrays.3 4The first paired run used NumPy's timestamped ``savez`` container and included5wall-clock/process metadata. This utility rewrites exactly the saved arrays6with fixed ZIP metadata, removes only nondeterministic metadata, corrects the7immutable source identifier, and then requires the two complete run8directories to be byte-identical.9"""10 11from __future__ import annotations12 13import argparse14import hashlib15import io16import json17import zipfile18from pathlib import Path19 20import numpy as np21 22 23ARRAY_ORDER = ("H1", "Y", "W1", "W2", "W3", "W4", "W5")24 25 26def digest(path: Path) -> str:27 return hashlib.sha256(path.read_bytes()).hexdigest()28 29 30def deterministic_npz(path: Path, arrays: dict[str, np.ndarray]) -> None:31 with zipfile.ZipFile(path, "w", compression=zipfile.ZIP_STORED) as archive:32 for name in ARRAY_ORDER:33 payload = io.BytesIO()34 np.lib.format.write_array(35 payload, np.asanyarray(arrays[name]), allow_pickle=False36 )37 info = zipfile.ZipInfo(f"{name}.npy", (1980, 1, 1, 0, 0, 0))38 info.compress_type = zipfile.ZIP_STORED39 info.external_attr = 0o600 << 1640 archive.writestr(info, payload.getvalue())41 42 43def normalize(directory: Path) -> dict[str, str]:44 state_path = directory / "final_state.npz"45 result_path = directory / "training_results.json"46 with np.load(state_path) as loaded:47 if set(loaded.files) != set(ARRAY_ORDER):48 raise RuntimeError(f"unexpected state keys in {state_path}")49 arrays = {name: loaded[name].copy() for name in ARRAY_ORDER}50 if not all(np.isfinite(array).all() for array in arrays.values()):51 raise RuntimeError(f"non-finite state in {state_path}")52 deterministic_npz(state_path, arrays)53 54 result = json.loads(result_path.read_text(encoding="utf-8"))55 result.pop("runtime_seconds", None)56 implementation = result["implementation"]57 implementation.pop("pid", None)58 result["paper"]["title"] = "Unifying Low Dimensional Spectra in Deep Learning"59 result["paper"]["source_revision"] = "arXiv:2404.06106v1"60 result["final_state_sha256"] = digest(state_path)61 result_path.write_text(62 json.dumps(result, indent=2, sort_keys=True) + "\n", encoding="utf-8"63 )64 return {65 "final_state.npz": digest(state_path),66 "training_results.json": digest(result_path),67 }68 69 70def main() -> None:71 parser = argparse.ArgumentParser()72 parser.add_argument("run_a", type=Path)73 parser.add_argument("run_b", type=Path)74 args = parser.parse_args()75 hashes_a = normalize(args.run_a)76 hashes_b = normalize(args.run_b)77 if hashes_a != hashes_b:78 raise RuntimeError(79 f"paired replay is not byte-identical: A={hashes_a}, B={hashes_b}"80 )81 print(82 json.dumps(83 {"status": "PASS", "paired_byte_identical": True, "sha256": hashes_a},84 indent=2,85 sort_keys=True,86 )87 )88 89 90if __name__ == "__main__":91 main()92 