iuna

iuna

iuna - experimental mainnet-candidate protocol
git clone https://getiuna.org/git/iuna.git
Log | Files | Refs | README | LICENSE

commit 5b13de07a30dc98da2cdb720a536c1bba9975beb
parent e8e763cfd17b0e22e2ba634ed059e5211bdc5ec5
Author: Joris Hartog <jorishartog@hotmail.com>
Date:   Sat,  5 Sep 2026 19:19:07 +0200

Preserve P2P partition release evidence

Diffstat:
M.gitignore | 1+
MPLAN.md | 2+-
MROADMAP.md | 5+++--
Mdeployment.sh | 3++-
Mdocs/security-review.md | 1+
Me2e/README.md | 20++++++++++++++++++++
Me2e/iuna_e2e.py | 169++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++---
7 files changed, 192 insertions(+), 9 deletions(-)

diff --git a/.gitignore b/.gitignore @@ -4,6 +4,7 @@ /.docker-build /.iuna /e2e/.runtime +/release-evidence /src-tauri/target /src-tauri/gen /src-tauri/binaries/iuna-sidecar-* diff --git a/PLAN.md b/PLAN.md @@ -13,7 +13,7 @@ test is nodig, maar sluit een live-soak gate niet automatisch. - [x] Herstart een node met zijn bestaande persistente data. - [x] Verifieer dat na recovery weer een normaal ticketblock wordt geproduceerd. - [x] Neem het scenario op in de verplichte post-activation release-gate. -- [ ] Bewaar bij falen en bij een candidate release de relevante node-logs en +- [x] Bewaar bij falen en bij een candidate release de relevante node-logs en tip/checkpoint-samenvatting. ## 2. Live recovery-bewijs diff --git a/ROADMAP.md b/ROADMAP.md @@ -130,8 +130,9 @@ A release intended for deployment must pass: `compact_snapshot`, `domain_json`, `stratum_request`, and `wallet_config` - `cargo test --locked --release --features e2e --test properties -- --ignored`, restoring the first objective checkpoint before exercising P2P, Stratum, and restarts -- `./e2e/iuna_e2e.py test post-activation --build`, which crosses height 1000, - checks objective finality, restarts all six nodes, and advances through 1007 +- `./e2e/iuna_e2e.py test post-activation --build --evidence-dir release-evidence`, + which crosses height 1000, checks objective finality, restarts all six nodes, + advances through 1007, and preserves process-level partition recovery evidence Normal local development may skip ignored long-running property tests and long fuzzing sessions, but deployment must run the release gate smoke checks. diff --git a/deployment.sh b/deployment.sh @@ -153,7 +153,8 @@ run_release_tests() { cargo run --locked --manifest-path fuzz/Cargo.toml --bin wallet_config -- -runs="$fuzz_runs" fuzz/corpus/wallet_config cargo run --locked --manifest-path fuzz/Cargo.toml --bin vdf_proof -- -runs="$vdf_fuzz_runs" fuzz/corpus/vdf_proof cargo test --locked --release --features e2e --test properties -- --ignored - ./e2e/iuna_e2e.py test post-activation --build + local release_evidence_dir="${IUNA_RELEASE_EVIDENCE_DIR:-release-evidence}" + ./e2e/iuna_e2e.py test post-activation --build --evidence-dir "$release_evidence_dir" } update_versions() { diff --git a/docs/security-review.md b/docs/security-review.md @@ -179,6 +179,7 @@ cargo run --locked --manifest-path fuzz/Cargo.toml --bin wallet_config -- -runs= cargo run --locked --manifest-path fuzz/Cargo.toml --bin vdf_proof -- -runs=16 fuzz/corpus/vdf_proof cargo test --locked domain::adversarial_tests:: -- --ignored cargo test --locked --release --test properties -- --ignored +./e2e/iuna_e2e.py test post-activation --build --evidence-dir release-evidence ``` On macOS hosts with the optional Python/C++ `chiavdf` package installed, also diff --git a/e2e/README.md b/e2e/README.md @@ -60,6 +60,13 @@ Tests can also be selected individually: ./e2e/iuna_e2e.py test partition-recovery ``` +Preserve a machine-readable phase report and complete container logs: + +```sh +./e2e/iuna_e2e.py test partition-recovery \ + --evidence-dir release-evidence +``` + `snapshots` is fast and does not need Docker. It verifies checksums, manifest metadata, the six stored chain databases, canonical heights and tips, and the profile of every committed checkpoint. The network scenarios restore the @@ -88,6 +95,19 @@ while leaving the other island members out of the recovery race. It restores the normal 50% configuration before healing, avoiding both a recovery-less small island and unrestricted fallback-ticket production. +Each evidence run is stored in a timestamped directory with `report.json` and +`nodes.log`. The report records the base Git revision, dirty-worktree flag, an +exact SHA-256 fingerprint of all tracked working-tree contents, per-phase node +tips and checkpoints, both partition recovery heights, the canonical recovery +block, the restarted service, and the resumed rank-0 ticket. The tree fingerprint +also identifies the tested state while deployment has staged version changes +that are committed and tagged only after all gates pass. +It is updated after every completed phase so a failed run remains useful. Raw +node logs contain public node/wallet addresses but no configuration files, +passwords, wallet ciphertext, or recovery phrases. `deployment.sh` enables this +automatically under `release-evidence/`; set `IUNA_RELEASE_EVIDENCE_DIR` to copy +release evidence to another retained location. + Use `--keep` on `smoke` to leave a failed or successful network running. Stop a network without deleting its mutable data with: diff --git a/e2e/iuna_e2e.py b/e2e/iuna_e2e.py @@ -5,6 +5,7 @@ from __future__ import annotations import argparse from dataclasses import dataclass +from datetime import datetime, timezone import hashlib import http.cookiejar import json @@ -138,6 +139,104 @@ def compose(*args: str, check: bool = True) -> subprocess.CompletedProcess[str]: ) +def create_evidence_run(base: Path | None, scenario: str) -> tuple[Path | None, dict]: + started_at = datetime.now(timezone.utc) + report = { + "format": 1, + "scenario": scenario, + "started_at": started_at.isoformat(), + "git_commit": subprocess.run( + ["git", "rev-parse", "HEAD"], + cwd=ROOT, + check=True, + capture_output=True, + text=True, + ).stdout.strip(), + "git_dirty": bool( + subprocess.run( + ["git", "status", "--porcelain"], + cwd=ROOT, + check=True, + capture_output=True, + text=True, + ).stdout.strip() + ), + "tracked_tree_sha256": tracked_tree_sha256(), + "phases": {}, + "outcome": "running", + } + if base is None: + return None, report + run = base.resolve() / f"{started_at.strftime('%Y%m%dT%H%M%S.%fZ')}-{scenario}" + run.mkdir(parents=True, exist_ok=False) + write_evidence_report(run, report) + return run, report + + +def tracked_tree_sha256() -> str: + tracked = subprocess.run( + ["git", "ls-files", "-z"], + cwd=ROOT, + check=True, + capture_output=True, + ).stdout.split(b"\0") + digest = hashlib.sha256() + for encoded_path in tracked: + if not encoded_path: + continue + path = ROOT / os.fsdecode(encoded_path) + digest.update(encoded_path) + digest.update(b"\0") + if path.exists(): + contents = path.read_bytes() + digest.update(len(contents).to_bytes(8, "big")) + digest.update(contents) + else: + digest.update(b"missing") + return digest.hexdigest() + + +def write_evidence_report(run: Path | None, report: dict) -> None: + if run is None: + return + temporary = run / "report.json.tmp" + temporary.write_text(json.dumps(report, indent=2, sort_keys=True) + "\n") + temporary.replace(run / "report.json") + + +def capture_evidence_logs(run: Path | None) -> None: + if run is None: + return + result = subprocess.run( + compose_command("logs", "--no-color", *SERVICES), + cwd=ROOT, + env=compose_env(), + check=False, + capture_output=True, + text=True, + ) + (run / "nodes.log").write_text(result.stdout + result.stderr) + + +def evidence_block(block: dict) -> dict: + return { + key: block.get(key) + for key in ( + "height", + "hash", + "prev_hash", + "timestamp_ms", + "finalizer_mode", + "finalizer_rank", + "miner", + ) + } + + +def evidence_statuses(statuses: dict[str, dict]) -> dict[str, dict]: + return {service: compact_status(status) for service, status in statuses.items()} + + def container_command( service: str, *args: str, check: bool = True ) -> subprocess.CompletedProcess[str]: @@ -890,8 +989,11 @@ def run_scenario( compose("down", "--remove-orphans", check=False) -def run_partition_recovery_scenario(timeout: float, build: bool, keep: bool) -> None: +def run_partition_recovery_scenario( + timeout: float, build: bool, keep: bool, evidence_dir: Path | None +) -> None: name = "partition-recovery" + evidence_run, evidence = create_evidence_run(evidence_dir, name) print( "running e2e scenario partition-recovery: physical 3-3 split, heal and restart", flush=True, @@ -899,7 +1001,9 @@ def run_partition_recovery_scenario(timeout: float, build: bool, keep: bool) -> restore_snapshot("first-objective-checkpoint") try: start(build) - wait_for_height(1_001, timeout, converge=True) + initial = wait_for_height(1_001, timeout, converge=True) + evidence["phases"]["initial"] = evidence_statuses(initial) + write_evidence_report(evidence_run, evidence) apply_partition() partitioned_start = all_statuses() left, right = PARTITION_GROUPS @@ -909,9 +1013,19 @@ def run_partition_recovery_scenario(timeout: float, build: bool, keep: bool) -> ) for label, group in (("left", left), ("right", right)) } + evidence["phases"]["partition_started"] = { + "boundaries": partition_boundaries, + "nodes": evidence_statuses(partitioned_start), + } + write_evidence_report(evidence_run, evidence) partitioned, recovery_heights = wait_for_partition_recovery( partition_boundaries, timeout ) + evidence["phases"]["partition_recovery"] = { + "recovery_heights": recovery_heights, + "nodes": evidence_statuses(partitioned), + } + write_evidence_report(evidence_run, evidence) print( "partition recovery observed: " + json.dumps(recovery_heights, sort_keys=True), @@ -937,21 +1051,52 @@ def run_partition_recovery_scenario(timeout: float, build: bool, keep: bool) -> canonical_recovery_height = max( int(block["height"]) for block in canonical_recoveries ) + canonical_recovery = next( + block + for block in canonical_recoveries + if int(block["height"]) == canonical_recovery_height + ) + evidence["phases"]["healed"] = { + "canonical_recovery": evidence_block(canonical_recovery), + "nodes": evidence_statuses(healed), + } + write_evidence_report(evidence_run, evidence) restart_height = max(status["chain"]["height"] for status in healed.values()) compose("restart", "node6") _OPENERS.pop("node6", None) + evidence["phases"]["restart"] = { + "service": "node6", + "after_height": restart_height, + } + write_evidence_report(evidence_run, evidence) resumed, ticket = wait_for_ticket_after( max(canonical_recovery_height, restart_height), timeout ) assert_converged(resumed) assert_api_health(resumed, int(ticket["height"])) + evidence["phases"]["resumed"] = { + "ticket": evidence_block(ticket), + "nodes": evidence_statuses(resumed), + } + evidence["outcome"] = "passed" + evidence["finished_at"] = datetime.now(timezone.utc).isoformat() + write_evidence_report(evidence_run, evidence) + capture_evidence_logs(evidence_run) print( f"e2e scenario {name} passed: recovery at {canonical_recovery_height}, " f"rank-0 ticket resumed at {ticket['height']}", flush=True, ) - except Exception: + except Exception as error: + evidence["outcome"] = "failed" + evidence["finished_at"] = datetime.now(timezone.utc).isoformat() + evidence["error"] = { + "type": type(error).__name__, + "message": str(error), + } + write_evidence_report(evidence_run, evidence) + capture_evidence_logs(evidence_run) compose("logs", "--tail", "300", *SERVICES, check=False) raise finally: @@ -960,7 +1105,9 @@ def run_partition_recovery_scenario(timeout: float, build: bool, keep: bool) -> compose("down", "--remove-orphans", check=False) -def run_tests(name: str, timeout: float, build: bool, keep: bool) -> None: +def run_tests( + name: str, timeout: float, build: bool, keep: bool, evidence_dir: Path | None +) -> None: if name in ("snapshots", "all"): test_snapshots() if name == "all": @@ -978,6 +1125,7 @@ def run_tests(name: str, timeout: float, build: bool, keep: bool) -> None: timeout, build and index == 0, scenario_keep, + evidence_dir, ) else: run_scenario( @@ -1046,6 +1194,11 @@ def parser() -> argparse.ArgumentParser: tests.add_argument("--timeout", type=float, default=600) tests.add_argument("--build", action="store_true") tests.add_argument("--keep", action="store_true") + tests.add_argument( + "--evidence-dir", + type=Path, + help="write partition recovery report and node logs below this directory", + ) return result @@ -1076,7 +1229,13 @@ def main() -> int: elif args.command == "smoke": smoke(args.snapshot, args.through, args.timeout, args.build, args.keep) elif args.command == "test": - run_tests(args.scenario, args.timeout, args.build, args.keep) + run_tests( + args.scenario, + args.timeout, + args.build, + args.keep, + args.evidence_dir, + ) return 0 except (E2EError, OSError, sqlite3.Error, subprocess.CalledProcessError) as error: print(f"e2e error: {error}", file=sys.stderr)