commit 4e89ca375ba8bdc20e9beade9faf5aba071cb5aa
parent 2060155b8211dd9c8f798cab8e1ec282124a7631
Author: Joris Hartog <jorishartog@hotmail.com>
Date: Sat, 5 Sep 2026 20:12:03 +0200
Add process-level sync resilience gate
Diffstat:
5 files changed, 296 insertions(+), 25 deletions(-)
diff --git a/PLAN.md b/PLAN.md
@@ -33,12 +33,19 @@ recovery-candidates bevat.
## 3. Sync-bewijs
-- [ ] Test een lege node die zonder handwerk vanaf genesis synchroniseert.
-- [ ] Test een stale node vanaf een oud snapshot via range/fork sync.
-- [ ] Test onderbreking en herstart tijdens beide sync-paden.
-- [ ] Archiveer hoogtes, tip-hashes, doorlooptijden en foutlogs als
+- [x] Test een lege node die zonder handwerk vanaf genesis synchroniseert.
+- [x] Test een stale node vanaf een oud snapshot via range/fork sync.
+- [x] Test onderbreking en herstart tijdens beide sync-paden.
+- [x] Archiveer hoogtes, tip-hashes, doorlooptijden en foutlogs als
release-evidence.
+Resultaat: de versnelde zeven-node `sync-resilience`-gate onderbrak een lege
+bootstrap vóór de eerste persistente snapshot en hervatte tot zeven-node
+convergentie op hoogte 1032. Daarna synchroniseerde dezelfde node vanaf hoogte
+299, werd tijdens actieve range-validatie met een persistente tussenstand op
+hoogte 811 afgebroken, en convergeerde na herstart met alle nodes op hoogte
+1059.
+
## 4. Release- en security-sign-off
- [ ] Draai alle gates uit `docs/security-review.md` op exact dezelfde revision.
diff --git a/ROADMAP.md b/ROADMAP.md
@@ -14,8 +14,8 @@ The current goal is to operate the candidate without unplanned resets, collect l
- [x] Block, transaction, ticket, VDF, recovery, fork-choice, and peer compatibility rules are documented.
- [x] Long-running testnet has stayed stable with independent nodes for an agreed window.
- [x] Mainnet-candidate network has launched from a fresh genesis using release artifacts.
-- [ ] New nodes can sync from genesis without manual intervention.
-- [ ] Stale nodes can reconnect and catch up from old snapshots/range sync.
+- [x] New nodes can sync from genesis without manual intervention.
+- [x] Stale nodes can reconnect and catch up from old snapshots/range sync.
- [ ] Post-activation network partitions have an implemented objective checkpoint-based recovery rule; live soak evidence is still required to close this gate.
- [ ] Recovery blocks restore liveness when selected finalizers disappear.
- [ ] Multiple recovery candidates converge safely.
@@ -139,7 +139,8 @@ A release intended for deployment must pass:
restoring the first objective checkpoint before exercising P2P, Stratum, and restarts
- `./e2e/iuna_e2e.py test post-activation --build --evidence-dir release-evidence`,
which crosses height 1000, checks objective finality, restarts all six nodes,
- advances through 1007, and preserves process-level partition recovery evidence
+ advances through 1007, interrupts empty and stale node sync, and preserves
+ process-level sync and partition recovery evidence
Normal local development may skip ignored long-running property tests and long fuzzing sessions, but deployment must run the release gate smoke checks.
diff --git a/e2e/README.md b/e2e/README.md
@@ -3,7 +3,8 @@
This harness runs the six-node network with a deliberately isolated consensus
profile, `iuna-local-e2e-5s-v1`. Its target block time is 5 seconds, its Docker
subnet is `172.29.0.0/24`, and its management ports are `28661` through `28666`.
-The regular local testnet keeps its 10-minute target and can run alongside it.
+The optional sync-test node uses `28667`. The regular local testnet keeps its
+10-minute target and can run alongside it.
The e2e binary refuses to start unless `IUNA_LOCAL_TESTNET=true`. Its profile ID
is part of the launch-profile hash, so a 5-second checkpoint cannot be loaded by
@@ -48,7 +49,8 @@ Run the standard post-activation gate used by deployment:
This verifies the committed checkpoints, crosses 999 through 1001, then restores
the first objective checkpoint, advances the restarted network through 1007,
-and runs a physical 3-3 P2P partition/recovery scenario.
+interrupts both an empty-node bootstrap and stale range sync, and runs a physical
+3-3 P2P partition/recovery scenario.
Tests can also be selected individually:
@@ -57,13 +59,14 @@ Tests can also be selected individually:
./e2e/iuna_e2e.py test fallback-activation
./e2e/iuna_e2e.py test objective-finality
./e2e/iuna_e2e.py test checkpoint-restart
+./e2e/iuna_e2e.py test sync-resilience
./e2e/iuna_e2e.py test partition-recovery
```
Preserve a machine-readable phase report and complete container logs:
```sh
-./e2e/iuna_e2e.py test partition-recovery \
+./e2e/iuna_e2e.py test sync-resilience \
--evidence-dir release-evidence
```
@@ -95,13 +98,24 @@ while leaving the other island members out of the recovery race. It restores the
normal 50% configuration before healing, avoiding both a recovery-less small
island and unrestricted fallback-ticket production.
+`sync-resilience` restores the mature six-node checkpoint and adds a seventh,
+non-finalizing `syncnode` with a fresh data directory. It interrupts that node
+before its fetched chain is persisted, then requires a successful restart and
+seven-node convergence. Next it gives the same node the height-299 chain/UI
+fixture, waits until the management API reports active incremental range
+validation, interrupts it again, and requires the persisted stale node to
+resume and converge. The six finalizing reference nodes remain untouched, so
+their consensus participation is not conflated with the sync failure being
+tested.
+
Each evidence run is stored in a timestamped directory with `report.json` and
`nodes.log`. The report records the base Git revision, dirty-worktree flag, an
-exact SHA-256 fingerprint of all tracked working-tree contents, per-phase node
-tips and checkpoints, both partition recovery heights, the canonical recovery
-block, the restarted service, and the resumed rank-0 ticket. The tree fingerprint
-also identifies the tested state while deployment has staged version changes
-that are committed and tagged only after all gates pass.
+exact SHA-256 fingerprint of all tracked working-tree contents, and per-phase
+node tips and checkpoints. Scenario-specific phases add the sync start,
+validated and target heights or both partition recovery heights, the canonical
+recovery block, the restarted service, and the resumed rank-0 ticket. The tree
+fingerprint also identifies the tested state while deployment has staged
+version changes that are committed and tagged only after all gates pass.
It is updated after every completed phase so a failed run remains useful. Raw
node logs contain public node/wallet addresses but no configuration files,
passwords, wallet ciphertext, or recovery phrases. `deployment.sh` enables this
@@ -119,7 +133,8 @@ Runtime data lives in `e2e/.runtime` and is ignored by Git. Set
`IUNA_E2E_RUNTIME_DIR`, `IUNA_E2E_SNAPSHOTS_DIR`, `IUNA_E2E_PROJECT`, or the
`IUNA_E2E_*_PORT` variables when a test needs independent paths or ports.
`reset` only removes the six service directories beneath that configured runtime
-directory; committed checkpoints are never touched.
+directory plus the disposable sync-node directory; committed checkpoints are
+never touched.
## Building mature checkpoints
diff --git a/e2e/docker-compose.e2e.yml b/e2e/docker-compose.e2e.yml
@@ -126,6 +126,38 @@ services:
iuna-testnet:
ipv4_address: 172.29.0.15
+ syncnode:
+ build:
+ context: .
+ dockerfile: Dockerfile.local-testnet
+ args:
+ IUNA_CARGO_FEATURES: "--features e2e"
+ image: iuna-local-e2e:latest
+ init: true
+ restart: "no"
+ environment:
+ IUNA_WALLET_PASSWORD: ${IUNA_TESTNET_PASSWORD:-testtesttest}
+ IUNA_SETUP_COMPLETE: "true"
+ IUNA_LOCAL_TESTNET: "true"
+ IUNA_AUTOMATIC_BURN_ENABLED: "false"
+ IUNA_POW_MINING_ENABLED: "false"
+ entrypoint: ["/bin/sh", "-c"]
+ command:
+ - exec iuna --join 172.29.0.13:9444 --data-dir /data --http 0.0.0.0:18661 --p2p 0.0.0.0:9444 --p2p-announce 172.29.0.16:9444 --debug
+ healthcheck:
+ test: ["CMD", "curl", "-fsS", "http://127.0.0.1:18661/api/auth/status"]
+ interval: 1s
+ timeout: 3s
+ retries: 900
+ start_period: 2s
+ ports:
+ - "${IUNA_E2E_SYNCNODE_PORT:-28667}:18661"
+ volumes:
+ - "${IUNA_E2E_RUNTIME_DIR:?set IUNA_E2E_RUNTIME_DIR}/syncnode:/data"
+ networks:
+ iuna-testnet:
+ ipv4_address: 172.29.0.16
+
networks:
iuna-testnet:
ipam:
diff --git a/e2e/iuna_e2e.py b/e2e/iuna_e2e.py
@@ -27,6 +27,7 @@ ROOT = Path(__file__).resolve().parent.parent
E2E_DIR = ROOT / "e2e"
COMPOSE_FILES = (ROOT / "docker-compose.yml", E2E_DIR / "docker-compose.e2e.yml")
SERVICES = ("bootstrap", "node2", "node3", "node4", "node5", "node6")
+SYNC_SERVICE = "syncnode"
SERVICE_IPS = {
"bootstrap": "172.29.0.10",
"node2": "172.29.0.11",
@@ -43,6 +44,7 @@ DEFAULT_PORTS = {
"node4": 28664,
"node5": 28665,
"node6": 28666,
+ SYNC_SERVICE: 28667,
}
PORT_ENV = {
"bootstrap": "IUNA_E2E_BOOTSTRAP_PORT",
@@ -51,6 +53,7 @@ PORT_ENV = {
"node4": "IUNA_E2E_NODE4_PORT",
"node5": "IUNA_E2E_NODE5_PORT",
"node6": "IUNA_E2E_NODE6_PORT",
+ SYNC_SERVICE: "IUNA_E2E_SYNCNODE_PORT",
}
SNAPSHOT_FILES = ("chain.sqlite3", "ui_data.sqlite3", "wallet.json", "config.json")
EXPECTED_PROFILE = "iuna-local-e2e-5s-v1"
@@ -90,7 +93,7 @@ SCENARIOS = {
leader_burn_minimum_height=1_002,
),
}
-SPECIAL_SCENARIOS = ("partition-recovery",)
+SPECIAL_SCENARIOS = ("sync-resilience", "partition-recovery")
POST_ACTIVATION_SCENARIOS = (
"objective-finality",
"checkpoint-restart",
@@ -204,11 +207,13 @@ def write_evidence_report(run: Path | None, report: dict) -> None:
temporary.replace(run / "report.json")
-def capture_evidence_logs(run: Path | None) -> None:
+def capture_evidence_logs(
+ run: Path | None, services: tuple[str, ...] = SERVICES
+) -> None:
if run is None:
return
result = subprocess.run(
- compose_command("logs", "--no-color", *SERVICES),
+ compose_command("logs", "--no-color", *services),
cwd=ROOT,
env=compose_env(),
check=False,
@@ -610,8 +615,10 @@ def assert_leader_uses_burn_from_height(
)
-def assert_api_health(statuses: dict[str, dict], through: int) -> None:
- for service in SERVICES:
+def assert_api_health(
+ statuses: dict[str, dict], through: int, services: tuple[str, ...] = SERVICES
+) -> None:
+ for service in services:
blocks = node_json(service, "/api/blocks?limit=3")
health = node_json(service, "/api/network/health")
peers = node_json(service, "/api/peers?limit=100")
@@ -647,6 +654,69 @@ def read_chain_metadata(path: Path) -> dict:
return {"height": row[0], "tip_hash": row[1], "updated_at_ms": row[2]}
+def remove_node_databases(service: str) -> None:
+ directory = runtime_dir() / service
+ for filename in ("chain.sqlite3", "ui_data.sqlite3"):
+ database = directory / filename
+ for suffix in ("", "-wal", "-shm", "-journal"):
+ candidate = Path(f"{database}{suffix}")
+ if candidate.exists():
+ candidate.unlink()
+
+
+def restore_node_databases(
+ service: str, snapshot: str, fixture_service: str | None = None
+) -> dict:
+ source = snapshot_path(snapshot)
+ verify_snapshot(source)
+ target = runtime_dir() / service
+ source_service = fixture_service or service
+ remove_node_databases(service)
+ for filename in ("chain.sqlite3", "ui_data.sqlite3"):
+ shutil.copy2(source / source_service / filename, target / filename)
+ return read_chain_metadata(target / "chain.sqlite3")
+
+
+def wait_for_active_sync(service: str, minimum_target: int, timeout: float) -> dict:
+ deadline = time.monotonic() + timeout
+ last_summary = "node unavailable"
+ while time.monotonic() < deadline:
+ try:
+ health = node_json(service, "/api/network/health")
+ start = health.get("sync_start_height")
+ validated = health.get("sync_validated_height")
+ target = health.get("sync_target_height")
+ last_summary = (
+ f"local={health.get('local_height')}, start={start}, "
+ f"validated={validated}, target={target}"
+ )
+ if (
+ isinstance(start, int)
+ and isinstance(validated, int)
+ and isinstance(target, int)
+ and target >= minimum_target
+ and validated < target
+ ):
+ return {
+ "local_height": health.get("local_height"),
+ "start_height": start,
+ "validated_height": validated,
+ "target_height": target,
+ }
+ except (
+ E2EError,
+ OSError,
+ KeyError,
+ ValueError,
+ urllib.error.URLError,
+ ) as error:
+ last_summary = str(error)
+ time.sleep(0.01)
+ raise E2EError(
+ f"timed out waiting for active range sync on {service}; {last_summary}"
+ )
+
+
def wait_for_persisted_height(target: int, timeout: float) -> dict:
deadline = time.monotonic() + timeout
database = runtime_dir() / "bootstrap" / "chain.sqlite3"
@@ -689,12 +759,17 @@ def materialize_chain_checkpoint(
)
-def wait_for_height(target: int, timeout: float, converge: bool) -> dict[str, dict]:
+def wait_for_height(
+ target: int,
+ timeout: float,
+ converge: bool,
+ services: tuple[str, ...] = SERVICES,
+) -> dict[str, dict]:
deadline = time.monotonic() + timeout
last_summary = "nodes unavailable"
while time.monotonic() < deadline:
try:
- statuses = all_statuses()
+ statuses = {service: node_status(service) for service in services}
assert_e2e_profile(statuses)
tips = {
(status["chain"]["height"], status["chain"]["tip_hash"])
@@ -907,7 +982,7 @@ def restore_snapshot(name: str) -> None:
def reset_runtime() -> None:
compose("down", "--remove-orphans", check=False)
base = runtime_dir()
- for service in SERVICES:
+ for service in (*SERVICES, SYNC_SERVICE):
target = base / service
if target.exists():
shutil.rmtree(target)
@@ -989,6 +1064,140 @@ def run_scenario(
compose("down", "--remove-orphans", check=False)
+def run_sync_resilience_scenario(
+ timeout: float, build: bool, keep: bool, evidence_dir: Path | None
+) -> None:
+ name = "sync-resilience"
+ service = SYNC_SERVICE
+ sync_services = (*SERVICES, service)
+ evidence_run, evidence = create_evidence_run(evidence_dir, name)
+ print(
+ "running e2e scenario sync-resilience: interrupted empty bootstrap and stale range sync",
+ flush=True,
+ )
+ restore_snapshot("first-objective-checkpoint")
+ try:
+ start(build)
+ initial = wait_for_height(1_001, timeout, converge=True)
+ initial_height = min(
+ status["chain"]["height"] for status in initial.values()
+ )
+ evidence["phases"]["initial"] = evidence_statuses(initial)
+ write_evidence_report(evidence_run, evidence)
+
+ compose("rm", "--force", "--stop", service, check=False)
+ _OPENERS.pop(service, None)
+ sync_directory = runtime_dir() / service
+ if sync_directory.exists():
+ shutil.rmtree(sync_directory)
+ sync_directory.mkdir(parents=True)
+ chain_database = sync_directory / "chain.sqlite3"
+ evidence["phases"]["empty_bootstrap_started"] = {
+ "service": service,
+ "source_height": initial_height,
+ "chain_snapshot_present": False,
+ }
+ write_evidence_report(evidence_run, evidence)
+ compose("up", "--detach", "--no-deps", service)
+ time.sleep(0.05)
+ compose("kill", "--signal", "SIGKILL", service)
+ _OPENERS.pop(service, None)
+ try:
+ interrupted = read_chain_metadata(chain_database)
+ except (E2EError, sqlite3.Error):
+ interrupted = None
+ if interrupted is not None and interrupted["height"] >= initial_height:
+ raise E2EError(
+ "empty bootstrap completed before it could be interrupted; "
+ "increase the fixture height"
+ )
+ evidence["phases"]["empty_bootstrap_interrupted"] = {
+ "signal": "SIGKILL",
+ "persisted_height": (
+ interrupted["height"] if interrupted is not None else None
+ ),
+ }
+ write_evidence_report(evidence_run, evidence)
+ compose("start", service)
+ empty_synced = wait_for_height(
+ initial_height, timeout, converge=True, services=sync_services
+ )
+ assert_converged(empty_synced)
+ evidence["phases"]["empty_bootstrap_resumed"] = {
+ "interrupted_before_target_persisted": True,
+ "nodes": evidence_statuses(empty_synced),
+ }
+ write_evidence_report(evidence_run, evidence)
+
+ stale_target = min(empty_synced[node]["chain"]["height"] for node in SERVICES)
+ compose("stop", service)
+ _OPENERS.pop(service, None)
+ stale = restore_node_databases(
+ service, "pre-fallback-invalidation", fixture_service="node6"
+ )
+ if stale["height"] >= stale_target:
+ raise E2EError(
+ f"stale fixture height {stale['height']} is not below target {stale_target}"
+ )
+ evidence["phases"]["stale_range_started"] = {
+ "service": service,
+ "stale_snapshot": "pre-fallback-invalidation",
+ "stale_height": stale["height"],
+ "stale_tip_hash": stale["tip_hash"],
+ "minimum_target_height": stale_target,
+ }
+ write_evidence_report(evidence_run, evidence)
+ compose("up", "--detach", "--no-deps", service)
+ _OPENERS.pop(service, None)
+ progress = wait_for_active_sync(service, stale_target, timeout)
+ compose("kill", "--signal", "SIGKILL", service)
+ _OPENERS.pop(service, None)
+ interrupted = read_chain_metadata(chain_database)
+ if interrupted["height"] >= progress["target_height"]:
+ raise E2EError(
+ "stale range sync reached its target before process interruption"
+ )
+ evidence["phases"]["stale_range_interrupted"] = {
+ **progress,
+ "signal": "SIGKILL",
+ "persisted_height": interrupted["height"],
+ "persisted_tip_hash": interrupted["tip_hash"],
+ }
+ write_evidence_report(evidence_run, evidence)
+ compose("start", service)
+ stale_synced = wait_for_height(
+ stale_target, timeout, converge=True, services=sync_services
+ )
+ assert_converged(stale_synced)
+ assert_api_health(stale_synced, stale_target, services=sync_services)
+ evidence["phases"]["stale_range_resumed"] = {
+ "nodes": evidence_statuses(stale_synced),
+ }
+ evidence["outcome"] = "passed"
+ evidence["finished_at"] = datetime.now(timezone.utc).isoformat()
+ write_evidence_report(evidence_run, evidence)
+ capture_evidence_logs(evidence_run, sync_services)
+ print(
+ f"e2e scenario {name} passed: empty bootstrap and range sync "
+ f"resumed through at least height {stale_target}",
+ flush=True,
+ )
+ except Exception as error:
+ evidence["outcome"] = "failed"
+ evidence["finished_at"] = datetime.now(timezone.utc).isoformat()
+ evidence["error"] = {
+ "type": type(error).__name__,
+ "message": str(error),
+ }
+ write_evidence_report(evidence_run, evidence)
+ capture_evidence_logs(evidence_run, sync_services)
+ compose("logs", "--tail", "300", *sync_services, check=False)
+ raise
+ finally:
+ if not keep:
+ compose("down", "--remove-orphans", check=False)
+
+
def run_partition_recovery_scenario(
timeout: float, build: bool, keep: bool, evidence_dir: Path | None
) -> None:
@@ -1127,6 +1336,13 @@ def run_tests(
scenario_keep,
evidence_dir,
)
+ elif scenario_name == "sync-resilience":
+ run_sync_resilience_scenario(
+ timeout,
+ build and index == 0,
+ scenario_keep,
+ evidence_dir,
+ )
else:
run_scenario(
scenario_name,
@@ -1197,7 +1413,7 @@ def parser() -> argparse.ArgumentParser:
tests.add_argument(
"--evidence-dir",
type=Path,
- help="write partition recovery report and node logs below this directory",
+ help="write scenario phase reports and node logs below this directory",
)
return result