From 79b1789cf1a02a740d221765dca7c44900ab6aea Mon Sep 17 00:00:00 2001 From: Moskyera Date: Tue, 28 Jul 2026 00:30:43 +0200 Subject: [PATCH 1/6] fix(node): a failed sync batch no longer ends the sync for good One line was the difference between a node that recovers and a node that is dead without saying so: if let Err(e) = res { println!("{}", e); return; } A failed batch printed and returned, and nothing ever asked for another. The node then sat at that height indefinitely, answering its RPC and looking healthy, while the chain moved on. That is how a single [Block Sync Warning] insert N failed: diamond status HTAKES not found became hours of silent nothing. It cost an operator an evening of mining against blocks the network had settled years earlier. It now resumes, from where the chain ACTUALLY is rather than from this batch's end, because a partial batch may have inserted some of its blocks and a wrong start height is refused by do_synchronize and would fail again at once. The ten second pause is not cosmetic: without it a permanently bad block becomes a tight loop that floods the peer and the log. With it a transient failure recovers on its own, and a permanent one keeps saying so out loud, which beats silence. While here, a correction to what the earlier commit assumed. fast_sync does not skip writing state. It reaches execution through ChainInfo and relaxes two CHECKS: protocol/src/context/context.rs:185 trusts a Type3 transaction's declared signers instead of verifying signatures, and protocol/src/action/macro.rs:125 runs precheck_runtime_action_fast_sync, which validates tx type and AST depth but skips the exec_from gating the full path applies. So the likely mechanism is that fast_sync ADMITS a block full validation would have rejected, leaving a chain a later block cannot build on. Narrowed, not proven, and stated that way. Co-Authored-By: Claude Opus 5 --- node/src/core/protocol.rs | 20 ++++++++++++++++++++ 1 file changed, 20 insertions(+) diff --git a/node/src/core/protocol.rs b/node/src/core/protocol.rs index 02f28ae..2596b97 100644 --- a/node/src/core/protocol.rs +++ b/node/src/core/protocol.rs @@ -311,7 +311,27 @@ pub(crate) async fn receive_blocks(hdl: &MsgHandler, peer: Arc, mut buf: V .await .unwrap(); if let Err(e) = res { + // A failed batch used to end the sync outright: print, return, and never + // ask for anything again. The node then sat at that height forever, + // answering its RPC and looking healthy, while the chain moved on. That + // is how a single + // [Block Sync Warning] insert N failed: diamond status HTAKES not found + // turned into a node that was dead for hours without saying so. + // + // One failure is not a reason to stop asking. Resume from where the + // chain ACTUALLY is rather than from this batch's end, because a partial + // batch may have inserted some of its blocks, and a wrong start height + // is refused by do_synchronize and would fail again immediately. + // + // The pause matters: without it a permanently bad block becomes a tight + // loop that floods the peer and the log. With it, a transient failure + // recovers on its own and a permanent one keeps saying so out loud, + // which is the outcome to prefer over silence. println!("{}", e); + let head = hdl.engine.latest_block().height().uint(); + println!("[P2P] sync failed at height {}; retrying from {} in 10s", end_hei, head + 1); + tokio::time::sleep(std::time::Duration::from_secs(10)).await; + send_req_block_msg(hdl, peer, head + 1).await; return; } println!("ok."); From 2532814a7b1ef672e1fb2d588ec874b26d679c73 Mon Sep 17 00:00:00 2001 From: Moskyera Date: Tue, 28 Jul 2026 00:47:51 +0200 Subject: [PATCH 2/6] fix: correct a false claim I made about fast_sync I wrote, in three shipped config comments and a release tag, that fast_sync = true builds a chain that cannot be extended. That is not true, and a controlled test run today says so plainly: a clean sync from zero with fast_sync = true reached the tip at 768,147, logged "all blocks sync finished", and produced zero warnings of any kind. What actually repaired the damaged chain yesterday was the clean RESYNC. I had one sample with the flag off, no sample with it on, and concluded causation anyway. That is the exact error this project spent the day catching elsewhere, and leaving it in a config comment would have taught it to whoever read the file next. The setting stays false, for the reason that survives testing rather than the one that did not. fast_sync reaches execution through ChainInfo and relaxes checks: protocol/src/context/context.rs trusts a Type3 transaction's declared signers instead of verifying signatures, and protocol/src/action/macro.rs runs a lighter action precheck. For a node serving a pool that pays other people, verifying every signature is worth more than a faster first sync. So the real cause of yesterday's stall remains unknown. What IS fixed and verified is the node giving up on it: a failed batch used to end the sync permanently, and now retries. Co-Authored-By: Claude Opus 5 --- deploy/node/hacash.config.ini | 20 +++++++++++++------- mainnet-configs/hacash.config.mainnet.ini | 20 +++++++++++++------- miner-panel/src/hacash_config.rs | 18 +++++++++++------- 3 files changed, 37 insertions(+), 21 deletions(-) diff --git a/deploy/node/hacash.config.ini b/deploy/node/hacash.config.ini index ba171db..e88b2ba 100644 --- a/deploy/node/hacash.config.ini +++ b/deploy/node/hacash.config.ini @@ -14,13 +14,19 @@ listen = 3337 boots = 54.193.49.59:3337, 182.92.163.225:3337, 54.219.80.127:3337 not_find_nodes = false fast_sync = false -; NOT true. Measured 2026-07-27: a chain synced with fast_sync = true -; stops dead at a block whose state it never wrote, with -; [Block Sync Warning] insert N failed: diamond status HTAKES not found -; and nothing retries, so the node sits there forever looking healthy. -; Turning the flag off afterwards does not repair it: the state was never -; written. A clean sync with it OFF reached the tip in seven minutes with -; no errors, which is the only reason this is not still true. +; Deliberately false, and NOT for the reason first written here. An earlier +; version of this comment claimed fast_sync corrupts the chain. That was wrong: +; a controlled test on 2026-07-28 synced this chain from zero with fast_sync = true +; and reached the tip with no errors at all. The corruption seen on 2026-07-27 was +; repaired by the clean RESYNC, not by the flag, and concluding otherwise was +; reading causation out of one sample with no control. +; +; The real reason to leave it off is what it actually relaxes. fast_sync reaches +; execution through ChainInfo and skips checks: protocol/src/context/context.rs +; trusts a Type3 transaction's declared signers instead of verifying signatures, +; and protocol/src/action/macro.rs runs a lighter action precheck. For a node that +; will serve a pool paying other people, validating every signature is worth more +; than a faster first sync. ; No [mint] section. chain_id defaults to 0, which is mainnet. diff --git a/mainnet-configs/hacash.config.mainnet.ini b/mainnet-configs/hacash.config.mainnet.ini index d9dcc0a..93b28d8 100644 --- a/mainnet-configs/hacash.config.mainnet.ini +++ b/mainnet-configs/hacash.config.mainnet.ini @@ -20,13 +20,19 @@ boots = 54.193.49.59:3337, 182.92.163.225:3337, 54.219.80.127:3337 ; MUST be false to join live network (true = isolated local chain) not_find_nodes = false fast_sync = false -; NOT true. Measured 2026-07-27: a chain synced with fast_sync = true -; stops dead at a block whose state it never wrote, with -; [Block Sync Warning] insert N failed: diamond status HTAKES not found -; and nothing retries, so the node sits there forever looking healthy. -; Turning the flag off afterwards does not repair it: the state was never -; written. A clean sync with it OFF reached the tip in seven minutes with -; no errors, which is the only reason this is not still true. +; Deliberately false, and NOT for the reason first written here. An earlier +; version of this comment claimed fast_sync corrupts the chain. That was wrong: +; a controlled test on 2026-07-28 synced this chain from zero with fast_sync = true +; and reached the tip with no errors at all. The corruption seen on 2026-07-27 was +; repaired by the clean RESYNC, not by the flag, and concluding otherwise was +; reading causation out of one sample with no control. +; +; The real reason to leave it off is what it actually relaxes. fast_sync reaches +; execution through ChainInfo and skips checks: protocol/src/context/context.rs +; trusts a Type3 transaction's declared signers instead of verifying signatures, +; and protocol/src/action/macro.rs runs a lighter action precheck. For a node that +; will serve a pool paying other people, validating every signature is worth more +; than a faster first sync. [mint] ; Mainnet consensus defaults (do not override difficulty / block time for mainnet) diff --git a/miner-panel/src/hacash_config.rs b/miner-panel/src/hacash_config.rs index b7a1b9a..24d5227 100644 --- a/miner-panel/src/hacash_config.rs +++ b/miner-panel/src/hacash_config.rs @@ -259,13 +259,17 @@ fn ensure_mainnet_node_section(content: &str) -> String { if has_node { return content.to_string(); } - // fast_sync is deliberately FALSE. Measured 2026-07-27: a chain synced with - // it on stops dead at a block whose state it never wrote, with - // [Block Sync Warning] insert N failed: diamond status HTAKES not found - // and nothing retries, so the node sits at that height forever while looking - // perfectly healthy. Turning the flag off afterwards does not repair it, - // because the state was never written. A clean sync with it OFF reached the - // tip in seven minutes with no errors at all. + // fast_sync is deliberately FALSE, and NOT for the reason first written here. + // This comment used to claim the flag corrupts the chain. It does not: a + // controlled test on 2026-07-28 synced from zero with fast_sync = true and + // reached the tip with no errors. What repaired the damaged chain seen the day + // before was the clean RESYNC, not the flag, and saying otherwise was reading + // causation out of a single sample with no control. + // + // The real reason is what it relaxes: fast_sync trusts a Type3 transaction's + // declared signers instead of verifying signatures, and runs a lighter action + // precheck. A panel user mining to their own wallet deserves the full checks; + // a faster first sync is not worth trading them for. let node = "[node]\nname = rust_node\nlisten = 3337\nboots = 54.193.49.59:3337, 182.92.163.225:3337, 54.219.80.127:3337\nnot_find_nodes = false\nfast_sync = false\n\n"; format!("{node}{content}") } From 12fdb5ca7f64f4793dfeb19b08bef3c83865d499 Mon Sep 17 00:00:00 2001 From: Moskyera Date: Tue, 28 Jul 2026 00:48:22 +0200 Subject: [PATCH 3/6] chore: 0.5.3 Co-Authored-By: Claude Opus 5 --- Cargo.lock | 2 +- Cargo.toml | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index 899db98..44de559 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -2129,7 +2129,7 @@ dependencies = [ [[package]] name = "hacash" -version = "0.5.2" +version = "0.5.3" dependencies = [ "app", "basis", diff --git a/Cargo.toml b/Cargo.toml index 5e4a7ee..013cb7e 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -1,7 +1,7 @@ [package] name = "hacash" default-run = "hacash" -version = "0.5.2" +version = "0.5.3" edition = "2024" [workspace] From ae1c2c0c26849cc9ee94826cbf29e560feb86ec9 Mon Sep 17 00:00:00 2001 From: Moskyera Date: Wed, 29 Jul 2026 00:04:34 +0200 Subject: [PATCH 4/6] fix(pool): eleven money defects, found by running the pool rather than reading it An adversarial audit raised 35 findings; 11 survived an independent refutation pass and merged into eight. A separate reviewer then found three of those eight fixes incomplete, two of which had quietly reintroduced the defect they claimed to close. A full settlement cycle on a purpose-built rig then found three more. All of it is fixed here, and the whole money path has now run end to end. THE THREE THAT STOPPED AN OPERATOR A payout the node had broadcast and then forgotten, which a routine restart causes, was treated as never paid and re-signed, so both could confirm and the operator funded the second. node_holds was already recorded and read by no decision anywhere; it now decides, and a payout the node still holds is re-broadcast byte-identically instead of re-signed. A submit that timed out is no longer read as a refusal. Shares carried no time, so a miner could withhold a block's worth and dump them at the settlement tick, evicting every honest miner from the window and taking the whole payout. The pool published the schedule itself in /terms. Credit is now residence, not headcount: a burst has earned nothing at the moment it lands, and the shares it evicts keep what they had already earned, so submitting late is strictly worse than submitting promptly. A node that failed to answer was valued as a wallet holding zero, because transport failure, a bad body and the node's own error object all collapsed to the same empty string and balance_units("") returned a confident Some(0). Every miner polling /earnings was told it was owed nothing, every thirty seconds, with no log line at all. Answering is now distinguishable from answering zero. THE REST A rejected settlement chunk was forgotten while the log promised the next cycle would re-issue it; there is now an owed ledger, paid first, persisted, and the message says what actually happens. Only the coinbase subsidy was held for maturity, so a block's transaction fees were distributable at zero confirmations and an orphan could take back money already paid. The share-cost bound was checked at startup and never again. The settlement thread called a function that can mint a wallet or process::exit mid-flight. flush_state reported success for a snapshot it never wrote, and the fsync it rested on was discarded. AND THREE LIES THE POOL TOLD ITS OPERATOR A stall alarm counted settlement cycles, which have no relation to block time, so on mainnet defaults it fired on cycles 3 to 6 of every payout forever, and blamed a cause it never checked: that the pool's blocks carry only a coinbase. Measured false on the rig, where the pool's own block carried four transactions and a later one carried the payout itself. It now counts blocks, and states what it measured. A restart re-stamped the template mid-height, so every connected worker kept hashing a dead header: 6,320 consecutive rejects, and 1,281 copies of a message asserting a permanent fault in the worker's software. The stamp now survives a restart, and the diagnostic reports what was observed instead of diagnosing a fault the pool cannot see. One stdout line per accepted share, 7.6 MB in four minutes, sharing a println with the block-found notice, so the one line that must never be missed was buried in the one that does not matter. WHAT WAS PROVEN, ON HARDWARE each_block_target_time sets a private chain's resting difficulty, because ASERT is an equilibrium controller. At 455 seconds this box reached 34 bits in 44 minutes, which is the first configuration the pool will serve AND a GPU can win blocks on. No consensus code was touched, so mainnet is byte-identical by construction. On it: a block carrying four fee-paying transactions had its fees held back, not released; two workers at different hashrates were paid 29 and 5 units, matching largest-remainder on their residence credit to the unit; and two hard kills across five process lifetimes produced exactly three payout transactions totalling exactly what the two miners received. Nothing paid twice, nothing lost. NOT PROVEN, AND SAID SO RATHER THAN LEFT IMPLIED The owed ledger was never non-empty on the rig, because the node could not be made to definitively reject a payout, so that path rests on unit tests alone. No mainnet block fixture exists; the fee fixtures are real bytes from a real node, but a testnet one. And fee_got equalled fee throughout, so the gas-refund case is untested against a node. Co-Authored-By: Claude Opus 5 --- .grok/workflows/review-miner-pool-delta.rhai | 496 +++ .grok/workflows/review-miner-pool.rhai | 301 ++ app/src/block_mining_runtime.rs | 13 +- app/src/diaworker.rs | 7 +- app/src/mining_batch.rs | 20 +- app/src/opencl_gpu/block.rs | 22 +- app/src/poworker.rs | 4 +- docs/MINER-POOL-DELTA-REVIEW.md | 138 + docs/MINER-POOL-WORKFLOW-REVIEW.md | 122 + hbit-pool/examples/fee_probe.rs | 42 + hbit-pool/examples/local_chain_watch.rs | 162 + hbit-pool/examples/rig_tx.rs | 100 + hbit-pool/examples/testnet_rig_plan.rs | 224 + hbit-pool/src/asert_check.rs | 19 +- hbit-pool/src/difficulty.rs | 28 +- hbit-pool/src/lib.rs | 1475 ++++++- hbit-pool/src/payout.rs | 316 +- hbit-pool/src/pool_core.rs | 555 ++- hbit-pool/src/server.rs | 3622 +++++++++++++++-- hbit-pool/src/settle.rs | 47 +- .../tests/block_fee_holdback_node_fixtures.rs | 465 +++ hbit-pool/tests/fixtures/node/README.md | 54 + .../fixtures/node/block_307_intro_0tx.json | 1 + .../fixtures/node/block_308_intro_2tx.json | 1 + .../fixtures/node/block_309_intro_3tx.json | 1 + .../fixtures/node/block_missing_error.json | 1 + .../tests/fixtures/node/tx_308_fee_1_239.json | 1 + .../tests/fixtures/node/tx_308_fee_1_247.json | 1 + .../tests/fixtures/node/tx_309_fee_1_246.json | 1 + .../tests/fixtures/node/tx_309_fee_3_245.json | 1 + .../tests/fixtures/node/tx_309_fee_7_244.json | 1 + .../fixtures/node/tx_309_fee_unit_mei.json | 1 + .../fixtures/node/tx_bad_hash_error.json | 1 + .../tests/fixtures/node/tx_missing_error.json | 1 + .../fixtures/node/unknown_route_body.txt | 0 .../fixtures/node/unknown_route_headers.txt | 4 + .../tests/upgrade_from_an_old_state_file.rs | 306 ++ miner-panel/src/connect.rs | 3 +- miner-panel/src/dashboard.rs | 38 +- miner-panel/src/fleet.rs | 3 +- miner-panel/src/hacash_config.rs | 20 +- miner-panel/src/i18n.rs | 8 +- miner-panel/src/main.rs | 2 +- miner-panel/src/mining_control.rs | 2 +- miner-panel/src/node_sync.rs | 19 +- miner-panel/src/public_pool.rs | 10 +- miner-panel/src/stats_poll.rs | 26 +- miner-pool/src/config.rs | 6 +- miner-pool/src/job.rs | 11 +- miner-pool/src/rpc_proxy.rs | 15 +- miner-pool/src/upstream.rs | 20 +- mint/src/check/difficulty_lwma.rs | 149 + node/src/core/protocol.rs | 6 +- scripts/hbit-local-chain-rig/.gitignore | 3 + scripts/hbit-local-chain-rig/Run-Rig.ps1 | 259 ++ .../hbit-local-chain-rig/node.ini.template | 92 + .../poworker.ini.template | 55 + x16rs-cuda/build.rs | 5 +- x16rs-cuda/src/lib.rs | 57 +- 59 files changed, 8717 insertions(+), 646 deletions(-) create mode 100644 .grok/workflows/review-miner-pool-delta.rhai create mode 100644 .grok/workflows/review-miner-pool.rhai create mode 100644 docs/MINER-POOL-DELTA-REVIEW.md create mode 100644 docs/MINER-POOL-WORKFLOW-REVIEW.md create mode 100644 hbit-pool/examples/fee_probe.rs create mode 100644 hbit-pool/examples/local_chain_watch.rs create mode 100644 hbit-pool/examples/rig_tx.rs create mode 100644 hbit-pool/examples/testnet_rig_plan.rs create mode 100644 hbit-pool/tests/block_fee_holdback_node_fixtures.rs create mode 100644 hbit-pool/tests/fixtures/node/README.md create mode 100644 hbit-pool/tests/fixtures/node/block_307_intro_0tx.json create mode 100644 hbit-pool/tests/fixtures/node/block_308_intro_2tx.json create mode 100644 hbit-pool/tests/fixtures/node/block_309_intro_3tx.json create mode 100644 hbit-pool/tests/fixtures/node/block_missing_error.json create mode 100644 hbit-pool/tests/fixtures/node/tx_308_fee_1_239.json create mode 100644 hbit-pool/tests/fixtures/node/tx_308_fee_1_247.json create mode 100644 hbit-pool/tests/fixtures/node/tx_309_fee_1_246.json create mode 100644 hbit-pool/tests/fixtures/node/tx_309_fee_3_245.json create mode 100644 hbit-pool/tests/fixtures/node/tx_309_fee_7_244.json create mode 100644 hbit-pool/tests/fixtures/node/tx_309_fee_unit_mei.json create mode 100644 hbit-pool/tests/fixtures/node/tx_bad_hash_error.json create mode 100644 hbit-pool/tests/fixtures/node/tx_missing_error.json create mode 100644 hbit-pool/tests/fixtures/node/unknown_route_body.txt create mode 100644 hbit-pool/tests/fixtures/node/unknown_route_headers.txt create mode 100644 hbit-pool/tests/upgrade_from_an_old_state_file.rs create mode 100644 scripts/hbit-local-chain-rig/.gitignore create mode 100644 scripts/hbit-local-chain-rig/Run-Rig.ps1 create mode 100644 scripts/hbit-local-chain-rig/node.ini.template create mode 100644 scripts/hbit-local-chain-rig/poworker.ini.template diff --git a/.grok/workflows/review-miner-pool-delta.rhai b/.grok/workflows/review-miner-pool-delta.rhai new file mode 100644 index 0000000..8202dcd --- /dev/null +++ b/.grok/workflows/review-miner-pool-delta.rhai @@ -0,0 +1,496 @@ +// Delta review: re-check prior confirmed bugs + scan for new ones. +// Compares against the 15 findings from review-miner-pool (2026-07-24). + +let meta = #{ + name: "review-miner-pool-delta", + description: "Re-verify prior miner/pool bugs and hunt for new ones; produce fixed vs open vs new report", + when_to_use: "After miner/pool fixes; track regression and remaining open issues", + phases: [ + #{ title: "Recheck", detail: "adversarially re-verify each prior confirmed bug" }, + #{ title: "Fresh scan", detail: "parallel hunt for NEW bugs not in the prior list" }, + #{ title: "Synthesize", detail: "delta report: fixed / still open / new" }, + ], +}; + +let status_schema = #{ + "type": "object", + "required": ["status", "reason", "evidence"], + "properties": #{ + "status": #{ "type": "string" }, + "reason": #{ "type": "string" }, + "evidence": #{ "type": "string" }, + }, +}; + +let findings_schema = #{ + "type": "object", + "required": ["findings"], + "properties": #{ + "findings": #{ + "type": "array", + "maxItems": 8, + "items": #{ + "type": "object", + "required": ["severity", "file", "issue", "impact"], + "properties": #{ + "severity": #{ "type": "string" }, + "file": #{ "type": "string" }, + "issue": #{ "type": "string" }, + "impact": #{ "type": "string" }, + }, + }, + }, + }, +}; + +let verdict_schema = #{ + "type": "object", + "required": ["real", "reason", "evidence"], + "properties": #{ + "real": #{ "type": "boolean" }, + "reason": #{ "type": "string" }, + "evidence": #{ "type": "string" }, + }, +}; + +let report_schema = #{ + "type": "object", + "required": ["summary", "fixed_count", "still_open_count", "new_count", "markdown"], + "properties": #{ + "summary": #{ "type": "string" }, + "fixed_count": #{ "type": "integer" }, + "still_open_count": #{ "type": "integer" }, + "new_count": #{ "type": "integer" }, + "markdown": #{ "type": "string" }, + }, +}; + +let root = "C:/Users/KQHEX/Documents/hacash-fullnodedev"; +if args != () && args.root != () { + root = args.root; +} + +log("delta review root=" + root); + +// Prior confirmed bugs from review-miner-pool (adversarially verified). +// Each recheck agent must set status to: fixed | still_open | unclear +let prior = [ + #{ + id: "P1", + severity: "high", + file: "app/src/block_mining_runtime.rs winners coalesce", + issue: "Winners coalesced only by height; same-height reorg stale result can out-rank and replace a live valid solution; no epoch/template id on BlockMiningResult; submit without live-template revalidation.", + }, + #{ + id: "P2", + severity: "high", + file: "app/src/poworker.rs template install", + issue: "Template install only when pending_height > curr_hei or same-height intro_changed; reorg that LOWERs pending height is ignored so workers never job-switch.", + }, + #{ + id: "P3", + severity: "high", + file: "x16rs/opencl/x16rs_diamond.cl reduction", + issue: "Diamond OpenCL reduction initializes best_hash=0 and leaves best_name uninitialized; scans i=1..; wrong winners vs block kernel seed-from-index pattern.", + }, + #{ + id: "P4", + severity: "high", + file: "x16rs/opencl/x16rs_diamond.cl fences", + issue: "Diamond kernel uses only CLK_LOCAL_MEM_FENCE while hashes live in global memory; block kernel uses LOCAL|GLOBAL.", + }, + #{ + id: "P5", + severity: "high", + file: "app/src/diaworker.rs diamond submit", + issue: "push_diamond_mining_success treats any HTTP Ok body as final; non-JSON/missing tx_hash fails permanently without retry; no durable requeue after drain.", + }, + #{ + id: "P6", + severity: "medium", + file: "app/src/poworker.rs LAST_PENDING_INTRO", + issue: "LAST_PENDING_INTRO overwritten BEFORE set_pending_block_stuff succeeds; failed same-height install never retried.", + }, + #{ + id: "P7", + severity: "medium", + file: "app/src/block_mining_runtime.rs send before job-switch", + issue: "Job-switch height/epoch checked only AFTER full batch and send(); stale results always enter channel; drain does not filter by live epoch.", + }, + #{ + id: "P8", + severity: "medium", + file: "app/src/block_mining_runtime.rs worker rollover", + issue: "Worker rollover uses check_hei > mining_hei (advance-only) plus epoch; height decrease without epoch path may not stop workers.", + }, + #{ + id: "P9", + severity: "medium", + file: "x16rs/opencl/sha3_256.cl diamond length", + issue: "Host builds 61-byte stuff for diamond numbers <= 20000 but sha3_256_hash_diamond always pads as 93-byte; GPU SHA3 diverges from CPU/consensus.", + }, + #{ + id: "P10", + severity: "medium", + file: "app/src/opencl_dia.rs medium hash verify", + issue: "Diamond GPU success path never recomputes x16rs_hash and byte-compares GPU medium hash (unlike block verify_gpu_best_result).", + }, + #{ + id: "P11", + severity: "medium", + file: "x16rs-cuda batch launch", + issue: "CUDA batch mining launches with fixed local_size 256 and never calls clamped_block_size; single-hash path clamps.", + }, + #{ + id: "P12", + severity: "medium", + file: "app/src/diaworker.rs submit durability", + issue: "After MAX_SUBMIT_ATTEMPTS network failure or soft parse failure, drained DiamondMint is dropped with only a log; no local save/requeue.", + }, + #{ + id: "P13", + severity: "medium", + file: "x16rs/opencl/x16rs_diamond.cl unit0", + issue: "Per-work-item reduction never diamond_hashs unit index 0 into best_name before comparisons.", + }, + #{ + id: "P14", + severity: "low", + file: "app/src/block_mining_runtime.rs total_nonce_space", + issue: "Drain aggregate adds planned res.nonce_space even when recovery only mined partial window; telemetry misleading.", + }, + #{ + id: "P15", + severity: "low", + file: "app/src/opencl_gpu/block.rs stuff length", + issue: "OpenCL block upload accepts stuff length <= 512 without enforcing 89-byte block intro that CUDA requires.", + }, +]; + +// Also re-check consensus items that should already be fixed from earlier sessions +let prior_extra = [ + #{ + id: "C1", + severity: "critical", + file: "x16rs/src/diamond.rs", + issue: "Testnet hack: DMD_L=4 DMD_M=10 instead of mainnet DMD_L=10 DMD_M=16.", + }, + #{ + id: "C2", + severity: "critical", + file: "mint/src/action/diamond_mint.rs", + issue: "Diamond height%5 rule disabled with if false.", + }, + #{ + id: "C3", + severity: "high", + file: "app/src mining winners/equal target/GPU fatal", + issue: "Old bugs: only single global best submit; equal-to-target dropped; silent CPU fallback on GPU fail; CUDA no verify; full CPU recovery.", + }, + #{ + id: "C4", + severity: "high", + file: "miner-pool rpc_proxy/stratum", + issue: "Non-JSON upstream body reported as ret:0 success; stratum used substring ret check.", + }, +]; + +// --- Phase 1: recheck all prior items --- +phase("Recheck"); + +let recheck_jobs = []; +let all_prior = []; +for p in prior { + all_prior.push(p); +} +for p in prior_extra { + all_prior.push(p); +} + +for p in all_prior { + let rp = ""; + rp += "READ-ONLY recheck of a previously reported bug in " + root + ".\n"; + rp += "Bug id: " + p.id + "\n"; + rp += "Prior severity: " + p.severity + "\n"; + rp += "File/area: " + p.file + "\n"; + rp += "Claimed issue: " + p.issue + "\n\n"; + rp += "Open the relevant source with read_file/grep. Decide status:\n"; + rp += "- fixed: the bug is no longer present; code clearly handles the case (quote the fix).\n"; + rp += "- still_open: the buggy logic is still there (quote the evidence).\n"; + rp += "- unclear: cannot determine (explain what was missing).\n"; + rp += "status MUST be exactly one of: fixed, still_open, unclear.\n"; + rp += "evidence must cite path and what you saw. reason is one short sentence.\n"; + recheck_jobs.push(#{ + prompt: rp, + label: "recheck:" + p.id, + capability_mode: "read-only", + output_schema: status_schema, + }); +} + +let recheck_results = parallel(recheck_jobs); + +let fixed_list = []; +let open_list = []; +let unclear_list = []; +let ri = 0; +for r in recheck_results { + let p = all_prior[ri]; + let st = "unclear"; + let reason = "recheck agent failed"; + let evidence = ""; + if r != () && r.success && r.output.status != () { + st = r.output.status; + if r.output.reason != () { + reason = r.output.reason; + } + if r.output.evidence != () { + evidence = r.output.evidence; + } + } + // normalize + if st != "fixed" && st != "still_open" && st != "unclear" { + // allow minor variants + if st == "FIXED" || st == "Fixed" { + st = "fixed"; + } else if st == "still open" || st == "open" || st == "STILL_OPEN" { + st = "still_open"; + } else { + st = "unclear"; + } + } + let row = #{ + id: p.id, + severity: p.severity, + file: p.file, + issue: p.issue, + status: st, + reason: reason, + evidence: evidence, + }; + if st == "fixed" { + fixed_list.push(row); + } else if st == "still_open" { + open_list.push(row); + } else { + unclear_list.push(row); + } + ri += 1; +} + +log("recheck: fixed=" + fixed_list.len().to_string() + + " open=" + open_list.len().to_string() + + " unclear=" + unclear_list.len().to_string()); + +// --- Phase 2: fresh scan for NEW bugs --- +phase("Fresh scan"); + +let areas = [ + #{ + id: "block-miner", + paths: "app/src/block_mining_runtime.rs, app/src/mining_batch.rs, app/src/poworker.rs, app/src/mining_runtime.rs, app/src/hash_util.rs", + focus: "reorg job install, winner epoch tagging, equal target, nonce advance, GPU fail-closed, races", + }, + #{ + id: "gpu-diamond", + paths: "x16rs/opencl/x16rs_diamond.cl, x16rs/opencl/x16rs_main.cl, x16rs/opencl/sha3_256.cl, app/src/opencl_dia.rs, app/src/opencl_gpu/, app/src/mining_batch.rs, x16rs-cuda/", + focus: "diamond/block kernel reduction, fences, SHA3 length, CUDA clamp, integrity verify, buffer sizes", + }, + #{ + id: "hacd-pool", + paths: "app/src/diaworker.rs, x16rs/src/diamond.rs, mint/src/action/diamond_mint.rs, mint/src/check/block_build.rs, miner-pool/src/", + focus: "diamond consensus mainnet, submit durability, pool ret handling, stratum parse", + }, +]; + +// Build exclusion summary so scanners avoid re-reporting known items +let known_summary = ""; +for p in all_prior { + known_summary += p.id + ": " + p.issue + " | "; +} + +let scan_jobs = []; +for a in areas { + let sp = ""; + sp += "READ-ONLY bug hunt for NEW issues only in " + root + ".\n"; + sp += "Area: " + a.id + "\n"; + sp += "Paths: " + a.paths + "\n"; + sp += "Focus: " + a.focus + "\n\n"; + sp += "ALREADY REPORTED (do NOT re-list these unless you found a DISTINCT new facet):\n"; + sp += known_summary + "\n\n"; + sp += "Use read_file/grep. Prefer real correctness bugs. severity: critical|high|medium|low.\n"; + sp += "Max 8 findings. Empty list valid after inspection. Return findings {severity,file,issue,impact}.\n"; + scan_jobs.push(#{ + prompt: sp, + label: "scan:" + a.id, + capability_mode: "read-only", + output_schema: findings_schema, + }); +} + +let scan_results = parallel(scan_jobs); + +let new_raw = []; +let si = 0; +for r in scan_results { + let area = areas[si].id; + if r != () && r.success && r.output.findings != () { + for f in r.output.findings { + new_raw.push(#{ + area: area, + severity: f.severity, + file: f.file, + issue: f.issue, + impact: f.impact, + }); + } + } + si += 1; +} + +log("fresh scan raw new candidates: " + new_raw.len().to_string()); + +// Verify new candidates (cap 12) +let MAX_NEW_V = 12; +let new_to_v = []; +let ndrop = 0; +for f in new_raw { + if new_to_v.len() < MAX_NEW_V { + new_to_v.push(f); + } else { + ndrop += 1; + } +} +if ndrop > 0 { + log("capped new verify list, dropped " + ndrop.to_string()); +} + +let new_confirmed = []; +if new_to_v.len() > 0 { + let vjobs = []; + for f in new_to_v { + let vp = ""; + vp += "Adversarially verify this alleged NEW bug under " + root + ".\n"; + vp += "Area: " + f.area + "\nFile: " + f.file + "\nIssue: " + f.issue + "\nImpact: " + f.impact + "\n"; + vp += "Prior known bugs (must not confirm duplicates of these):\n" + known_summary + "\n"; + vp += "Set real=true only if: (1) bug still exists in code, (2) it is NOT the same as a prior known bug, (3) evidence is concrete.\n"; + vp += "Set real=false if fixed, duplicate of known list, wrong, or style-only.\n"; + vjobs.push(#{ + prompt: vp, + label: "verify-new:" + f.area, + capability_mode: "read-only", + output_schema: verdict_schema, + }); + } + let vres = parallel(vjobs); + let vi = 0; + for v in vres { + let f = new_to_v[vi]; + if v != () && v.success && v.output.real == true + && v.output.evidence != () && v.output.evidence != "" { + new_confirmed.push(#{ + area: f.area, + severity: f.severity, + file: f.file, + issue: f.issue, + impact: f.impact, + evidence: v.output.evidence, + reason: v.output.reason, + }); + } + vi += 1; + } +} + +log("new confirmed bugs: " + new_confirmed.len().to_string()); + +// --- Phase 3: synthesize --- +phase("Synthesize"); + +let fixed_json = json_encode(fixed_list); +let open_json = json_encode(open_list); +let unclear_json = json_encode(unclear_list); +let new_json = json_encode(new_confirmed); + +let yp = ""; +yp += "Write a delta bug review report in GitHub-flavored markdown for " + root + ".\n\n"; +yp += "FIXED prior bugs (JSON):\n" + fixed_json + "\n\n"; +yp += "STILL OPEN prior bugs (JSON):\n" + open_json + "\n\n"; +yp += "UNCLEAR prior rechecks (JSON):\n" + unclear_json + "\n\n"; +yp += "NEW confirmed bugs (JSON):\n" + new_json + "\n\n"; +yp += "Sections required:\n"; +yp += "1. Summary (counts: fixed / still open / unclear / new)\n"; +yp += "2. Fixed (table id|severity|file|note)\n"; +yp += "3. Still open (table id|severity|file|issue)\n"; +yp += "4. New bugs (table severity|file|issue|impact)\n"; +yp += "5. Unclear (if any)\n"; +yp += "6. Recommended next fix order\n"; +yp += "Return JSON: summary, fixed_count, still_open_count, new_count, markdown.\n"; +yp += "fixed_count=" + fixed_list.len().to_string() + + " still_open_count=" + open_list.len().to_string() + + " new_count=" + new_confirmed.len().to_string() + "\n"; + +let synth = agent(yp, #{ + label: "synthesize-delta", + capability_mode: "read-only", + output_schema: report_schema, +}); + +let md = "# Miner+Pool Delta Review\n\n"; +let summary = "Delta review complete."; +let fc = fixed_list.len(); +let oc = open_list.len(); +let nc = new_confirmed.len(); + +if synth != () && synth.success && synth.output.markdown != () { + md = synth.output.markdown; + if synth.output.summary != () { + summary = synth.output.summary; + } + if synth.output.fixed_count != () { + fc = synth.output.fixed_count; + } + if synth.output.still_open_count != () { + oc = synth.output.still_open_count; + } + if synth.output.new_count != () { + nc = synth.output.new_count; + } +} else { + md += "## Summary\n\nFixed " + fc.to_string() + ", still open " + oc.to_string() + + ", unclear " + unclear_list.len().to_string() + + ", new " + nc.to_string() + ".\n\n"; + md += "## Fixed\n\n"; + for x in fixed_list { + md += "- **" + x.id + "** (" + x.severity + ") `" + x.file + "` — " + x.reason + "\n"; + } + md += "\n## Still open\n\n"; + for x in open_list { + md += "- **" + x.id + "** (" + x.severity + ") `" + x.file + "` — " + x.issue + "\n"; + } + md += "\n## New\n\n"; + for x in new_confirmed { + md += "- **" + x.severity + "** `" + x.file + "` — " + x.issue + "\n"; + } + md += "\n## Unclear\n\n"; + for x in unclear_list { + md += "- **" + x.id + "** — " + x.reason + "\n"; + } + summary = "Fixed " + fc.to_string() + " / open " + oc.to_string() + " / new " + nc.to_string(); +} + +let path = write_scratch_file("miner-pool-delta-review.md", md); +log("delta report: " + path); + +complete(#{ + summary: summary, + fixed_count: fc, + still_open_count: oc, + unclear_count: unclear_list.len(), + new_count: nc, + path: path, + fixed: fixed_list, + still_open: open_list, + unclear: unclear_list, + new_bugs: new_confirmed, +}); diff --git a/.grok/workflows/review-miner-pool.rhai b/.grok/workflows/review-miner-pool.rhai new file mode 100644 index 0000000..8e06e85 --- /dev/null +++ b/.grok/workflows/review-miner-pool.rhai @@ -0,0 +1,301 @@ +// Multi-agent bug review of the Hacash miner + pool stack. +// Parallel dimension reviewers → adversarial verify → synthesis report. + +let meta = #{ + name: "review-miner-pool", + description: "Parallel bug review of miner (poworker/diaworker/OpenCL/CUDA) and pool (hac-pool/stratum), with adversarial verification", + when_to_use: "After miner or pool changes; pre-release audit of mining correctness and consensus", + phases: [ + #{ title: "Review", detail: "one reviewer per area of the miner/pool stack" }, + #{ title: "Verify", detail: "adversarial check of each claimed finding" }, + #{ title: "Synthesize", detail: "single ranked bug report" }, + ], +}; + +let findings_schema = #{ + "type": "object", + "required": ["findings"], + "properties": #{ + "findings": #{ + "type": "array", + "maxItems": 10, + "items": #{ + "type": "object", + "required": ["severity", "file", "issue", "impact"], + "properties": #{ + "severity": #{ "type": "string" }, + "file": #{ "type": "string" }, + "issue": #{ "type": "string" }, + "impact": #{ "type": "string" }, + }, + }, + }, + }, +}; + +let verdict_schema = #{ + "type": "object", + "required": ["real", "reason", "evidence"], + "properties": #{ + "real": #{ "type": "boolean" }, + "reason": #{ "type": "string" }, + "evidence": #{ "type": "string" }, + }, +}; + +let report_schema = #{ + "type": "object", + "required": ["summary", "confirmed_count", "open_count", "markdown"], + "properties": #{ + "summary": #{ "type": "string" }, + "confirmed_count": #{ "type": "integer" }, + "open_count": #{ "type": "integer" }, + "markdown": #{ "type": "string" }, + }, +}; + +// --- args --- +let root = "C:/Users/KQHEX/Documents/hacash-fullnodedev"; +if args != () && args.root != () { + root = args.root; +} +let max_findings = 8; +if args != () && args.max_findings != () { + max_findings = args.max_findings; +} + +log("review-miner-pool root=" + root); + +// --- Phase 1: parallel area reviews (read-only) --- +phase("Review"); + +let areas = [ + #{ + id: "block-miner", + paths: "app/src/block_mining_runtime.rs, app/src/mining_batch.rs, app/src/poworker.rs, app/src/hash_util.rs, app/src/mining_runtime.rs", + focus: "block PoW: job switch, nonce advance, winner submit (all heights), equal-inclusive target, GPU init fail-closed, hashrate accounting, race conditions", + }, + #{ + id: "gpu-backends", + paths: "app/src/opencl_gpu/, app/src/mining_batch.rs, app/src/cuda_pow.rs, app/src/opencl_dia.rs, x16rs/opencl/, x16rs-cuda/", + focus: "OpenCL/CUDA integrity verify, buffer OOB, bounded recovery vs full CPU fallback, custom_nonce gating for diamonds, kernel/host mismatch", + }, + #{ + id: "diamond-consensus", + paths: "x16rs/src/diamond.rs, app/src/diaworker.rs, mint/src/action/diamond_mint.rs, mint/src/check/block_build.rs, mint/src/check/block_accept.rs, mint/src/api/", + focus: "mainnet DMD_L=10 DMD_M=16, height%5 diamond rule, name extraction via check_diamond_hash_result (not hardcoded slices), submit retries, testnet hacks left behind", + }, + #{ + id: "pool", + paths: "miner-pool/src/ (main, rpc_proxy, stratum, upstream, job, config), pool-spike/src/ if present", + focus: "submit success/failure semantics (non-JSON must not be ret:0), stratum ret parsing, auth token, share attribution, stale job handling", + }, +]; + +let review_jobs = []; +for a in areas { + let p = ""; + p += "You are a senior mining-protocol code reviewer doing a READ-ONLY bug hunt.\n"; + p += "Repository root: " + root + "\n"; + p += "Area id: " + a.id + "\n"; + p += "Focus: " + a.focus + "\n"; + p += "Primary paths (relative to root): " + a.paths + "\n\n"; + p += "REQUIRED process:\n"; + p += "1. Use read_file and grep on the real files under the root. Do NOT invent findings from memory.\n"; + p += "2. Prefer REAL correctness/security bugs over style nits.\n"; + p += "3. For each finding set severity to one of: critical, high, medium, low.\n"; + p += "4. file must be a concrete path like app/src/foo.rs:LINE or path without line if range.\n"; + p += "5. At most " + max_findings.to_string() + " findings. Empty findings is valid ONLY after you inspected the paths.\n"; + p += "6. If code already has an explicit fix comment for a historical bug, do not re-report it as open unless the bug still exists.\n"; + p += "7. Return JSON matching the schema: findings array of {severity, file, issue, impact}.\n"; + review_jobs.push(#{ + prompt: p, + label: "review:" + a.id, + capability_mode: "read-only", + output_schema: findings_schema, + }); +} + +let review_results = parallel(review_jobs); + +let all_findings = []; +let area_i = 0; +for r in review_results { + let area_id = areas[area_i].id; + if r == () || !r.success { + log("review panel failed or empty for " + area_id); + } else if r.output.findings != () { + for f in r.output.findings { + // Tag area for later synthesis + let item = #{ + area: area_id, + severity: f.severity, + file: f.file, + issue: f.issue, + impact: f.impact, + }; + all_findings.push(item); + } + } + area_i += 1; +} + +log("raw findings collected: " + all_findings.len().to_string()); + +if all_findings.len() == 0 { + let empty_md = "# Miner + Pool Review\n\nNo findings after parallel area reviews (block-miner, gpu-backends, diamond-consensus, pool).\n"; + let path = write_scratch_file("miner-pool-review.md", empty_md); + complete(#{ + summary: "No findings from area reviewers.", + confirmed_count: 0, + open_count: 0, + path: path, + report: empty_md, + }); +} + +// Cap verification fan-out to keep budget headroom for synthesis +let MAX_VERIFY = 16; +let to_verify = []; +let dropped = 0; +let fi = 0; +for f in all_findings { + if to_verify.len() < MAX_VERIFY { + to_verify.push(f); + } else { + dropped += 1; + } + fi += 1; +} +if dropped > 0 { + log("capped verify list: dropped " + dropped.to_string() + " extra raw findings"); +} + +// --- Phase 2: adversarial verify --- +phase("Verify"); + +let vjobs = []; +for f in to_verify { + let vp = ""; + vp += "Adversarially verify this alleged bug against the REAL code under " + root + ".\n"; + vp += "Claimed area: " + f.area + "\n"; + vp += "File: " + f.file + "\n"; + vp += "Issue: " + f.issue + "\n"; + vp += "Impact claimed: " + f.impact + "\n"; + vp += "Severity claimed: " + f.severity + "\n\n"; + vp += "You MUST open the file(s) with read_file/grep and check whether the bug still exists.\n"; + vp += "Set real=true ONLY if you independently confirm the bug is still present with concrete evidence (quote or describe the exact logic).\n"; + vp += "Set real=false if the code already fixes it, the claim is wrong, speculative, or style-only.\n"; + vp += "evidence must cite path and what you saw. reason is a short verdict sentence.\n"; + vjobs.push(#{ + prompt: vp, + label: "verify:" + f.area, + capability_mode: "read-only", + output_schema: verdict_schema, + }); +} + +let verdicts = parallel(vjobs); + +let confirmed = []; +let rejected = []; +let vi = 0; +for v in verdicts { + let f = to_verify[vi]; + if v != () && v.success && v.output.real == true + && v.output.evidence != () && v.output.evidence != "" { + confirmed.push(#{ + area: f.area, + severity: f.severity, + file: f.file, + issue: f.issue, + impact: f.impact, + evidence: v.output.evidence, + reason: v.output.reason, + }); + } else { + let why = "unverified or not real"; + if v != () && v.success && v.output.reason != () { + why = v.output.reason; + } + rejected.push(#{ + area: f.area, + file: f.file, + issue: f.issue, + why: why, + }); + } + vi += 1; +} + +log("confirmed " + confirmed.len().to_string() + " / verified " + to_verify.len().to_string()); + +// --- Phase 3: synthesize markdown report --- +phase("Synthesize"); + +let conf_json = json_encode(confirmed); +let rej_json = json_encode(rejected); + +let sp = ""; +sp += "Synthesize a final miner+pool bug review report in GitHub-flavored markdown.\n"; +sp += "Repository: " + root + "\n\n"; +sp += "CONFIRMED findings (JSON, already adversarially verified — treat as open bugs):\n"; +sp += conf_json + "\n\n"; +sp += "REJECTED / unverified claims (JSON — do not list as open bugs; may mention briefly as closed):\n"; +sp += rej_json + "\n\n"; +sp += "Write markdown with sections:\n"; +sp += "1. Summary (2-4 sentences)\n"; +sp += "2. Confirmed open bugs table: severity | file | issue | impact\n"; +sp += "3. Notes on already-fixed / rejected claims (short)\n"; +sp += "4. Recommended fix order\n"; +sp += "Return JSON: summary, confirmed_count, open_count (same as confirmed), markdown (full report body).\n"; +sp += "confirmed_count and open_count must equal " + confirmed.len().to_string() + ".\n"; + +let synth = agent(sp, #{ + label: "synthesize-report", + capability_mode: "read-only", + output_schema: report_schema, +}); + +let md = "# Miner + Pool Review\n\n"; +let summary = "Review completed."; +let conf_n = confirmed.len(); +let open_n = confirmed.len(); + +if synth != () && synth.success && synth.output.markdown != () { + md = synth.output.markdown; + if synth.output.summary != () { + summary = synth.output.summary; + } + if synth.output.confirmed_count != () { + conf_n = synth.output.confirmed_count; + } + if synth.output.open_count != () { + open_n = synth.output.open_count; + } +} else { + // Fallback local markdown if synthesis agent fails + md += "## Summary\n\n"; + md += "Confirmed " + confirmed.len().to_string() + " findings after adversarial verification.\n\n"; + md += "## Confirmed open bugs\n\n"; + for c in confirmed { + md += "- **" + c.severity + "** `" + c.file + "` — " + c.issue + " (impact: " + c.impact + ")\n"; + } + md += "\n## Rejected\n\n"; + for r in rejected { + md += "- `" + r.file + "` — " + r.issue + " — " + r.why + "\n"; + } + summary = "Confirmed " + confirmed.len().to_string() + " open bugs (fallback report)."; +} + +let path = write_scratch_file("miner-pool-review.md", md); +log("report written: " + path); + +complete(#{ + summary: summary, + confirmed_count: conf_n, + open_count: open_n, + path: path, + confirmed: confirmed, + rejected_count: rejected.len(), +}); diff --git a/app/src/block_mining_runtime.rs b/app/src/block_mining_runtime.rs index 5359595..95cb470 100644 --- a/app/src/block_mining_runtime.rs +++ b/app/src/block_mining_runtime.rs @@ -13,12 +13,12 @@ use crate::efficiency::*; use crate::hash_util::{hash_left_zero_pad3, hash_more_power}; // The panic firewall is shared with the diamond (HACD) worker, which has exactly // the same "one result thread owns every submission" shape. -use crate::mining_guard::guard_mining_iteration; #[cfg(feature = "cuda")] use crate::mining_batch::CudaBlockBackend; #[cfg(feature = "ocl")] use crate::mining_batch::OpenclBlockBackend; use crate::mining_batch::{BatchCtx, BlockMinerBackend, CpuBlockBackend}; +use crate::mining_guard::guard_mining_iteration; use basis::difficulty::*; use basis::interface::*; @@ -1434,15 +1434,8 @@ fn backend_nonce_space(_cnf: &PoWorkConf, backend: &MinerBackend) -> u32 { // Match run_batch: the planned window must reflect the same effective // work-groups (OOM/error backoff) and thermal cap the batch will use, // otherwise the nonce accounting overstates what the GPU covered. - let thermal = _cnf - .runtime - .thermal_workgroups_cap() - .unwrap_or(u32::MAX); - let wg = res - .effective_wg() - .min(_cnf.workgroups) - .min(thermal) - .max(1); + let thermal = _cnf.runtime.thermal_workgroups_cap().unwrap_or(u32::MAX); + let wg = res.effective_wg().min(_cnf.workgroups).min(thermal).max(1); wg.saturating_mul(x16rs_cuda::DEFAULT_LOCAL_SIZE) .saturating_mul(res.unit_size) .max(1) diff --git a/app/src/diaworker.rs b/app/src/diaworker.rs index 97e22cc..5e803c2 100644 --- a/app/src/diaworker.rs +++ b/app/src/diaworker.rs @@ -478,7 +478,12 @@ pub fn diaworker_with_stop(stop_flag: Option>) { return; } guard_mining_iteration("diamond mining worker", || { - run_diamond_worker_thread(&cnf2, thrid, rstx.clone(), &stop_flag_worker); + run_diamond_worker_thread( + &cnf2, + thrid, + rstx.clone(), + &stop_flag_worker, + ); }); delay_continue_ms!(9); } diff --git a/app/src/mining_batch.rs b/app/src/mining_batch.rs index e0b62f0..a42a1c8 100644 --- a/app/src/mining_batch.rs +++ b/app/src/mining_batch.rs @@ -902,8 +902,15 @@ mod tests { let mut wrong_hash = good; wrong_hash.0 = 12; assert!( - verify_gpu_shares(height, &block_intro, 0, 256, &easiest_target, &[good, wrong_hash]) - .is_err() + verify_gpu_shares( + height, + &block_intro, + 0, + 256, + &easiest_target, + &[good, wrong_hash] + ) + .is_err() ); // A nonce outside the batch window. assert!( @@ -920,9 +927,7 @@ mod tests { // An honest hash the card listed even though it is ABOVE the target it // was told to filter on: the compare is broken, so nothing is forwarded. let strict_target = [0u8; 32]; - assert!( - verify_gpu_shares(height, &block_intro, 0, 256, &strict_target, &[good]).is_err() - ); + assert!(verify_gpu_shares(height, &block_intro, 0, 256, &strict_target, &[good]).is_err()); } #[test] @@ -1005,7 +1010,10 @@ mod tests { thermal_wg_cap: None, share_target: None, }; - assert_eq!(x16rs_cuda::share_capacity_for(solo.share_target.as_ref()), 0); + assert_eq!( + x16rs_cuda::share_capacity_for(solo.share_target.as_ref()), + 0 + ); let pooled_target = [0x0fu8; 32]; assert_eq!( diff --git a/app/src/opencl_gpu/block.rs b/app/src/opencl_gpu/block.rs index 73bc42b..839b24b 100644 --- a/app/src/opencl_gpu/block.rs +++ b/app/src/opencl_gpu/block.rs @@ -260,17 +260,28 @@ mod gpu_tests { pool.share_hits, BATCH_NONCES as u64, "the counter must see every hit, not only the stored ones" ); - assert_eq!(pool.shares.len(), SHARE_LIST_CAPACITY.min(BATCH_NONCES as usize)); + assert_eq!( + pool.shares.len(), + SHARE_LIST_CAPACITY.min(BATCH_NONCES as usize) + ); let mut seen: Vec = pool.shares.iter().map(|(nonce, _)| *nonce).collect(); seen.sort_unstable(); seen.dedup(); - assert_eq!(seen.len(), pool.shares.len(), "no nonce may be listed twice"); + assert_eq!( + seen.len(), + pool.shares.len(), + "no nonce may be listed twice" + ); for (nonce, hash) in &pool.shares { assert!( (NONCE_START..NONCE_START + BATCH_NONCES).contains(nonce), "share nonce {nonce} is outside the batch window" ); - assert_eq!(*hash, cpu_hash(&intro, *nonce), "share hash must match the CPU"); + assert_eq!( + *hash, + cpu_hash(&intro, *nonce), + "share hash must match the CPU" + ); } // 3. POOL, a target only three nonces beat: exactly those three, and @@ -297,6 +308,9 @@ mod gpu_tests { assert_eq!(strict.share_hits, 3); let mut got: Vec = strict.shares.iter().map(|(nonce, _)| *nonce).collect(); got.sort_unstable(); - assert_eq!(got, expected, "the kernel must list exactly the payable nonces"); + assert_eq!( + got, expected, + "the kernel must list exactly the payable nonces" + ); } } diff --git a/app/src/poworker.rs b/app/src/poworker.rs index f3961ef..8d7afa6 100644 --- a/app/src/poworker.rs +++ b/app/src/poworker.rs @@ -1255,7 +1255,9 @@ mod tests { assert_eq!(upstream_stale_reason(&serde_json::json!({})), None); assert_eq!(upstream_stale_reason(&serde_json::json!({"err": ""})), None); assert_eq!( - upstream_stale_reason(&serde_json::json!({"err": "no job yet; wait for upstream fullnode"})), + upstream_stale_reason( + &serde_json::json!({"err": "no job yet; wait for upstream fullnode"}) + ), None ); assert_eq!(upstream_stale_reason(&serde_json::json!({"err": 7})), None); diff --git a/docs/MINER-POOL-DELTA-REVIEW.md b/docs/MINER-POOL-DELTA-REVIEW.md new file mode 100644 index 0000000..b0fd700 --- /dev/null +++ b/docs/MINER-POOL-DELTA-REVIEW.md @@ -0,0 +1,138 @@ +# Delta Bug Review Report + +**Repo:** `C:/Users/KQHEX/Documents/hacash-fullnodedev` +**Scope:** Block miner, diamond GPU, hacd-pool / miner-pool +**Date:** 2026-07-24 + +--- + +## 1. Summary + +| Category | Count | +|----------|------:| +| **Fixed** | 5 | +| **Still open** | 14 | +| **Unclear** | 0 | +| **New** | 11 | + +**Net residual risk:** 25 open issues (14 prior + 11 new). Critical consensus/testnet hacks (**C1**, **C2**) and the major block-mining correctness suite (**C3**, **C4**, **P11**) are fixed. Remaining work clusters on **template/reorg lifecycle**, **OpenCL diamond kernel correctness**, and **submit durability**. + +--- + +## 2. Fixed + +| ID | Severity | File | Note | +|----|----------|------|------| +| **P11** | medium | `x16rs-cuda` batch launch | Batch path calls `clamped_block_size` at `cuda_init_miner` and refuses devices that cannot launch 256-thread blocks. | +| **C1** | critical | `x16rs/src/diamond.rs` | Mainnet `DMD_L=10` / `DMD_M=16` hardcoded; no testnet 4/10 override. | +| **C2** | critical | `mint/src/action/diamond_mint.rs` | Height `% 5` diamond mint rule enforced; no `if false` bypass. | +| **C3** | high | `app/src` mining winners / equal target / GPU fatal | Multi-winner submit, equal-inclusive target, logged GPU fail, CUDA verify, capped CPU recovery — implemented and unit-tested. | +| **C4** | high | `miner-pool` `rpc_proxy` / `stratum` | Non-JSON upstream → `ret:1`; stratum accepts only parseable JSON with `ret==0`. | + +--- + +## 3. Still open + +| ID | Severity | File | Issue | +|----|----------|------|-------| +| **P1** | high | `app/src/block_mining_runtime.rs` | Winners coalesced without epoch/template id; submit/drain skip live-template revalidation; same-height reorg stale result can out-rank live work. | +| **P2** | high | `app/src/poworker.rs` | Template install only on height advance or same-height `intro_changed`; reorg that **lowers** pending height is ignored. | +| **P3** | high | `x16rs/opencl/x16rs_diamond.cl` | Diamond OpenCL reduction seeds `best_hash=0` with uninitialized `best_name`; scans `i=1..` (unlike block kernel seed-from-index). | +| **P4** | high | `x16rs/opencl/x16rs_diamond.cl` | Diamond barriers use only `CLK_LOCAL_MEM_FENCE` while hashes live in global memory; block kernel uses `LOCAL\|GLOBAL`. | +| **P5** | high | `app/src/diaworker.rs` | `push_diamond_mining_success` treats any HTTP Ok as final; non-JSON / missing `tx_hash` fails permanently; no requeue after drain. | +| **P6** | medium | `app/src/poworker.rs` | `LAST_PENDING_INTRO` written **before** `set_pending_block_stuff` succeeds; failed same-height install never retried. | +| **P7** | medium | `app/src/block_mining_runtime.rs` | Job-switch height/epoch checked only **after** full batch + `send()`; drain does not filter by live epoch. | +| **P8** | medium | `app/src/block_mining_runtime.rs` | Worker rollover is advance-only (`check_hei > mining_hei`) + epoch; height decrease without epoch may not stop workers. | +| **P9** | medium | `x16rs/opencl/sha3_256.cl` | Host builds 61-byte diamond stuff for number ≤ 20000; GPU always pads as 93-byte. | +| **P10** | medium | `app/src/opencl_dia.rs` | Diamond GPU success trusts medium hash; no `x16rs_hash` recompute + byte-compare (unlike block `verify_gpu_best_result`). | +| **P12** | medium | `app/src/diaworker.rs` | After `MAX_SUBMIT_ATTEMPTS` or soft parse failure, drained `DiamondMint` is logged and dropped — no durable save/requeue. | +| **P13** | medium | `x16rs/opencl/x16rs_diamond.cl` | Per-work-item reduction never seeds `best_name` from unit index 0 before `i=1..` comparisons. | +| **P14** | low | `app/src/block_mining_runtime.rs` | Drain aggregates planned `res.nonce_space` even when recovery only mined a partial window — telemetry overstated. | +| **P15** | low | `app/src/opencl_gpu/block.rs` | OpenCL accepts stuff length ≤ 512 without enforcing the 89-byte block intro CUDA requires. | + +--- + +## 4. New bugs + +| Severity | File | Issue | Impact | +|----------|------|-------|--------| +| **high** | `app/src/poworker.rs` | Notice long-poll only breaks when notice height ≥ `pending_height`; fullnode notice reports chain tip (typically pending−1), so timeouts and same-height tip reorgs never re-fetch pending. `intro_changed` path is effectively dead on fullnode. | After tip reorg, workers hash orphaned parent for up to a full block interval; PoW cannot land on main chain. | +| **high** | `app/src/poworker.rs` | `leave_upstream_stale()` runs as soon as `block_intro` is present, before `set_pending_block_stuff` succeeds and even when install is skipped by the height/intro gate. | After upstream-stale outage, workers can resume grinding a previous dead template if recovery install fails/skips. | +| **high** | `app/src/diaworker.rs` | `pull_and_push_diamond` only advances when `next_num > mining_num`; never rolls number backward or refreshes `prev_hash` / `born.hash` after diamond reorg. | Workers mine invalid `(number, prev_hash)` until chain mints past stale number; successes fail node validation. | +| **high** | `miner-pool/src/job.rs` | `JobHub::update` always sets `job_id=h{height}` only; stratum dedup keys solely on `job_id`, so same-height template changes never emit `mining.notify`. | Stratum miners stay on orphaned/obsolete template while HTTP path can serve new job; shares miss live tip. | +| **medium** | `app/src/mining_batch.rs` | OpenCL batch/integrity failures only trigger `on_batch_error` + bounded CPU recovery; no consecutive-failure budget or session GPU-disable (CUDA has both at 20). | Permanently failing OpenCL device never fail-closes; miner stuck on tiny CPU recovery, masking hardware death. | +| **medium** | `app/src/block_mining_runtime.rs` | `set_pending_block_stuff` does not require JSON height == `block_intro.height()`; mining uses JSON height for x16rs repeat while consensus uses intro-embedded height. | Mismatched upstream height → wrong-repeat mining and/or submit rejection. | +| **medium** | `app/src/block_mining_runtime.rs` | Result drain thread returns immediately on `stop_flag` without draining the result channel; only submit queue is wound down. | Clean shutdown/restart can discard target-meeting winners still in the result channel (lost shares/blocks). | +| **medium** | `x16rs/opencl/x16rs_diamond.cl` | Kernel reduction ranks solely by `diamond_more_power` (more leading zeros); consensus requires **exactly** `DMD_L` zeros — overshoots are invalid. | Valid 10-zero diamond lost when an invalid 11+ overshoot is in the same unit/WG. | +| **medium** | `app/src/opencl_dia.rs` | Diamond GPU post-process never bounds-checks nonces against `[nonce_start, nonce_start+nonce_space)` (block path does). | Corrupted/out-of-window nonces can pass partial checks → false success or wasted submits. | +| **medium** | `app/src/opencl_dia.rs` | On `check_diamer_success`, function returns before `needs_queue_finish` / `queue.finish()`; block OpenCL always finishes on RDNA4/duplicate ICD. | After diamond success, AMD queue may not drain → stale/out-of-order batches. | +| **low** | `x16rs/opencl/util.cl` | `block_t` is 88 bytes; GPU SHA3 hardcodes pad lane forcing `intro[88]==0`; CPU/consensus use full 89-byte intro (`witness_stage`). | Latent while `witness_stage` is zero; non-zero 89th byte makes GPU SHA3 diverge from consensus. | + +--- + +## 5. Unclear + +None. Prior rechecks produced no unclear outcomes. + +--- + +## 6. Recommended next fix order + +Priority groups by **payout / correctness risk**, then **cluster affinity** (fix one area together). + +### Tier 0 — Template / reorg correctness (blocks payout path) + +1. **New: notice long-poll / tip vs pending** (`poworker.rs` + fullnode `miner_notice`) — unlocks same-height reorg path; without this, **P2**/`intro_changed` barely matter on fullnode. +2. **New: `leave_upstream_stale` before install** (`poworker.rs`) — stop grinding dead templates after outage recovery. +3. **P2** — install on height decrease (reorg to lower pending). +4. **P6** — write `LAST_PENDING_INTRO` only after successful `set_pending_block_stuff`. +5. **P1 + P7 + P8** together — epoch/template id on `BlockMiningResult`; filter drain; stop workers on height decrease; revalidate before submit. +6. **New: height vs intro.height mismatch** (`set_pending_block_stuff`) — cheap invariant, prevents wrong-repeat mining. + +### Tier 1 — Pool / diamond job identity + +7. **New: stratum `job_id=h{height}` only** (`miner-pool/job.rs` + stratum notify) — include intro/content hash so same-height reorgs notify. +8. **New: diamond number/prev_hash never roll back** (`diaworker.rs` `pull_and_push_diamond`) — reorg-safe diamond job refresh. + +### Tier 2 — OpenCL diamond correctness (find loss) + +9. **P3 + P13** — seed reduction from unit index 0 / current work-item (match block kernel pattern). +10. **P4** — `CLK_LOCAL_MEM_FENCE | CLK_GLOBAL_MEM_FENCE` on diamond barriers. +11. **New: diamond_more_power overshoot** — prefer valid exact-`DMD_L` names over stronger invalid overshoots (or validate before reduce). +12. **P9** — 61-byte vs 93-byte diamond SHA3 padding for number ≤ 20000. +13. **P10 + New: nonce window + finish-on-success** (`opencl_dia.rs`) — recompute medium hash, bounds-check nonces, always `queue.finish` when required. + +### Tier 3 — Submit durability & fail-close + +14. **P5 + P12** — durable diamond submit requeue / save on network and soft parse failure. +15. **New: result channel abandon-on-stop** — final drain before result thread exit. +16. **New: OpenCL consecutive-failure GPU disable** — parity with CUDA session latch. + +### Tier 4 — Low / latent + +17. **P14** — report actual mined nonce space, not planned. +18. **P15** — enforce 89-byte block intro on OpenCL upload. +19. **New: block_t 88-byte / intro[88]** — load 89th byte into SHA3 when `witness_stage` can be non-zero. + +--- + +### Cluster map (for parallel workstreams) + +| Stream | Items | +|--------|--------| +| **A. Block job lifecycle** | Notice long-poll, leave_upstream_stale, P2, P6, P1, P7, P8, height==intro.height | +| **B. Pool / diamond jobs** | Stratum job_id, diamond number rollback | +| **C. Diamond GPU kernel** | P3, P4, P13, overshoot, P9 | +| **D. Diamond host post** | P10, nonce window, finish-on-success, P5, P12 | +| **E. Ops / telemetry** | OpenCL GPU disable, shutdown drain, P14, P15, block_t 89th byte | + +--- + +### Severity rollup (open only) + +| Severity | Still open | New | Total | +|----------|----------:|----:|------:| +| high | 5 | 4 | **9** | +| medium | 7 | 6 | **13** | +| low | 2 | 1 | **3** | +| **Total** | **14** | **11** | **25** | diff --git a/docs/MINER-POOL-WORKFLOW-REVIEW.md b/docs/MINER-POOL-WORKFLOW-REVIEW.md new file mode 100644 index 0000000..49dc330 --- /dev/null +++ b/docs/MINER-POOL-WORKFLOW-REVIEW.md @@ -0,0 +1,122 @@ +# Miner + Pool Bug Review Report + +**Repository:** `C:/Users/KQHEX/Documents/hacash-fullnodedev` +**Scope:** block miner runtime / poworker, GPU OpenCL+CUDA backends, diamond worker + consensus submit +**Status:** 15 confirmed open bugs (adversarially verified) + +--- + +## 1. Summary + +Review of the miner and related GPU/diamond paths confirmed **15 open bugs**. The most severe cluster is in **block mining job lifecycle and winner selection**: same-height reorgs can allow stale in-flight results to out-rank and replace a currently valid solution; depth-reducing reorgs never install a lower pending height; and workers only stop on height *advance* or epoch bump. A second high-severity cluster is in **diamond OpenCL** (uninitialized/wrong reduction seed + LOCAL-only fences on global-backed hashes) and **diamond submit durability** (HTTP-200 noise treated as final; no requeue/save after drain). Medium issues include failed same-height install not retried, late job-switch checks feeding the result channel, SHA3 length mismatch for low diamond numbers, missing GPU medium-hash re-verify, and CUDA batch launch without `clamped_block_size`. Lower-severity items cover nonce-span telemetry after recovery and OpenCL block stuff length acceptance. + +--- + +## 2. Confirmed open bugs + +| Severity | File | Issue | Impact | +|---|---|---|---| +| **high** | `app/src/block_mining_runtime.rs:709` | Winners coalesced only by height (keep strongest `result_hash`). After same-height reorg/epoch change, an in-flight stale-template result can still meet its own `target_hash` and out-rank a weaker but currently valid solution; only the stale entry is submitted. No epoch/template id on `BlockMiningResult`; `push_block_mining_success` submits the coalesced winner without live-template revalidation. | A real block acceptable for the current template can be silently discarded; miner loses payout while the node rejects the orphaned solution. | +| **high** | `app/src/poworker.rs:264` | Template install only when `pending_height > curr_hei` or `(same height && intro_changed)`. A reorg that **lowers** pending height is ignored, so `MINING_BLOCK_HEIGHT`/epoch never update and workers never job-switch. | After a depth-reducing reorg the miner can grind a non-existent height indefinitely, producing only rejected work and missing the new tip. | +| **high** | `x16rs/opencl/x16rs_diamond.cl:122` | Per-thread diamond reduction initializes `best_hash = 0` and leaves `best_name` uninitialized, then scans only `i=1..unit_size-1`. Block/CUDA paths already seed from `index` (see `x16rs_main.cl:84`, `block_miner.cu:71-76`). | For `local_id>0`, `best_hash` can point into another thread’s slots; work-group reduction propagates wrong diamond winners—missed finds and inconsistent nonce/hash pairs. | +| **high** | `x16rs/opencl/x16rs_diamond.cl:96` | Diamond kernel stores hashes in global memory (`local_hashes = global_hashes + …`) but post-SHA3 / reduction barriers use only `CLK_LOCAL_MEM_FENCE`. Block kernel uses `CLK_LOCAL_MEM_FENCE \| CLK_GLOBAL_MEM_FENCE` at matching points. | No guaranteed cross-item visibility of global hash writes before reduction; intermittent wrong diamond reductions (lost finds / corrupted best nonce-hash) on some devices. | +| **high** | `app/src/diaworker.rs:890` | `push_diamond_mining_success` treats any HTTP transport `Ok` body as final: breaks immediately, then fails permanently on non-JSON / missing `tx_hash` without retry. Block path (`poworker`) retries unrecognized HTTP-200 bodies; diamond only retries `reqwest` `Err`. | A rare mined diamond can be discarded after proxy/HTML 200, truncated body, or gateway noise within the timeout window. No durable requeue after drain. | +| **medium** | `app/src/poworker.rs:255` | `LAST_PENDING_INTRO` is overwritten **before** `set_pending_block_stuff` succeeds. On same-height install failure (bad target/coinbase/mkrl), the new intro is already remembered → `intro_changed` stays false and install is never retried while height is unchanged. | Miner can remain stuck on an orphaned same-height template until a later height advance, wasting hashrate and missing blocks after reorg/partial RPC payload. | +| **medium** | `app/src/block_mining_runtime.rs:617` | Job-switch (height/epoch) is checked only **after** a full batch and after `send()`. Stale `BlockMiningResult` values always enter the result channel; `deal_block_mining_results` does not filter by current epoch/target before `push_block_mining_success`. Stale submits block the single result thread (HTTP timeouts × attempts). | Feeds same-height winner coalescing failures; delays fresher queue items behind useless submit attempts. | +| **medium** | `app/src/block_mining_runtime.rs:643` | Worker rollover uses `check_hei > mining_hei` (advance-only) plus epoch; correctness for non-monotonic height depends entirely on epoch bumps from `set_pending`, which `pull_pending` may never call on height decrease. | Defense-in-depth gap: if height goes down without an epoch publish path, workers do not stop; compounds the reorg job-switch bug. | +| **medium** | `x16rs/opencl/sha3_256.cl:182` | Host builds 61-byte stuff when custom message is gated empty for diamond numbers ≤ 20000 (`opencl_dia.rs:31-54`), but `sha3_256_hash_diamond` always applies fixed 93-byte SHA3 padding. Consensus/CPU hash true length. | GPU SHA3 diverges from CPU/consensus for low diamond numbers; finds cannot verify—OpenCL diamond mining useless on that range (testnets / early numbers). | +| **medium** | `app/src/opencl_dia.rs:107` | Diamond GPU success path only runs `calculate_hash(stuff)` + `check_diamer_success`; never recomputes `x16rs_hash(repeat, ssshash)` and byte-compares to the GPU medium hash (unlike block `verify_gpu_best_result`). | GPU medium hash can pass independent name/difficulty checks without being the x16rs of the claimed nonce’s SHA3; false local successes under reduction bugs; consensus re-mines and rejects. | +| **medium** | `x16rs-cuda/src/lib.rs:472` | Batch mining launches `x16rs_cuda_main` with fixed `miner.local_size` (256) and never calls `clamped_block_size`. Single-hash path clamps to `maxThreadsPerBlock` to avoid `cudaErrorInvalidConfiguration`. Failures hit 100k-nonce CPU recovery. | If batch kernel `maxThreadsPerBlock` < 256, every CUDA batch fails; hashrate collapses despite a usable GPU. | +| **medium** | `app/src/diaworker.rs:905` | After `MAX_SUBMIT_ATTEMPTS` (5, ~7.5s backoff) network failure, or after a soft parse failure, the drained `DiamondMint` is dropped with only a log—no local save or later resubmit. Recovery curl is commented out. | Node restart, brief RPC outage, or auth blip during submit permanently loses the find even though PoW work completed. | +| **medium** | `x16rs/opencl/x16rs_diamond.cl:122` | Per-work-item reduction leaves `diamond_t best_name` uninitialized and starts comparison at `i=1`, so unit index 0 is never `diamond_hash`’d into `best_name` before comparisons. (Host re-checks candidates; `DiaWorkConf` currently forces `useopencl=false` for HACD.) | If HACD OpenCL is re-enabled, valid nonces can be discarded and hashrate/find rate understated. No false mint while host revalidates and OpenCL is disabled. | +| **low** | `app/src/block_mining_runtime.rs:691` | Drain aggregate adds planned `res.nonce_space` into `total_nonce_space` for the status line even when GPU/CPU recovery reports only a partial window; hashrate EWMA uses partial counts. | Misleading nonce-span / efficiency telemetry after OOM/integrity recovery; can skew operator decisions (not consensus). | +| **low** | `app/src/opencl_gpu/block.rs:20` | OpenCL block upload accepts any stuff length ≤ 512 (`write_stuff_to_gpu`) and never enforces the 89-byte block intro that CUDA requires (`STUFF_BYTES`). Kernel SHA3 assumes fixed 89-byte padded layout. | Short/long intro desyncs GPU vs CPU hashes; integrity verify fails every batch → 100k-nonce CPU recovery until template fixed. Wrong solutions not submitted, but GPU work wasted. | + +### Count by severity + +| Severity | Count | +|---|---| +| high | 5 | +| medium | 8 | +| low | 2 | +| **total open** | **15** | + +### Count by area + +| Area | Count | +|---|---| +| block-miner | 6 | +| gpu-backends | 6 | +| diamond-consensus | 3 | + +--- + +## 3. Notes on already-fixed / rejected claims + +**Rejected (do not treat as open bugs):** + +- **`MINING_BLOCK_HEIGHT` / `EPOCH` Relaxed atomics without Acquire on worker job-switch** (`app/src/block_mining_runtime.rs:316`) — Height/epoch act as Relaxed cancel flags only; template payload is correctly synced via `RwLock`. Missing Acquire/Release does **not** establish an extra real stale-batch bug beyond the confirmed job-switch and coalesce issues above. + +**Related “already fixed elsewhere” notes (still open on diamond path):** + +- Block OpenCL (`x16rs_main.cl`) and CUDA block miner already seed reduction from `index` and use LOCAL\|GLOBAL fences; diamond OpenCL still has the old reduction/fence pattern. +- Block mining has full `verify_gpu_best_result` recompute; diamond OpenCL success does not recompute/equality-check the medium hash. +- Block submit retries unrecognized HTTP-200 bodies; diamond submit does not. + +--- + +## 4. Recommended fix order + +1. **Block reorg job install + worker stop (high, root cause of grinding dead height)** + - In `poworker.rs`, install templates when `pending_height != curr_hei` (or explicitly handle `pending_height < curr_hei`), not only advance / same-height intro change. + - Always bump `MINING_BLOCK_EPOCH` on any template change including height decrease. + - Worker rollover: stop on **any** height change (`check_hei != mining_hei`) or epoch change, not advance-only. + +2. **Winner selection / submit revalidation (high, silent payout loss)** + - Tag `BlockMiningResult` with epoch (and/or template id / stuff hash). + - Coalesce or accept winners only for the **live** epoch/template; re-check result against live `MINING_BLOCK_STUFF` target before `push_block_mining_success`. + - Prefer filtering **before** `send()` (or drop in drain) so stale same-height results cannot suppress a live win. + +3. **Diamond OpenCL reduction + fences (high, correctness if/when GPU HACD is used)** + - Seed `best_hash = index`, hash slot 0 into `best_name` before the loop (mirror `x16rs_main.cl` / CUDA). + - Use `CLK_LOCAL_MEM_FENCE \| CLK_GLOBAL_MEM_FENCE` after global hash writes and during reduction. + - Keep host revalidation; re-enable only after kernel + SHA3 length fixes. + +4. **Diamond submit durability (high → medium)** + - Align with block path: retry unrecognized/non-JSON HTTP-200 bodies. + - On exhausted attempts or soft parse failure: **local save + requeue** the drained `DiamondMint` (do not only log). + - Do not treat every transport `Ok` as terminal success. + +5. **Same-height install atomicity (medium)** + - Update `LAST_PENDING_INTRO` **only after** successful `set_pending_block_stuff`, or roll back on `Err` so `intro_changed` can retry. + +6. **Stale result pipeline (medium)** + - Check height/epoch **before** building/sending `BlockMiningResult`. + - Drain path: drop winners whose epoch/target no longer match live template before blocking HTTP submit. + +7. **Diamond SHA3 length + host verify (medium)** + - Make `sha3_256_hash_diamond` length-aware (61 vs 93) consistent with consensus/CPU for numbers ≤ 20000. + - Recompute `x16rs_hash` and require equality to GPU medium hash (parity with block `verify_gpu_best_result`). + +8. **CUDA batch `clamped_block_size` (medium)** + - Launch batch kernel with clamped block size like the single-hash path to avoid permanent invalid-config → 100k CPU recovery collapse. + +9. **OpenCL block stuff length (low)** + - Enforce 89-byte intro (or reject) on OpenCL upload to match CUDA/`STUFF_BYTES` and fixed kernel pad layout. + +10. **Nonce-space telemetry (low)** + - Aggregate status `total_nonce_space` from actual recovered `gpu_nonce_space`/`cpu_nonce_space` when partial recovery is reported, not planned `res.nonce_space` alone. + +### Suggested patch grouping + +| PR | Scope | Severity | +|---|---|---| +| A | Reorg install gate + epoch bump + worker `!=` height stop | high | +| B | Result epoch tag + live-template winner filter + pre-send job check | high | +| C | Diamond CL reduction seed + GLOBAL fences + SHA3 length | high/medium | +| D | Diamond submit retry parity + durable requeue/save | high/medium | +| E | `LAST_PENDING_INTRO` only-after-success; CUDA clamp; OpenCL stuff len; telemetry | medium/low | + +--- + +*End of report — 15 confirmed open bugs; 1 rejected claim.* \ No newline at end of file diff --git a/hbit-pool/examples/fee_probe.rs b/hbit-pool/examples/fee_probe.rs new file mode 100644 index 0000000..e4136b7 --- /dev/null +++ b/hbit-pool/examples/fee_probe.rs @@ -0,0 +1,42 @@ +//! Ask a LIVE node what one of the pool's own blocks paid it in transaction +//! fees, using the exact function the settlement path uses. +//! +//! This exists to make the fee hold-back falsifiable. `block_fees` is the one +//! step of the money path that cannot be exercised without a node: it reads +//! `/query/block/intro?tx_hash_list=true` and then one `/query/transaction` per +//! packed transaction, and every one of those field names is a promise about a +//! node this crate does not build. Run it against a height the pool really won +//! and the answer is either the fee total in units of 0.1 HAC, or the refusal +//! the settlement turns into "settle nothing this cycle". +//! +//! usage: fee_probe + +use hbit_pool::{BlockFees, block_fees, http_client}; + +fn main() { + let a: Vec = std::env::args().collect(); + if a.len() < 4 { + eprintln!("usage: fee_probe "); + std::process::exit(2); + } + let node = a[1].trim_end_matches('/').to_string(); + let height: u64 = a[2].parse().expect("height"); + let hash = a[3].to_lowercase(); + + let client = http_client(); + match block_fees(&client, &node, height, &hash) { + BlockFees::Counted(u) => { + println!("Counted({u}) units of 0.1 HAC"); + } + BlockFees::NotOnChain => { + println!("NotOnChain (the chain does not hold our block at that height)"); + std::process::exit(1); + } + BlockFees::Unknown(why) => { + println!("Unknown: {why}"); + // The settlement turns this into "nothing is settled this cycle", + // so a probe that answered 0 here would be worse than useless. + std::process::exit(1); + } + } +} diff --git a/hbit-pool/examples/local_chain_watch.rs b/hbit-pool/examples/local_chain_watch.rs new file mode 100644 index 0000000..680a36e --- /dev/null +++ b/hbit-pool/examples/local_chain_watch.rs @@ -0,0 +1,162 @@ +//! Watch a local (non-mainnet) chain climb to a difficulty the HBIT pool will serve. +//! +//! This is the MEASURING instrument for the rig in +//! `scripts/hbit-local-chain-rig/`. It does not model anything: it polls the +//! node and reports the difficulty the chain actually stored, converted to +//! leading zero bits by the SAME function the pool's own admission check uses +//! (`pool_core::share_cost_bits` over `pool_core::network_target_hash`), so the +//! number printed here is the number `check_share_target` will see. +//! +//! Exit code is the result, because that is the only part a script may trust: +//! 0 the chain reached `--want` bits inside the deadline +//! 1 the deadline passed first (prints the best it managed) +//! 2 bad arguments / the node could not be read +//! +//! Usage: +//! cargo run --release -p hbit-pool --example local_chain_watch -- \ +//! [poll_secs] + +use hbit_pool::pool_core::{ + achieved_share_factor, network_target_hash, share_cost_bits, share_target_hash, +}; +use hbit_pool::{find_u64, get_json, http_client}; +use std::time::{Duration, Instant}; + +/// server.rs MIN_SHARE_FACTOR: how much easier a share may be than a block. +const MIN_SHARE_FACTOR: u32 = 18; +/// server.rs MIN_SHARE_COST_BITS: what the share itself must cost. +const MIN_SHARE_COST_BITS: u32 = 16; + +/// Reproduce the pool's own admission gate for a given network difficulty. +/// +/// server.rs derives exactly these two numbers from exactly these two calls and +/// hands them to `check_share_target`, so if this says yes the pool says yes. +/// `check_share_target` is private to that binary, which is why the inputs are +/// recomputed here rather than the decision being imported. +fn pool_would_serve(difficulty: u32, share_bits: u32) -> (bool, u32, u32) { + let network = network_target_hash(difficulty); + let share = share_target_hash(difficulty, share_bits); + let achieved = achieved_share_factor(&network, &share); + let cost = share_cost_bits(&share); + ( + achieved >= MIN_SHARE_FACTOR && cost >= MIN_SHARE_COST_BITS, + achieved, + cost, + ) +} + +fn main() { + let a: Vec = std::env::args().collect(); + let node = a + .get(1) + .cloned() + .unwrap_or_else(|| "http://127.0.0.1:8080".to_string()); + let node = node.trim_end_matches('/').to_string(); + let want: u32 = a.get(2).and_then(|s| s.parse().ok()).unwrap_or(34); + let deadline: u64 = a.get(3).and_then(|s| s.parse().ok()).unwrap_or(3600); + let poll: u64 = a.get(4).and_then(|s| s.parse().ok()).unwrap_or(5); + + let client = http_client(); + let started = Instant::now(); + + // Read the tip's height, timestamp and stored difficulty. + let tip = |c: &reqwest::blocking::Client| -> Option<(u64, u64, u32)> { + let h = find_u64(&get_json(c, &format!("{node}/query/latest")), "height")?; + let b = get_json(c, &format!("{node}/query/block/intro?height={h}")); + Some(( + h, + find_u64(&b, "timestamp")?, + find_u64(&b, "difficulty")? as u32, + )) + }; + + let Some((h0, _, d0)) = tip(&client) else { + eprintln!("could not read the chain tip from {node}"); + std::process::exit(2); + }; + println!("== local chain watch =="); + println!("node = {node}"); + println!("want >= {want} network leading-zero bits"); + println!("deadline = {deadline}s, poll every {poll}s"); + println!( + "start = height {h0}, difficulty {d0}, {} bits\n", + share_cost_bits(&network_target_hash(d0)) + ); + println!( + "{:>8} {:>8} {:>6} {:>12} {:>9} {:>10}", + "elapsed", "height", "bits", "difficulty", "blk_secs", "blocks" + ); + + let mut last_height = h0; + let mut last_ts: Option = None; + let mut best_bits = share_cost_bits(&network_target_hash(d0)); + // First time each milestone was seen, as (bits, elapsed_secs, height). + let mut milestones: Vec<(u32, u64, u64)> = Vec::new(); + + loop { + let elapsed = started.elapsed().as_secs(); + if let Some((h, ts, d)) = tip(&client) { + let bits = share_cost_bits(&network_target_hash(d)); + let blk_secs = match last_ts { + Some(prev) if h > last_height => { + format!("{}", ts.saturating_sub(prev) / (h - last_height).max(1)) + } + _ => "-".to_string(), + }; + if h != last_height || bits != best_bits { + println!( + "{:>7}s {:>8} {:>6} {:>12} {:>9} {:>10}", + elapsed, + h, + bits, + d, + blk_secs, + h.saturating_sub(h0) + ); + } + if bits > best_bits { + best_bits = bits; + milestones.push((bits, elapsed, h)); + } + last_height = h; + last_ts = Some(ts); + if bits >= want { + println!("\nREACHED {want} bits at height {h} after {elapsed}s."); + print_milestones(&milestones, want); + println!( + "\nA block at {bits} bits costs 2^{bits} hashes on average.\n\ + The pool's own admission gate at this difficulty ({d}):" + ); + for sb in [MIN_SHARE_FACTOR, 20, 24] { + let (ok, achieved, cost) = pool_would_serve(d, sb); + let verdict = if ok { "SERVES" } else { "REFUSES" }; + println!( + " share_bits {sb:>2}: achieved {achieved:>2} (min {MIN_SHARE_FACTOR}), \ + cost bits {cost:>2} (min {MIN_SHARE_COST_BITS}) -> {verdict}" + ); + } + std::process::exit(0); + } + } else { + println!("{:>7}s (node not answering yet)", elapsed); + } + if elapsed >= deadline { + println!("\nDEADLINE: only reached {best_bits} bits (wanted {want}) in {elapsed}s."); + print_milestones(&milestones, want); + std::process::exit(1); + } + std::thread::sleep(Duration::from_secs(poll.max(1))); + } +} + +fn print_milestones(ms: &[(u32, u64, u64)], want: u32) { + if ms.is_empty() { + println!("(difficulty never moved)"); + return; + } + println!("\nfirst sighting of each bit level:"); + for (bits, secs, hei) in ms { + let mark = if *bits >= want { " <- pool servable" } else { "" }; + println!(" {bits:>2} bits at {secs:>6}s height {hei}{mark}"); + } +} diff --git a/hbit-pool/examples/rig_tx.rs b/hbit-pool/examples/rig_tx.rs new file mode 100644 index 0000000..35ce26e --- /dev/null +++ b/hbit-pool/examples/rig_tx.rs @@ -0,0 +1,100 @@ +//! Rig helper: derive an address from a throwaway secret, and push real +//! transfers into a node's mempool so a mined block CONTAINS TRANSACTIONS. +//! +//! This exists because the pool's transaction-fee hold-back can only be +//! exercised by a block that actually carries fee-paying transactions, and a +//! private rig chain has no other traffic on it. +//! +//! TESTNET ONLY. It takes a raw private key on the command line. +//! +//! usage: +//! rig_tx addr +//! rig_tx send [count] +//! +//! `addr` prints the readable address for a secret so it can be funded. +//! `send` builds, signs and submits `count` transfers (each with a distinct +//! timestamp so the hashes differ), printing every node answer. + +use basis::interface::*; +use field::*; +use protocol::action::HacToTrs; +use protocol::transaction::TransactionType2; +use sys::*; + +use hbit_pool::{get_json, http_client, post_hex}; + +fn secret_from_hex(s: &str) -> [u8; 32] { + let b = hex::decode(s).expect("secret must be 64 hex chars"); + assert_eq!(b.len(), 32, "secret must be 32 bytes"); + let mut k = [0u8; 32]; + k.copy_from_slice(&b); + k +} + +fn main() { + let a: Vec = std::env::args().collect(); + match a.get(1).map(|s| s.as_str()) { + Some("addr") => { + let acc = Account::create_by_secret_key_value(secret_from_hex(&a[2])).expect("account"); + println!("{}", acc.readable()); + } + Some("send") => { + let base = a[2].trim_end_matches('/').to_string(); + let acc = Account::create_by_secret_key_value(secret_from_hex(&a[3])).expect("account"); + let to = Address::from_readable(&a[4]).expect("to_address"); + let amt = Amount::from(&a[5]).expect("hac amount"); + let fee = Amount::from(&a[6]).expect("fee amount"); + let count: u64 = a.get(7).and_then(|s| s.parse().ok()).unwrap_or(1); + + let client = http_client(); + let main = Address::from(*acc.address()); + println!("from = {}", acc.readable()); + println!("to = {}", a[4]); + println!( + "amount = {} each, fee = {} each, count = {count}", + a[5], a[6] + ); + + let base_ts = curtimes(); + for i in 0..count { + // Distinct timestamps make the hashes differ. They go BACKWARDS: + // the node refuses a transaction stamped later than its own + // clock, so counting up rejects everything after the first. + let mut tx = + TransactionType2::new_by(main.clone(), fee.clone(), base_ts.saturating_sub(i)); + let mut act = HacToTrs::new(); + act.to = AddrOrPtr::from_addr(to.clone()); + act.hacash = amt.clone(); + tx.push_action(Box::new(act)).expect("push action"); + tx.fill_sign(&acc).expect("fill_sign"); + let hash = hex::encode(tx.hash().serialize()); + let body = hex::encode(tx.serialize()); + let resp = post_hex( + &client, + &format!("{base}/submit/transaction?hexbody=true"), + &body, + ); + println!("submit {hash} -> {resp}"); + } + // Report what the node thinks it now holds, so "submitted" is never + // confused with "in the mempool". + std::thread::sleep(std::time::Duration::from_millis(800)); + let pend = get_json( + &client, + &format!("{base}/query/miner/pending?detail=true&transaction=true&stuff=true"), + ); + let n = pend + .get("data") + .and_then(|d| d.get("transactions")) + .and_then(|v| v.as_array()) + .map(|a| a.len()); + println!("node template now carries transactions: {n:?}"); + } + _ => { + eprintln!( + "usage:\n rig_tx addr \n rig_tx send [count]" + ); + std::process::exit(2); + } + } +} diff --git a/hbit-pool/examples/testnet_rig_plan.rs b/hbit-pool/examples/testnet_rig_plan.rs new file mode 100644 index 0000000..d317797 --- /dev/null +++ b/hbit-pool/examples/testnet_rig_plan.rs @@ -0,0 +1,224 @@ +//! Pick (and prove) the `[mint]` settings for a local chain the HBIT pool will +//! actually serve. +//! +//! WHY THIS EXISTS +//! +//! The pool refuses to hand out work when a share is not worth counting: +//! `share_cost_bits >= 16` AND `achieved_share_factor >= 18`, and those two add +//! up to the leading zero bits of the NETWORK target. So the pool needs a chain +//! sitting at >= 34 leading zero bits. A freshly started non-mainnet chain does +//! not: off mainnet ASERT anchors at `difficulty_adjust_blocks + 2` with the +//! fixed constant 0xe9cfffff, which is exactly 22 leading zero bits. +//! +//! The only knob that moves where the chain SETTLES is `each_block_target_time`. +//! ASERT is an equilibrium controller: it drives the target until a block takes +//! `each_block_target_time` seconds, so the resting difficulty of a private +//! chain is whatever your own hashrate can do in that time. Pick the target time +//! and you pick the difficulty. That is the whole trick, and it needs no code +//! change at all, so mainnet consensus is untouched by construction. +//! +//! This example does not model that with a formula. It runs the REAL difficulty +//! function (`hbit_pool::difficulty::next_difficulty`, which `hbit-asert-check` +//! has already proven byte-identical to the node against real mainnet history) +//! block by block, and reports the wall clock. Compare its answer against a real +//! run; if they disagree, believe the run. +//! +//! USAGE +//! cargo run -p hbit-pool --example testnet_rig_plan -- \ +//! [adjust_blocks] [target_time_secs] [max_sim_secs] [min_block_secs] +//! +//! With no target time it SEARCHES for one and prints a table, which is how you +//! choose the number to put in the ini. +//! +//! Note on hashrate: x16rs repeats `height/50000 + 1` times (capped at 16), so a +//! private chain below height 50000 runs at repeat=1. Use your repeat=1 figure +//! here, not the mainnet repeat=16 one. They differ by more than 10x. + +use hbit_pool::difficulty::{ChainParams, next_difficulty}; +use hbit_pool::pool_core::share_cost_bits; + +/// What the pool demands of the NETWORK target before it will serve work. +/// Mirrors MIN_SHARE_FACTOR (18) + MIN_SHARE_COST_BITS (16) in server.rs. +const POOL_MIN_NETWORK_BITS: u32 = 34; +/// Above this a block stops being winnable in a sitting; not a hard rule, just +/// the top of the band this rig aims for. +const BAND_MAX_BITS: u32 = 38; + +struct Outcome { + /// Seconds of wall clock until the chain first reaches POOL_MIN_NETWORK_BITS. + reach_secs: Option, + /// Height at that moment. + reach_height: Option, + /// Seconds spent inside [POOL_MIN_NETWORK_BITS, BAND_MAX_BITS] within the sim. + band_secs: u64, + /// Bits at the end of the simulated window. + final_bits: u32, + /// Expected seconds for one block at the final difficulty. + final_block_secs: f64, + /// True if the sim ran out of time before reaching the band. + timed_out: bool, +} + +/// Walk the chain forward through the real difficulty rule. +/// +/// The model has exactly one assumption: a block at B leading zero bits takes +/// 2^B / hashrate seconds, floored at 1 because `chain/src/verify.rs` rejects +/// `blk_time <= prev_blk_time` in a release build. Everything else is the +/// shipped consensus code. +fn simulate( + hashrate: f64, + adjust_blocks: u64, + target_time: u64, + max_secs: u64, + min_block_secs: u64, +) -> Outcome { + let p = ChainParams::testnet(adjust_blocks, target_time); + // Bootstrap heights are LOWEST_DIFFICULTY (every hash wins) and the anchor + // block itself is the fixed start target; both are one second each because + // of the timestamp floor. + let anchor_time: u64 = p.asert_height; // one second per bootstrap block from t=0 + let mut clock = anchor_time; + let mut height = p.asert_height; + let mut prev_diff = next_difficulty(&p, p.asert_height, anchor_time, 0, 0).0; + + let mut reach_secs = None; + let mut reach_height = None; + let mut band_secs = 0u64; + let mut bits = share_cost_bits(&next_difficulty(&p, p.asert_height, anchor_time, 0, 0).1); + let mut block_secs = 2f64.powi(bits as i32) / hashrate; + + while clock - anchor_time < max_secs { + // How long this block takes at the difficulty now in force. + block_secs = 2f64.powi(bits as i32) / hashrate; + let step = (block_secs.round() as u64).max(min_block_secs); + if (POOL_MIN_NETWORK_BITS..=BAND_MAX_BITS).contains(&bits) { + band_secs += step; + } + clock += step; + height += 1; + let (num, hash) = next_difficulty(&p, height, clock, prev_diff, anchor_time); + prev_diff = num; + bits = share_cost_bits(&hash); + if bits >= POOL_MIN_NETWORK_BITS && reach_secs.is_none() { + reach_secs = Some(clock - anchor_time); + reach_height = Some(height); + } + } + Outcome { + reach_secs, + reach_height, + band_secs, + final_bits: bits, + final_block_secs: block_secs, + timed_out: reach_secs.is_none(), + } +} + +fn hms(s: u64) -> String { + format!("{:02}h{:02}m{:02}s", s / 3600, (s % 3600) / 60, s % 60) +} + +fn main() { + let a: Vec = std::env::args().collect(); + let mhs: f64 = a + .get(1) + .and_then(|s| s.parse().ok()) + .unwrap_or_else(|| { + eprintln!( + "usage: testnet_rig_plan [adjust_blocks] [target_time_secs] [max_sim_secs] [min_block_secs]" + ); + std::process::exit(2) + }); + let hashrate = mhs * 1e6; + let adjust_blocks: u64 = a.get(2).and_then(|s| s.parse().ok()).unwrap_or(8); + let explicit_tt: Option = a.get(3).and_then(|s| s.parse().ok()); + let max_secs: u64 = a.get(4).and_then(|s| s.parse().ok()).unwrap_or(6 * 3600); + // Floor on how long a block takes in practice. The consensus floor is 1 + // second (block_build.rs nextts = max(now, prev_ts+1)), but a real worker + // also spends time polling for a template and posting the solution, so the + // cheap early blocks land slower than the pure hash cost suggests. A run + // measured on this rig arrived at 34 bits about 20% later than the 1-second + // model; pass 2 here to see that bracket. + let min_block_secs: u64 = a.get(5).and_then(|s| s.parse().ok()).unwrap_or(1).max(1); + + let p = ChainParams::testnet(adjust_blocks, explicit_tt.unwrap_or(300)); + let anchor_bits = share_cost_bits(&next_difficulty(&p, p.asert_height, 0, 0, 0).1); + + println!("== HBIT local-chain rig plan =="); + println!("hashrate = {mhs} MH/s (x16rs repeat=1)"); + println!("difficulty_adjust_blocks = {adjust_blocks} -> ASERT anchors at height {}", p.asert_height); + println!("anchor difficulty = {anchor_bits} leading zero bits (fixed constant 0xe9cfffff)"); + println!("pool needs >= {POOL_MIN_NETWORK_BITS} network bits (share_bits 18 + share cost 16)"); + println!("simulation window = {}\n", hms(max_secs)); + + // Equilibrium: a block at B bits takes 2^B/hashrate seconds, so the chain + // rests where that equals each_block_target_time. + let eq_bits = |tt: u64| (hashrate * tt as f64).log2(); + let tt_for = |bits: f64| (2f64.powf(bits) / hashrate).round() as u64; + + match explicit_tt { + Some(tt) => { + let o = simulate(hashrate, adjust_blocks, tt, max_secs, min_block_secs); + report(tt, eq_bits(tt), &o); + // The verdict is the exit code, not the text. A script must be able + // to reject an unviable target time without reading English. + if o.reach_secs.is_none() { + std::process::exit(1); + } + } + None => { + println!( + "{:>9} {:>8} {:>12} {:>8} {:>12} {:>10}", + "target_t", "eq_bits", "reach>=34", "at_hei", "in_band", "blk_at_end" + ); + println!("{}", "-".repeat(70)); + // Candidate target times that put equilibrium at 34..39 bits. + for b in POOL_MIN_NETWORK_BITS..=(BAND_MAX_BITS + 1) { + let tt = tt_for(b as f64).max(1); + let o = simulate(hashrate, adjust_blocks, tt, max_secs, min_block_secs); + println!( + "{:>9} {:>8.2} {:>12} {:>8} {:>12} {:>9.0}s", + tt, + eq_bits(tt), + o.reach_secs.map(hms).unwrap_or_else(|| "NEVER".into()), + o.reach_height + .map(|h| h.to_string()) + .unwrap_or_else(|| "-".into()), + hms(o.band_secs), + o.final_block_secs + ); + } + println!( + "\nPick the row with the smallest `reach>=34` that still leaves a long `in_band`,\n\ + put that target_t in [mint].each_block_target_time, then re-run this with it\n\ + as argument 3 for the full report." + ); + } + } +} + +fn report(tt: u64, eq: f64, o: &Outcome) { + println!("each_block_target_time = {tt}s (equilibrium ~{eq:.2} bits)"); + match (o.reach_secs, o.reach_height) { + (Some(s), Some(h)) => { + println!("reaches {POOL_MIN_NETWORK_BITS} bits after {} of mining, at height {h}", hms(s)); + println!( + "stays in the {POOL_MIN_NETWORK_BITS}..{BAND_MAX_BITS} band for {} of the simulated window", + hms(o.band_secs) + ); + println!( + "at the end of the window: {} bits, ~{:.0}s per block", + o.final_bits, o.final_block_secs + ); + println!("\nPLAN IS VIABLE."); + } + _ => { + println!( + "NEVER reaches {POOL_MIN_NETWORK_BITS} bits in the window (ends at {} bits, ~{:.0}s per block).", + o.final_bits, o.final_block_secs + ); + println!("timed_out={}", o.timed_out); + println!("\nPLAN IS NOT VIABLE at this target time."); + } + } +} diff --git a/hbit-pool/src/asert_check.rs b/hbit-pool/src/asert_check.rs index 0ba2def..9681b72 100644 --- a/hbit-pool/src/asert_check.rs +++ b/hbit-pool/src/asert_check.rs @@ -57,7 +57,10 @@ fn main() { println!("== HBIT ASERT check =="); println!("node = {node}"); - println!("chain = {chain} (ASERT anchor at height {})", params.asert_height); + println!( + "chain = {chain} (ASERT anchor at height {})", + params.asert_height + ); println!("tip = {tip}"); let anchor_time = find_u64( @@ -80,7 +83,10 @@ fn main() { println!("h={h} (missing block data, skipped)"); continue; }; - let pb = get_json(&client, &format!("{node}/query/block/intro?height={}", h - 1)); + let pb = get_json( + &client, + &format!("{node}/query/block/intro?height={}", h - 1), + ); let Some(prev_diff) = find_u64(&pb, "difficulty") else { println!("h={h} (missing parent, skipped)"); continue; @@ -107,13 +113,18 @@ fn main() { } Some(_) => { ok += 1; - println!("h={h} OK difficulty={stored} target={}", hex::encode(target)); + println!( + "h={h} OK difficulty={stored} target={}", + hex::encode(target) + ); } None => { // Without the block's hash only half the check ran; do not report // that as a pass. bad += 1; - println!("h={h} NO-HASH could not read the block's own hash to verify the target"); + println!( + "h={h} NO-HASH could not read the block's own hash to verify the target" + ); } } } diff --git a/hbit-pool/src/difficulty.rs b/hbit-pool/src/difficulty.rs index 8918b53..c55910f 100644 --- a/hbit-pool/src/difficulty.rs +++ b/hbit-pool/src/difficulty.rs @@ -176,7 +176,10 @@ mod tests { // Documented defaults and mainnet still parse. let d = ChainParams::parse("testnet").expect("bare testnet"); assert_eq!((d.asert_height, d.target_time), (290, 10)); - assert_eq!(ChainParams::parse("mainnet").expect("mainnet").asert_height, 738654); + assert_eq!( + ChainParams::parse("mainnet").expect("mainnet").asert_height, + 738654 + ); // Anything we cannot mine is refused rather than silently guessed. assert!(ChainParams::parse("regtest").is_none()); assert!(ChainParams::parse("testnet:8").is_none()); @@ -201,7 +204,10 @@ mod tests { let p = ChainParams::testnet(288, 10); let (num, hash) = next_difficulty(&p, 290, 9_999, LOWEST_DIFFICULTY, 0); assert_eq!(num, ASERT_START_TARGET_NUM); - assert_eq!(hash, DifficultyTarget::from_num(ASERT_START_TARGET_NUM).hash); + assert_eq!( + hash, + DifficultyTarget::from_num(ASERT_START_TARGET_NUM).hash + ); // mainnet anchors at 738654 let m = ChainParams::mainnet(); assert_eq!( @@ -232,11 +238,25 @@ mod tests { ); // ahead of schedule (mined too fast) -> smaller target (harder) let fast = DifficultyTarget::from_num( - next_difficulty(&p, height, on_time - 600, ASERT_START_TARGET_NUM, anchor_time).0, + next_difficulty( + &p, + height, + on_time - 600, + ASERT_START_TARGET_NUM, + anchor_time, + ) + .0, ); // behind schedule -> larger target (easier), capped at 2x the parent let slow = DifficultyTarget::from_num( - next_difficulty(&p, height, on_time + 600, ASERT_START_TARGET_NUM, anchor_time).0, + next_difficulty( + &p, + height, + on_time + 600, + ASERT_START_TARGET_NUM, + anchor_time, + ) + .0, ); assert!(fast.big < base.big, "faster blocks must tighten the target"); assert!(slow.big > base.big, "slower blocks must ease the target"); diff --git a/hbit-pool/src/lib.rs b/hbit-pool/src/lib.rs index b55f0f2..09f699b 100644 --- a/hbit-pool/src/lib.rs +++ b/hbit-pool/src/lib.rs @@ -68,10 +68,98 @@ pub fn find_value<'a>(v: &'a Value, key: &str) -> Option<&'a Value> { } } -/// The recipient's "hacash" balance string (e.g. "1:248"), or "" if none. -pub fn balance(client: &reqwest::blocking::Client, base: &str, addr: &str) -> String { - let j = get_json(client, &format!("{base}/query/balance?address={addr}")); - find_str(&j, "hacash").unwrap_or_default() +/// What the node said when it was asked for an address's balance. +/// +/// Three states, not two. This used to be a bare `String` that folded every +/// failure into "", which `balance_units` then valued as a confident zero. A +/// node that was down, restarting, or answering with its own error object read +/// exactly like a wallet holding nothing: settlement published "matured = 0" and +/// flagged it CURRENT, so every miner polling `/earnings` was told it was owed +/// nothing for as long as the outage lasted, and the template loop re-poisoned +/// the same figure every 30 seconds. Miners whose shares rolled out of the PPLNS +/// window during the outage were never paid for that work. +/// +/// A wallet holding nothing is NOT this state: the node always emits the +/// `hacash` field and renders an empty wallet as "0:0", which is +/// [`Reported`](BalanceAnswer::Reported) and values as a real zero. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum BalanceAnswer { + /// The node gave a balance string for the address, e.g. "1:248" or "0:0". + Reported(String), + /// The node answered, but not with a balance: its own `{"ret":1,...}` error + /// object, or a body carrying no `hacash` field at all. + Refused(String), + /// Nothing usable came back: connection refused, a timeout, a proxy's error + /// page. The wallet is UNKNOWN, not empty. + NoAnswer(String), +} + +impl BalanceAnswer { + /// The balance in whole units of 0.1 HAC, or `None` when the pool must not + /// act on this answer at all. Anything but a reported balance is `None`: + /// callers already treat `None` as "skip this cycle, keep the last good + /// figure and mark it stale", which is exactly right for a silent node. + pub fn units(&self) -> Option { + match self { + BalanceAnswer::Reported(s) => balance_units(s), + BalanceAnswer::Refused(_) | BalanceAnswer::NoAnswer(_) => None, + } + } +} + +impl std::fmt::Display for BalanceAnswer { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + BalanceAnswer::Reported(s) => write!(f, "{s}"), + BalanceAnswer::Refused(s) => write!(f, "the node refused to report a balance: {s}"), + BalanceAnswer::NoAnswer(s) => write!(f, "no answer from the node: {s}"), + } + } +} + +/// How much of an unusable answer goes into a log line. A non-JSON body can be a +/// whole HTML error page, and a log that scrolls the real message away is a log +/// nobody can read during the outage it is describing. +const ANSWER_EXCERPT_CHARS: usize = 200; + +/// Truncate on a CHARACTER boundary: this text comes off the wire and slicing it +/// by bytes would panic the caller on any multi-byte error message. +fn excerpt(s: &str) -> String { + s.chars().take(ANSWER_EXCERPT_CHARS).collect() +} + +/// Classify a `/query/balance` response. Fails SAFE: only an answer that really +/// carries a balance becomes [`BalanceAnswer::Reported`], and everything else is +/// a state the caller must refuse to pay on. +pub fn balance_answer(j: &Value) -> BalanceAnswer { + // get_json encodes a transport failure as {"http_error": "..."} and a + // non-JSON body as a bare string. Neither is the node speaking. + if let Some(e) = j.get("http_error").and_then(|v| v.as_str()) { + return BalanceAnswer::NoAnswer(excerpt(e)); + } + if !j.is_object() { + return BalanceAnswer::NoAnswer(excerpt(&j.to_string())); + } + // The node answered, and its answer is "no": a bad address, too many + // addresses, an unreadable state. There is no balance in it to pay on. + if find_u64(j, "ret").is_some_and(|r| r != 0) { + return BalanceAnswer::Refused(excerpt(&j.to_string())); + } + match find_str(j, "hacash") { + Some(s) if !s.trim().is_empty() => BalanceAnswer::Reported(s), + // ret=0 with no `hacash` is a shape this pool does not recognise. The + // node always emits the field, so its absence means we are not talking + // to one - never that the wallet is empty. + _ => BalanceAnswer::Refused(excerpt(&j.to_string())), + } +} + +/// The address's "hacash" balance as the node reported it, or why it did not. +pub fn balance(client: &reqwest::blocking::Client, base: &str, addr: &str) -> BalanceAnswer { + balance_answer(&get_json( + client, + &format!("{base}/query/balance?address={addr}"), + )) } /// The largest balance the pool will act on, in units of 0.1 HAC. Hacash's whole @@ -92,12 +180,15 @@ pub const MAX_PLAUSIBLE_UNITS: u64 = 1_000_000_000_000; /// larger than any real wallet: the caller must SKIP settlement rather than pay /// out on it. Saturating to u64::MAX here (as this used to) means "infinite /// money" to `distributable_units` and `split_payout`, which then plan a payout -/// of the whole u64 range off one malformed response. An EMPTY string is not an -/// error: the node simply omits the field for an address holding nothing. +/// of the whole u64 range off one malformed response. +/// +/// An EMPTY string is one of those refusals, and it is the important one. The +/// node always emits the `hacash` field and renders a wallet holding nothing as +/// "0:0", so "" is never something it reported: it is what the old reader +/// produced when there was no answer at all. Valuing it as `Some(0)` told the +/// settlement that a wallet it could not see was empty, and told every miner +/// polling `/earnings` that it was owed nothing for the length of the outage. pub fn balance_units(bal: &str) -> Option { - if bal.trim().is_empty() { - return Some(0); - } let (m, u) = bal.split_once(':')?; let (Ok(m), Ok(u)) = (m.trim().parse::(), u.trim().parse::()) else { return None; @@ -118,13 +209,204 @@ pub fn balance_units(bal: &str) -> Option { (units <= MAX_PLAUSIBLE_UNITS).then_some(units) } -/// The coinbase subsidy of the block at `height`, in units of 0.1 HAC. The pool -/// mines coinbase-only blocks, so this is the entire income a found block brings -/// into the wallet (`block_reward` is a whole number of HAC = unit 248). +/// The coinbase subsidy of the block at `height`, in units of 0.1 HAC +/// (`block_reward` is a whole number of HAC = unit 248). +/// +/// This is NOT the whole income a found block brings in. The pool packs the +/// node's transactions, and the chain credits the sum of their fees to the +/// coinbase address as well - the same wallet the pool settles from. See +/// [`block_fees`] for that half of it. pub fn block_reward_units(height: u64) -> u64 { mint::genesis::block_reward_number(height) as u64 * 10 } +/// Fine steps in one payout unit of 0.1 HAC; one step is 10^-9 HAC. +/// +/// A transaction fee is routinely a thousandth of a payout unit, so a block's +/// fees are summed on this finer scale and rounded up to whole units only once, +/// at the end. Rounding each fee up on its own would hold back a whole unit per +/// transaction and freeze real money for a whole maturity window. +const FEE_FINE_PER_UNIT: u128 = 100_000_000; + +/// The largest fee total this pool will believe, on the fine scale. Past the +/// whole coin supply the answer is corrupt or hostile, not a rich block, and +/// turning it into a hold-back would stop every payout the pool ever makes. +const MAX_FEE_FINE: u128 = MAX_PLAUSIBLE_UNITS as u128 * FEE_FINE_PER_UNIT; + +/// A node "mantissa:unit" amount on the fine scale, ROUNDED UP. +/// +/// Rounds UP because this number becomes money the pool refuses to pay out yet. +/// Rounding a fee down to nothing is exactly how the fee ends up distributed at +/// zero confirmations, which is the failure this exists to stop. +/// +/// `None` is "this is not an amount I can value" - a negative mantissa, a +/// missing separator, an exponent no wallet could hold - and the caller must +/// then refuse to settle rather than read it as a zero fee. +pub fn fin_fine_ceil(amount: &str) -> Option { + let (m, u) = amount.split_once(':')?; + let (Ok(m), Ok(u)) = (m.trim().parse::(), u.trim().parse::()) else { + return None; + }; + if !(0..=255).contains(&u) { + return None; // the chain's unit is a u8; anything else is not its answer + } + if m == 0 { + return Some(0); // a real zero, at any unit + } + // value = m * 10^(u-248) HAC, and one fine step is 10^-9 HAC. + let exp = u - 239; + let fine = if exp >= 0 { + let scale = u32::try_from(exp) + .ok() + .and_then(|e| 10u128.checked_pow(e))?; + m.checked_mul(scale)? + } else { + match u32::try_from(-exp).ok().and_then(|e| 10u128.checked_pow(e)) { + Some(d) => m.div_ceil(d), + // Finer than a fine step by more orders of magnitude than a u128 can + // express. It is still money, so it still counts as one step. + None => 1, + } + }; + (fine <= MAX_FEE_FINE).then_some(fine) +} + +/// Fine steps as whole payout units of 0.1 HAC, rounded UP. +/// +/// The wallet balance the pool settles against is itself floored to whole units, +/// and a fee that straddles a unit boundary can push that floor up by one. The +/// ceiling is what makes the hold-back cover that case instead of leaving one +/// unit payable out of income a reorg can still revoke. +pub fn fine_to_units_ceil(fine: u128) -> u64 { + fine.div_ceil(FEE_FINE_PER_UNIT) + .min(MAX_PLAUSIBLE_UNITS as u128) as u64 +} + +/// What the node says about the transaction fees one of OUR blocks credited. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum BlockFees { + /// The chain holds our block at that height, and it credited this much fee + /// income to the pool wallet, in units of 0.1 HAC rounded up. + Counted(u64), + /// The node answered, and the chain does NOT hold our block at that height. + /// It credited nothing there - no subsidy and no fee - so there is no fee + /// income to hold back. + NotOnChain, + /// No usable answer. This is NOT a zero fee: the wallet may be holding fee + /// income the pool cannot value, so the caller must refuse to settle. + Unknown(String), +} + +/// Which transactions the chain says are in our block, or why it cannot say. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum BlockTxs { + /// The chain holds OUR block at that height, and these are the hashes of the + /// transactions in it. The coinbase is not among them: it pays no fee. + Ours(Vec), + /// The node answered, and the chain does not hold our block there. + NotOnChain, + /// No usable answer. + Unknown(String), +} + +/// Read a `/query/block/intro?tx_hash_list=true` answer for one of OUR blocks. +/// +/// Split out from [`block_fees`] so the decision - price it, ignore it, or stop +/// settling - is testable without a node. Fails SAFE: only an answer that really +/// carries our block's transaction list is [`BlockTxs::Ours`]. +pub fn block_txs_of(j: &Value, our_hash_hex: &str) -> BlockTxs { + // get_json encodes a transport failure as {"http_error": "..."} and a + // non-JSON body as a bare string. Neither is the node speaking. + if !j.is_object() || j.get("http_error").is_some() { + return BlockTxs::Unknown(excerpt(&j.to_string())); + } + let Some(ret) = find_u64(j, "ret") else { + return BlockTxs::Unknown(excerpt(&j.to_string())); + }; + if ret != 0 { + // The node is up and has no block at that height: ours was refused, or + // has not been inserted yet. Either way it has credited nothing. + return BlockTxs::NotOnChain; + } + let Some(hash) = find_str(j, "hash") else { + return BlockTxs::Unknown(excerpt(&j.to_string())); + }; + if !hash.eq_ignore_ascii_case(our_hash_hex) { + return BlockTxs::NotOnChain; // another block won that height + } + // ret=0 for our block but no list at all is an answer this pool does not + // recognise - never "the block had no transactions". A node that quietly + // dropped the field would otherwise read as a zero fee on every block. + let Some(list) = find_value(j, "tx_hash_list").and_then(|v| v.as_array()) else { + return BlockTxs::Unknown(excerpt(&j.to_string())); + }; + match list + .iter() + .map(|h| h.as_str().map(|s| s.to_string())) + .collect::>>() + { + Some(hs) => BlockTxs::Ours(hs), + None => BlockTxs::Unknown(excerpt(&j.to_string())), + } +} + +/// The `fee_got` in an answer to `/query/transaction`, on the fine scale. +/// +/// `fee_got` and not `fee`: what the chain adds to the coinbase address is the +/// fee the transaction actually PAID for its place in the block, which a +/// fee-raise or a gas refund can make smaller than the fee it declared. +pub fn fee_got_fine(j: &Value) -> Option { + if !j.is_object() || j.get("http_error").is_some() { + return None; + } + if find_u64(j, "ret") != Some(0) { + return None; + } + fin_fine_ceil(&find_str(j, "fee_got")?) +} + +/// What the chain credited the pool wallet in TRANSACTION FEES for our block at +/// `height`, in units of 0.1 HAC rounded up. +/// +/// The figure cannot be taken from the block the pool built: its transaction +/// bodies are raw bytes the pool deliberately has no codec for, and +/// `/submit/block` answers only `{"ok":true}`. So it is read back off the node +/// once the block exists, one `/query/transaction` per packed transaction. +/// +/// Answers `Unknown` on anything short of a definitive reply, because the caller +/// turns that into "settle nothing this cycle". The alternative - treating an +/// unreachable node as a zero fee - pays that fee income out at zero +/// confirmations, and an orphan then leaves the operator funding a payout out of +/// a block the chain no longer has. +pub fn block_fees( + client: &reqwest::blocking::Client, + node: &str, + height: u64, + our_hash_hex: &str, +) -> BlockFees { + let j = get_json( + client, + &format!("{node}/query/block/intro?height={height}&tx_hash_list=true"), + ); + let hashes = match block_txs_of(&j, our_hash_hex) { + BlockTxs::Ours(hs) => hs, + BlockTxs::NotOnChain => return BlockFees::NotOnChain, + BlockTxs::Unknown(why) => return BlockFees::Unknown(why), + }; + let mut fine: u128 = 0; + for h in &hashes { + let t = get_json(client, &format!("{node}/query/transaction?hash={h}")); + let Some(f) = fee_got_fine(&t) else { + return BlockFees::Unknown(format!("transaction {h}: {}", excerpt(&t.to_string()))); + }; + fine = fine.saturating_add(f); + if fine > MAX_FEE_FINE { + return BlockFees::Unknown(format!("fees at height {height} exceed any real block")); + } + } + BlockFees::Counted(fine_to_units_ceil(fine)) +} + /// How deep a payout transaction must be buried before the pool stops tracking /// it. The node keeps up to `unstable_block` (4) blocks reorg-able, so a payout /// that is only 1-3 confirmations deep can still come back to the mempool; @@ -227,11 +509,7 @@ pub fn admission_of(j: &Value) -> Admission { /// Ask the node whether it really holds `txhash`, retrying while it has not made /// up its mind. The insert runs on a background task, so an immediate "not /// found" only becomes a verdict once the node has had time to do it. -pub fn verify_admitted( - client: &reqwest::blocking::Client, - node: &str, - txhash: &str, -) -> Admission { +pub fn verify_admitted(client: &reqwest::blocking::Client, node: &str, txhash: &str) -> Admission { let mut last = Admission::Unresolved; for attempt in 0..ADMIT_POLL_TRIES { let j = get_json(client, &format!("{node}/query/transaction?hash={txhash}")); @@ -250,10 +528,14 @@ pub fn verify_admitted( /// reorg could still take back, MINUS the fee reserve. `None` means "nothing /// spendable, do not settle this cycle". /// -/// `immature_units` is the coinbase of blocks the pool found that are not yet -/// buried deep enough to be final. Distributing that and then losing the block -/// to a reorg is an unrecoverable operator loss: the income disappears from the -/// canonical chain while the payout transaction that spent it stays valid. +/// `immature_units` is the WHOLE income of blocks the pool found that are not +/// yet buried deep enough to be final. Whole, because the chain credits the +/// coinbase address both the subsidy and the sum of the fees of every +/// transaction in the block, and this pool packs the node's transactions: a +/// hold-back of the subsidy alone leaves the fees payable here. Distributing +/// either and then losing the block to a reorg is an unrecoverable operator +/// loss: the income disappears from the canonical chain while the payout +/// transaction that spent it stays valid. /// /// All arithmetic saturates, so an out-of-range reserve can never wrap the /// guard open the way `reserve + 1` used to. @@ -271,7 +553,8 @@ pub fn distributable_units( /// Atomic file write (temp + optional fsync + rename) so a crash or a full disk /// mid-write can never leave a truncated or corrupt file behind. `durable` -/// fsyncs before the rename. +/// fsyncs the bytes before the rename and the directory after it, and FAILS if +/// either fsync fails. pub fn atomic_write(path: &str, body: &[u8], durable: bool) -> std::io::Result<()> { use std::io::Write; let tmp = format!("{path}.tmp.{}", std::process::id()); @@ -279,10 +562,53 @@ pub fn atomic_write(path: &str, body: &[u8], durable: bool) -> std::io::Result<( let mut f = std::fs::File::create(&tmp)?; f.write_all(body)?; if durable { - let _ = f.sync_all(); + // This used to be `let _ = f.sync_all()`. `durable` is the promise + // the settlement path broadcasts a payout on, and a discarded error + // here answers "recorded" for bytes that only ever reached the page + // cache - a full disk, an I/O error, or a network mount that went + // away all report themselves at flush time and nowhere else. Lose + // power in the seconds that follow and the pool restarts with no + // memory of the transaction it signed, so the next cycle signs a + // SECOND payout for the same PPLNS window and the operator funds + // the difference out of their own wallet. + f.sync_all()?; } } - std::fs::rename(&tmp, path) + std::fs::rename(&tmp, path)?; + if durable { + // The bytes can be on the platter while the directory entry pointing at + // them is not: the rename is its own metadata change and is lost on its + // own. That leaves the PREVIOUS state file in place - the one without + // the payout hash or without the immature hold-back - which costs + // exactly what the paragraph above costs. + fsync_parent_dir(path)?; + } + Ok(()) +} + +/// fsync the directory that holds `path`, so a rename into it survives a power +/// cut rather than only the bytes it points at. +/// +/// A bare filename (a relative `wallet_file` in the config) has no directory +/// component and must fall back to the working directory: treating that as a +/// failure would make every durable write fail and stop the pool paying anyone. +#[cfg(unix)] +fn fsync_parent_dir(path: &str) -> std::io::Result<()> { + let dir = match std::path::Path::new(path).parent() { + Some(d) if !d.as_os_str().is_empty() => d.to_path_buf(), + _ => std::path::PathBuf::from("."), + }; + std::fs::File::open(dir)?.sync_all() +} + +/// Windows has no directory fsync: a directory handle cannot be opened for +/// `FlushFileBuffers`, so there is nothing to call and the durability of the +/// rename is NTFS's own metadata journal. Reporting that as a failure would +/// refuse every settlement the pool ever tried, which is worse than the gap it +/// would be reporting. The data fsync above still holds on this platform. +#[cfg(not(unix))] +fn fsync_parent_dir(_path: &str) -> std::io::Result<()> { + Ok(()) } /// The pool's accounting file for `wallet_file`. The auto-settle server and the @@ -317,13 +643,100 @@ pub fn load_pending_payout_txs(state_file: &str) -> Vec { /// Rolling PPLNS window: the last N accepted shares decide the payout split. pub const PPLNS_WINDOW: usize = 4096; -/// Rebuild the PPLNS share counts from the pool's own accounting file. +/// How long a share may go on earning credit, and how long the credit an evicted +/// share already earned survives, expressed in settlement intervals. +/// +/// One interval is the unit that matters because that is the longest a miner can +/// usefully sit on shares: the pool pins a template for a block interval, and it +/// publishes the settlement interval in `/terms`. Anything shorter would let a +/// hoarder time its dump; much longer and the split stops tracking who is mining +/// now. +const PPLNS_HORIZON_INTERVALS: u64 = 1; + +/// The documented default settlement interval. `hbit-pool-server`'s `usage()` +/// quotes it and `hbit-pool-payout` falls back to it when it has to read an +/// accounting file that does not record the interval the server was running. +pub const DEFAULT_SETTLE_SECS: u64 = 300; + +/// The credit horizon in milliseconds for a pool settling every `settle_secs`. +pub fn pplns_horizon_ms(settle_secs: u64) -> u64 { + settle_secs + .saturating_mul(PPLNS_HORIZON_INTERVALS) + .saturating_mul(1_000) + .max(1_000) +} + +/// Read the persisted share window, accepting BOTH the timestamped form this +/// pool writes now and the bare list of worker ids older builds wrote. +/// +/// An older file carries no arrival times at all. Every share in it is given the +/// SAME stamp, `fallback_ms`, because that is the only assumption that treats +/// every miner alike: credit is proportional, so one common start time preserves +/// the split exactly, while inventing different ages would silently move money +/// between miners on a restart. +pub fn parse_share_order(j: &Value, fallback_ms: u64) -> Vec<(String, u64)> { + j.get("order") + .and_then(|v| v.as_array()) + .map(|a| { + a.iter() + .filter_map(|x| { + if let Some(s) = x.as_str() { + return Some((s.to_string(), fallback_ms)); + } + let row = x.as_array()?; + let w = row.first()?.as_str()?.to_string(); + let at = row.get(1)?.as_u64().unwrap_or(fallback_ms); + Some((w, at)) + }) + .collect() + }) + .unwrap_or_default() +} + +/// Read the banked credit of shares that have already left the window. Absent in +/// a file written before shares were timed, which reads as "none banked" rather +/// than as a corrupt file. +pub fn parse_banked_credit(j: &Value) -> Vec<(u64, Vec<(String, u64)>)> { + j.get("banked") + .and_then(|v| v.as_array()) + .map(|a| { + a.iter() + .filter_map(|x| { + let at = x.get("at").and_then(|v| v.as_u64())?; + let rows = x + .get("rows") + .and_then(|v| v.as_array()) + .map(|r| { + r.iter() + .filter_map(|e| { + let row = e.as_array()?; + Some(( + row.first()?.as_str()?.to_string(), + row.get(1)?.as_u64()?, + )) + }) + .collect() + }) + .unwrap_or_default(); + Some((at, rows)) + }) + .collect() + }) + .unwrap_or_default() +} + +/// Rebuild the PPLNS payout credit from the pool's own accounting file. /// /// The manual payout tool needs this because the server holds the wallet's /// settlement lock for its whole run: if the tool is able to settle at all then /// the server is stopped, so its `/stats` endpoint cannot answer and the file it /// left behind is the authority on who is owed what. -pub fn load_pplns_counts(state_file: &str) -> Vec<(String, u64)> { +/// +/// It returns CREDIT, not share counts, for the same reason the server settles on +/// credit: a headcount taken at the instant of a payout is a number one miner can +/// own outright by dumping a window's worth of withheld shares, and this tool +/// signs the same money. +pub fn load_pplns_credit(state_file: &str) -> Vec<(String, u64)> { let Some(j) = read_state_json(state_file) else { return Vec::new(); }; @@ -331,36 +744,119 @@ pub fn load_pplns_counts(state_file: &str) -> Vec<(String, u64)> { .get("window") .and_then(|v| v.as_u64()) .unwrap_or(PPLNS_WINDOW as u64) as usize; - let order: Vec = j - .get("order") - .and_then(|v| v.as_array()) - .map(|a| { - a.iter() - .filter_map(|x| x.as_str().map(|s| s.to_string())) - .collect() - }) - .unwrap_or_default(); - if order.is_empty() { + // The horizon the SERVER was running, so the manual tool splits money the + // same way the automatic settlement would have. + let horizon = j + .get("credit_horizon_ms") + .and_then(|v| v.as_u64()) + .filter(|h| *h > 0) + .unwrap_or_else(|| pplns_horizon_ms(DEFAULT_SETTLE_SECS)); + let at = credit_anchor_ms(&j, pool_core::now_ms()); + // A file with no arrival times is read as if every share landed one horizon + // before that instant: they are all treated alike, so the split is the one + // the old build would have made, and no share reads as newer than it is. + let order = parse_share_order(&j, at.saturating_sub(horizon)); + let banked = parse_banked_credit(&j); + if order.is_empty() && banked.is_empty() { return Vec::new(); } - pool_core::Pplns::restore(window, order).counts() + pool_core::Pplns::restore(window, horizon, order, banked).credit(at) +} + +/// The instant a stored share window is worth valuing at: the last moment the +/// pool that wrote the file was actually accounting. +/// +/// NOT the wall clock. Credit is residence in the window, and nothing enters or +/// leaves that window while the server is stopped - which it always is when this +/// tool runs, because the tool can only get the settlement lock if it is. Valuing +/// at the wall clock let the whole window age together while nothing happened, +/// and past one horizon that undoes the fix this file exists to carry: every +/// share caps at the horizon, so the split flattens back to a HEADCOUNT, and the +/// banked credit of the miners a dump evicted expires entirely. A miner that +/// withheld a window's worth and dumped it before the server stopped would then +/// take the lot - the exact attack, back again, on the settler an operator +/// reaches for when the server is down. Five minutes between stopping the pool +/// and running the payout is all it took. +/// +/// Anchoring here also makes the payout DETERMINISTIC: the same file settles the +/// same way whether the operator runs the tool immediately or an hour later. +/// +/// The tool's own clock is deliberately not consulted when the file has times of +/// its own. Every credit figure is a DIFFERENCE against this instant, so a +/// consistent anchor taken from the file makes the split independent of what the +/// machine running the settlement thinks the time is. +/// +/// Falls back to `now_ms` when the file carries no times at all (a window written +/// before shares were stamped), because there is nothing to anchor to and the +/// fallback stamp one horizon back then weighs every share alike, as it must. +pub fn credit_anchor_ms(j: &Value, now_ms: u64) -> u64 { + let newest_share = j + .get("order") + .and_then(|v| v.as_array()) + .and_then(|a| a.iter().filter_map(|x| x.as_array()?.get(1)?.as_u64()).max()); + let newest_bank = j + .get("banked") + .and_then(|v| v.as_array()) + .and_then(|a| a.iter().filter_map(|x| x.get("at")?.as_u64()).max()); + newest_share + .into_iter() + .chain(newest_bank) + .max() + .unwrap_or(now_ms) } -/// Total held-back (not yet final) block income recorded by the pool server, in -/// units of 0.1 HAC. The manual payout tool reads it so it applies the SAME -/// maturity gate as the automatic settlement instead of paying at the tip. -pub fn load_immature_units(state_file: &str) -> u64 { +/// One block of not-yet-final income the pool server is holding back. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct ImmatureBlock { + pub height: u64, + /// OUR block's hash at that height, hex. The income is only real while the + /// chain still holds this hash there. + pub hash: String, + /// What it put into the pool wallet so far, in units of 0.1 HAC. + pub units: u64, + /// Are that block's TRANSACTION FEES already inside `units`? + /// + /// False means `units` is the coinbase subsidy alone, and the block's fees - + /// which the chain credits to the very same wallet - are still sitting in + /// the balance unaccounted for. Paying against that balance hands those fees + /// out at zero confirmations. + pub fees_counted: bool, +} + +/// Every block of not-yet-final income the pool server recorded. The manual +/// payout tool reads it so it applies the SAME maturity gate as the automatic +/// settlement instead of paying at the tip. +/// +/// `fees_counted` defaults to FALSE when the field is absent, because a file +/// written by a build that held back only the subsidy really does carry the +/// subsidy alone. Defaulting the other way would silently distribute those +/// blocks' fees on the first settlement after an upgrade. +pub fn load_immature_blocks(state_file: &str) -> Vec { let Some(j) = read_state_json(state_file) else { - return 0; + return Vec::new(); }; j.get("immature") .and_then(|v| v.as_array()) .map(|a| { a.iter() - .filter_map(|x| x.get("units").and_then(|v| v.as_u64())) - .sum() + .filter_map(|x| { + Some(ImmatureBlock { + height: x.get("height").and_then(|v| v.as_u64())?, + hash: x + .get("hash") + .and_then(|v| v.as_str()) + .unwrap_or_default() + .to_string(), + units: x.get("units").and_then(|v| v.as_u64())?, + fees_counted: x + .get("fees_counted") + .and_then(|v| v.as_bool()) + .unwrap_or(false), + }) + }) + .collect() }) - .unwrap_or(0) + .unwrap_or_default() } /// Replace `settle_pending_txs` in the pool state file, preserving every other @@ -436,6 +932,17 @@ pub struct PayoutRecord { /// submitted but the node's verdict could not be read: it may well be in /// flight, so it stays tracked, but nothing about it is claimed. pub node_holds: bool, + /// The exact signed bytes that were submitted, hex-encoded. + /// + /// Kept because a node that once HELD this transaction also relayed it, and + /// the mempool is memory-only: a routine node restart empties it, and the + /// pool then asks about a hash the node no longer knows. Re-splitting and + /// re-signing that window makes a DIFFERENT transaction (fresh timestamp, + /// so a different hash); replay protection on this chain is by hash alone, + /// so both can be mined and the operator pays the same miners twice out of + /// its own wallet. With the bytes here the pool re-broadcasts the identical + /// transaction, which can only ever be included once. + pub body_hex: String, /// (worker address, units of 0.1 HAC) exactly as the transaction pays them. pub rows: Vec<(String, u64)>, } @@ -443,7 +950,10 @@ pub struct PayoutRecord { impl PayoutRecord { /// Total this transaction pays, in units of 0.1 HAC. pub fn units(&self) -> u64 { - self.rows.iter().map(|(_, u)| *u).fold(0u64, |a, b| a.saturating_add(b)) + self.rows + .iter() + .map(|(_, u)| *u) + .fold(0u64, |a, b| a.saturating_add(b)) } /// What this transaction pays ONE worker. @@ -460,6 +970,7 @@ impl PayoutRecord { "hash": self.hash, "at": self.at, "node_holds": self.node_holds, + "body_hex": self.body_hex, "rows": self.rows.iter() .map(|(w, u)| serde_json::json!([w, u])) .collect::>(), @@ -490,6 +1001,15 @@ impl PayoutRecord { .get("node_holds") .and_then(|x| x.as_bool()) .unwrap_or(false), + // A record written before the bytes were kept reads as "no bytes", + // which `gone_action` treats as un-rebroadcastable rather than as + // safe to re-issue. Missing evidence must never become permission + // to sign the same window again. + body_hex: v + .get("body_hex") + .and_then(|x| x.as_str()) + .unwrap_or_default() + .to_string(), rows, }) } @@ -632,6 +1152,232 @@ pub fn drop_payout(records: &mut Vec, hash: &str) -> Option) -> GoneAction { + match rec { + Some(r) if !r.body_hex.is_empty() => GoneAction::Rebroadcast, + Some(r) if r.node_holds => GoneAction::Stuck, + _ => GoneAction::Forget, + } +} + +/// What `/submit/transaction` said about a payout we just posted. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum SubmitVerdict { + /// `ret=0`: the API took the bytes. Still not proof the node holds it - only + /// [`verify_admitted`] is that. + Accepted, + /// The node itself answered with a non-zero `ret`: it refused the + /// transaction during synchronous validation, so it never inserted it into + /// the mempool and never relayed it. + Rejected, + /// No verdict at all: [`post_hex`] returns the plain string + /// `"http_error: ..."` when the request times out or the connection drops, + /// and a timeout happens AFTER the node may have already taken and relayed + /// the transaction. Reading that as a rejection and forgetting the hash is + /// how a payout gets issued a second time. + Unresolved, +} + +/// Classify a `/submit/transaction` response body. +/// +/// Fails SAFE: anything that is not the node speaking a `ret` we can read is +/// `Unresolved`, which keeps the payout tracked for the next cycle's poll. +pub fn submit_verdict(resp: &str) -> SubmitVerdict { + let Ok(j) = serde_json::from_str::(resp) else { + return SubmitVerdict::Unresolved; // "http_error: ..." lands here + }; + if j.get("http_error").is_some() { + return SubmitVerdict::Unresolved; + } + match find_u64(&j, "ret") { + Some(0) => SubmitVerdict::Accepted, + Some(_) => SubmitVerdict::Rejected, + None => SubmitVerdict::Unresolved, + } +} + +/* --------------------------------------------------------------------------- + * The owed ledger. + * + * When a settlement chunk definitively does not happen, the money it carried + * does not simply return to the pot: it is owed to the exact miners that chunk + * named. Dropping the rows and letting the next cycle re-split the whole balance + * over the live PPLNS window hands that money to whoever is mining now - + * including the miners whose chunks DID go through, who are then paid twice for + * the same window while the miners in the failed chunk are paid once, or never. + * + * These rows are therefore persisted and paid FIRST, before a single unit of + * fresh income is split. + * ------------------------------------------------------------------------- */ + +/// Add the rows of a payout that definitively did not happen to the owed ledger. +pub fn owe_rows(owed: &mut Vec<(String, u64)>, rows: &[(String, u64)]) { + for (w, u) in rows { + if *u == 0 { + continue; + } + match owed.iter_mut().find(|(x, _)| x == w) { + Some(e) => e.1 = e.1.saturating_add(*u), + None => owed.push((w.clone(), *u)), + } + } +} + +/// Take off the owed ledger what a payout the pool has now RECORDED carries. +/// +/// Saturating, and only ever downward: a row paying more than is owed (an owed +/// row and a fresh share to the same miner, merged into one action) clears the +/// debt and no more. +pub fn deduct_owed(owed: &mut Vec<(String, u64)>, rows: &[(String, u64)]) { + for (w, u) in rows { + if let Some(e) = owed.iter_mut().find(|(x, _)| x == w) { + e.1 = e.1.saturating_sub(*u); + } + } + owed.retain(|(_, u)| *u > 0); +} + +/// The rows this cycle must pay BEFORE it splits any fresh income, and what is +/// left to split after them. +/// +/// Owed rows are taken in order and partially where the balance runs out, so one +/// large debt cannot starve while smaller ones keep being paid around it. What is +/// not taken stays on the ledger for the next cycle. +pub fn take_owed(owed: &[(String, u64)], distributable: u64) -> (Vec<(String, u64)>, u64) { + let mut left = distributable; + let mut rows: Vec<(String, u64)> = Vec::new(); + for (w, u) in owed { + if left == 0 { + break; + } + let pay = (*u).min(left); + if pay == 0 { + continue; + } + left -= pay; + rows.push((w.clone(), pay)); + } + (rows, left) +} + +/// Fold rows paying the same address into one action, keeping first-seen order. +/// +/// An owed row and a fresh share for the same miner would otherwise be two +/// actions in one transaction, and each action costs against the node's +/// TX_ACTIONS_MAX limit that `PAYOUT_CHUNK` is sized against. +pub fn merge_payout_rows(rows: &mut Vec<(String, u64)>) { + let mut at: HashMap = HashMap::with_capacity(rows.len()); + let mut out: Vec<(String, u64)> = Vec::with_capacity(rows.len()); + for (w, u) in rows.drain(..) { + match at.get(&w) { + Some(i) => out[*i].1 = out[*i].1.saturating_add(u), + None => { + at.insert(w.clone(), out.len()); + out.push((w, u)); + } + } + } + *rows = out; +} + +/// The owed ledger as it is stored in the pool state file. +pub fn owed_to_json(owed: &[(String, u64)]) -> Value { + Value::Array( + owed.iter() + .map(|(w, u)| serde_json::json!([w, u])) + .collect(), + ) +} + +/// Read the owed ledger out of an already-parsed state document. A row the file +/// cannot describe is dropped rather than guessed at: an unreadable amount must +/// never become a payment. +pub fn parse_owed(j: &Value) -> Vec<(String, u64)> { + j.get("owed") + .and_then(|v| v.as_array()) + .map(|a| { + a.iter() + .filter_map(|r| { + let x = r.as_array()?; + let w = x.first()?.as_str()?.to_string(); + let u = x.get(1)?.as_u64()?; + (!w.is_empty() && u > 0).then_some((w, u)) + }) + .collect() + }) + .unwrap_or_default() +} + +/// The owed ledger. Losing it means the miners in a failed chunk are never paid +/// for that window, so it is persisted with everything else. +pub fn load_owed(state_file: &str) -> Vec<(String, u64)> { + let Some(j) = read_state_json(state_file) else { + return Vec::new(); + }; + parse_owed(&j) +} + /// The per-transaction rows of every payout this pool has in flight. pub fn load_payout_records(state_file: &str) -> Vec { let Some(j) = read_state_json(state_file) else { @@ -661,21 +1407,25 @@ pub fn parse_paid_ledger(j: &Value) -> PaidLedger { j.get("paid").map(PaidLedger::from_json).unwrap_or_default() } -/// Replace the WHOLE settlement ledger (pending hashes, per-transaction rows and -/// confirmed totals) in the pool state file, preserving every other field. +/// Replace the WHOLE settlement ledger (pending hashes, per-transaction rows, +/// what is still owed, and confirmed totals) in the pool state file, preserving +/// every other field. /// -/// One write, because the three move together: a payout leaves the in-flight -/// rows at the same instant it enters the paid totals, and a crash between the -/// two would either lose a payment or count it twice. +/// One write, because the four move together: a payout leaves the in-flight rows +/// at the same instant it enters the paid totals, and a chunk that failed leaves +/// them at the same instant its rows become owed. A crash between any two would +/// either lose a payment or count it twice. pub fn save_settlement_ledger( state_file: &str, hashes: &[String], records: &[PayoutRecord], + owed: &[(String, u64)], paid: &PaidLedger, ) -> std::io::Result<()> { let mut j = read_state_json(state_file).unwrap_or_else(|| serde_json::json!({})); j["settle_pending_txs"] = serde_json::json!(hashes); j["payouts_inflight"] = Value::Array(records.iter().map(|r| r.to_json()).collect()); + j["owed"] = owed_to_json(owed); j["paid"] = paid.to_json(); atomic_write(state_file, j.to_string().as_bytes(), true) } @@ -903,9 +1653,14 @@ fn encrypt_key_hex(key_hex: &str, pass: &str) -> Result { } fn envelope_u32(j: &Value, key: &str, default: u32, max: u32) -> Result { - let v = j.get(key).and_then(|v| v.as_u64()).unwrap_or(default as u64); + let v = j + .get(key) + .and_then(|v| v.as_u64()) + .unwrap_or(default as u64); if v == 0 || v > max as u64 { - return Err(format!("its `{key}` is outside the range this build accepts")); + return Err(format!( + "its `{key}` is outside the range this build accepts" + )); } Ok(v as u32) } @@ -970,14 +1725,23 @@ fn decrypt_key_hex(body: &str, pass: &str) -> Result, Envelope let key = wallet_derive_key( pass, &salt, - envelope_u32(&j, "kdf_m_cost_kb", WALLET_KDF_M_COST_KB, WALLET_KDF_MAX_M_COST_KB) - .map_err(EnvelopeError::Shape)?, + envelope_u32( + &j, + "kdf_m_cost_kb", + WALLET_KDF_M_COST_KB, + WALLET_KDF_MAX_M_COST_KB, + ) + .map_err(EnvelopeError::Shape)?, envelope_u32(&j, "kdf_t_cost", WALLET_KDF_T_COST, WALLET_KDF_MAX_T_COST) .map_err(EnvelopeError::Shape)?, envelope_u32(&j, "kdf_p_cost", WALLET_KDF_P_COST, WALLET_KDF_MAX_P_COST) .map_err(EnvelopeError::Shape)?, ) - .map_err(|e| shape(format!("its key-derivation settings cannot be used here ({e})")))?; + .map_err(|e| { + shape(format!( + "its key-derivation settings cannot be used here ({e})" + )) + })?; let cipher = Aes256Gcm::new_from_slice(&*key) .map_err(|e| shape(format!("its cipher key could not be set up ({e})")))?; // The one check that cannot say WHY it failed. @@ -1372,7 +2136,9 @@ fn migrate_key_file_to_encrypted(path: &str, key_hex: &str, pass: &str) -> Resul match decrypt_key_hex(&body, pass) { Ok(back) if back.trim().eq_ignore_ascii_case(key_hex) => {} _ => { - eprintln!("[wallet] WARNING: the encrypted form of {path} did not verify; leaving it as-is."); + eprintln!( + "[wallet] WARNING: the encrypted form of {path} did not verify; leaving it as-is." + ); return Ok(()); } } @@ -1587,7 +2353,9 @@ fn windows_sid_of(principal: &str) -> Option { #[cfg(windows)] fn windows_verify_owner_only(path: &str, name: &str, sid: &str) -> std::io::Result<()> { let unverified = |why: &str| { - eprintln!("[wallet] WARNING: could not verify the ACL of {path} ({why}); check it manually."); + eprintln!( + "[wallet] WARNING: could not verify the ACL of {path} ({why}); check it manually." + ); }; let Ok(out) = std::process::Command::new("icacls").arg(path).output() else { unverified("icacls did not run"); @@ -1701,6 +2469,61 @@ pub struct Template { pub txs: Arc, } +/// The header timestamp a pool has ALREADY handed out for a height, so a restart +/// can reproduce the same header bytes instead of inventing new ones. +/// +/// The stamp is `max(now, prev_ts + 1)` at the moment the template is first +/// fetched, and it lives in the 89-byte header every worker hashes. A pool pins +/// one template per height, so a restart part-way through a height would +/// otherwise serve a DIFFERENT header for the SAME height: measured on a rig, a +/// restart at height 350 served a stamp 68 seconds later than the one already in +/// flight. `/query/miner/notice` signals only a HEIGHT change, so nothing tells a +/// worker to reload - it goes on hashing the dead header until its current scan +/// pass ends, and every share it finds in the meantime is thrown away. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct StampPin { + pub height: u64, + pub prevhash: Hash, + pub timestamp: u64, +} + +/// The header timestamp to serve for `height` on `prevhash`. +/// +/// `pin` is honoured ONLY when it describes this exact block and would produce a +/// header the node still accepts. Both guards cost a whole block reward when they +/// are wrong: `chain::verify::block_verify` rejects a block whose timestamp is +/// `<= prev_blk_time` or `> curtimes()`, and a rejected block is reported +/// asynchronously, so the pool would mine a full round into nothing and only +/// notice because the tip never reached its height. +/// +/// The upper guard is `fresh`, not `now`: a pin may never move the stamp FORWARD +/// past what a fresh fetch would produce, so a stale or tampered state file +/// cannot talk the pool into mining a future-stamped block. Moving it BACKWARD to +/// a stamp this pool already served is exactly what a pool that never restarted +/// would be serving. +pub fn template_timestamp( + pin: Option<&StampPin>, + height: u64, + prevhash: &Hash, + prev_ts: u64, + now: u64, +) -> u64 { + let fresh = std::cmp::max(now, prev_ts.saturating_add(1)); + let Some(pin) = pin else { + return fresh; + }; + // A pin from another block says nothing about this one. Height alone is not + // enough: after a same-height reorg the parent differs, and the old stamp was + // computed against a parent whose timestamp this block no longer follows. + if pin.height != height || pin.prevhash != *prevhash { + return fresh; + } + if pin.timestamp <= prev_ts || pin.timestamp > fresh { + return fresh; + } + pin.timestamp +} + /// Read the chain tip and build a template for the next block, computing the /// next difficulty off-node with the same rule the node will validate against. /// @@ -1712,6 +2535,18 @@ pub fn fetch_template( base: &str, coinbase_addr: &str, params: &ChainParams, +) -> Option