diff --git a/.grok/workflows/review-miner-pool-delta.rhai b/.grok/workflows/review-miner-pool-delta.rhai new file mode 100644 index 00000000..8202dcd7 --- /dev/null +++ b/.grok/workflows/review-miner-pool-delta.rhai @@ -0,0 +1,496 @@ +// Delta review: re-check prior confirmed bugs + scan for new ones. +// Compares against the 15 findings from review-miner-pool (2026-07-24). + +let meta = #{ + name: "review-miner-pool-delta", + description: "Re-verify prior miner/pool bugs and hunt for new ones; produce fixed vs open vs new report", + when_to_use: "After miner/pool fixes; track regression and remaining open issues", + phases: [ + #{ title: "Recheck", detail: "adversarially re-verify each prior confirmed bug" }, + #{ title: "Fresh scan", detail: "parallel hunt for NEW bugs not in the prior list" }, + #{ title: "Synthesize", detail: "delta report: fixed / still open / new" }, + ], +}; + +let status_schema = #{ + "type": "object", + "required": ["status", "reason", "evidence"], + "properties": #{ + "status": #{ "type": "string" }, + "reason": #{ "type": "string" }, + "evidence": #{ "type": "string" }, + }, +}; + +let findings_schema = #{ + "type": "object", + "required": ["findings"], + "properties": #{ + "findings": #{ + "type": "array", + "maxItems": 8, + "items": #{ + "type": "object", + "required": ["severity", "file", "issue", "impact"], + "properties": #{ + "severity": #{ "type": "string" }, + "file": #{ "type": "string" }, + "issue": #{ "type": "string" }, + "impact": #{ "type": "string" }, + }, + }, + }, + }, +}; + +let verdict_schema = #{ + "type": "object", + "required": ["real", "reason", "evidence"], + "properties": #{ + "real": #{ "type": "boolean" }, + "reason": #{ "type": "string" }, + "evidence": #{ "type": "string" }, + }, +}; + +let report_schema = #{ + "type": "object", + "required": ["summary", "fixed_count", "still_open_count", "new_count", "markdown"], + "properties": #{ + "summary": #{ "type": "string" }, + "fixed_count": #{ "type": "integer" }, + "still_open_count": #{ "type": "integer" }, + "new_count": #{ "type": "integer" }, + "markdown": #{ "type": "string" }, + }, +}; + +let root = "C:/Users/KQHEX/Documents/hacash-fullnodedev"; +if args != () && args.root != () { + root = args.root; +} + +log("delta review root=" + root); + +// Prior confirmed bugs from review-miner-pool (adversarially verified). +// Each recheck agent must set status to: fixed | still_open | unclear +let prior = [ + #{ + id: "P1", + severity: "high", + file: "app/src/block_mining_runtime.rs winners coalesce", + issue: "Winners coalesced only by height; same-height reorg stale result can out-rank and replace a live valid solution; no epoch/template id on BlockMiningResult; submit without live-template revalidation.", + }, + #{ + id: "P2", + severity: "high", + file: "app/src/poworker.rs template install", + issue: "Template install only when pending_height > curr_hei or same-height intro_changed; reorg that LOWERs pending height is ignored so workers never job-switch.", + }, + #{ + id: "P3", + severity: "high", + file: "x16rs/opencl/x16rs_diamond.cl reduction", + issue: "Diamond OpenCL reduction initializes best_hash=0 and leaves best_name uninitialized; scans i=1..; wrong winners vs block kernel seed-from-index pattern.", + }, + #{ + id: "P4", + severity: "high", + file: "x16rs/opencl/x16rs_diamond.cl fences", + issue: "Diamond kernel uses only CLK_LOCAL_MEM_FENCE while hashes live in global memory; block kernel uses LOCAL|GLOBAL.", + }, + #{ + id: "P5", + severity: "high", + file: "app/src/diaworker.rs diamond submit", + issue: "push_diamond_mining_success treats any HTTP Ok body as final; non-JSON/missing tx_hash fails permanently without retry; no durable requeue after drain.", + }, + #{ + id: "P6", + severity: "medium", + file: "app/src/poworker.rs LAST_PENDING_INTRO", + issue: "LAST_PENDING_INTRO overwritten BEFORE set_pending_block_stuff succeeds; failed same-height install never retried.", + }, + #{ + id: "P7", + severity: "medium", + file: "app/src/block_mining_runtime.rs send before job-switch", + issue: "Job-switch height/epoch checked only AFTER full batch and send(); stale results always enter channel; drain does not filter by live epoch.", + }, + #{ + id: "P8", + severity: "medium", + file: "app/src/block_mining_runtime.rs worker rollover", + issue: "Worker rollover uses check_hei > mining_hei (advance-only) plus epoch; height decrease without epoch path may not stop workers.", + }, + #{ + id: "P9", + severity: "medium", + file: "x16rs/opencl/sha3_256.cl diamond length", + issue: "Host builds 61-byte stuff for diamond numbers <= 20000 but sha3_256_hash_diamond always pads as 93-byte; GPU SHA3 diverges from CPU/consensus.", + }, + #{ + id: "P10", + severity: "medium", + file: "app/src/opencl_dia.rs medium hash verify", + issue: "Diamond GPU success path never recomputes x16rs_hash and byte-compares GPU medium hash (unlike block verify_gpu_best_result).", + }, + #{ + id: "P11", + severity: "medium", + file: "x16rs-cuda batch launch", + issue: "CUDA batch mining launches with fixed local_size 256 and never calls clamped_block_size; single-hash path clamps.", + }, + #{ + id: "P12", + severity: "medium", + file: "app/src/diaworker.rs submit durability", + issue: "After MAX_SUBMIT_ATTEMPTS network failure or soft parse failure, drained DiamondMint is dropped with only a log; no local save/requeue.", + }, + #{ + id: "P13", + severity: "medium", + file: "x16rs/opencl/x16rs_diamond.cl unit0", + issue: "Per-work-item reduction never diamond_hashs unit index 0 into best_name before comparisons.", + }, + #{ + id: "P14", + severity: "low", + file: "app/src/block_mining_runtime.rs total_nonce_space", + issue: "Drain aggregate adds planned res.nonce_space even when recovery only mined partial window; telemetry misleading.", + }, + #{ + id: "P15", + severity: "low", + file: "app/src/opencl_gpu/block.rs stuff length", + issue: "OpenCL block upload accepts stuff length <= 512 without enforcing 89-byte block intro that CUDA requires.", + }, +]; + +// Also re-check consensus items that should already be fixed from earlier sessions +let prior_extra = [ + #{ + id: "C1", + severity: "critical", + file: "x16rs/src/diamond.rs", + issue: "Testnet hack: DMD_L=4 DMD_M=10 instead of mainnet DMD_L=10 DMD_M=16.", + }, + #{ + id: "C2", + severity: "critical", + file: "mint/src/action/diamond_mint.rs", + issue: "Diamond height%5 rule disabled with if false.", + }, + #{ + id: "C3", + severity: "high", + file: "app/src mining winners/equal target/GPU fatal", + issue: "Old bugs: only single global best submit; equal-to-target dropped; silent CPU fallback on GPU fail; CUDA no verify; full CPU recovery.", + }, + #{ + id: "C4", + severity: "high", + file: "miner-pool rpc_proxy/stratum", + issue: "Non-JSON upstream body reported as ret:0 success; stratum used substring ret check.", + }, +]; + +// --- Phase 1: recheck all prior items --- +phase("Recheck"); + +let recheck_jobs = []; +let all_prior = []; +for p in prior { + all_prior.push(p); +} +for p in prior_extra { + all_prior.push(p); +} + +for p in all_prior { + let rp = ""; + rp += "READ-ONLY recheck of a previously reported bug in " + root + ".\n"; + rp += "Bug id: " + p.id + "\n"; + rp += "Prior severity: " + p.severity + "\n"; + rp += "File/area: " + p.file + "\n"; + rp += "Claimed issue: " + p.issue + "\n\n"; + rp += "Open the relevant source with read_file/grep. Decide status:\n"; + rp += "- fixed: the bug is no longer present; code clearly handles the case (quote the fix).\n"; + rp += "- still_open: the buggy logic is still there (quote the evidence).\n"; + rp += "- unclear: cannot determine (explain what was missing).\n"; + rp += "status MUST be exactly one of: fixed, still_open, unclear.\n"; + rp += "evidence must cite path and what you saw. reason is one short sentence.\n"; + recheck_jobs.push(#{ + prompt: rp, + label: "recheck:" + p.id, + capability_mode: "read-only", + output_schema: status_schema, + }); +} + +let recheck_results = parallel(recheck_jobs); + +let fixed_list = []; +let open_list = []; +let unclear_list = []; +let ri = 0; +for r in recheck_results { + let p = all_prior[ri]; + let st = "unclear"; + let reason = "recheck agent failed"; + let evidence = ""; + if r != () && r.success && r.output.status != () { + st = r.output.status; + if r.output.reason != () { + reason = r.output.reason; + } + if r.output.evidence != () { + evidence = r.output.evidence; + } + } + // normalize + if st != "fixed" && st != "still_open" && st != "unclear" { + // allow minor variants + if st == "FIXED" || st == "Fixed" { + st = "fixed"; + } else if st == "still open" || st == "open" || st == "STILL_OPEN" { + st = "still_open"; + } else { + st = "unclear"; + } + } + let row = #{ + id: p.id, + severity: p.severity, + file: p.file, + issue: p.issue, + status: st, + reason: reason, + evidence: evidence, + }; + if st == "fixed" { + fixed_list.push(row); + } else if st == "still_open" { + open_list.push(row); + } else { + unclear_list.push(row); + } + ri += 1; +} + +log("recheck: fixed=" + fixed_list.len().to_string() + + " open=" + open_list.len().to_string() + + " unclear=" + unclear_list.len().to_string()); + +// --- Phase 2: fresh scan for NEW bugs --- +phase("Fresh scan"); + +let areas = [ + #{ + id: "block-miner", + paths: "app/src/block_mining_runtime.rs, app/src/mining_batch.rs, app/src/poworker.rs, app/src/mining_runtime.rs, app/src/hash_util.rs", + focus: "reorg job install, winner epoch tagging, equal target, nonce advance, GPU fail-closed, races", + }, + #{ + id: "gpu-diamond", + paths: "x16rs/opencl/x16rs_diamond.cl, x16rs/opencl/x16rs_main.cl, x16rs/opencl/sha3_256.cl, app/src/opencl_dia.rs, app/src/opencl_gpu/, app/src/mining_batch.rs, x16rs-cuda/", + focus: "diamond/block kernel reduction, fences, SHA3 length, CUDA clamp, integrity verify, buffer sizes", + }, + #{ + id: "hacd-pool", + paths: "app/src/diaworker.rs, x16rs/src/diamond.rs, mint/src/action/diamond_mint.rs, mint/src/check/block_build.rs, miner-pool/src/", + focus: "diamond consensus mainnet, submit durability, pool ret handling, stratum parse", + }, +]; + +// Build exclusion summary so scanners avoid re-reporting known items +let known_summary = ""; +for p in all_prior { + known_summary += p.id + ": " + p.issue + " | "; +} + +let scan_jobs = []; +for a in areas { + let sp = ""; + sp += "READ-ONLY bug hunt for NEW issues only in " + root + ".\n"; + sp += "Area: " + a.id + "\n"; + sp += "Paths: " + a.paths + "\n"; + sp += "Focus: " + a.focus + "\n\n"; + sp += "ALREADY REPORTED (do NOT re-list these unless you found a DISTINCT new facet):\n"; + sp += known_summary + "\n\n"; + sp += "Use read_file/grep. Prefer real correctness bugs. severity: critical|high|medium|low.\n"; + sp += "Max 8 findings. Empty list valid after inspection. Return findings {severity,file,issue,impact}.\n"; + scan_jobs.push(#{ + prompt: sp, + label: "scan:" + a.id, + capability_mode: "read-only", + output_schema: findings_schema, + }); +} + +let scan_results = parallel(scan_jobs); + +let new_raw = []; +let si = 0; +for r in scan_results { + let area = areas[si].id; + if r != () && r.success && r.output.findings != () { + for f in r.output.findings { + new_raw.push(#{ + area: area, + severity: f.severity, + file: f.file, + issue: f.issue, + impact: f.impact, + }); + } + } + si += 1; +} + +log("fresh scan raw new candidates: " + new_raw.len().to_string()); + +// Verify new candidates (cap 12) +let MAX_NEW_V = 12; +let new_to_v = []; +let ndrop = 0; +for f in new_raw { + if new_to_v.len() < MAX_NEW_V { + new_to_v.push(f); + } else { + ndrop += 1; + } +} +if ndrop > 0 { + log("capped new verify list, dropped " + ndrop.to_string()); +} + +let new_confirmed = []; +if new_to_v.len() > 0 { + let vjobs = []; + for f in new_to_v { + let vp = ""; + vp += "Adversarially verify this alleged NEW bug under " + root + ".\n"; + vp += "Area: " + f.area + "\nFile: " + f.file + "\nIssue: " + f.issue + "\nImpact: " + f.impact + "\n"; + vp += "Prior known bugs (must not confirm duplicates of these):\n" + known_summary + "\n"; + vp += "Set real=true only if: (1) bug still exists in code, (2) it is NOT the same as a prior known bug, (3) evidence is concrete.\n"; + vp += "Set real=false if fixed, duplicate of known list, wrong, or style-only.\n"; + vjobs.push(#{ + prompt: vp, + label: "verify-new:" + f.area, + capability_mode: "read-only", + output_schema: verdict_schema, + }); + } + let vres = parallel(vjobs); + let vi = 0; + for v in vres { + let f = new_to_v[vi]; + if v != () && v.success && v.output.real == true + && v.output.evidence != () && v.output.evidence != "" { + new_confirmed.push(#{ + area: f.area, + severity: f.severity, + file: f.file, + issue: f.issue, + impact: f.impact, + evidence: v.output.evidence, + reason: v.output.reason, + }); + } + vi += 1; + } +} + +log("new confirmed bugs: " + new_confirmed.len().to_string()); + +// --- Phase 3: synthesize --- +phase("Synthesize"); + +let fixed_json = json_encode(fixed_list); +let open_json = json_encode(open_list); +let unclear_json = json_encode(unclear_list); +let new_json = json_encode(new_confirmed); + +let yp = ""; +yp += "Write a delta bug review report in GitHub-flavored markdown for " + root + ".\n\n"; +yp += "FIXED prior bugs (JSON):\n" + fixed_json + "\n\n"; +yp += "STILL OPEN prior bugs (JSON):\n" + open_json + "\n\n"; +yp += "UNCLEAR prior rechecks (JSON):\n" + unclear_json + "\n\n"; +yp += "NEW confirmed bugs (JSON):\n" + new_json + "\n\n"; +yp += "Sections required:\n"; +yp += "1. Summary (counts: fixed / still open / unclear / new)\n"; +yp += "2. Fixed (table id|severity|file|note)\n"; +yp += "3. Still open (table id|severity|file|issue)\n"; +yp += "4. New bugs (table severity|file|issue|impact)\n"; +yp += "5. Unclear (if any)\n"; +yp += "6. Recommended next fix order\n"; +yp += "Return JSON: summary, fixed_count, still_open_count, new_count, markdown.\n"; +yp += "fixed_count=" + fixed_list.len().to_string() + + " still_open_count=" + open_list.len().to_string() + + " new_count=" + new_confirmed.len().to_string() + "\n"; + +let synth = agent(yp, #{ + label: "synthesize-delta", + capability_mode: "read-only", + output_schema: report_schema, +}); + +let md = "# Miner+Pool Delta Review\n\n"; +let summary = "Delta review complete."; +let fc = fixed_list.len(); +let oc = open_list.len(); +let nc = new_confirmed.len(); + +if synth != () && synth.success && synth.output.markdown != () { + md = synth.output.markdown; + if synth.output.summary != () { + summary = synth.output.summary; + } + if synth.output.fixed_count != () { + fc = synth.output.fixed_count; + } + if synth.output.still_open_count != () { + oc = synth.output.still_open_count; + } + if synth.output.new_count != () { + nc = synth.output.new_count; + } +} else { + md += "## Summary\n\nFixed " + fc.to_string() + ", still open " + oc.to_string() + + ", unclear " + unclear_list.len().to_string() + + ", new " + nc.to_string() + ".\n\n"; + md += "## Fixed\n\n"; + for x in fixed_list { + md += "- **" + x.id + "** (" + x.severity + ") `" + x.file + "` — " + x.reason + "\n"; + } + md += "\n## Still open\n\n"; + for x in open_list { + md += "- **" + x.id + "** (" + x.severity + ") `" + x.file + "` — " + x.issue + "\n"; + } + md += "\n## New\n\n"; + for x in new_confirmed { + md += "- **" + x.severity + "** `" + x.file + "` — " + x.issue + "\n"; + } + md += "\n## Unclear\n\n"; + for x in unclear_list { + md += "- **" + x.id + "** — " + x.reason + "\n"; + } + summary = "Fixed " + fc.to_string() + " / open " + oc.to_string() + " / new " + nc.to_string(); +} + +let path = write_scratch_file("miner-pool-delta-review.md", md); +log("delta report: " + path); + +complete(#{ + summary: summary, + fixed_count: fc, + still_open_count: oc, + unclear_count: unclear_list.len(), + new_count: nc, + path: path, + fixed: fixed_list, + still_open: open_list, + unclear: unclear_list, + new_bugs: new_confirmed, +}); diff --git a/.grok/workflows/review-miner-pool.rhai b/.grok/workflows/review-miner-pool.rhai new file mode 100644 index 00000000..8e06e857 --- /dev/null +++ b/.grok/workflows/review-miner-pool.rhai @@ -0,0 +1,301 @@ +// Multi-agent bug review of the Hacash miner + pool stack. +// Parallel dimension reviewers → adversarial verify → synthesis report. + +let meta = #{ + name: "review-miner-pool", + description: "Parallel bug review of miner (poworker/diaworker/OpenCL/CUDA) and pool (hac-pool/stratum), with adversarial verification", + when_to_use: "After miner or pool changes; pre-release audit of mining correctness and consensus", + phases: [ + #{ title: "Review", detail: "one reviewer per area of the miner/pool stack" }, + #{ title: "Verify", detail: "adversarial check of each claimed finding" }, + #{ title: "Synthesize", detail: "single ranked bug report" }, + ], +}; + +let findings_schema = #{ + "type": "object", + "required": ["findings"], + "properties": #{ + "findings": #{ + "type": "array", + "maxItems": 10, + "items": #{ + "type": "object", + "required": ["severity", "file", "issue", "impact"], + "properties": #{ + "severity": #{ "type": "string" }, + "file": #{ "type": "string" }, + "issue": #{ "type": "string" }, + "impact": #{ "type": "string" }, + }, + }, + }, + }, +}; + +let verdict_schema = #{ + "type": "object", + "required": ["real", "reason", "evidence"], + "properties": #{ + "real": #{ "type": "boolean" }, + "reason": #{ "type": "string" }, + "evidence": #{ "type": "string" }, + }, +}; + +let report_schema = #{ + "type": "object", + "required": ["summary", "confirmed_count", "open_count", "markdown"], + "properties": #{ + "summary": #{ "type": "string" }, + "confirmed_count": #{ "type": "integer" }, + "open_count": #{ "type": "integer" }, + "markdown": #{ "type": "string" }, + }, +}; + +// --- args --- +let root = "C:/Users/KQHEX/Documents/hacash-fullnodedev"; +if args != () && args.root != () { + root = args.root; +} +let max_findings = 8; +if args != () && args.max_findings != () { + max_findings = args.max_findings; +} + +log("review-miner-pool root=" + root); + +// --- Phase 1: parallel area reviews (read-only) --- +phase("Review"); + +let areas = [ + #{ + id: "block-miner", + paths: "app/src/block_mining_runtime.rs, app/src/mining_batch.rs, app/src/poworker.rs, app/src/hash_util.rs, app/src/mining_runtime.rs", + focus: "block PoW: job switch, nonce advance, winner submit (all heights), equal-inclusive target, GPU init fail-closed, hashrate accounting, race conditions", + }, + #{ + id: "gpu-backends", + paths: "app/src/opencl_gpu/, app/src/mining_batch.rs, app/src/cuda_pow.rs, app/src/opencl_dia.rs, x16rs/opencl/, x16rs-cuda/", + focus: "OpenCL/CUDA integrity verify, buffer OOB, bounded recovery vs full CPU fallback, custom_nonce gating for diamonds, kernel/host mismatch", + }, + #{ + id: "diamond-consensus", + paths: "x16rs/src/diamond.rs, app/src/diaworker.rs, mint/src/action/diamond_mint.rs, mint/src/check/block_build.rs, mint/src/check/block_accept.rs, mint/src/api/", + focus: "mainnet DMD_L=10 DMD_M=16, height%5 diamond rule, name extraction via check_diamond_hash_result (not hardcoded slices), submit retries, testnet hacks left behind", + }, + #{ + id: "pool", + paths: "miner-pool/src/ (main, rpc_proxy, stratum, upstream, job, config), pool-spike/src/ if present", + focus: "submit success/failure semantics (non-JSON must not be ret:0), stratum ret parsing, auth token, share attribution, stale job handling", + }, +]; + +let review_jobs = []; +for a in areas { + let p = ""; + p += "You are a senior mining-protocol code reviewer doing a READ-ONLY bug hunt.\n"; + p += "Repository root: " + root + "\n"; + p += "Area id: " + a.id + "\n"; + p += "Focus: " + a.focus + "\n"; + p += "Primary paths (relative to root): " + a.paths + "\n\n"; + p += "REQUIRED process:\n"; + p += "1. Use read_file and grep on the real files under the root. Do NOT invent findings from memory.\n"; + p += "2. Prefer REAL correctness/security bugs over style nits.\n"; + p += "3. For each finding set severity to one of: critical, high, medium, low.\n"; + p += "4. file must be a concrete path like app/src/foo.rs:LINE or path without line if range.\n"; + p += "5. At most " + max_findings.to_string() + " findings. Empty findings is valid ONLY after you inspected the paths.\n"; + p += "6. If code already has an explicit fix comment for a historical bug, do not re-report it as open unless the bug still exists.\n"; + p += "7. Return JSON matching the schema: findings array of {severity, file, issue, impact}.\n"; + review_jobs.push(#{ + prompt: p, + label: "review:" + a.id, + capability_mode: "read-only", + output_schema: findings_schema, + }); +} + +let review_results = parallel(review_jobs); + +let all_findings = []; +let area_i = 0; +for r in review_results { + let area_id = areas[area_i].id; + if r == () || !r.success { + log("review panel failed or empty for " + area_id); + } else if r.output.findings != () { + for f in r.output.findings { + // Tag area for later synthesis + let item = #{ + area: area_id, + severity: f.severity, + file: f.file, + issue: f.issue, + impact: f.impact, + }; + all_findings.push(item); + } + } + area_i += 1; +} + +log("raw findings collected: " + all_findings.len().to_string()); + +if all_findings.len() == 0 { + let empty_md = "# Miner + Pool Review\n\nNo findings after parallel area reviews (block-miner, gpu-backends, diamond-consensus, pool).\n"; + let path = write_scratch_file("miner-pool-review.md", empty_md); + complete(#{ + summary: "No findings from area reviewers.", + confirmed_count: 0, + open_count: 0, + path: path, + report: empty_md, + }); +} + +// Cap verification fan-out to keep budget headroom for synthesis +let MAX_VERIFY = 16; +let to_verify = []; +let dropped = 0; +let fi = 0; +for f in all_findings { + if to_verify.len() < MAX_VERIFY { + to_verify.push(f); + } else { + dropped += 1; + } + fi += 1; +} +if dropped > 0 { + log("capped verify list: dropped " + dropped.to_string() + " extra raw findings"); +} + +// --- Phase 2: adversarial verify --- +phase("Verify"); + +let vjobs = []; +for f in to_verify { + let vp = ""; + vp += "Adversarially verify this alleged bug against the REAL code under " + root + ".\n"; + vp += "Claimed area: " + f.area + "\n"; + vp += "File: " + f.file + "\n"; + vp += "Issue: " + f.issue + "\n"; + vp += "Impact claimed: " + f.impact + "\n"; + vp += "Severity claimed: " + f.severity + "\n\n"; + vp += "You MUST open the file(s) with read_file/grep and check whether the bug still exists.\n"; + vp += "Set real=true ONLY if you independently confirm the bug is still present with concrete evidence (quote or describe the exact logic).\n"; + vp += "Set real=false if the code already fixes it, the claim is wrong, speculative, or style-only.\n"; + vp += "evidence must cite path and what you saw. reason is a short verdict sentence.\n"; + vjobs.push(#{ + prompt: vp, + label: "verify:" + f.area, + capability_mode: "read-only", + output_schema: verdict_schema, + }); +} + +let verdicts = parallel(vjobs); + +let confirmed = []; +let rejected = []; +let vi = 0; +for v in verdicts { + let f = to_verify[vi]; + if v != () && v.success && v.output.real == true + && v.output.evidence != () && v.output.evidence != "" { + confirmed.push(#{ + area: f.area, + severity: f.severity, + file: f.file, + issue: f.issue, + impact: f.impact, + evidence: v.output.evidence, + reason: v.output.reason, + }); + } else { + let why = "unverified or not real"; + if v != () && v.success && v.output.reason != () { + why = v.output.reason; + } + rejected.push(#{ + area: f.area, + file: f.file, + issue: f.issue, + why: why, + }); + } + vi += 1; +} + +log("confirmed " + confirmed.len().to_string() + " / verified " + to_verify.len().to_string()); + +// --- Phase 3: synthesize markdown report --- +phase("Synthesize"); + +let conf_json = json_encode(confirmed); +let rej_json = json_encode(rejected); + +let sp = ""; +sp += "Synthesize a final miner+pool bug review report in GitHub-flavored markdown.\n"; +sp += "Repository: " + root + "\n\n"; +sp += "CONFIRMED findings (JSON, already adversarially verified — treat as open bugs):\n"; +sp += conf_json + "\n\n"; +sp += "REJECTED / unverified claims (JSON — do not list as open bugs; may mention briefly as closed):\n"; +sp += rej_json + "\n\n"; +sp += "Write markdown with sections:\n"; +sp += "1. Summary (2-4 sentences)\n"; +sp += "2. Confirmed open bugs table: severity | file | issue | impact\n"; +sp += "3. Notes on already-fixed / rejected claims (short)\n"; +sp += "4. Recommended fix order\n"; +sp += "Return JSON: summary, confirmed_count, open_count (same as confirmed), markdown (full report body).\n"; +sp += "confirmed_count and open_count must equal " + confirmed.len().to_string() + ".\n"; + +let synth = agent(sp, #{ + label: "synthesize-report", + capability_mode: "read-only", + output_schema: report_schema, +}); + +let md = "# Miner + Pool Review\n\n"; +let summary = "Review completed."; +let conf_n = confirmed.len(); +let open_n = confirmed.len(); + +if synth != () && synth.success && synth.output.markdown != () { + md = synth.output.markdown; + if synth.output.summary != () { + summary = synth.output.summary; + } + if synth.output.confirmed_count != () { + conf_n = synth.output.confirmed_count; + } + if synth.output.open_count != () { + open_n = synth.output.open_count; + } +} else { + // Fallback local markdown if synthesis agent fails + md += "## Summary\n\n"; + md += "Confirmed " + confirmed.len().to_string() + " findings after adversarial verification.\n\n"; + md += "## Confirmed open bugs\n\n"; + for c in confirmed { + md += "- **" + c.severity + "** `" + c.file + "` — " + c.issue + " (impact: " + c.impact + ")\n"; + } + md += "\n## Rejected\n\n"; + for r in rejected { + md += "- `" + r.file + "` — " + r.issue + " — " + r.why + "\n"; + } + summary = "Confirmed " + confirmed.len().to_string() + " open bugs (fallback report)."; +} + +let path = write_scratch_file("miner-pool-review.md", md); +log("report written: " + path); + +complete(#{ + summary: summary, + confirmed_count: conf_n, + open_count: open_n, + path: path, + confirmed: confirmed, + rejected_count: rejected.len(), +}); diff --git a/Cargo.lock b/Cargo.lock index 899db985..a5f08deb 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -2129,7 +2129,7 @@ dependencies = [ [[package]] name = "hacash" -version = "0.5.2" +version = "0.5.5" dependencies = [ "app", "basis", diff --git a/Cargo.toml b/Cargo.toml index 5e4a7ee3..f8b06c3b 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -1,7 +1,7 @@ [package] name = "hacash" default-run = "hacash" -version = "0.5.2" +version = "0.5.5" edition = "2024" [workspace] diff --git a/app/src/bench_mainnet_repeat16.rs b/app/src/bench_mainnet_repeat16.rs index ed8c5913..15c5c88d 100644 --- a/app/src/bench_mainnet_repeat16.rs +++ b/app/src/bench_mainnet_repeat16.rs @@ -1,6 +1,6 @@ // bench_mainnet_repeat16.rs // -// ADDITIVE MODULE — does not replace or modify any existing file. +// ADDITIVE MODULE: does not replace or modify any existing file. // Drop-in benchmark that measures GPU block-mining throughput at the SAME // X16RS round count the live Hacash mainnet uses (repeat = 16), and prints // every raw number a reviewer needs to reproduce/verify the figure: @@ -197,7 +197,7 @@ pub fn measure_at_height( Ok(used) }; - // Warm-up (JIT / clocks / caches) — not counted. + // Warm-up (JIT / clocks / caches), not counted. let mut nonce = 0u32; for w in 0..WARMUP_BATCHES { run_batch(nonce).map_err(|e| format!("warm-up batch {} failed: {e}", w + 1))?; @@ -262,8 +262,8 @@ pub fn run_repeat_comparison( unitsize: &u32, seconds: u64, ) { - println!("[repeat16] Mainnet-representative benchmark (x16rs repeat=16)."); - println!( + wlogln!("[repeat16] Mainnet-representative benchmark (x16rs repeat=16)."); + wlogln!( "[repeat16] Every measured batch is CPU-verified with x16rs::block_hash before it counts." ); @@ -281,11 +281,11 @@ pub fn run_repeat_comparison( false, ); if resources.is_empty() { - println!("[repeat16] No OpenCL devices."); + wlogln!("[repeat16] No OpenCL devices."); return; } if resources.len() != 1 { - println!( + wlogln!( "[repeat16] Detected {} devices. Run one device at a time (set [gpu] device_ids to a single id) so the number is unambiguous.", resources.len() ); @@ -311,34 +311,34 @@ pub fn run_repeat_comparison( seconds, ); - println!("\n================ REPEAT-16 (MAINNET) ================"); + wlogln!("\n================ REPEAT-16 (MAINNET) ================"); match &r16 { - Ok(rep) => println!("{}", rep.render()), - Err(e) => println!(" REJECTED: {e}"), + Ok(rep) => wlogln!("{}", rep.render()), + Err(e) => wlogln!(" REJECTED: {e}"), } - println!("\n================ REPEAT-1 (auto-tune reference) ===="); + wlogln!("\n================ REPEAT-1 (auto-tune reference) ===="); match &r1 { - Ok(rep) => println!("{}", rep.render()), - Err(e) => println!(" REJECTED: {e}"), + Ok(rep) => wlogln!("{}", rep.render()), + Err(e) => wlogln!(" REJECTED: {e}"), } if let (Ok(a), Ok(b)) = (&r16, &r1) { if a.nonces_per_sec > 0.0 { let ratio = b.nonces_per_sec / a.nonces_per_sec; - println!("\n================ SUMMARY ============================"); - println!( + wlogln!("\n================ SUMMARY ============================"); + wlogln!( " repeat=1 : {}\n repeat=16 : {}\n ratio : {:.2}x (expected ~16x; the repeat=1 figure is NOT a mainnet rate)", fmt_rate(b.nonces_per_sec), fmt_rate(a.nonces_per_sec), ratio ); - println!( + wlogln!( " On the live 16-round mainnet, THIS rig produces {} of block hashes.", fmt_rate(a.nonces_per_sec) ); } } - println!("===================================================="); + wlogln!("===================================================="); } /// Optional zero-touch trigger: if HACASH_REPEAT16_BENCH_SECONDS is set to a diff --git a/app/src/block_mining_runtime.rs b/app/src/block_mining_runtime.rs index 53595954..8a24cd6b 100644 --- a/app/src/block_mining_runtime.rs +++ b/app/src/block_mining_runtime.rs @@ -13,12 +13,12 @@ use crate::efficiency::*; use crate::hash_util::{hash_left_zero_pad3, hash_more_power}; // The panic firewall is shared with the diamond (HACD) worker, which has exactly // the same "one result thread owns every submission" shape. -use crate::mining_guard::guard_mining_iteration; #[cfg(feature = "cuda")] use crate::mining_batch::CudaBlockBackend; #[cfg(feature = "ocl")] use crate::mining_batch::OpenclBlockBackend; use crate::mining_batch::{BatchCtx, BlockMinerBackend, CpuBlockBackend}; +use crate::mining_guard::guard_mining_iteration; use basis::difficulty::*; use basis::interface::*; @@ -292,7 +292,7 @@ fn report_share_undersampling(dropped: u64) { { return; } - eprintln!( + wlogerr!( "\n[Mining] UNDERSAMPLING: the GPU share list filled up, so {dropped} payable nonces from this batch were never submitted ({total} so far this session). This is lost income, and it means the pool's share target is far too easy for this card: ask the operator to LOWER share_bits, which makes each share harder. Raising it makes shares easier and loses more. Nothing is wrong with the GPU. This line is rate limited to one per 30 seconds, so it undercounts how often this happens; the session figure is the number that matters." ); } @@ -363,7 +363,7 @@ fn send_mining_result( if result_meets_target(&res) { return result_ch_tx.send(res).is_ok(); } - eprintln!( + wlogerr!( "[Mining] Result queue full, dropped a statistics-only batch at height {}.", res.height ); @@ -582,7 +582,7 @@ fn submit_block_mining_success(cnf: &PoWorkConf, success: &BlockMiningResult) -> // duplicate, busy, invalid): resending this exact // submission cannot help, so stop attempting. _ => { - println!( + wlogln!( "[submit] node rejected height {}: {}", success.height, err ); @@ -596,7 +596,7 @@ fn submit_block_mining_success(cnf: &PoWorkConf, success: &BlockMiningResult) -> // node decision, so treat it as transient and retry: a // winning block is not discarded on a front-end hiccup. let snippet: String = body.chars().take(120).collect(); - println!( + wlogln!( "[submit] attempt {}/{} unrecognized response, retrying: {}", attempt, MAX_SUBMIT_ATTEMPTS, snippet ); @@ -608,7 +608,7 @@ fn submit_block_mining_success(cnf: &PoWorkConf, success: &BlockMiningResult) -> } Err(e) => { last = format!("transport error: {e}"); - println!( + wlogln!( "[submit] attempt {}/{} failed: {e}", attempt, MAX_SUBMIT_ATTEMPTS ); @@ -618,10 +618,10 @@ fn submit_block_mining_success(cnf: &PoWorkConf, success: &BlockMiningResult) -> } } } - println!("{} {}", &urlapi_success, last); + wlogln!("{} {}", &urlapi_success, last); match verdict { SubmitVerdict::Accepted => { - println!( + wlogln!( "\n\n████████████████ [MINING SUCCESS] Find a block height {},\n██ hash {} to submit.", success.height, success.result_hash.to_hex() @@ -632,28 +632,28 @@ fn submit_block_mining_success(cnf: &PoWorkConf, success: &BlockMiningResult) -> // "check the node/connection" alarm for these would cry wolf on every // single share a pool miner earns. SubmitVerdict::ShareCredited => { - println!( + wlogln!( "[submit] pool credited a share at height {} (hash {}).", success.height, success.result_hash.to_hex() ); } SubmitVerdict::UpstreamBusy => { - println!( + wlogln!( "[submit] pool is busy at height {}: pausing submits for it for {}s.", success.height, POOL_BUSY_COOLDOWN.as_secs() ); } SubmitVerdict::DuplicateSubmission => { - println!( + wlogln!( "[submit] height {} was already credited for this nonce (hash {}).", success.height, success.result_hash.to_hex() ); } _ => { - println!( + wlogln!( "\n\n████████████████ [MINING SUBMIT FAILED] block height {} was NOT confirmed accepted\n██ after {} attempts (hash {}). Check the node/connection.", success.height, MAX_SUBMIT_ATTEMPTS, @@ -661,7 +661,7 @@ fn submit_block_mining_success(cnf: &PoWorkConf, success: &BlockMiningResult) -> ); } } - println!("▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔"); + wlogln!("▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔"); verdict } @@ -870,7 +870,7 @@ impl SubmitGate { // report: the submit path already printed its own outcome. if entry.suppressed > 0 { if entry.state == TemplateSubmitState::Settled { - println!( + wlogln!( "\n[Mining] height {} {}, suppressed {} redundant winners.", key.0, entry.outcome, entry.suppressed ); @@ -879,7 +879,7 @@ impl SubmitGate { // held back while a submission was in flight or while the // upstream asked for a pause. Say so instead of calling them // redundant. - println!( + wlogln!( "\n[Mining] height {} {}, held back {} further winners.", key.0, entry.outcome, entry.suppressed ); @@ -945,7 +945,7 @@ fn queue_block_mining_success( match submit_tx.try_send(win.clone()) { Ok(()) => {} Err(mpsc::TrySendError::Full(win)) => { - eprintln!( + wlogerr!( "[Mining] Submit queue full, submitting height {} inline.", win.height ); @@ -1284,7 +1284,7 @@ fn build_gpu_backends(cnf: &PoWorkConf) -> Vec { let cuda_resources = super::initialize_cuda(cnf.cudadevice, cnf.workgroups, cnf.unitsize); if !cuda_resources.is_empty() { - println!( + wlogln!( "\n[Start] Create CUDA block miner worker #{}.", cuda_resources.len() ); @@ -1295,7 +1295,7 @@ fn build_gpu_backends(cnf: &PoWorkConf) -> Vec { } #[cfg(not(feature = "cuda"))] { - println!( + wlogln!( "\n[Warn] use_cuda=true but app built without `cuda` feature, fallback to CPU miner." ); } @@ -1316,7 +1316,7 @@ fn build_gpu_backends(cnf: &PoWorkConf) -> Vec { false, ); if !opencl_resources.is_empty() { - println!( + wlogln!( "\n[Start] Create GPU block miner worker #{}.", opencl_resources.len() ); @@ -1351,7 +1351,7 @@ fn build_gpu_backends(cnf: &PoWorkConf) -> Vec { #[cfg(not(feature = "ocl"))] { - println!( + wlogln!( "\n[Warn] use_opencl=true but app built without `ocl` feature, fallback to CPU miner." ); } @@ -1375,7 +1375,7 @@ fn build_miner_backends( break; } let wait = GPU_INIT_RETRY_DELAYS_SECS[attempt - 1]; - println!( + wlogln!( "\n[Start] GPU not ready yet (attempt {attempt}/{attempts}). Waiting {wait}s and trying again - this is normal shortly after a boot, a resume, or a driver reset." ); if sleep_unless_stopped(stop_flag, Duration::from_secs(wait)) { @@ -1387,7 +1387,7 @@ fn build_miner_backends( // work running instead of taking the whole rig to zero. let assist_threads = cpu_assist_thread_count(cnf, backends.len()); if assist_threads > 0 { - println!( + wlogln!( "\n[Start] Create #{} Ryzen CPU assist threads (hybrid GPU+CPU, active={}).", assist_threads, cnf.runtime.active_cpu_assist.load(Relaxed) @@ -1401,13 +1401,13 @@ fn build_miner_backends( if backends.is_empty() { if gpu_requested { - eprintln!( + wlogerr!( "[Fatal] a GPU miner was requested but no usable GPU backend initialized after {attempts} attempt(s); refusing silent CPU fallback (you would pay for GPU power while mining slowly on the CPU). If the GPU works in other software, wait a minute and press Start again; otherwise check the driver/CUDA runtime, or set the backend to CPU." ); return backends; } let thrnum = cnf.efficiency.clamp_supervene(cnf.supervene.max(1)) as usize; - println!( + wlogln!( "\n[Start] Create #{} CPU block miner worker thread.", thrnum ); @@ -1434,15 +1434,8 @@ fn backend_nonce_space(_cnf: &PoWorkConf, backend: &MinerBackend) -> u32 { // Match run_batch: the planned window must reflect the same effective // work-groups (OOM/error backoff) and thermal cap the batch will use, // otherwise the nonce accounting overstates what the GPU covered. - let thermal = _cnf - .runtime - .thermal_workgroups_cap() - .unwrap_or(u32::MAX); - let wg = res - .effective_wg() - .min(_cnf.workgroups) - .min(thermal) - .max(1); + let thermal = _cnf.runtime.thermal_workgroups_cap().unwrap_or(u32::MAX); + let wg = res.effective_wg().min(_cnf.workgroups).min(thermal).max(1); wg.saturating_mul(x16rs_cuda::DEFAULT_LOCAL_SIZE) .saturating_mul(res.unit_size) .max(1) @@ -1487,7 +1480,7 @@ fn run_block_mining_item( let stuff = match MINING_BLOCK_STUFF.read() { Ok(stuff) => stuff.clone(), Err(e) => { - eprintln!("[Mining] Block state lock failed: {e}"); + wlogerr!("[Mining] Block state lock failed: {e}"); return; } }; @@ -1500,7 +1493,7 @@ fn run_block_mining_item( let mut coinbase_nonce = [0u8; HASH_WIDTH]; if let Err(e) = getrandom::fill(&mut coinbase_nonce) { - eprintln!("[Mining] Secure random nonce failed: {e}"); + wlogerr!("[Mining] Secure random nonce failed: {e}"); return; } let coinbase_nonce = Hash::from(coinbase_nonce); @@ -1705,7 +1698,7 @@ fn drain_winners_for_shutdown( && admit_for_submit(gate, &res) { if let Err(e) = submit_tx.try_send(res.clone()) { - eprintln!( + wlogerr!( "[Mining] Shutdown could not queue a winning result at height {}: {e}", res.height ); @@ -1796,7 +1789,7 @@ fn deal_block_mining_results( *most_hash = most.result_hash.clone(); } let Ok(tarhx) = most.target_hash.clone().try_into() else { - eprintln!("[Mining] Ignoring result with invalid target hash length."); + wlogerr!("[Mining] Ignoring result with invalid target hash length."); return; }; let target_rates = hash_to_rates(&tarhx, TARGET_BLOCK_TIME); @@ -1820,7 +1813,7 @@ fn deal_block_mining_results( .maybe_adjust_supervene(&cnf.efficiency, gpu_nonce_space, cpu_nonce_space); if should_pause_for_profit(&cnf.efficiency, hac1day, &cnf.gpu_profile, active_cpu) { cnf.runtime.paused_unprofitable.store(true, Relaxed); - println!( + wlogln!( "\n[efficiency] Mining paused: estimated cost exceeds HAC revenue. Set pause_if_unprofitable=false or lower power draw." ); } else { @@ -1893,12 +1886,12 @@ pub(crate) fn may_print_turn_to_nex_block_mining(curr_hei: u64, most_hash: Optio *most_hash = vec![255u8; 32]; } let Ok(stuff) = MINING_BLOCK_STUFF.read() else { - eprintln!("[Mining] Cannot read block state."); + wlogerr!("[Mining] Cannot read block state."); return; }; let tarhx = hash_left_zero_pad3(&stuff.target_hash.as_bytes()).to_hex(); - println!( + wlogln!( "\n[{}] req height {} target {} to mining ... ", &ctshow()[5..], mining_hei, diff --git a/app/src/cuda_pow.rs b/app/src/cuda_pow.rs index 2a9d54fd..80d9cf2b 100644 --- a/app/src/cuda_pow.rs +++ b/app/src/cuda_pow.rs @@ -45,7 +45,7 @@ impl CudaMiningResources { /// a clean batch, so throughput recovers once memory pressure clears. pub fn record_success(&self) { if self.quarantine.record_success() { - println!( + wlogln!( "[CUDA] GPU RECOVERED: the re-probe succeeded, quarantine cleared and GPU mining has resumed." ); } @@ -71,7 +71,7 @@ impl CudaMiningResources { // smallest, most likely to succeed grid. self.eff_wg .store(self.floor_wg.max(1), std::sync::atomic::Ordering::Relaxed); - eprintln!( + wlogerr!( "[CUDA] ALERT GPU quarantined after {} consecutive failed batches ({} this session): no GPU work for {}, then the card is re-probed automatically. Mining continues on capped CPU recovery. {} Check the driver, cooling, power and the PCIe riser.", report.consecutive_failures, report.total_failures, @@ -93,7 +93,7 @@ impl CudaMiningResources { notify, } => { if notify { - eprintln!( + wlogerr!( "[CUDA] GPU QUARANTINED (level {level}, {total_failures} failed batches): no GPU work for another {}, mining continues on capped CPU recovery. The card is re-probed automatically, no restart needed.", crate::gpu_oom::format_backoff(retry_in) ); @@ -106,7 +106,7 @@ impl CudaMiningResources { } => { let wg = self.floor_wg.max(1); self.eff_wg.store(wg, std::sync::atomic::Ordering::Relaxed); - println!( + wlogln!( "[CUDA] GPU quarantine (level {level}, {total_failures} failed batches) expired: re-probing the device at work_groups={wg}." ); false @@ -131,7 +131,7 @@ pub fn initialize_cuda( unit_size: u32, ) -> Vec> { if !CudaMiner::is_available() { - eprintln!( + wlogerr!( "[CUDA] x16rs-cuda built without kernels; rebuild with: cargo build -p poworker --features cuda" ); return Vec::new(); @@ -140,21 +140,21 @@ pub fn initialize_cuda( match CudaMiner::list_devices() { Ok(devices) => { for d in &devices { - println!( + wlogln!( "[CUDA] Device #{}: {} (SM {}.{}, MP={})", d.index, d.name, d.compute_major, d.compute_minor, d.multiprocessor_count ); } } Err(e) => { - eprintln!("[CUDA] enumerate devices failed: {e}"); + wlogerr!("[CUDA] enumerate devices failed: {e}"); return Vec::new(); } } match CudaMiner::new(device_index, workgroups, unit_size) { Ok(miner) => { - println!( + wlogln!( "[CUDA] Initialized device #{device_index} work_groups={workgroups} unit_size={unit_size}" ); vec![Arc::new(CudaMiningResources { @@ -167,7 +167,7 @@ pub fn initialize_cuda( })] } Err(e) => { - eprintln!("[CUDA] init failed: {e}"); + wlogerr!("[CUDA] init failed: {e}"); Vec::new() } } diff --git a/app/src/diabider.rs b/app/src/diabider.rs index 2ca420b0..c567b4b9 100644 --- a/app/src/diabider.rs +++ b/app/src/diabider.rs @@ -25,7 +25,7 @@ pub fn start_diamond_auto_bidding(mut worker: Worker, hnode: Arc) { macro_rules! printerr { ( $f: expr, $( $v: expr ),+ ) => { - println!("\n\n{} {}\n\n", + wlogln!("\n\n{} {}\n\n", "[Diamond Auto Bid Config Warning]", format!($f, $( $v ),+) ); @@ -45,7 +45,7 @@ pub fn start_diamond_auto_bidding(mut worker: Worker, hnode: Arc) { return; } - println!( + wlogln!( "[Diamond Auto Bidding] start with account {} min fee {} and max fee {}.", &cnf.dmer_bid_account.readable(), &bidmin, @@ -99,7 +99,7 @@ fn check_bidding_step( macro_rules! printerr { ( $f: expr, $( $v: expr ),+ ) => { - println!("\n\n{} {}\n\n", + wlogln!("\n\n{} {}\n\n", "[Diamond Auto Build Error]", format!($f, $( $v ),+) ); diff --git a/app/src/diaworker.rs b/app/src/diaworker.rs index 97e22cc9..5b11f768 100644 --- a/app/src/diaworker.rs +++ b/app/src/diaworker.rs @@ -74,7 +74,7 @@ impl DiaWorkConf { ) }) || ini_must_u64(sec, "workgroups", 0) > 0; if wants_gpu { - println!( + wlogln!( "[diamond] NOTE: HACD (diamond) mining is CPU / full-node only; the GPU keys \ in this config (useopencl / usecuda / workgroups) are ignored." ); @@ -259,7 +259,7 @@ fn send_diamond_result( if res.is_success.is_some() { return result_ch_tx.send(res).is_ok(); } - eprintln!( + wlogerr!( "[Mining] Diamond result queue full, dropped a statistics-only batch at number {}.", res.number ); @@ -293,7 +293,7 @@ fn queue_diamond_mining_success( match submit_tx.try_send(success) { Ok(()) => {} Err(mpsc::TrySendError::Full(success)) => { - eprintln!( + wlogerr!( "[Mining] Diamond submit queue full, submitting number {} inline.", *success.d.number ); @@ -360,7 +360,7 @@ pub fn diaworker_with_stop(stop_flag: Option>) { }; #[cfg(feature = "ocl")] if cnf.useopencl && opencl_resources.is_empty() { - eprintln!( + wlogerr!( "[Fatal] OpenCL was requested but no usable GPU backend initialized; stopping HACD worker." ); return; @@ -422,7 +422,7 @@ pub fn diaworker_with_stop(stop_flag: Option>) { #[cfg(feature = "ocl")] { // Initialize OpenCL - println!( + wlogln!( "\n[Start] Create GPU diamond miner worker #{}.", opencl_resources.len() ); @@ -463,11 +463,11 @@ pub fn diaworker_with_stop(stop_flag: Option>) { } #[cfg(not(feature = "ocl"))] { - println!( + wlogln!( "[Warning] use_opencl=true but app built without feature 'ocl'; fallback to CPU mining." ); let thrnum = cnf.efficiency.clamp_supervene(cnf.supervene) as usize; - println!("\n[Start] Create #{} diamond miner worker thread.", thrnum); + wlogln!("\n[Start] Create #{} diamond miner worker thread.", thrnum); for thrid in 0..thrnum { let cnf2 = cnf.clone(); let rstx = res_tx.clone(); @@ -478,7 +478,12 @@ pub fn diaworker_with_stop(stop_flag: Option>) { return; } guard_mining_iteration("diamond mining worker", || { - run_diamond_worker_thread(&cnf2, thrid, rstx.clone(), &stop_flag_worker); + run_diamond_worker_thread( + &cnf2, + thrid, + rstx.clone(), + &stop_flag_worker, + ); }); delay_continue_ms!(9); } @@ -490,7 +495,7 @@ pub fn diaworker_with_stop(stop_flag: Option>) { #[cfg(feature = "ocl")] { let thrnum = cnf.efficiency.spawn_supervene(cnf.supervene) as usize; - println!( + wlogln!( "\n[Start] Create #{} Ryzen CPU assist threads for diamonds (hybrid).", thrnum ); @@ -519,7 +524,7 @@ pub fn diaworker_with_stop(stop_flag: Option>) { } } else { let thrnum = cnf.efficiency.clamp_supervene(cnf.supervene) as usize; - println!("\n[Start] Create #{} diamond miner worker thread.", thrnum); + wlogln!("\n[Start] Create #{} diamond miner worker thread.", thrnum); for thrid in 0..thrnum { let cnf2 = cnf.clone(); let rstx = res_tx.clone(); @@ -621,7 +626,7 @@ fn deal_diamond_mining_results( .maybe_adjust_supervene(&cnf.efficiency, gpu_nonce_space, cpu_nonce_space); if should_pause_for_diamond_profit(&cnf.efficiency, &cnf.gpu_profile, active_cpu) { cnf.runtime.paused_unprofitable.store(true, Relaxed); - println!( + wlogln!( "\n[efficiency] HACD mining paused: daily power cost exceeds configured revenue target (hac_price)." ); } else { @@ -680,7 +685,7 @@ fn may_print_turn_to_nex_diamond_mining( *most_dia_str = [b'W'; DIAMOND_HASH_LEN]; // reset } - println!( + wlogln!( "\n[{}] req next number {} to mining ... ", &ctshow()[5..], mining_number @@ -717,7 +722,7 @@ fn run_diamond_worker_thread( // start mining let mut custom_nonce = [0u8; HASH_WIDTH]; if let Err(e) = getrandom::fill(&mut custom_nonce) { - eprintln!("[Mining] Secure random nonce failed: {e}"); + wlogerr!("[Mining] Secure random nonce failed: {e}"); return; } let custom_nonce = Hash::from(custom_nonce); @@ -737,7 +742,7 @@ fn run_diamond_worker_thread( return; } let ctn = Instant::now(); - // println!("- nonce_start: {}", nonce_start); + // wlogln!("- nonce_start: {}", nonce_start); let mut result = do_diamond_group_mining( current_mining_number, ¤t_mining_block_hash, @@ -746,7 +751,7 @@ fn run_diamond_worker_thread( nonce_start, nonce_space, ); - // println!("do_diamond_group_mining: {:?}", &result); + // wlogln!("do_diamond_group_mining: {:?}", &result); let use_secs = Instant::now().duration_since(ctn).as_millis() as f64 / 1000.0; result.use_secs = use_secs; result.is_gpu = false; @@ -793,7 +798,7 @@ fn run_diamond_worker_thread_opencl( let mut custom_nonce = [0u8; HASH_WIDTH]; if let Err(e) = getrandom::fill(&mut custom_nonce) { - eprintln!("[Mining] Secure random nonce failed: {e}"); + wlogerr!("[Mining] Secure random nonce failed: {e}"); return; } let custom_nonce = Hash::from(custom_nonce); @@ -1004,7 +1009,7 @@ fn load_init(cnf: &mut DiaWorkConf) { match crate::rpc_http::get_text(&HTTP_CLIENT, &urlapi_pending, &cnf.api_token, None) { Ok(t) => t, Err(e) => { - println!( + wlogln!( "Error: cannot init diamond miner from {}: {}", &urlapi_pending, e ); @@ -1012,26 +1017,26 @@ fn load_init(cnf: &mut DiaWorkConf) { } }; let Ok(res) = serde_json::from_str::(&body) else { - println!("Error: invalid JSON from {urlapi_pending}"); + wlogln!("Error: invalid JSON from {urlapi_pending}"); delay_continue!(30); }; let jstr = |k| res[k].as_str().unwrap_or(""); let err = jstr("err"); if err.len() > 0 { - println!("{} Error: {}", &urlapi_pending, err); + wlogln!("{} Error: {}", &urlapi_pending, err); delay_continue!(30); } let adr1 = jstr("bid_address"); let Ok(bid_addr) = Address::from_readable(&adr1) else { - println!("Error: bid_address '{}' format invalid", &adr1); + wlogln!("Error: bid_address '{}' format invalid", &adr1); delay_continue!(30); }; let adr2 = jstr("reward_address"); let Ok(rwd_addr) = Address::from_readable(&adr2) else { - println!("Error: reward_address '{}' format invalid", &adr2); + wlogln!("Error: reward_address '{}' format invalid", &adr2); delay_continue!(30); }; - println!( + wlogln!( "[Config] query diamond miner bid address: {}, reward address: {}", &adr1, &adr2 ); @@ -1048,24 +1053,24 @@ fn pull_and_push_diamond(cnf: &DiaWorkConf) { let urlapi_latest = format!("http://{}/query/latest", &cnf.rpcaddr); // get next number - // println!("urlapi_latest: {}", &urlapi_latest); + // wlogln!("urlapi_latest: {}", &urlapi_latest); let body = match crate::rpc_http::get_text(&HTTP_CLIENT, &urlapi_latest, &cnf.api_token, None) { Ok(t) => t, Err(e) => { - println!("Error: cannot get latest from {}: {}", &urlapi_latest, e); + wlogln!("Error: cannot get latest from {}: {}", &urlapi_latest, e); delay_return!(30); } }; let Ok(res) = serde_json::from_str::(&body) else { - println!("Error: invalid JSON from {urlapi_latest}"); + wlogln!("Error: invalid JSON from {urlapi_latest}"); delay_return!(30); }; - // println!("get latest: {:?}", &res); + // wlogln!("get latest: {:?}", &res); let jnum = |k| res[k].as_u64().unwrap_or(0); let next_num = jnum("diamond") as u32 + 1; - // println!("mining next num: {} {}", &mining_num, &next_num); + // wlogln!("mining next num: {} {}", &mining_num, &next_num); if next_num == 1 { - // println!("get latest: next_num == 1"); + // wlogln!("get latest: next_num == 1"); install_diamond_job(next_num, genesis_block_hash()); return; // first mining } @@ -1075,7 +1080,7 @@ fn pull_and_push_diamond(cnf: &DiaWorkConf) { // Refresh prev_hash for the same number when the node reorged the tip. // Cheap GET; skip heavy work only when hash is unchanged. } else if next_num < mining_num { - println!( + wlogln!( "[HACD] diamond tip reorg: number {} -> {}, refreshing job", mining_num, next_num ); @@ -1086,23 +1091,23 @@ fn pull_and_push_diamond(cnf: &DiaWorkConf) { &cnf.rpcaddr, next_num - 1 ); - // println!("urlapi_diamond: {}", &urlapi_diamond); + // wlogln!("urlapi_diamond: {}", &urlapi_diamond); let body = match crate::rpc_http::get_text(&HTTP_CLIENT, &urlapi_diamond, &cnf.api_token, None) { Ok(t) => t, Err(e) => { - println!("Error: cannot get diamond from {}: {}", &urlapi_diamond, e); + wlogln!("Error: cannot get diamond from {}: {}", &urlapi_diamond, e); delay_return!(30); } }; let Ok(res) = serde_json::from_str::(&body) else { - println!("Error: invalid JSON from {urlapi_diamond}"); + wlogln!("Error: invalid JSON from {urlapi_diamond}"); delay_return!(30); }; - // println!("query diamond: {:?}", &res); + // wlogln!("query diamond: {:?}", &res); let prev_hash = res["born"]["hash"].as_str().unwrap_or(""); let Ok(hx) = hex::decode(&prev_hash) else { - println!( + wlogln!( "Error: cannot get born.hash from {}: {:?}", &urlapi_diamond, &res ); @@ -1148,7 +1153,7 @@ fn push_diamond_mining_success(cnf: &DiaWorkConf, success: DiamondMint) { last = body.chars().take(200).collect(); let Ok(res) = serde_json::from_str::(&body) else { let snippet: String = body.chars().take(120).collect(); - println!( + wlogln!( "[HACD submit] attempt {attempt}/{MAX_SUBMIT_ATTEMPTS} unrecognized response, retrying: {snippet}" ); if attempt < MAX_SUBMIT_ATTEMPTS { @@ -1161,7 +1166,7 @@ fn push_diamond_mining_success(cnf: &DiaWorkConf, success: DiamondMint) { let jstr = |k: &str| res[k].as_str().unwrap_or(""); let tx_err = jstr("err"); if !tx_err.is_empty() { - println!( + wlogln!( "ㄨㄨㄨㄨ Failed submit tx diamond mint to mainnet\n ERROR: {}\n", tx_err ); @@ -1169,7 +1174,7 @@ fn push_diamond_mining_success(cnf: &DiaWorkConf, success: DiamondMint) { } let tx_hash = jstr("tx_hash"); if tx_hash.len() == 64 { - println!( + wlogln!( "Success submit tx diamond mint {} ({}) to mainnet, \n get tx hash: {}\n", success.d.diamond.to_readable(), *success.d.number, @@ -1177,8 +1182,8 @@ fn push_diamond_mining_success(cnf: &DiaWorkConf, success: DiamondMint) { ); return; } - // JSON but no usable tx_hash — treat as transient front-end noise. - println!( + // JSON but no usable tx_hash: treat as transient front-end noise. + wlogln!( "[HACD submit] attempt {attempt}/{MAX_SUBMIT_ATTEMPTS} missing tx_hash, retrying" ); if attempt < MAX_SUBMIT_ATTEMPTS { @@ -1187,7 +1192,7 @@ fn push_diamond_mining_success(cnf: &DiaWorkConf, success: DiamondMint) { } Err(e) => { last = format!("transport error: {e}"); - println!( + wlogln!( "Error: attempt {attempt}/{MAX_SUBMIT_ATTEMPTS} cannot submit diamond success to {urlapi_success}: {e}" ); if attempt < MAX_SUBMIT_ATTEMPTS { @@ -1196,7 +1201,7 @@ fn push_diamond_mining_success(cnf: &DiaWorkConf, success: DiamondMint) { } } } - println!( + wlogln!( "ㄨㄨㄨㄨ Failed submit tx diamond mint after {MAX_SUBMIT_ATTEMPTS} attempts ({last}). Check the node/connection." ); } @@ -1438,16 +1443,16 @@ fn run_diamond_mining_benchmark(cnf: &DiaWorkConf, config_path: &str) { #[cfg(not(feature = "ocl"))] { let _ = (cnf, config_path); - println!("[benchmark] Rebuild diaworker with --features ocl"); + wlogln!("[benchmark] Rebuild diaworker with --features ocl"); return; } #[cfg(feature = "ocl")] { if !cnf.useopencl { - println!("[benchmark] HACD is CPU-only; Auto Tune applies to the HAC poworker."); + wlogln!("[benchmark] HACD is CPU-only; Auto Tune applies to the HAC poworker."); return; } - println!( + wlogln!( "[benchmark] HACD: GPU tuning uses same profiles as HAC; run poworker benchmark or share ini." ); let scan = crate::opencl_diag::scan_opencl(); @@ -1483,7 +1488,7 @@ fn run_diamond_mining_benchmark(cnf: &DiaWorkConf, config_path: &str) { max_us, ); if candidates.is_empty() { - println!("[benchmark] HACD: no safe tuning candidates"); + wlogln!("[benchmark] HACD: no safe tuning candidates"); return; } let per = @@ -1535,7 +1540,7 @@ fn run_diamond_mining_benchmark(cnf: &DiaWorkConf, config_path: &str) { EfficiencyMode::Max => hps, _ => kh_per_j, }; - println!( + wlogln!( "[benchmark] HACD {}: {} ({:.1} kH/J, wg={}, unit_size={})", pick.profile, rates_to_show(hps), @@ -1555,7 +1560,7 @@ fn run_diamond_mining_benchmark(cnf: &DiaWorkConf, config_path: &str) { if let Some((pick, _)) = best { let _ = apply_benchmark_pick(config_path, &pick); } else { - println!("[benchmark] HACD: all tuning points failed; config unchanged"); + wlogln!("[benchmark] HACD: all tuning points failed; config unchanged"); } } } diff --git a/app/src/efficiency.rs b/app/src/efficiency.rs index 56a8da2f..5307a917 100644 --- a/app/src/efficiency.rs +++ b/app/src/efficiency.rs @@ -657,7 +657,7 @@ pub fn apply_benchmark_pick(path: &str, pick: &BenchmarkPick) -> std::io::Result out.push('\n'); } atomic_write_private(Path::new(path), out.as_bytes())?; - println!( + wlogln!( "[benchmark] Applied gpu_profile={} (work_groups={}, unit_size={}) to {}", pick.profile, pick.workgroups, pick.unitsize, path ); @@ -954,6 +954,11 @@ enum GpuTempSensorSource { args: Vec, parser: GpuTempParser, }, + /// The AMD display driver's own library, which is the only one of these + /// that exists on a consumer Windows install. Bound to one ADL adapter at + /// detection time, so every later read is the same physical card. + #[cfg(windows)] + AmdDriver { adapter_index: i32 }, } /// A sensor source selected once for one exact GPU and reused by the monitor. @@ -988,6 +993,10 @@ impl GpuTempSensorBackend { } } } + #[cfg(windows)] + GpuTempSensorSource::AmdDriver { adapter_index } => { + crate::gpu_temp_adl::temperature_c(*adapter_index).and_then(valid_gpu_temp) + } } } } @@ -1086,6 +1095,31 @@ fn nvidia_sensor(gpu_index: u32) -> GpuTempSensorBackend { ) } +/// The AMD driver's own sensor, on Windows, where no `*-smi` tool exists. +/// +/// Bound only when exactly one card answers. ADL numbers its adapters in its own +/// order, which is not the OpenCL device order (measured: on a machine with an +/// integrated Radeon and an RX 9070 XT, ADL adapter 0 is the integrated one +/// while OpenCL device 0 is the 9070 XT), so with several cards answering there +/// is no honest way to say which reading belongs to `gpu_index`, and a +/// neighbouring card's temperature is a wrong number rather than a missing one. +/// The card that is bound is named in the sensor label so the operator can see +/// which one the gauge is showing. +#[cfg(windows)] +fn amd_driver_sensor(_gpu_index: u32) -> Option<(GpuTempSensorBackend, f32)> { + let reporting = crate::gpu_temp_adl::reporting_gpus(); + let [gpu] = reporting.as_slice() else { + return None; + }; + let sensor = GpuTempSensorBackend { + label: format!("AMD driver (ADL) {}", gpu.name), + source: GpuTempSensorSource::AmdDriver { + adapter_index: gpu.adapter_index, + }, + }; + Some((sensor, gpu.temp_c)) +} + pub(crate) fn detect_gpu_temp_sensor( thermal_file: &str, gpu_index: u32, @@ -1100,6 +1134,14 @@ pub(crate) fn detect_gpu_temp_sensor( } match vendor { + // The driver library is tried first on Windows: it is the only source + // that exists there, it answers in under a millisecond, and each + // `*-smi` candidate behind it costs a process spawn and up to two + // seconds of timeout before failing. + #[cfg(windows)] + crate::gpu_arch::GpuVendor::Amd => amd_driver_sensor(gpu_index) + .or_else(|| detect_first_sensor(amd_sensor_candidates(gpu_index))), + #[cfg(not(windows))] crate::gpu_arch::GpuVendor::Amd => detect_first_sensor(amd_sensor_candidates(gpu_index)), crate::gpu_arch::GpuVendor::Nvidia => { let sensor = nvidia_sensor(gpu_index); diff --git a/app/src/fullnode.rs b/app/src/fullnode.rs index af296d38..d8c28812 100644 --- a/app/src/fullnode.rs +++ b/app/src/fullnode.rs @@ -77,7 +77,7 @@ impl FullnodeRuntime { if panic_count > 0 { return errf!("{} thread panicked", panic_count); } - println!("[Exit] Hacash fullnode closed."); + wlogln!("[Exit] Hacash fullnode closed."); Ok(()) } } diff --git a/app/src/gpu_oom.rs b/app/src/gpu_oom.rs index 0e14b081..80cde465 100644 --- a/app/src/gpu_oom.rs +++ b/app/src/gpu_oom.rs @@ -17,7 +17,7 @@ pub const OOM_RECOVERY_BATCHES: u32 = 16; /// whole session after a single transient CL_OUT_OF_RESOURCES. pub const OOM_SLOW_RAMP_BATCHES: u32 = OOM_RECOVERY_BATCHES * 4; -/// Per-device work_groups state — lives on [`crate::opencl_gpu::OpenclGpuHandle`]. +/// Per-device work_groups state, lives on [`crate::opencl_gpu::OpenclGpuHandle`]. pub struct GpuOomState { base_workgroups: u32, effective_workgroups: AtomicU32, @@ -83,7 +83,7 @@ impl GpuOomState { let floor = self.oom_floor_wg.max(1); let next = (cur / 2).max(floor); if next < cur { - eprintln!( + wlogerr!( "[efficiency] OpenCL error - reducing work_groups {} -> {} (floor={})", cur, next, floor ); @@ -134,7 +134,7 @@ impl GpuOomState { self.effective_workgroups.store(base, Relaxed); self.oom_reduced.store(false, Relaxed); self.success_batches_since_oom.store(0, Relaxed); - println!("[efficiency] GPU stable - restored work_groups to {}", base); + wlogln!("[efficiency] GPU stable - restored work_groups to {}", base); } return; } @@ -151,7 +151,7 @@ impl GpuOomState { if next >= base { self.oom_reduced.store(false, Relaxed); } - println!( + wlogln!( "[efficiency] GPU stable for {} batches - raising work_groups {} -> {}", n, cur, next ); diff --git a/app/src/gpu_temp_adl.rs b/app/src/gpu_temp_adl.rs new file mode 100644 index 00000000..6ea294b2 --- /dev/null +++ b/app/src/gpu_temp_adl.rs @@ -0,0 +1,364 @@ +//! GPU temperature on Windows, read from the AMD display driver's own library. +//! +//! Why this file exists. The thermal gauge was built on `rocm-smi` and +//! `amd-smi`, and on Windows neither of those exists: ROCm is a Linux stack and +//! `amd-smi` is not part of the consumer Adrenalin package. The Linux +//! `thermal_file` hwmon path is absent too. So on a normal Windows box with a +//! Radeon card, which is most of this miner's audience, the whole feature could +//! only ever print +//! +//! [Thermal] No GPU temperature to report: no exact temperature sensor for +//! Amd GPU 0; supported: amd-smi, rocm-smi, or thermal_file +//! +//! What Windows does have is `atiadlxx.dll`, the AMD Display Library. It ships +//! with every Adrenalin driver and lives in `System32`, so it is loaded here at +//! runtime with `LoadLibrary`: no build dependency, no import library, nothing +//! bundled, and on a machine without an AMD driver the load simply fails and +//! this module reports nothing at all. +//! +//! Measured on the machine this was written for, an RX 9070 XT (RDNA 4, Navi 48) +//! on driver 32.0.31035.1003, while it was mining: +//! +//! ADL2_Main_Control_Create ~10 ms, once +//! ADL2_New_QueryPMLogData_Get 0.2 to 1.1 ms per read +//! sensors: edge 60 C, memory 68 C, hotspot 84 C, fan 1032 rpm, +//! activity 99%, gfx clock 3302 MHz, gfx voltage 1123 mV +//! ADL2_OverdriveN_Temperature_Get ADL_ERR_NOT_SUPPORTED (-8) +//! ADL2_Overdrive6_Temperature_Get ADL_ERR_NOT_SUPPORTED (-8) +//! +//! So on current hardware PMLog is the only one of the three that answers, and +//! it answers cheaply. The Overdrive entry points are deliberately not used. +//! +//! About the sensor indices. They come from AMD's published `ADLSensorType` +//! enum, and the live dump above is what confirms them rather than a comment +//! claiming they are right: at index 1 a 3302 MHz core clock, at 2 a 2505 MHz +//! memory clock, at 14 a 1032 rpm fan, at 15 a fan percentage, at 19 a 99% load +//! on a card at full tilt, at 21 a 1123 mV core voltage. Every one of those +//! landed exactly where the enum says it should, which is what makes the three +//! temperature indices next to them trustworthy. A sensor whose `supported` +//! flag is clear, or whose value is outside a plausible range, is dropped by the +//! caller rather than reported. + +use std::ffi::{CString, c_char, c_int, c_void}; +use std::sync::{Mutex, OnceLock}; + +#[link(name = "kernel32")] +unsafe extern "system" { + fn LoadLibraryA(name: *const c_char) -> *mut c_void; + fn GetProcAddress(module: *mut c_void, name: *const c_char) -> *mut c_void; +} + +// ADL allocates through a caller-supplied callback. It hands the pointer back +// to us and we own it, so this must be the same allocator the C runtime frees +// with; `std::alloc` with its size-and-align contract is not, because ADL frees +// nothing and we free with `free`. +unsafe extern "C" { + fn malloc(size: usize) -> *mut c_void; +} + +unsafe extern "C" fn adl_malloc(size: c_int) -> *mut c_void { + if size <= 0 { + return std::ptr::null_mut(); + } + unsafe { malloc(size as usize) } +} + +const ADL_OK: c_int = 0; +const ADL_MAX_PATH: usize = 256; +const PMLOG_SENSOR_COUNT: usize = 256; + +/// `ADLSensorType` indices for the three temperatures `rocm-smi --showtemp` +/// prints on Linux, so the number this module reports means the same thing on +/// both platforms. +const PMLOG_TEMPERATURE_EDGE: usize = 8; +const PMLOG_TEMPERATURE_MEM: usize = 9; +const PMLOG_TEMPERATURE_HOTSPOT: usize = 27; + +/// `AdapterInfo` from `adl_structures.h`. The last five fields are Windows-only +/// in the header and this file is Windows-only, so all of them are present. The +/// struct is passed by us and filled by ADL, so its size must match exactly: +/// 1572 bytes, asserted below and confirmed against the library at runtime by +/// the fact that the bus and device numbers come back correct. +#[repr(C)] +#[derive(Clone, Copy)] +struct AdapterInfo { + size: c_int, + adapter_index: c_int, + udid: [c_char; ADL_MAX_PATH], + bus_number: c_int, + device_number: c_int, + function_number: c_int, + vendor_id: c_int, + adapter_name: [c_char; ADL_MAX_PATH], + display_name: [c_char; ADL_MAX_PATH], + present: c_int, + exist: c_int, + driver_path: [c_char; ADL_MAX_PATH], + driver_path_ext: [c_char; ADL_MAX_PATH], + pnp_string: [c_char; ADL_MAX_PATH], + os_display_index: c_int, +} + +const _: () = assert!(std::mem::size_of::() == 1572); + +#[repr(C)] +#[derive(Clone, Copy, Default)] +struct SingleSensorData { + supported: c_int, + value: c_int, +} + +#[repr(C)] +struct PmLogDataOutput { + size: c_int, + sensors: [SingleSensorData; PMLOG_SENSOR_COUNT], +} + +type FnMainControlCreate = unsafe extern "C" fn( + unsafe extern "C" fn(c_int) -> *mut c_void, + c_int, + *mut *mut c_void, +) -> c_int; +type FnNumberOfAdapters = unsafe extern "C" fn(*mut c_void, *mut c_int) -> c_int; +type FnAdapterInfoGet = unsafe extern "C" fn(*mut c_void, *mut AdapterInfo, c_int) -> c_int; +type FnQueryPmLogData = unsafe extern "C" fn(*mut c_void, c_int, *mut PmLogDataOutput) -> c_int; + +struct Adl { + context: *mut c_void, + adapter_info_get: FnAdapterInfoGet, + number_of_adapters: FnNumberOfAdapters, + query_pmlog: FnQueryPmLogData, +} + +// The handles are process-lifetime and every use goes through the `Mutex` +// below; ADL's own context is not documented as thread safe, which is exactly +// why nothing here is reachable without holding that lock. +unsafe impl Send for Adl {} + +/// One AMD GPU as ADL sees it, and the temperature it answered with. +#[derive(Clone, Debug, PartialEq)] +pub struct AdlGpuTemp { + pub adapter_index: i32, + /// PCI bus, device and function. Several ADL adapters share one physical + /// card (one per display output), so this is what identifies the card. + pub pci: (i32, i32, i32), + pub name: String, + pub temp_c: f32, +} + +fn plausible_temp(value: c_int) -> Option { + // Same window the rest of the thermal code uses. A GPU below zero or above + // 120 C is a sensor answering with something that is not a temperature. + let value = value as f32; + (value > 0.0 && value < 120.0).then_some(value) +} + +impl Adl { + fn load() -> Option> { + // SAFETY: every pointer below is either checked for null before use or + // handed straight back to the library that produced it. + unsafe { + let name = CString::new("atiadlxx.dll").ok()?; + let library = LoadLibraryA(name.as_ptr()); + if library.is_null() { + return None; + } + // The library is intentionally never freed: the context created + // below outlives every read, and unloading a driver library while + // its worker threads run is how a shutdown crash is bought. + let symbol = |symbol_name: &str| -> Option<*mut c_void> { + let symbol_name = CString::new(symbol_name).ok()?; + let address = GetProcAddress(library, symbol_name.as_ptr()); + (!address.is_null()).then_some(address) + }; + + let create: FnMainControlCreate = + std::mem::transmute(symbol("ADL2_Main_Control_Create")?); + let number_of_adapters: FnNumberOfAdapters = + std::mem::transmute(symbol("ADL2_Adapter_NumberOfAdapters_Get")?); + let adapter_info_get: FnAdapterInfoGet = + std::mem::transmute(symbol("ADL2_Adapter_AdapterInfo_Get")?); + let query_pmlog: FnQueryPmLogData = + std::mem::transmute(symbol("ADL2_New_QueryPMLogData_Get")?); + + let mut context: *mut c_void = std::ptr::null_mut(); + // Second argument 1: enumerate connected adapters only. + if create(adl_malloc, 1, &mut context) != ADL_OK || context.is_null() { + return None; + } + Some(Mutex::new(Adl { + context, + adapter_info_get, + number_of_adapters, + query_pmlog, + })) + } + } + + fn adapters(&self) -> Vec { + // SAFETY: the buffer is sized from the count ADL just reported, and the + // byte length passed is the one it fills. + unsafe { + let mut count: c_int = 0; + if (self.number_of_adapters)(self.context, &mut count) != ADL_OK || count <= 0 { + return Vec::new(); + } + let count = count as usize; + let mut infos: Vec = vec![std::mem::zeroed(); count]; + let bytes = count * std::mem::size_of::(); + let Ok(bytes) = c_int::try_from(bytes) else { + return Vec::new(); + }; + if (self.adapter_info_get)(self.context, infos.as_mut_ptr(), bytes) != ADL_OK { + return Vec::new(); + } + infos + } + } + + fn temperature(&self, adapter_index: i32) -> Option { + // SAFETY: the output struct is fully owned here and ADL only writes into + // it. A bad adapter index is refused by the library with a non-zero + // return (measured: -5 for an index that does not exist, -8 for an + // integrated GPU that has no such sensor), never with a stale value. + unsafe { + let mut out: PmLogDataOutput = std::mem::zeroed(); + if (self.query_pmlog)(self.context, adapter_index as c_int, &mut out) != ADL_OK { + return None; + } + [ + PMLOG_TEMPERATURE_EDGE, + PMLOG_TEMPERATURE_MEM, + PMLOG_TEMPERATURE_HOTSPOT, + ] + .into_iter() + .filter(|index| out.sensors[*index].supported != 0) + .filter_map(|index| plausible_temp(out.sensors[index].value)) + .reduce(f32::max) + } + } +} + +fn adl() -> Option<&'static Mutex> { + static ADL: OnceLock>> = OnceLock::new(); + ADL.get_or_init(Adl::load).as_ref() +} + +fn c_string(bytes: &[c_char]) -> String { + let raw: Vec = bytes.iter().map(|byte| *byte as u8).collect(); + let end = raw.iter().position(|byte| *byte == 0).unwrap_or(raw.len()); + String::from_utf8_lossy(&raw[..end]).trim().to_string() +} + +/// Every distinct AMD card that answers with a temperature right now. +/// +/// Distinct means by PCI address: ADL lists one adapter per display output, so +/// a single card shows up several times. Cards that do not answer are left out +/// rather than reported as zero, which is what keeps an integrated GPU (it +/// returns ADL_ERR_NOT_SUPPORTED) from being offered as a mining sensor. +pub fn reporting_gpus() -> Vec { + let Some(adl) = adl() else { + return Vec::new(); + }; + let adl = adl.lock().unwrap_or_else(|error| error.into_inner()); + let mut found: Vec = Vec::new(); + for info in adl.adapters() { + let pci = (info.bus_number, info.device_number, info.function_number); + if found.iter().any(|gpu| gpu.pci == pci) { + continue; + } + if let Some(temp_c) = adl.temperature(info.adapter_index) { + found.push(AdlGpuTemp { + adapter_index: info.adapter_index, + pci, + name: c_string(&info.adapter_name), + temp_c, + }); + } + } + found +} + +/// Read one adapter that `reporting_gpus` already bound to. +pub fn temperature_c(adapter_index: i32) -> Option { + let adl = adl()?; + let adl = adl.lock().unwrap_or_else(|error| error.into_inner()); + adl.temperature(adapter_index) +} + +/// Why there is no ADL temperature, in words an operator can act on. +/// +/// Only called on the failure path, and it says what was actually observed +/// rather than guessing: the driver library missing, or present but with more +/// than one card answering, which is the case this module refuses to resolve. +pub fn unavailable_reason(gpu_index: u32) -> String { + if adl().is_none() { + return "the AMD display driver library (atiadlxx.dll) did not load; \ + install or repair the AMD driver" + .to_string(); + } + let reporting = reporting_gpus(); + match reporting.len() { + 0 => "the AMD display driver library loaded but no card answered with a temperature" + .to_string(), + // Reachable only if the one card stopped answering between binding and + // this call, since one card is exactly the case that does bind. + 1 => format!( + "the AMD display driver stopped reporting a temperature for {}", + reporting[0].name + ), + _ => format!( + "the AMD display driver reports {} cards ({}) and cannot say which one is OpenCL \ + device {}; set thermal_file, or run one card per worker", + reporting.len(), + reporting + .iter() + .map(|gpu| gpu.name.as_str()) + .collect::>() + .join(", "), + gpu_index + ), + } +} + +#[cfg(test)] +mod tests { + use super::*; + + /// Runs on any Windows machine. On one without an AMD driver every call + /// must come back empty instead of panicking or blocking, because that is + /// the path a CPU-only or NVIDIA miner takes on every start. + #[test] + fn a_machine_without_the_amd_driver_reports_nothing_and_does_not_panic() { + let gpus = reporting_gpus(); + for gpu in &gpus { + assert!( + gpu.temp_c > 0.0 && gpu.temp_c < 120.0, + "a reported temperature must be a plausible one, got {}", + gpu.temp_c + ); + assert!(!gpu.name.is_empty(), "a bound card must be named"); + } + // Two consecutive enumerations must agree on how many cards answer: + // the count is what decides whether a temperature may be attributed to + // a GPU at all. + assert_eq!(gpus.len(), reporting_gpus().len()); + assert!(!unavailable_reason(0).is_empty()); + } + + #[test] + fn an_adapter_index_that_cannot_exist_yields_no_temperature() { + // ADL answers a bad index with an error code, and the wrapper must turn + // that into "no reading" rather than into the zero in the output buffer. + assert_eq!(temperature_c(i32::MAX), None); + assert_eq!(temperature_c(-1), None); + } + + #[test] + fn implausible_sensor_values_are_not_temperatures() { + assert_eq!(plausible_temp(0), None); + assert_eq!(plausible_temp(-40), None); + assert_eq!(plausible_temp(120), None); + assert_eq!(plausible_temp(60), Some(60.0)); + } +} diff --git a/app/src/lib.rs b/app/src/lib.rs index 03e7da04..ffe6e0b7 100644 --- a/app/src/lib.rs +++ b/app/src/lib.rs @@ -1,8 +1,18 @@ include! {"version.rs"} +// First, and `macro_use`, because `wlogln!` and `wlogerr!` replace `println!` +// and `eprintln!` throughout this crate: they must be in scope for every module +// declared below. +#[macro_use] +pub mod worker_log; + pub mod efficiency; pub mod gpu_arch; pub mod gpu_oom; +/// Windows only: the AMD driver's own library is the only GPU temperature +/// source that exists on a consumer Windows install. +#[cfg(windows)] +pub mod gpu_temp_adl; pub mod hash_util; pub mod mining_batch; pub mod mining_guard; diff --git a/app/src/mining_batch.rs b/app/src/mining_batch.rs index e0b62f03..5c56d593 100644 --- a/app/src/mining_batch.rs +++ b/app/src/mining_batch.rs @@ -431,7 +431,7 @@ impl BlockMinerBackend for OpenclBlockBackend { match gpu_result { Err(e) => { - eprintln!("[efficiency] GPU batch failed: {}", e.display()); + wlogerr!("[efficiency] GPU batch failed: {}", e.display()); self.gpu .on_batch_error(e, self.oom_fallback, ctx.configured_wg, &self.runtime); cpu_gpu_error_recovery( @@ -469,7 +469,7 @@ impl BlockMinerBackend for OpenclBlockBackend { Err(message) => { let integrity_error = GpuBatchError::Other(format!("GPU integrity error: {message}")); - eprintln!("[OpenCL] {}", integrity_error.display()); + wlogerr!("[OpenCL] {}", integrity_error.display()); self.gpu.on_batch_error( integrity_error, false, @@ -566,7 +566,7 @@ impl BlockMinerBackend for CudaBlockBackend { ctx.share_target.as_ref(), ) { Err(e) => { - eprintln!("[CUDA] batch failed: {e}"); + wlogerr!("[CUDA] batch failed: {e}"); self.runtime.record_gpu_error_event(); let report = self.cuda.note_batch_failure(); let action = cuda_failure_action(e.is_sticky()); @@ -574,7 +574,7 @@ impl BlockMinerBackend for CudaBlockBackend { // Back off the effective work-groups so the next batch tries a // smaller, likely-runnable size instead of failing forever. let reduced = self.cuda.record_error(); - eprintln!("[CUDA] reducing work_groups to {reduced} after error"); + wlogerr!("[CUDA] reducing work_groups to {reduced} after error"); } self.cuda .announce_quarantine(&report, &format!("Last error: {e}")); @@ -617,7 +617,7 @@ impl BlockMinerBackend for CudaBlockBackend { let shares = match verified { Ok(shares) => shares, Err(message) => { - eprintln!("[CUDA] GPU integrity error: {message}"); + wlogerr!("[CUDA] GPU integrity error: {message}"); self.runtime.record_gpu_error_event(); // A card returning hashes the CPU cannot reproduce is as // dead as one that fails to launch, so it shares the same @@ -902,8 +902,15 @@ mod tests { let mut wrong_hash = good; wrong_hash.0 = 12; assert!( - verify_gpu_shares(height, &block_intro, 0, 256, &easiest_target, &[good, wrong_hash]) - .is_err() + verify_gpu_shares( + height, + &block_intro, + 0, + 256, + &easiest_target, + &[good, wrong_hash] + ) + .is_err() ); // A nonce outside the batch window. assert!( @@ -920,9 +927,7 @@ mod tests { // An honest hash the card listed even though it is ABOVE the target it // was told to filter on: the compare is broken, so nothing is forwarded. let strict_target = [0u8; 32]; - assert!( - verify_gpu_shares(height, &block_intro, 0, 256, &strict_target, &[good]).is_err() - ); + assert!(verify_gpu_shares(height, &block_intro, 0, 256, &strict_target, &[good]).is_err()); } #[test] @@ -1005,7 +1010,10 @@ mod tests { thermal_wg_cap: None, share_target: None, }; - assert_eq!(x16rs_cuda::share_capacity_for(solo.share_target.as_ref()), 0); + assert_eq!( + x16rs_cuda::share_capacity_for(solo.share_target.as_ref()), + 0 + ); let pooled_target = [0x0fu8; 32]; assert_eq!( diff --git a/app/src/mining_guard.rs b/app/src/mining_guard.rs index 4d0df781..e5c46c14 100644 --- a/app/src/mining_guard.rs +++ b/app/src/mining_guard.rs @@ -20,7 +20,7 @@ pub(crate) fn panic_reason(payload: &(dyn std::any::Any + Send)) -> &str { /// panic, log it, and let the loop keep running. pub(crate) fn guard_mining_iteration(label: &str, body: impl FnOnce()) { if let Err(payload) = std::panic::catch_unwind(std::panic::AssertUnwindSafe(body)) { - eprintln!( + wlogerr!( "[Mining] {} panicked and was contained: {}", label, panic_reason(&*payload) diff --git a/app/src/mining_runtime.rs b/app/src/mining_runtime.rs index 1af63fdb..771a5a1c 100644 --- a/app/src/mining_runtime.rs +++ b/app/src/mining_runtime.rs @@ -5,6 +5,7 @@ use std::sync::atomic::{ AtomicBool, AtomicU32, AtomicU64, Ordering::{AcqRel, Acquire, Relaxed, Release}, }; +use std::time::{Duration, Instant}; use crate::efficiency::EfficiencyConf; @@ -18,14 +19,29 @@ pub struct MiningRuntimeState { pub adjust_counter: AtomicU64, /// When non-zero, caps per-GPU effective work_groups (thermal throttle). thermal_cap_wg: AtomicU32, - /// Latest OOM-adjusted work_groups across GPU workers (0 = unknown). - oom_work_groups_min: AtomicU32, + /// Work_groups the OOM fallback ALLOWS, worst case across GPU workers + /// (0 = no GPU has reported yet). On a card that never hit an out-of-memory + /// batch this is the configured count, so it is never a measure of how much + /// was lost: the loss is the gap below `configured_wg`. + oom_allowed_work_groups_min: AtomicU32, /// Latest effective work_groups across GPU workers (OOM ∩ thermal ∩ configured). effective_work_groups_min: AtomicU32, /// Last writer for backward-compatible single-GPU panel reads. pub reported_effective_wg: AtomicU32, /// Mining/result/thermal threads that have not acknowledged shutdown yet. active_mining_threads: AtomicU32, + /// Hottest GPU temperature the monitor really read, in hundredths of a + /// degree Celsius. 0 means no sensor has ever answered, which is not the + /// same fact as a cold card and must never reach the panel as a zero. + gpu_temp_c100: AtomicU32, + /// Wall clock of that reading, so a sensor that stops answering stops being + /// quoted. 0 = never read. + gpu_temp_unix_ms: AtomicU64, + /// How old a reading may be before it stops being offered to the panel, in + /// milliseconds. Set from the cadence the monitor really polls at AND from + /// what a pass on this rig can cost, so the window and the rate readings can + /// actually arrive at can never contradict each other. + gpu_temp_fresh_ms: AtomicU64, } pub(crate) struct MiningThreadGuard { @@ -48,13 +64,56 @@ impl MiningRuntimeState { paused_unprofitable: AtomicBool::new(false), adjust_counter: AtomicU64::new(0), thermal_cap_wg: AtomicU32::new(0), - oom_work_groups_min: AtomicU32::new(0), + oom_allowed_work_groups_min: AtomicU32::new(0), effective_work_groups_min: AtomicU32::new(0), reported_effective_wg: AtomicU32::new(0), active_mining_threads: AtomicU32::new(0), + gpu_temp_c100: AtomicU32::new(0), + gpu_temp_unix_ms: AtomicU64::new(0), + // Before a monitor starts there is nothing to be stale, so the + // narrowest cadence with a single sensor is the safe placeholder. + gpu_temp_fresh_ms: AtomicU64::new(temp_freshness_ms(ThermalCadence::GUARDED, 1)), }) } + /// Tell the freshness window which cadence the monitor is running at, and + /// over how many sensors, because a pass over eight of them is itself + /// longer than the window a single sensor needs. + fn set_temp_freshness_window(&self, cadence: ThermalCadence, sensor_count: usize) { + self.gpu_temp_fresh_ms + .store(temp_freshness_ms(cadence, sensor_count), Relaxed); + } + + /// Publish one temperature the monitor actually read. Values outside the + /// range a GPU sensor can produce are dropped rather than published, on the + /// same rule the sensor parsers already apply. + pub fn record_gpu_temperature(&self, temp_c: f32) { + if !temp_c.is_finite() || temp_c <= 0.0 || temp_c >= 120.0 { + return; + } + self.gpu_temp_c100 + .store((temp_c * 100.0).round() as u32, Relaxed); + self.gpu_temp_unix_ms + .store(crate::mining_stats::unix_ms_now(), Relaxed); + } + + /// The GPU temperature this process measured, or `None` when no sensor has + /// answered recently. Never 0: where there is no sensor there is no reading + /// to report, and the panel is told exactly that. + pub fn gpu_temp_c(&self) -> Option { + let hundredths = self.gpu_temp_c100.load(Relaxed); + let taken_at = self.gpu_temp_unix_ms.load(Relaxed); + if hundredths == 0 || taken_at == 0 { + return None; + } + if crate::mining_stats::unix_ms_now().saturating_sub(taken_at) + > self.gpu_temp_fresh_ms.load(Relaxed) + { + return None; + } + Some(hundredths as f32 / 100.0) + } + pub(crate) fn track_mining_thread(self: &Arc) -> MiningThreadGuard { self.active_mining_threads.fetch_add(1, AcqRel); MiningThreadGuard { @@ -80,16 +139,16 @@ impl MiningRuntimeState { if cap == 0 { None } else { Some(cap) } } - pub fn oom_work_groups(&self) -> u32 { - self.oom_work_groups_min.load(Relaxed) + pub fn oom_allowed_work_groups(&self) -> u32 { + self.oom_allowed_work_groups_min.load(Relaxed) } pub fn effective_work_groups(&self) -> u32 { self.effective_work_groups_min.load(Relaxed) } - fn compute_effective(oom_wg: u32, thermal_cap: Option, configured_wg: u32) -> u32 { - let mut effective = oom_wg.max(1); + fn compute_effective(oom_allowed_wg: u32, thermal_cap: Option, configured_wg: u32) -> u32 { + let mut effective = oom_allowed_wg.max(1); if let Some(cap) = thermal_cap { if cap > 0 { effective = effective.min(cap); @@ -119,10 +178,17 @@ impl MiningRuntimeState { }); } - /// Report per-GPU OOM work_groups; aggregates latest across devices for panel stats. - pub fn report_gpu_workgroups(&self, oom_wg: u32, thermal_cap: Option, configured_wg: u32) { - let effective = Self::compute_effective(oom_wg, thermal_cap, configured_wg); - Self::store_workgroup_stat(&self.oom_work_groups_min, oom_wg); + /// Report the work_groups one GPU is currently allowed to run; aggregates the + /// worst case across devices for panel stats. `oom_allowed_wg` is what the OOM + /// fallback permits right now, which on a healthy card is the configured count. + pub fn report_gpu_workgroups( + &self, + oom_allowed_wg: u32, + thermal_cap: Option, + configured_wg: u32, + ) { + let effective = Self::compute_effective(oom_allowed_wg, thermal_cap, configured_wg); + Self::store_workgroup_stat(&self.oom_allowed_work_groups_min, oom_allowed_wg); Self::store_workgroup_stat(&self.effective_work_groups_min, effective); self.reported_effective_wg.store(effective, Relaxed); } @@ -176,7 +242,7 @@ impl MiningRuntimeState { let prev = self.thermal_cap_wg.load(Relaxed); self.thermal_cap_wg.store(wg, Relaxed); if !self.throttled.swap(true, Relaxed) || prev != wg { - println!( + wlogln!( "[efficiency] Thermal {}C >= {}C - cap work_groups to {} (configured {})", temp_c, max_temp_c, @@ -189,7 +255,7 @@ impl MiningRuntimeState { if self.throttled.load(Relaxed) && temp_c + 5 < max_temp_c { self.thermal_cap_wg.store(0, Relaxed); self.throttled.store(false, Relaxed); - println!( + wlogln!( "[efficiency] Thermal OK ({}C) - removed work_groups thermal cap", temp_c ); @@ -212,7 +278,7 @@ impl MiningRuntimeState { let (cap, _) = self.ensure_thermal_cap(0, configured_wg); self.throttled.store(true, Relaxed); if !self.thermal_paused.swap(true, Relaxed) { - eprintln!( + wlogerr!( "[Thermal] Mining paused fail-closed: {reason}; conservative work_groups cap={cap}" ); } @@ -236,13 +302,13 @@ impl MiningRuntimeState { let (cap, cap_changed) = self.ensure_thermal_cap(throttle_wg, configured_wg); let newly_throttled = !self.throttled.swap(true, Relaxed); if newly_throttled || cap_changed { - eprintln!( + wlogerr!( "[Thermal] {:.1}C >= {:.1}C: cap work_groups to {}", temp_c, max_temp, cap ); } if !self.thermal_paused.swap(true, Relaxed) { - eprintln!( + wlogerr!( "[Thermal] CRITICAL {:.1}C >= {:.1}C: mining paused until <= {:.1}C", temp_c, critical_temp, recovery_temp ); @@ -253,7 +319,7 @@ impl MiningRuntimeState { if temp_c >= max_temp { let (cap, cap_changed) = self.ensure_thermal_cap(throttle_wg, configured_wg); if !self.throttled.swap(true, Relaxed) || cap_changed { - eprintln!( + wlogerr!( "[Thermal] {:.1}C >= {:.1}C: cap work_groups to {}", temp_c, max_temp, cap ); @@ -266,7 +332,7 @@ impl MiningRuntimeState { let was_throttled = self.throttled.swap(false, Relaxed); let old_cap = self.thermal_cap_wg.swap(0, Relaxed); if was_paused || was_throttled || old_cap > 0 { - println!( + wlogln!( "[Thermal] Recovered at {:.1}C (<= {:.1}C): mining resumed, cap removed", temp_c, recovery_temp ); @@ -285,7 +351,7 @@ impl MiningRuntimeState { } let (cap, _) = self.ensure_thermal_cap(throttle_wg, configured_wg); self.throttled.store(true, Relaxed); - eprintln!( + wlogerr!( "[Thermal] Sensor missed 3 consecutive samples; preserving safety state and capping work_groups to {cap}" ); } @@ -314,7 +380,114 @@ impl MiningRuntimeState { } } -const THERMAL_POLL_INTERVAL: std::time::Duration = std::time::Duration::from_millis(2_500); +/// What one sensor pass costs. +/// +/// A pass probes every selected GPU, one after another, and each probe is a +/// subprocess (or, with `thermal_file`, one file read). A probe that answers +/// costs milliseconds; a probe that stalls is bounded by SENSOR_COMMAND_TIMEOUT +/// (2s) plus the bounded kill of a process that ignored it, which is the 2.5s +/// per GPU used here. So a healthy 8-GPU rig finishes a pass in well under a +/// second and a sick one spends 20s in it, and every interval below is chosen +/// against that number rather than against a hoped-for one. +const SENSOR_PASS_WORST_CASE_PER_GPU: Duration = Duration::from_millis(2_500); + +fn worst_case_pass(sensor_count: usize) -> Duration { + let sensors = u32::try_from(sensor_count.max(1)).unwrap_or(u32::MAX); + SENSOR_PASS_WORST_CASE_PER_GPU.saturating_mul(sensors) +} + +/// The idle the monitor leaves between the END of one pass and the start of the +/// next, whatever that pass cost. +/// +/// Before the start-to-start change the loop simply slept this long after every +/// pass. Keeping it as a floor is what stops a rig whose passes run longer than +/// its interval from polling with zero idle: back-to-back subprocess spawns, +/// forever, on a machine that is mining. On the guarded path the floor and the +/// interval are the same 2.5s, so the guarded cadence is arithmetically the one +/// that shipped before this work (pass, then 2.5s idle, always) and a guarded +/// rig cannot come out of it spending a larger share of its wall clock probing +/// sensors than it did going in. +const THERMAL_MIN_IDLE: Duration = Duration::from_millis(2_500); + +/// Guarded cadence (`max_temp_c > 0`). Here the temperature is a safety input: +/// a late sample is a late pause, so the target stays at 2.5s. The interval is +/// deliberately NOT stretched to cover a slow pass; THERMAL_MIN_IDLE is what +/// bounds the duty cycle, so a rig whose sensors are slow samples as often as +/// those sensors allow and no oftener, exactly as it used to. +const THERMAL_POLL_INTERVAL: Duration = Duration::from_millis(2_500); + +/// Reporting-only cadence (`max_temp_c == 0`, the shipped default). This is a +/// number on a screen, not a safety input. 30s is comfortably longer than the +/// worst-case pass above (~20s on an 8-GPU rig), keeps the duty cycle low on a +/// healthy rig (sub-second pass, then idle), and still refreshes the gauge +/// twice a minute, which is faster than a GPU changes temperature in any way an +/// operator who set no limit needs to watch. +const THERMAL_REPORTING_POLL_INTERVAL: Duration = Duration::from_secs(30); + +/// How the monitor spaces one sensor pass from the next. +/// +/// Two numbers, because one cannot express both halves of a duty cycle. The +/// interval is start-to-start, so a pass that runs long eats into the wait +/// instead of being added on top of it. The idle floor is end-to-start, so a +/// pass that runs longer than the interval still cannot close the gap to zero. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +struct ThermalCadence { + poll_interval: Duration, + min_idle: Duration, +} + +impl ThermalCadence { + const GUARDED: ThermalCadence = ThermalCadence { + poll_interval: THERMAL_POLL_INTERVAL, + min_idle: THERMAL_MIN_IDLE, + }; + + const REPORTING: ThermalCadence = ThermalCadence { + poll_interval: THERMAL_REPORTING_POLL_INTERVAL, + min_idle: THERMAL_MIN_IDLE, + }; + + /// When the next pass may begin, given when this one began and ended. + fn next_poll_after(&self, pass_started: Instant, pass_ended: Instant) -> Instant { + (pass_started + self.poll_interval).max(pass_ended + self.min_idle) + } + + /// The longest wall clock that can separate two published readings on a rig + /// with this many sensors: one whole wait, which a worst-case pass pushes + /// out to the idle floor, and then one whole worst-case pass before the + /// next value exists. + fn worst_case_sample_gap(&self, sensor_count: usize) -> Duration { + let pass = worst_case_pass(sensor_count); + self.poll_interval + .max(pass.saturating_add(self.min_idle)) + .saturating_add(pass) + } +} + +/// How old a reading may get before the panel is told there is none. +/// +/// A gauge still showing the last temperature of a sensor that went silent +/// would be the same lie as a gauge showing 0, but a window shorter than the +/// rig's own sampling period blanks the gauge between two perfectly healthy +/// samples. So it is two whole worst-case sampling periods plus slack, and +/// never below the historical 15s. +/// +/// The sampling period is the cadence AND the pass. That distinction is the +/// whole point: on a guarded rig the interval is 2.5s, but a pass over eight +/// slow-but-healthy sensors takes longer than the old fixed 15s window all by +/// itself, so the window has to be measured against the pass or a working rig +/// blanks its own gauge. This is a display window only; the safety path acts on +/// each reading as it arrives and never consults it. +fn temp_freshness_ms(cadence: ThermalCadence, sensor_count: usize) -> u64 { + let gap = u64::try_from(cadence.worst_case_sample_gap(sensor_count).as_millis()) + .unwrap_or(u64::MAX); + gap.saturating_mul(2).saturating_add(5_000).max(15_000) +} + +/// A sensor that flaps gets at most one failure line, and later one recovery +/// line, per this window. Without it a sensor answering every other sample +/// writes a pair of lines every few seconds into a 4 MiB worker log. +const THERMAL_MISS_LOG_INTERVAL: Duration = Duration::from_secs(300); fn thermal_monitor_should_stop(stop_flag: &Option>) -> bool { stop_flag @@ -323,24 +496,143 @@ fn thermal_monitor_should_stop(stop_flag: &Option>) -> bool { - let deadline = std::time::Instant::now() + THERMAL_POLL_INTERVAL; +fn wait_for_thermal_poll( + deadline: Instant, + stop_flag: &Option>, +) -> bool { loop { if thermal_monitor_should_stop(stop_flag) { return true; } - let now = std::time::Instant::now(); + let now = Instant::now(); if now >= deadline { return false; } std::thread::sleep( deadline .saturating_duration_since(now) - .min(std::time::Duration::from_millis(100)), + .min(Duration::from_millis(100)), ); } } +/// Log state for one monitor thread, kept across samples so a flapping sensor +/// cannot turn into a permanent log writer. +struct ThermalSampleLog { + consecutive_misses: u32, + /// Whether the current miss streak actually printed a line. A streak that + /// was rate-limited into silence gets no recovery line either, so the two + /// always appear as a pair or not at all. + miss_streak_logged: bool, + last_miss_log: Option, +} + +impl ThermalSampleLog { + fn new() -> Self { + ThermalSampleLog { + consecutive_misses: 0, + miss_streak_logged: false, + last_miss_log: None, + } + } + + fn should_log_miss(&mut self) -> bool { + let now = Instant::now(); + if self + .last_miss_log + .is_some_and(|at| now.duration_since(at) < THERMAL_MISS_LOG_INTERVAL) + { + return false; + } + self.last_miss_log = Some(now); + self.miss_streak_logged = true; + true + } +} + +/// Everything one sample does, with the reading already taken. Split out from +/// the loop so the reporting-only guarantees can be tested on the real branch +/// instead of on an early return. +fn handle_thermal_sample( + runtime: &MiningRuntimeState, + reading: Option, + max_temp_c: u32, + throttle_wg: u32, + configured_wg: u32, + log: &mut ThermalSampleLog, +) { + let reporting_only = max_temp_c == 0; + match reading { + Some(temp) => { + if log.miss_streak_logged { + wlogln!( + "[Thermal] Sensor recovered after {} missed sample(s)", + log.consecutive_misses + ); + log.miss_streak_logged = false; + } + log.consecutive_misses = 0; + runtime.record_gpu_temperature(temp); + runtime.observe_thermal_temperature(max_temp_c, throttle_wg, configured_wg, temp); + } + None => { + log.consecutive_misses = log.consecutive_misses.saturating_add(1); + if log.consecutive_misses == 1 && log.should_log_miss() { + if reporting_only { + wlogerr!( + "[Thermal] Sensor read failed; the reported temperature goes stale until it answers again" + ); + } else { + wlogerr!("[Thermal] Sensor read failed; preserving the current safety state"); + } + } + // A miss is not a safety event where no limit was set: the reading + // simply goes stale and stops being published. + if !reporting_only { + runtime.observe_thermal_sensor_miss( + log.consecutive_misses, + throttle_wg, + configured_wg, + ); + } + } + } +} + +fn run_thermal_monitor( + runtime: &MiningRuntimeState, + sensors: &[crate::efficiency::GpuTempSensorBackend], + max_temp_c: u32, + throttle_wg: u32, + configured_wg: u32, + cadence: ThermalCadence, + stop_flag: &Option>, +) { + let mut log = ThermalSampleLog::new(); + let mut next_poll = Instant::now() + cadence.poll_interval; + loop { + if wait_for_thermal_poll(next_poll, stop_flag) { + return; + } + // Both ends of the duty cycle, from the two clocks the pass itself + // provides: the interval is measured from where this pass began, so a + // slow pass is not added on top of a full wait; the idle floor is + // measured from where it ended, so a pass slower than the interval + // still cannot spawn its next subprocess immediately. + let pass_started = Instant::now(); + let reading = hottest_sensor_reading(sensors); + handle_thermal_sample( + runtime, + reading, + max_temp_c, + throttle_wg, + configured_wg, + &mut log, + ); + next_poll = cadence.next_poll_after(pass_started, Instant::now()); + } +} + fn hottest_sensor_reading(sensors: &[crate::efficiency::GpuTempSensorBackend]) -> Option { let mut hottest: Option = None; for sensor in sensors { @@ -355,7 +647,128 @@ fn hottest_sensor_reading(sensors: &[crate::efficiency::GpuTempSensorBackend]) - /// spawn failure is a transient OS resource problem worth retrying. const THERMAL_SPAWN_ATTEMPTS: u32 = 4; +/// Why no sensor answered, naming only what this platform can actually offer. +/// +/// The old wording advised `amd-smi` and `rocm-smi` on every platform. On +/// Windows neither of those can exist (ROCm is a Linux stack and amd-smi is not +/// part of the Adrenalin driver) and neither can the Linux hwmon `thermal_file`, +/// so the advice sent every Windows operator after tools they could not install. +fn no_sensor_reason(vendor: crate::gpu_arch::GpuVendor, gpu_index: u32) -> String { + let supported = match vendor { + #[cfg(windows)] + crate::gpu_arch::GpuVendor::Amd => crate::gpu_temp_adl::unavailable_reason(gpu_index), + #[cfg(not(windows))] + crate::gpu_arch::GpuVendor::Amd => "supported: amd-smi, rocm-smi, or thermal_file".into(), + crate::gpu_arch::GpuVendor::Nvidia => { + // nvidia-smi.exe is installed with the Windows driver, so unlike + // the AMD tools it is a fair thing to ask for on either platform. + "supported: nvidia-smi or thermal_file".to_string() + } + crate::gpu_arch::GpuVendor::Intel | crate::gpu_arch::GpuVendor::Unknown => { + #[cfg(windows)] + { + "this platform publishes no GPU temperature for that vendor".to_string() + } + #[cfg(not(windows))] + { + "supported: thermal_file".to_string() + } + } + }; + format!("no exact temperature sensor for {vendor:?} GPU {gpu_index}; {supported}") +} + +/// Probe one sensor per GPU identity and return them with the hottest initial +/// reading, or the reason there is no usable sensor. +/// +/// Every probe here is a subprocess (AMD tries five candidates), so on the +/// guarded path this runs on the caller before mining starts - the first +/// reading is a safety precondition there - and on the reporting-only path it +/// runs on the monitor thread instead, where nothing waits for it. +fn detect_thermal_sensors( + thermal_file: &str, + identities: &[(crate::gpu_arch::GpuVendor, u32)], +) -> Result<(Vec, f32), String> { + let mut sensors = Vec::with_capacity(identities.len()); + let mut hottest: Option = None; + for &(vendor, gpu_index) in identities { + // The single configured thermal_file describes the first identity only; + // the caller has already refused to let it stand in for several GPUs. + let file = if sensors.is_empty() { thermal_file } else { "" }; + let Some((sensor, temp)) = crate::efficiency::detect_gpu_temp_sensor(file, gpu_index, vendor) + else { + return Err(no_sensor_reason(vendor, gpu_index)); + }; + wlogln!( + "[Thermal] Monitoring {:?} GPU {} with {} (initial {:.1}C)", + vendor, + gpu_index, + sensor.label(), + temp + ); + hottest = Some(hottest.map_or(temp, |current: f32| current.max(temp))); + sensors.push(sensor); + } + match hottest { + Some(temp) => Ok((sensors, temp)), + None => Err("no initialized GPU identity".to_string()), + } +} + +/// Reporting-only monitor: sensor detection AND polling both happen on the +/// monitor thread, so the caller reaches its mining threads immediately and the +/// first temperature simply arrives a moment later. `configured_wg` is not +/// passed in at all: with no limit set there is nothing here that may cap. +fn spawn_reporting_only_thermal_monitor( + runtime: &Arc, + thermal_file: String, + identities: Vec<(crate::gpu_arch::GpuVendor, u32)>, + stop_flag: Option>, +) { + let monitor_runtime = runtime.clone(); + let guard_runtime = runtime.clone(); + let spawn_result = std::thread::Builder::new() + .name("hac-thermal-report".to_string()) + .spawn(move || { + let _monitor_guard = guard_runtime.track_mining_thread(); + let (sensors, initial) = match detect_thermal_sensors(&thermal_file, &identities) { + Ok(found) => found, + Err(reason) => { + // One line, then the thread ends: an operator who turned the + // guard off gets an empty gauge, not a stopped miner. + wlogln!("[Thermal] No GPU temperature to report: {reason}"); + return; + } + }; + monitor_runtime.record_gpu_temperature(initial); + run_thermal_monitor( + &monitor_runtime, + &sensors, + 0, + 0, + 0, + ThermalCadence::REPORTING, + &stop_flag, + ); + }); + if let Err(error) = spawn_result { + // No retry loop with backoff here: retrying on the caller is exactly the + // startup stall this path exists to avoid, and the only thing lost is a + // gauge reading. + wlogerr!("[Thermal] reporting-only monitor thread not started: {error}"); + } +} + /// Start a vendor-specific cached GPU sensor monitor before mining workers run. +/// +/// With `max_temp_c = 0` there is no thermal limit to enforce, but the sensor is +/// still the only place a real GPU temperature exists and the panel has a gauge +/// for it. The monitor then runs in reporting-only mode: it publishes what it +/// reads and does nothing else. It never pauses mining, never caps work groups, +/// and gives up quietly on a machine with no sensor, because none of those are +/// safety decisions an operator who turned the guard off asked for. That mode +/// is also the shipped default, so it must cost the caller nothing: it does all +/// of its work, sensor probes included, on its own thread and returns at once. pub fn start_thermal_monitor( runtime: &Arc, eff: &EfficiencyConf, @@ -363,9 +776,7 @@ pub fn start_thermal_monitor( devices: &[(crate::gpu_arch::GpuVendor, u32)], stop_flag: Option>, ) { - if eff.max_temp_c == 0 { - return; - } + let reporting_only = eff.max_temp_c == 0; let mut identities = Vec::new(); for identity in devices.iter().copied() { @@ -374,60 +785,52 @@ pub fn start_thermal_monitor( } } if identities.is_empty() { - runtime.fail_closed_thermal(configured_wg, "no initialized GPU identity"); + if !reporting_only { + runtime.fail_closed_thermal(configured_wg, "no initialized GPU identity"); + } return; } if !eff.thermal_file.trim().is_empty() && identities.len() != 1 { - runtime.fail_closed_thermal( - configured_wg, - "one thermal_file cannot safely identify multiple selected GPUs", - ); - return; - } - - let mut sensors = Vec::with_capacity(identities.len()); - let mut initial_hottest: Option = None; - for (vendor, gpu_index) in identities { - let thermal_file = if sensors.is_empty() { - eff.thermal_file.as_str() - } else { - "" - }; - let Some((sensor, temp)) = - crate::efficiency::detect_gpu_temp_sensor(thermal_file, gpu_index, vendor) - else { - let supported = match vendor { - crate::gpu_arch::GpuVendor::Amd => "amd-smi, rocm-smi, or thermal_file", - crate::gpu_arch::GpuVendor::Nvidia => "nvidia-smi or thermal_file", - crate::gpu_arch::GpuVendor::Intel | crate::gpu_arch::GpuVendor::Unknown => { - "thermal_file" - } - }; + if !reporting_only { runtime.fail_closed_thermal( configured_wg, - &format!( - "no exact temperature sensor for {:?} GPU {}; supported: {}", - vendor, gpu_index, supported - ), + "one thermal_file cannot safely identify multiple selected GPUs", ); - return; - }; - println!( - "[Thermal] Monitoring {:?} GPU {} with {} (initial {:.1}C)", - vendor, - gpu_index, - sensor.label(), - temp + } + return; + } + + if reporting_only { + // The gauge cadence, and with it the freshness window the panel is + // judged by, is much slower than the guarded one. Every identity gets + // its own sensor, so the count is what a pass will have to walk. + runtime.set_temp_freshness_window(ThermalCadence::REPORTING, identities.len()); + spawn_reporting_only_thermal_monitor( + runtime, + eff.thermal_file.clone(), + identities, + stop_flag, ); - initial_hottest = Some(initial_hottest.map_or(temp, |current| current.max(temp))); - sensors.push(sensor); + return; } + runtime.set_temp_freshness_window(ThermalCadence::GUARDED, identities.len()); + // Guarded path only: the first reading is a safety precondition, so it is + // taken here, on the caller, before any mining thread is spawned. + let (sensors, initial_hottest) = match detect_thermal_sensors(&eff.thermal_file, &identities) { + Ok(found) => found, + Err(reason) => { + runtime.fail_closed_thermal(configured_wg, &reason); + return; + } + }; + + runtime.record_gpu_temperature(initial_hottest); runtime.observe_thermal_temperature( eff.max_temp_c, eff.throttle_workgroups, configured_wg, - initial_hottest.unwrap_or(eff.max_temp_c as f32 + 5.0), + initial_hottest, ); let max_temp_c = eff.max_temp_c; @@ -446,42 +849,15 @@ pub fn start_thermal_monitor( .name("hac-thermal-monitor".to_string()) .spawn(move || { let _monitor_guard = guard_runtime.track_mining_thread(); - let mut consecutive_misses = 0u32; - loop { - if wait_for_thermal_poll(&stop_flag) { - return; - } - match hottest_sensor_reading(&sensors) { - Some(temp) => { - if consecutive_misses > 0 { - println!( - "[Thermal] Sensor recovered after {} missed sample(s)", - consecutive_misses - ); - } - consecutive_misses = 0; - monitor_runtime.observe_thermal_temperature( - max_temp_c, - throttle_wg, - configured_wg, - temp, - ); - } - None => { - consecutive_misses = consecutive_misses.saturating_add(1); - if consecutive_misses == 1 { - eprintln!( - "[Thermal] Sensor read failed; preserving the current safety state" - ); - } - monitor_runtime.observe_thermal_sensor_miss( - consecutive_misses, - throttle_wg, - configured_wg, - ); - } - } - } + run_thermal_monitor( + &monitor_runtime, + &sensors, + max_temp_c, + throttle_wg, + configured_wg, + ThermalCadence::GUARDED, + &stop_flag, + ); }); match spawn_result { Ok(_) => { @@ -489,17 +865,19 @@ pub fn start_thermal_monitor( break; } Err(error) => { - eprintln!( + wlogerr!( "[Thermal] monitor spawn attempt {attempt}/{THERMAL_SPAWN_ATTEMPTS} failed: {error}" ); if attempt < THERMAL_SPAWN_ATTEMPTS { - std::thread::sleep(std::time::Duration::from_millis(250u64 * attempt as u64)); + std::thread::sleep(Duration::from_millis(250u64 * attempt as u64)); } } } } // Only fail-close (pause all mining) after exhausting retries - a single - // transient spawn failure must not permanently halt a working miner. + // transient spawn failure must not permanently halt a working miner. This + // is the guarded path only; reporting-only returned above and never gets + // here, so no `reporting_only` check is needed (or wanted) around it. if !spawned { runtime.fail_closed_thermal( configured_wg, @@ -584,6 +962,304 @@ mod tests { assert_eq!(runtime.thermal_workgroups_cap(), Some(20)); } + #[test] + fn a_runtime_that_never_read_a_sensor_reports_no_temperature() { + let runtime = MiningRuntimeState::new(64, 0); + assert_eq!(runtime.gpu_temp_c(), None); + // Nothing a sensor cannot produce is ever published, so the panel can + // never be handed a 0 that looks like a cold card. + for impossible in [0.0, -5.0, 120.0, 999.0, f32::NAN] { + runtime.record_gpu_temperature(impossible); + assert_eq!(runtime.gpu_temp_c(), None, "{impossible}"); + } + runtime.record_gpu_temperature(67.5); + assert_eq!(runtime.gpu_temp_c(), Some(67.5)); + } + + #[test] + fn a_sensor_that_went_silent_stops_being_quoted() { + let runtime = MiningRuntimeState::new(64, 0); + runtime.record_gpu_temperature(71.25); + assert_eq!(runtime.gpu_temp_c(), Some(71.25)); + // Age the reading past the freshness window without touching the value: + // the last temperature of a dead sensor is not a current temperature. + let taken_at = runtime.gpu_temp_unix_ms.load(Relaxed); + let window = runtime.gpu_temp_fresh_ms.load(Relaxed); + runtime + .gpu_temp_unix_ms + .store(taken_at - window - 1, Relaxed); + assert_eq!(runtime.gpu_temp_c(), None); + } + + #[test] + fn freshness_window_follows_the_poll_cadence() { + // One constant that can contradict the interval is what let a 15s window + // expire on every single sample of a slower loop. + assert_eq!(temp_freshness_ms(ThermalCadence::GUARDED, 1), 20_000); + assert!( + temp_freshness_ms(ThermalCadence::REPORTING, 1) + > 2 * THERMAL_REPORTING_POLL_INTERVAL.as_millis() as u64 + ); + let runtime = MiningRuntimeState::new(64, 0); + assert_eq!(runtime.gpu_temp_fresh_ms.load(Relaxed), 20_000); + runtime.set_temp_freshness_window(ThermalCadence::REPORTING, 1); + assert_eq!(runtime.gpu_temp_fresh_ms.load(Relaxed), 70_000); + } + + #[test] + fn the_freshness_window_covers_a_slow_pass_on_a_guarded_multi_gpu_rig() { + // Eight guarded GPUs answering slowly but successfully: about 1.9s of + // subprocess each, so one healthy pass is 15.2s of wall clock. + let healthy_but_slow_pass_ms = 8 * 1_900u64; + assert!( + healthy_but_slow_pass_ms > 15_000, + "the old fixed 15s window expired inside a single healthy pass, \ + which blanked the gauge between two good samples" + ); + + let window = temp_freshness_ms(ThermalCadence::GUARDED, 8); + assert!( + window > 2 * healthy_but_slow_pass_ms, + "a guarded window of {window}ms must survive two slow passes" + ); + // And it is derived, not guessed: the window follows the sensor count + // because the pass does. + assert!(temp_freshness_ms(ThermalCadence::GUARDED, 8) > temp_freshness_ms(ThermalCadence::GUARDED, 1)); + + let runtime = MiningRuntimeState::new(64, 0); + runtime.set_temp_freshness_window(ThermalCadence::GUARDED, 8); + runtime.record_gpu_temperature(74.0); + let taken_at = runtime.gpu_temp_unix_ms.load(Relaxed); + // A reading that is one whole slow pass old is still the current + // reading of this rig, because this rig cannot produce one faster. + runtime + .gpu_temp_unix_ms + .store(taken_at - healthy_but_slow_pass_ms, Relaxed); + assert_eq!(runtime.gpu_temp_c(), Some(74.0)); + } + + #[test] + fn the_guarded_rig_keeps_the_idle_it_had_before_start_to_start() { + // The pre-fix loop slept 2.5s after every pass, however long the pass + // was. The guarded cadence must still do exactly that: a 20s pass on an + // 8-GPU rig may not be followed by an immediate respawn. + let started = Instant::now(); + for pass in [ + Duration::from_millis(1), + Duration::from_millis(400), + Duration::from_secs(3), + Duration::from_secs(20), + ] { + let ended = started + pass; + let next = ThermalCadence::GUARDED.next_poll_after(started, ended); + assert_eq!( + next, + ended + Duration::from_millis(2_500), + "guarded pass of {pass:?} must be followed by the full 2.5s idle" + ); + assert!( + next.duration_since(ended) >= THERMAL_MIN_IDLE, + "guarded pass of {pass:?} left less idle than the floor" + ); + } + } + + #[test] + fn the_reporting_loop_keeps_start_to_start_and_still_cannot_reach_zero_idle() { + let started = Instant::now(); + // The fix this is protecting: a 20s pass does not turn a 30s loop into + // a 50s one, so the wait absorbs it. + let slow = started + Duration::from_secs(20); + assert_eq!( + ThermalCadence::REPORTING.next_poll_after(started, slow), + started + THERMAL_REPORTING_POLL_INTERVAL + ); + // But absorbing it has a floor: a pass longer than the interval leaves + // idle rather than spawning subprocesses back to back forever. + let pathological = started + Duration::from_secs(45); + let next = ThermalCadence::REPORTING.next_poll_after(started, pathological); + assert_eq!(next, pathological + THERMAL_MIN_IDLE); + assert!(next.duration_since(pathological) >= THERMAL_MIN_IDLE); + } + + #[test] + fn a_worst_case_pass_is_counted_per_gpu() { + assert_eq!(worst_case_pass(0), SENSOR_PASS_WORST_CASE_PER_GPU); + assert_eq!(worst_case_pass(1), SENSOR_PASS_WORST_CASE_PER_GPU); + assert_eq!(worst_case_pass(8), Duration::from_secs(20)); + // Nothing here may overflow into a tiny window on an absurd rig. + assert!(worst_case_pass(usize::MAX) >= Duration::from_secs(20)); + assert!(temp_freshness_ms(ThermalCadence::GUARDED, usize::MAX) >= 90_000); + } + + #[test] + fn reporting_only_mode_never_pauses_or_caps_anything() { + // max_temp_c = 0 is the panel's own default: the guard is off. This + // drives the monitor's real per-sample body (the same call the thread + // makes every tick), not an early return, with the exact readings that + // pause and cap a guarded miner. + let runtime = MiningRuntimeState::new(64, 0); + runtime.reported_effective_wg.store(32, Relaxed); + let mut log = ThermalSampleLog::new(); + + for reading in [ + Some(40.0), + Some(79.9), + Some(80.0), // at the limit: guarded caps here + Some(95.0), // over critical: guarded pauses here + None, // sensor miss + None, // second miss + None, // third miss: guarded caps fail-closed here + None, // fourth + Some(110.0), // recovery straight into a critical reading + ] { + handle_thermal_sample(&runtime, reading, 0, 64, 64, &mut log); + assert!(!runtime.thermal_pause_active(), "{reading:?}"); + assert!(!runtime.throttled.load(Relaxed), "{reading:?}"); + assert_eq!(runtime.thermal_workgroups_cap(), None, "{reading:?}"); + } + // The readings were still published: reporting-only reports. + assert_eq!(runtime.gpu_temp_c(), Some(110.0)); + + // The same readings on a guarded runtime do pause and cap, which is what + // makes the assertions above a real guarantee rather than a dead branch. + let guarded = MiningRuntimeState::new(64, 0); + guarded.reported_effective_wg.store(32, Relaxed); + let mut guarded_log = ThermalSampleLog::new(); + handle_thermal_sample(&guarded, Some(95.0), 80, 64, 64, &mut guarded_log); + assert!(guarded.thermal_pause_active()); + assert_eq!(guarded.thermal_workgroups_cap(), Some(16)); + + let missing = MiningRuntimeState::new(64, 0); + missing.reported_effective_wg.store(32, Relaxed); + let mut missing_log = ThermalSampleLog::new(); + for _ in 0..3 { + handle_thermal_sample(&missing, None, 80, 64, 64, &mut missing_log); + } + assert!(missing.throttled.load(Relaxed)); + assert_eq!(missing.thermal_workgroups_cap(), Some(16)); + } + + #[test] + fn a_flapping_sensor_does_not_fill_the_worker_log() { + // Miss, hit, miss, hit ... used to print a failure line and a recovery + // line every few seconds forever. One pair per rate-limit window now. + let runtime = MiningRuntimeState::new(64, 0); + let mut log = ThermalSampleLog::new(); + + handle_thermal_sample(&runtime, None, 0, 0, 0, &mut log); + assert!(log.miss_streak_logged, "first failure is reported"); + handle_thermal_sample(&runtime, Some(55.0), 0, 0, 0, &mut log); + assert!(!log.miss_streak_logged, "recovery line pairs with it"); + + for _ in 0..50 { + handle_thermal_sample(&runtime, None, 0, 0, 0, &mut log); + assert!( + !log.miss_streak_logged, + "later streaks stay silent inside the window" + ); + handle_thermal_sample(&runtime, Some(55.0), 0, 0, 0, &mut log); + } + + // Once the window has passed, one more pair is allowed through. + log.last_miss_log = Some(Instant::now() - THERMAL_MISS_LOG_INTERVAL); + handle_thermal_sample(&runtime, None, 0, 0, 0, &mut log); + assert!(log.miss_streak_logged); + } + + #[test] + fn reporting_only_start_never_fail_closes_on_a_bad_configuration() { + // The two early returns: nothing to monitor, and one thermal_file that + // cannot stand for several GPUs. Both fail-close when a limit is set. + let efficiency = EfficiencyConf::from_ini(&sys::IniObj::new()); + assert_eq!(efficiency.max_temp_c, 0); + + let runtime = MiningRuntimeState::new(64, 0); + start_thermal_monitor(&runtime, &efficiency, 64, &[], None); + + let mut with_file = efficiency.clone(); + with_file.thermal_file = "one-sensor.txt".to_string(); + start_thermal_monitor( + &runtime, + &with_file, + 64, + &[ + (crate::gpu_arch::GpuVendor::Amd, 0), + (crate::gpu_arch::GpuVendor::Amd, 1), + ], + None, + ); + + // And a real identity with no sensor at all on this machine. + start_thermal_monitor( + &runtime, + &efficiency, + 64, + &[(crate::gpu_arch::GpuVendor::Unknown, 0)], + None, + ); + assert!(!runtime.thermal_pause_active()); + assert!(!runtime.throttled.load(Relaxed)); + assert_eq!(runtime.thermal_workgroups_cap(), None); + } + + #[test] + fn reporting_only_detects_and_publishes_off_the_caller_thread() { + // The name of this test is the claim that the caller does not pay for + // sensor probes, and the only thing that can prove it is ORDER: + // `start_thermal_monitor` must return while there is still no reading, + // and the reading must turn up afterwards. Asserting the published + // value alone proves nothing, because a synchronous implementation + // (detect on the caller, publish, then spawn) publishes exactly the + // same value, never pauses and never caps: it passes every other + // assertion here and fails only the one below. + // + // The order is decided by an OS thread start followed by a file open, + // read and parse on one side, against a handful of instructions and a + // return on the other, so this is not a coin flip. + let path = std::env::temp_dir().join(format!( + "hacash-thermal-monitor-test-{}.txt", + std::process::id() + )); + std::fs::write(&path, "61.5\n").expect("write test sensor file"); + + let runtime = MiningRuntimeState::new(64, 0); + let mut efficiency = EfficiencyConf::from_ini(&sys::IniObj::new()); + efficiency.thermal_file = path.to_string_lossy().to_string(); + let stop = Arc::new(AtomicBool::new(false)); + let called_at = Instant::now(); + start_thermal_monitor( + &runtime, + &efficiency, + 64, + &[(crate::gpu_arch::GpuVendor::Amd, 0)], + Some(stop.clone()), + ); + let returned_in = called_at.elapsed(); + let at_return = runtime.gpu_temp_c(); + + let deadline = Instant::now() + Duration::from_secs(10); + while runtime.gpu_temp_c().is_none() && Instant::now() < deadline { + std::thread::sleep(Duration::from_millis(20)); + } + stop.store(true, Relaxed); + let published = runtime.gpu_temp_c(); + let _ = std::fs::remove_file(&path); + + assert_eq!( + at_return, None, + "the sensor had already been read when start_thermal_monitor \ + returned, so detection was on the caller after all" + ); + assert!( + returned_in < Duration::from_millis(250), + "start_thermal_monitor took {returned_in:?} of the caller's time" + ); + assert_eq!(published, Some(61.5)); + assert!(!runtime.thermal_pause_active()); + assert_eq!(runtime.thermal_workgroups_cap(), None); + } + #[test] fn missing_initial_sensor_is_fail_closed() { let runtime = MiningRuntimeState::new(64, 0); @@ -595,6 +1271,179 @@ mod tests { assert_eq!(runtime.thermal_workgroups_cap(), Some(32)); } + #[test] + fn the_guarded_first_reading_and_first_safety_decision_happen_before_the_call_returns() { + // block_mining_runtime spawns its mining threads on the line after + // start_thermal_monitor returns. So on the guarded path everything that + // decides whether this rig may hash at all - the probe, the publish and + // the first observe - has to be finished by the time control comes back + // here. Moving any of it onto the monitor thread lets the miner run + // during the gap, and a mutation that did exactly that still passed + // every other test in this file. + let base = std::env::temp_dir(); + let cool = base.join(format!( + "hacash-guarded-first-reading-cool-{}.txt", + std::process::id() + )); + let critical = base.join(format!( + "hacash-guarded-first-reading-critical-{}.txt", + std::process::id() + )); + std::fs::write(&cool, "63.5\n").expect("write cool test sensor file"); + std::fs::write(&critical, "95.0\n").expect("write critical test sensor file"); + + let runtime = MiningRuntimeState::new(64, 0); + let mut efficiency = EfficiencyConf::from_ini(&sys::IniObj::new()); + efficiency.max_temp_c = 80; + efficiency.thermal_file = cool.to_string_lossy().to_string(); + let stop = Arc::new(AtomicBool::new(false)); + start_thermal_monitor( + &runtime, + &efficiency, + 64, + &[(crate::gpu_arch::GpuVendor::Amd, 0)], + Some(stop.clone()), + ); + // Read on this line, not after a wait: the claim is about the instant of + // return, so anything that arrives later is too late to count. + let cool_at_return = runtime.gpu_temp_c(); + let cool_gated_at_return = crate::efficiency::mining_is_gated(&runtime, &efficiency); + stop.store(true, Relaxed); + + // Same call on a rig that is already past critical: the pause must also + // be in place at the instant of return, because there is no later. + let hot_runtime = MiningRuntimeState::new(64, 0); + let mut hot_efficiency = efficiency.clone(); + hot_efficiency.thermal_file = critical.to_string_lossy().to_string(); + let hot_stop = Arc::new(AtomicBool::new(false)); + start_thermal_monitor( + &hot_runtime, + &hot_efficiency, + 64, + &[(crate::gpu_arch::GpuVendor::Amd, 0)], + Some(hot_stop.clone()), + ); + let hot_at_return = hot_runtime.gpu_temp_c(); + let hot_gated_at_return = + crate::efficiency::mining_is_gated(&hot_runtime, &hot_efficiency); + let hot_cap_at_return = hot_runtime.thermal_workgroups_cap(); + hot_stop.store(true, Relaxed); + + let _ = std::fs::remove_file(&cool); + let _ = std::fs::remove_file(&critical); + + assert_eq!( + cool_at_return, + Some(63.5), + "start_thermal_monitor returned with no guarded temperature yet, so \ + mining threads spawn before this rig has been read at all" + ); + assert!( + !cool_gated_at_return, + "a 63.5C rig under an 80C limit must be free to mine on return" + ); + assert_eq!( + hot_at_return, + Some(95.0), + "the critical rig had not been read when start_thermal_monitor returned" + ); + assert!( + hot_gated_at_return, + "a 95C rig under an 80C limit was still allowed to mine on return" + ); + assert_eq!(hot_cap_at_return, Some(32)); + } + + #[test] + fn a_guarded_rig_whose_sensor_is_gone_is_gated_before_the_call_returns() { + // The uncooled-machine case: a limit is configured and there is no + // sensor to enforce it with. mining_is_gated is the exact predicate + // block_mining_runtime's worker loop consults, so asserting on it is + // asserting that no hash is computed. + let path = std::env::temp_dir().join(format!( + "hacash-guarded-vanished-sensor-{}.txt", + std::process::id() + )); + // Written and then removed: this is a sensor that went away, not a path + // that was never plausible. + std::fs::write(&path, "63.5\n").expect("write test sensor file"); + std::fs::remove_file(&path).expect("remove test sensor file"); + + let runtime = MiningRuntimeState::new(64, 0); + let mut efficiency = EfficiencyConf::from_ini(&sys::IniObj::new()); + efficiency.max_temp_c = 80; + efficiency.thermal_file = path.to_string_lossy().to_string(); + + // Control: the gate is open beforehand, so whatever closes it below is + // the fail-close and not the idle schedule this machine happens to run + // the suite in. + assert!( + !crate::efficiency::mining_is_gated(&runtime, &efficiency), + "mining was already gated before the monitor started, so this test \ + would pass without any fail-close at all" + ); + + start_thermal_monitor( + &runtime, + &efficiency, + 64, + &[(crate::gpu_arch::GpuVendor::Amd, 0)], + None, + ); + let _ = std::fs::remove_file(&path); + + assert!( + crate::efficiency::mining_is_gated(&runtime, &efficiency), + "a rig with an 80C limit and no sensor was cleared to mine uncooled" + ); + assert!(runtime.thermal_pause_active()); + assert!(runtime.throttled.load(Relaxed)); + assert_eq!(runtime.thermal_workgroups_cap(), Some(32)); + + // Control: a vendor with no sensor of its own and no file gets the same + // answer, so the guarantee is about having no reading, not about files. + let no_sensor_at_all = MiningRuntimeState::new(64, 0); + let mut vendorless = EfficiencyConf::from_ini(&sys::IniObj::new()); + vendorless.max_temp_c = 80; + assert!(!crate::efficiency::mining_is_gated( + &no_sensor_at_all, + &vendorless + )); + start_thermal_monitor( + &no_sensor_at_all, + &vendorless, + 64, + &[(crate::gpu_arch::GpuVendor::Unknown, 0)], + None, + ); + assert!(crate::efficiency::mining_is_gated( + &no_sensor_at_all, + &vendorless + )); + assert_eq!(no_sensor_at_all.thermal_workgroups_cap(), Some(32)); + + // Control: the SAME absent sensor with the guard turned off must not + // gate anything. Without this the two asserts above would also pass if + // start_thermal_monitor simply paused every rig it ever saw. + let unguarded = MiningRuntimeState::new(64, 0); + let mut no_limit = efficiency.clone(); + no_limit.max_temp_c = 0; + start_thermal_monitor( + &unguarded, + &no_limit, + 64, + &[(crate::gpu_arch::GpuVendor::Amd, 0)], + None, + ); + assert!( + !crate::efficiency::mining_is_gated(&unguarded, &no_limit), + "max_temp_c = 0 is the operator saying there is no limit to enforce; \ + a missing sensor is then nothing to fail closed on" + ); + assert!(!unguarded.thermal_pause_active()); + assert_eq!(unguarded.thermal_workgroups_cap(), None); + } + #[test] fn one_thermal_file_cannot_claim_multi_gpu_coverage() { let runtime = MiningRuntimeState::new(64, 0); @@ -613,4 +1462,102 @@ mod tests { ); assert!(runtime.thermal_pause_active()); } + + #[test] + fn a_guarded_rig_that_heats_up_after_start_keeps_being_judged_by_the_monitor_thread() { + // The caller-side first reading is the FIRST of them, not the only one. + // A card that is cool at startup and cooks ten minutes in is the + // ordinary case, and every other test in this file stops at the instant + // start_thermal_monitor returns. That gap is real: a mutation that + // spawned the guarded monitor thread and let it exit immediately - one + // reading, ever, then an uncooled rig hashing forever - left all 173 of + // them green. + // + // So this drives the loop through the sensor it actually polls: cool, + // then hot, then cool again. The last leg matters as much as the first, + // because a monitor that took exactly one more sample would satisfy + // everything up to it. + fn wait_for(deadline: Duration, mut done: impl FnMut() -> bool) -> bool { + let limit = Instant::now() + deadline; + while Instant::now() < limit { + if done() { + return true; + } + std::thread::sleep(Duration::from_millis(25)); + } + done() + } + + let path = std::env::temp_dir().join(format!( + "hacash-guarded-heats-up-{}.txt", + std::process::id() + )); + std::fs::write(&path, "63.5\n").expect("write test sensor file"); + + let runtime = MiningRuntimeState::new(64, 0); + let mut efficiency = EfficiencyConf::from_ini(&sys::IniObj::new()); + efficiency.max_temp_c = 80; + efficiency.thermal_file = path.to_string_lossy().to_string(); + let stop = Arc::new(AtomicBool::new(false)); + start_thermal_monitor( + &runtime, + &efficiency, + 64, + &[(crate::gpu_arch::GpuVendor::Amd, 0)], + Some(stop.clone()), + ); + // Control: cool and free to mine at the instant of return, so the pause + // below cannot be the startup fail-close or the idle schedule. + let temp_at_return = runtime.gpu_temp_c(); + let gated_at_return = crate::efficiency::mining_is_gated(&runtime, &efficiency); + + std::fs::write(&path, "95.0\n").expect("heat the test sensor up"); + let paused = wait_for(Duration::from_secs(30), || runtime.thermal_pause_active()); + let hot_gated = crate::efficiency::mining_is_gated(&runtime, &efficiency); + let hot_temp = runtime.gpu_temp_c(); + let hot_cap = runtime.thermal_workgroups_cap(); + + std::fs::write(&path, "63.5\n").expect("cool the test sensor down"); + let resumed = wait_for(Duration::from_secs(30), || { + !runtime.thermal_pause_active() + }); + let cool_gated = crate::efficiency::mining_is_gated(&runtime, &efficiency); + let cool_cap = runtime.thermal_workgroups_cap(); + + stop.store(true, Relaxed); + let _ = std::fs::remove_file(&path); + + assert_eq!( + temp_at_return, + Some(63.5), + "the caller-side first reading is missing, so this test could not \ + tell a later sample from the first one" + ); + assert!( + !gated_at_return, + "the rig was already gated while still at 63.5C under an 80C limit" + ); + + assert!( + paused, + "the sensor went from 63.5C to 95.0C under an 80C limit and the \ + guarded monitor never paused mining: after start_thermal_monitor \ + returns, nothing is reading this rig" + ); + assert!( + hot_gated, + "thermal_pause_active was set but mining_is_gated still cleared the \ + worker loop to hash at 95.0C" + ); + assert_eq!(hot_temp, Some(95.0), "the 95.0C sample was never published"); + assert_eq!(hot_cap, Some(32), "no conservative work_groups cap at 95.0C"); + + assert!( + resumed, + "the rig cooled back to 63.5C and stayed paused: the monitor took \ + one more sample and stopped, so it is not a loop" + ); + assert!(!cool_gated); + assert_eq!(cool_cap, None, "the thermal cap outlived the heat"); + } } diff --git a/app/src/mining_stats.rs b/app/src/mining_stats.rs index 6006ceb0..ccc65416 100644 --- a/app/src/mining_stats.rs +++ b/app/src/mining_stats.rs @@ -24,8 +24,22 @@ pub struct MiningStatsSnapshot { pub gpu_profile: String, #[serde(default)] pub configured_work_groups: u32, - #[serde(default)] - pub oom_work_groups: u32, + /// The work-group count the OOM fallback currently ALLOWS, per GPU, taken at + /// its worst across the devices in this rig. + /// + /// It is a count, not a loss. On a card that has never hit an out-of-memory + /// batch this equals `configured_work_groups`, and a healthy rig therefore + /// publishes a large positive number here every second it mines. Anything + /// reading `> 0` as "an OOM clamp happened" is wrong about every rig it will + /// ever meet; the clamp is the gap, so the only true test is + /// `oom_allowed_work_groups < configured_work_groups`. + /// + /// Serialised under its old key as well, because a stats file written by an + /// already-installed worker has to keep loading. + #[serde(default, alias = "oom_work_groups")] + pub oom_allowed_work_groups: u32, + /// The thermal cap in force, or 0 when the thermal guard has capped nothing. + /// Unlike the field above, this one really is absent when it is zero. #[serde(default)] pub thermal_cap_work_groups: u32, #[serde(default)] @@ -33,6 +47,15 @@ pub struct MiningStatsSnapshot { pub gpu_hashrate_hps: f64, pub cpu_hashrate_hps: f64, pub gpu_hashrate_display: String, + /// The GPU temperature in Celsius, as read from the sensor this build + /// already talks to (`thermal_file`, `nvidia-smi`, `rocm-smi`, `amd-smi`). + /// + /// `None` where no sensor answered, and that is the whole point of the + /// Option: a machine with no readable sensor must publish nothing here. A + /// gauge reading 0 degrees looks like a working sensor on a cold card, and + /// is the one thing this field must never cause. + #[serde(default)] + pub gpu_temp_c: Option, pub active_cpu_threads: u32, pub paused_unprofitable: bool, pub mining_kind: String, @@ -63,7 +86,7 @@ pub fn emit_from_batch_aggregate( diamond_best: &str, stats_path: &str, ) { - let oom_wg = runtime.oom_work_groups(); + let oom_wg = runtime.oom_allowed_work_groups(); let thermal = runtime.thermal_workgroups_cap().unwrap_or(0); let effective = runtime.effective_work_groups(); let stats = if mining_kind == "hacd" { @@ -98,11 +121,13 @@ pub fn emit_from_batch_aggregate( effective, agg.gpu_hashrate, agg.cpu_hashrate, + runtime.gpu_temp_c(), ) }; write_mining_stats(stats_path, &stats); } +#[allow(clippy::too_many_arguments)] pub fn build_mining_stats( hashrate: f64, hac_per_day: f64, @@ -113,11 +138,12 @@ pub fn build_mining_stats( height: u64, paused: bool, configured_work_groups: u32, - oom_work_groups: u32, + oom_allowed_work_groups: u32, thermal_cap_work_groups: u32, effective_work_groups: u32, gpu_hashrate_hps: f64, cpu_hashrate_hps: f64, + gpu_temp_c: Option, ) -> MiningStatsSnapshot { let gpu_w = eff.estimate_gpu_watts(profile); let watts = gpu_w + active_cpu as f64 * eff.cpu_watts_per_thread; @@ -150,12 +176,13 @@ pub fn build_mining_stats( height, gpu_profile: profile.to_string(), configured_work_groups, - oom_work_groups, + oom_allowed_work_groups, thermal_cap_work_groups, effective_work_groups, gpu_hashrate_hps, cpu_hashrate_hps, gpu_hashrate_display: rates_to_show(gpu_hashrate_hps), + gpu_temp_c: sensor_temperature(gpu_temp_c), active_cpu_threads: active_cpu, paused_unprofitable: paused, mining_kind: "hac".to_string(), @@ -174,7 +201,7 @@ pub fn build_diamond_mining_stats( diamond_best: &str, paused: bool, _configured_work_groups: u32, - _oom_work_groups: u32, + _oom_allowed_work_groups: u32, _thermal_cap_work_groups: u32, _effective_work_groups: u32, _gpu_hashrate_hps: f64, @@ -210,12 +237,15 @@ pub fn build_diamond_mining_stats( height: diamond_number as u64, gpu_profile: String::new(), configured_work_groups: 0, - oom_work_groups: 0, + oom_allowed_work_groups: 0, thermal_cap_work_groups: 0, effective_work_groups: 0, gpu_hashrate_hps: 0.0, cpu_hashrate_hps, gpu_hashrate_display: rates_to_show(0.0), + // HACD mines on the CPU through the full node. There is no GPU under + // this snapshot, so there is no GPU temperature to report. + gpu_temp_c: None, active_cpu_threads: active_cpu, paused_unprofitable: paused, mining_kind: "hacd".to_string(), @@ -225,6 +255,13 @@ pub fn build_diamond_mining_stats( } } +/// The last gate a temperature passes before it is published. Anything a GPU +/// sensor cannot produce is dropped to `None` rather than written out, so a +/// broken reading reaches the panel as "no sensor" instead of as a number. +fn sensor_temperature(temp_c: Option) -> Option { + temp_c.filter(|c| c.is_finite() && *c > 0.0 && *c < 120.0) +} + pub fn write_mining_stats(path: &str, stats: &MiningStatsSnapshot) { if path.is_empty() { return; @@ -243,7 +280,7 @@ pub fn write_mining_stats(path: &str, stats: &MiningStatsSnapshot) { Ok(()) => WARNED.store(false, Relaxed), Err(e) => { if !WARNED.swap(true, Relaxed) { - eprintln!("[stats] cannot update stats file ({e}); suppressing until it recovers"); + wlogerr!("[stats] cannot update stats file ({e}); suppressing until it recovers"); } } } @@ -310,6 +347,116 @@ mod tests { std::fs::remove_file(path).unwrap(); } + fn hac_stats(gpu_temp_c: Option) -> MiningStatsSnapshot { + build_mining_stats( + 1_000_000.0, + 0.5, + 0.01, + &efficiency(), + "amd_profit", + 2, + 765_432, + false, + 1_536, + 0, + 0, + 1_536, + 900_000.0, + 100_000.0, + gpu_temp_c, + ) + } + + #[test] + fn a_machine_with_no_sensor_publishes_no_temperature_rather_than_zero() { + // A gauge reading 0C looks exactly like a working sensor on a cold + // card, so the absence has to survive all the way to the JSON. + let stats = hac_stats(None); + assert_eq!(stats.gpu_temp_c, None); + let json = serde_json::to_string(&stats).unwrap(); + let read: MiningStatsSnapshot = serde_json::from_str(&json).unwrap(); + assert_eq!(read.gpu_temp_c, None); + } + + #[test] + fn a_measured_temperature_is_published_and_survives_the_json() { + let stats = hac_stats(Some(67.5)); + assert_eq!(stats.gpu_temp_c, Some(67.5)); + let json = serde_json::to_string(&stats).unwrap(); + let read: MiningStatsSnapshot = serde_json::from_str(&json).unwrap(); + assert_eq!(read.gpu_temp_c, Some(67.5)); + } + + #[test] + fn a_reading_no_gpu_sensor_could_produce_is_dropped_not_published() { + for impossible in [0.0, -40.0, 500.0, f32::NAN, f32::INFINITY] { + assert_eq!( + hac_stats(Some(impossible)).gpu_temp_c, + None, + "{impossible} is not a GPU temperature" + ); + } + } + + #[test] + fn a_stats_file_written_before_the_sensor_existed_still_reads_back() { + // Older workers wrote no temperature field at all. That file must load + // as "no reading", not fail the whole snapshot and freeze the panel. + let legacy = r#"{"status":"mining","hashrate_hps":1.0,"hashrate_display":"1H/s", + "watts":10.0,"kh_per_j":0.1,"hac_per_day":0.0,"network_pct":0.0, + "daily_cost_eur":0.0,"daily_revenue_eur":0.0,"daily_net_eur":0.0,"height":7, + "gpu_profile":"amd_profit","gpu_hashrate_hps":1.0,"cpu_hashrate_hps":0.0, + "gpu_hashrate_display":"1H/s","active_cpu_threads":0,"paused_unprofitable":false, + "mining_kind":"hac","diamond_number":0,"diamond_best":"","updated_unix_ms":1}"#; + let read: MiningStatsSnapshot = serde_json::from_str(legacy).unwrap(); + assert_eq!(read.gpu_temp_c, None); + assert_eq!(read.height, 7); + } + + #[test] + fn a_healthy_rig_publishes_the_full_work_group_count_under_the_oom_field() { + // The operator's rig: 48 configured, 48 allowed, nothing clamped. This + // is the shape that made the panel draw a red ring, so it is written + // down here as the normal case it actually is. + let stats = build_mining_stats( + 1_000_000.0, + 0.5, + 0.01, + &efficiency(), + "amd_profit", + 2, + 768_566, + false, + 48, + 48, + 0, + 48, + 900_000.0, + 100_000.0, + None, + ); + assert_eq!(stats.oom_allowed_work_groups, stats.configured_work_groups); + assert_eq!(stats.thermal_cap_work_groups, 0); + } + + #[test] + fn a_stats_file_written_under_the_old_oom_key_still_loads() { + // The field was renamed because its old name read as a loss. A worker + // already installed on disk still writes the old key, and its snapshot + // must not come back as zero work groups. + let legacy = r#"{"status":"mining","hashrate_hps":1.0,"hashrate_display":"1H/s", + "watts":10.0,"kh_per_j":0.1,"hac_per_day":0.0,"network_pct":0.0, + "daily_cost_eur":0.0,"daily_revenue_eur":0.0,"daily_net_eur":0.0,"height":7, + "gpu_profile":"amd_profit","configured_work_groups":48,"oom_work_groups":48, + "thermal_cap_work_groups":0,"effective_work_groups":48, + "gpu_hashrate_hps":1.0,"cpu_hashrate_hps":0.0, + "gpu_hashrate_display":"1H/s","active_cpu_threads":0,"paused_unprofitable":false, + "mining_kind":"hac","diamond_number":0,"diamond_best":"","updated_unix_ms":1}"#; + let read: MiningStatsSnapshot = serde_json::from_str(legacy).unwrap(); + assert_eq!(read.oom_allowed_work_groups, 48); + assert_eq!(read.configured_work_groups, 48); + } + #[test] fn hacd_snapshot_is_cpu_only_even_with_legacy_gpu_inputs() { let stats = build_diamond_mining_stats( @@ -327,6 +474,7 @@ mod tests { 900_000.0, 1_000_000.0, ); + assert!(stats.gpu_temp_c.is_none()); assert_eq!(stats.watts, 32.0); assert!((stats.daily_cost_eur - 0.192).abs() < 0.000_001); assert_eq!(stats.gpu_hashrate_hps, 0.0); @@ -338,7 +486,7 @@ mod tests { } } -fn unix_ms_now() -> u64 { +pub(crate) fn unix_ms_now() -> u64 { SystemTime::now() .duration_since(UNIX_EPOCH) .map(|d| d.as_millis() as u64) diff --git a/app/src/opencl_dia.rs b/app/src/opencl_dia.rs index 6cf996a7..b820bbfc 100644 --- a/app/src/opencl_dia.rs +++ b/app/src/opencl_dia.rs @@ -54,7 +54,7 @@ pub(crate) fn do_diamond_group_mining_opencl( // refuse the batch instead of mining garbage. debug_assert!(stuff_len == 61 || stuff_len == 93); if stuff_len != 61 && stuff_len != 93 { - eprintln!( + wlogerr!( "[OpenCL] diamond pre-image length {} is neither 61 nor 93; skipping batch", stuff_len ); @@ -65,7 +65,7 @@ pub(crate) fn do_diamond_group_mining_opencl( let write_event = match write_stuff_to_gpu(opencl, &stuff, None) { Ok(ev) => ev, Err(e) => { - eprintln!("[OpenCL] stuff upload failed: {}", e); + wlogerr!("[OpenCL] stuff upload failed: {}", e); most.gpu_batch_ok = false; return most; } @@ -83,7 +83,7 @@ pub(crate) fn do_diamond_group_mining_opencl( ) { Ok(ev) => ev, Err(e) => { - eprintln!("[OpenCL] diamond kernel failed: {}", e.display()); + wlogerr!("[OpenCL] diamond kernel failed: {}", e.display()); most.gpu_batch_ok = false; return most; } @@ -152,11 +152,11 @@ pub(crate) fn do_diamond_group_mining_opencl( } } - // Always finish the queue when required — including the early success path — + // Always finish the queue when required, including the early success path, // so AMD RDNA/duplicate ICD does not leave work outstanding. if opencl.needs_queue_finish { if let Err(e) = opencl.queue.finish() { - eprintln!("[OpenCL] diamond queue finish: {}", e); + wlogerr!("[OpenCL] diamond queue finish: {}", e); // Report the driver failure, but NEVER discard `most.is_success`: that // DiamondMint was already verified on the CPU above (x16rs_hash // recompute, then check_diamond_hash_result + check_diamond_difficulty diff --git a/app/src/opencl_diag.rs b/app/src/opencl_diag.rs index 2d97cfde..a00c67d4 100644 --- a/app/src/opencl_diag.rs +++ b/app/src/opencl_diag.rs @@ -222,7 +222,7 @@ pub fn scan_opencl() -> OpenClScan { let limits = crate::gpu_arch::ArchLimits::for_slug(&sel.device_slug); if limits.is_experimental() && amd_platforms.len() > 1 { warnings.push( - "gfx1201 (RX 9070 XT): duplicate AMD OpenCL platforms — miner will cap work_groups to 64 on this architecture." + "gfx1201 (RX 9070 XT): duplicate AMD OpenCL platforms; miner will cap work_groups to 64 on this architecture." .to_string(), ); } @@ -330,7 +330,7 @@ pub fn resolve_opencl_selection( let Some(plat) = platforms.iter().find(|p| p.index == configured_platform) else { if let Some(rec) = recommend_opencl_device(platforms) { notes.push(format!( - "[OpenCL] platform_id={} invalid — using recommended platform {} device {} ({})", + "[OpenCL] platform_id={} invalid, using recommended platform {} device {} ({})", configured_platform, rec.platform_id, rec.device_id, rec.device_slug )); return (rec.platform_id, rec.device_id, notes); @@ -341,7 +341,7 @@ pub fn resolve_opencl_selection( let Some(dev) = plat.devices.iter().find(|d| d.index == configured_device) else { if let Some(rec) = recommend_opencl_device(platforms) { notes.push(format!( - "[OpenCL] device_id={} not found on platform {} — using {} ({})", + "[OpenCL] device_id={} not found on platform {}, using {} ({})", configured_device, configured_platform, rec.device_id, rec.device_slug )); return (rec.platform_id, rec.device_id, notes); @@ -369,14 +369,14 @@ pub fn resolve_opencl_selection( } if best_plat != configured_platform { notes.push(format!( - "[OpenCL] Auto-selected platform {} (AMD-APP {}) for {} — config had platform {} (AMD-APP {})", + "[OpenCL] Auto-selected platform {} (AMD-APP {}) for {}, config had platform {} (AMD-APP {})", best_plat, best_build, slug, configured_platform, plat.amd_app_build )); } if is_igpu_slug(&dev.slug, dev.compute_units, &dev.name) { if let Some(rec) = recommend_opencl_device(platforms) { notes.push(format!( - "[OpenCL] device {} looks like iGPU — switching to {} ({})", + "[OpenCL] device {} looks like iGPU, switching to {} ({})", dev.name, rec.device_name, rec.device_slug )); return (rec.platform_id, rec.device_id, notes); @@ -387,7 +387,7 @@ pub fn resolve_opencl_selection( if let Some(rec) = recommend_opencl_device(platforms) { notes.push(format!( - "[OpenCL] Configured device is not discrete — using {} ({})", + "[OpenCL] Configured device is not discrete, using {} ({})", rec.device_name, rec.device_slug )); return (rec.platform_id, rec.device_id, notes); @@ -397,9 +397,9 @@ pub fn resolve_opencl_selection( } pub fn print_scan_report(scan: &OpenClScan) { - println!("OpenCL diagnostic scan\n"); + wlogln!("OpenCL diagnostic scan\n"); for plat in &scan.platforms { - println!( + wlogln!( "Platform {}: {} vendor={} version={} AMD-APP build={}", plat.index, plat.name, plat.vendor, plat.version, plat.amd_app_build ); @@ -411,7 +411,7 @@ pub fn print_scan_report(scan: &OpenClScan) { } else { "other" }; - println!( + wlogln!( " device {}: {} ({}) CU={} VRAM={}MB max_wg={} [{}]", dev.index, dev.name, @@ -422,18 +422,18 @@ pub fn print_scan_report(scan: &OpenClScan) { kind ); } - println!(); + wlogln!(); } if let Some(rec) = &scan.recommended { - println!( + wlogln!( "Recommended: platform_id={} device_id={} {} ({}) AMD-APP {}", rec.platform_id, rec.device_id, rec.device_name, rec.device_slug, rec.amd_app_build ); } if !scan.warnings.is_empty() { - println!("\nWarnings:"); + wlogln!("\nWarnings:"); for w in &scan.warnings { - println!(" ! {}", w); + wlogln!(" ! {}", w); } } } diff --git a/app/src/opencl_gpu/block.rs b/app/src/opencl_gpu/block.rs index 73bc42be..839b24b8 100644 --- a/app/src/opencl_gpu/block.rs +++ b/app/src/opencl_gpu/block.rs @@ -260,17 +260,28 @@ mod gpu_tests { pool.share_hits, BATCH_NONCES as u64, "the counter must see every hit, not only the stored ones" ); - assert_eq!(pool.shares.len(), SHARE_LIST_CAPACITY.min(BATCH_NONCES as usize)); + assert_eq!( + pool.shares.len(), + SHARE_LIST_CAPACITY.min(BATCH_NONCES as usize) + ); let mut seen: Vec = pool.shares.iter().map(|(nonce, _)| *nonce).collect(); seen.sort_unstable(); seen.dedup(); - assert_eq!(seen.len(), pool.shares.len(), "no nonce may be listed twice"); + assert_eq!( + seen.len(), + pool.shares.len(), + "no nonce may be listed twice" + ); for (nonce, hash) in &pool.shares { assert!( (NONCE_START..NONCE_START + BATCH_NONCES).contains(nonce), "share nonce {nonce} is outside the batch window" ); - assert_eq!(*hash, cpu_hash(&intro, *nonce), "share hash must match the CPU"); + assert_eq!( + *hash, + cpu_hash(&intro, *nonce), + "share hash must match the CPU" + ); } // 3. POOL, a target only three nonces beat: exactly those three, and @@ -297,6 +308,9 @@ mod gpu_tests { assert_eq!(strict.share_hits, 3); let mut got: Vec = strict.shares.iter().map(|(nonce, _)| *nonce).collect(); got.sort_unstable(); - assert_eq!(got, expected, "the kernel must list exactly the payable nonces"); + assert_eq!( + got, expected, + "the kernel must list exactly the payable nonces" + ); } } diff --git a/app/src/opencl_gpu/compile.rs b/app/src/opencl_gpu/compile.rs index e662bdd6..710d4e0d 100644 --- a/app/src/opencl_gpu/compile.rs +++ b/app/src/opencl_gpu/compile.rs @@ -324,13 +324,13 @@ pub(crate) fn compile_program_from_source( amd_fast: bool, ) -> Option { if let Err(error) = newest_opencl_source_mtime(opencldir, kernel_path) { - eprintln!("[OpenCL] Unsafe or invalid kernel tree: {error}"); + wlogerr!("[OpenCL] Unsafe or invalid kernel tree: {error}"); return None; } let kernel_bytes = match read_bounded_regular_file(kernel_path, MAX_OPENCL_SOURCE_BYTES) { Ok(source) => source, Err(error) => { - eprintln!( + wlogerr!( "[OpenCL] Cannot read kernel {}: {error}", kernel_path.display() ); @@ -340,7 +340,7 @@ pub(crate) fn compile_program_from_source( let kernel_src = match String::from_utf8(kernel_bytes) { Ok(source) => source, Err(error) => { - eprintln!( + wlogerr!( "[OpenCL] Kernel {} is not valid UTF-8: {error}", kernel_path.display() ); @@ -350,14 +350,14 @@ pub(crate) fn compile_program_from_source( let include_option = match compiler_include_option(opencldir) { Ok(option) => option, Err(error) => { - eprintln!("[OpenCL] Invalid kernel directory: {error}"); + wlogerr!("[OpenCL] Invalid kernel directory: {error}"); return None; } }; let arch_defs = gpu_arch::compile_defines(vendor, arch_slug, amd_fast); // -cl-uniform-work-group-size is an optional OpenCL 2.0 optimization hint that - // NVIDIA's OpenCL compiler does not recognize — it rejects the whole build + // NVIDIA's OpenCL compiler does not recognize; it rejects the whole build // ("Don't understand command line argument ..."). Omit it on NVIDIA; AMD and // Intel keep it unchanged (their kernel build is byte-identical to before). let uniform_wg = if vendor == gpu_arch::GpuVendor::Nvidia { @@ -368,7 +368,7 @@ pub(crate) fn compile_program_from_source( let compile_options = format!( "-cl-std=CL2.0 -cl-fast-relaxed-math -cl-mad-enable{uniform_wg} {include_option}{arch_defs}" ); - println!("[OpenCL] compile opts:{arch_defs}"); + wlogln!("[OpenCL] compile opts:{arch_defs}"); let program = match Program::builder() .src(&kernel_src) .devices(device) @@ -377,7 +377,7 @@ pub(crate) fn compile_program_from_source( { Ok(program) => program, Err(error) => { - eprintln!("OpenCL program compilation error: {error}"); + wlogerr!("OpenCL program compilation error: {error}"); return None; } }; @@ -386,14 +386,14 @@ pub(crate) fn compile_program_from_source( match program.info(ProgramInfo::Binaries) { Ok(ProgramInfoResult::Binaries(binaries)) => { if let Some(binary) = binaries.first() { - println!("Saving OpenCL program in binary file..."); + wlogln!("Saving OpenCL program in binary file..."); if let Err(error) = write_cache_atomically(binary_path, binary) { - eprintln!("[OpenCL] Cannot cache {}: {error}", binary_path.display()); + wlogerr!("[OpenCL] Cannot cache {}: {error}", binary_path.display()); } } } - Ok(_) => eprintln!("[OpenCL] Driver returned no program binaries; cache disabled"), - Err(error) => eprintln!("[OpenCL] Cannot read compiled binary for cache: {error}"), + Ok(_) => wlogerr!("[OpenCL] Driver returned no program binaries; cache disabled"), + Err(error) => wlogerr!("[OpenCL] Cannot read compiled binary for cache: {error}"), } Some(program) diff --git a/app/src/opencl_gpu/handle.rs b/app/src/opencl_gpu/handle.rs index 25dac0f9..158789f9 100644 --- a/app/src/opencl_gpu/handle.rs +++ b/app/src/opencl_gpu/handle.rs @@ -91,7 +91,7 @@ impl OpenclGpuHandle { notify, } => { if notify { - eprintln!( + wlogerr!( "[OpenCL] GPU QUARANTINED (level {level}, {total_failures} failed batches): no GPU work for another {}, mining continues on capped CPU recovery. The card is re-probed automatically, no restart needed.", format_backoff(retry_in) ); @@ -103,7 +103,7 @@ impl OpenclGpuHandle { total_failures, } => { let wg = self.prepare_reprobe(); - println!( + wlogln!( "[OpenCL] GPU quarantine (level {level}, {total_failures} failed batches) expired: re-probing the device at work_groups={wg}." ); false @@ -157,7 +157,7 @@ impl OpenclGpuHandle { } return synced_wg; } - Err(e) => eprintln!("[OpenCL] re-probe context rebuild failed: {}", e), + Err(e) => wlogerr!("[OpenCL] re-probe context rebuild failed: {}", e), } } floor @@ -233,7 +233,7 @@ impl OpenclGpuHandle { soft_recover_opencl(&mut res); drop(res); runtime.report_gpu_workgroups(floor, runtime.thermal_workgroups_cap(), configured_wg); - eprintln!( + wlogerr!( "[OpenCL] ALERT GPU quarantined after {} consecutive failed batches ({} this session): no GPU work for {}, then the card is re-probed automatically. Mining continues on capped CPU recovery. Last error: {}. Check the driver, cooling, power and the PCIe riser.", report.consecutive_failures, report.total_failures, @@ -275,12 +275,12 @@ impl OpenclGpuHandle { oom.sync_effective(synced_wg); } self.consecutive_errors.store(0, Relaxed); - eprintln!( + wlogerr!( "[OpenCL] Rebuilt GPU context (errors={}, work_groups={})", n, rebuild_wg ); } - Err(e) => eprintln!("[OpenCL] Context rebuild failed: {}", e), + Err(e) => wlogerr!("[OpenCL] Context rebuild failed: {}", e), } } drop(res); @@ -292,7 +292,7 @@ impl OpenclGpuHandle { use std::sync::atomic::Ordering::Relaxed; self.consecutive_errors.store(0, Relaxed); if self.quarantine.record_success() { - println!( + wlogln!( "[OpenCL] GPU RECOVERED: the re-probe succeeded, quarantine cleared and GPU mining has resumed." ); } @@ -336,7 +336,7 @@ impl OpenclGpuHandle { if let Ok(mut oom) = self.oom.lock() { oom.sync_effective(synced_wg); } - println!("[OpenCL] Restored GPU context at work_groups={}", synced_wg); + wlogln!("[OpenCL] Restored GPU context at work_groups={}", synced_wg); } Err(e) => { // Clamp the OOM state back to what the live context can run, so a @@ -346,7 +346,7 @@ impl OpenclGpuHandle { if let Ok(mut oom) = self.oom.lock() { oom.sync_effective(capped); } - eprintln!( + wlogerr!( "[OpenCL] work_groups ramp-up rebuild failed, staying at {}: {}", capped, e ); diff --git a/app/src/opencl_gpu/init.rs b/app/src/opencl_gpu/init.rs index 2971ea71..95823fee 100644 --- a/app/src/opencl_gpu/init.rs +++ b/app/src/opencl_gpu/init.rs @@ -82,7 +82,7 @@ pub fn initialize_opencl( quiet: bool, ) -> Vec { if *localsize != 256 { - eprintln!( + wlogerr!( "[Warn] OpenCL local_size={} is incompatible with kernel fixed local arrays(256), fallback to CPU miner.", localsize ); @@ -91,7 +91,7 @@ pub fn initialize_opencl( let opencl_path = Path::new(opencldir); if !opencl_path.is_dir() { - eprintln!("[OpenCL] Kernel directory not found: {opencldir}"); + wlogerr!("[OpenCL] Kernel directory not found: {opencldir}"); return Vec::new(); } let kernel_path = opencl_kernel_path(opencl_path, diamond_mining); @@ -114,7 +114,7 @@ pub fn initialize_opencl( } }; for w in &scan.warnings { - eprintln!("[OpenCL] {}", w); + wlogerr!("[OpenCL] {}", w); } let mut cnf_devices = match parse_device_ids(deviceids) { @@ -129,7 +129,7 @@ pub fn initialize_opencl( ids } Err(error) => { - eprintln!("[OpenCL] Invalid device_ids configuration: {error}"); + wlogerr!("[OpenCL] Invalid device_ids configuration: {error}"); return Vec::new(); } }; @@ -138,7 +138,7 @@ pub fn initialize_opencl( crate::opencl_diag::resolve_opencl_selection(&scan.platforms, *platformid, primary_device); if !quiet { for n in ¬es { - println!("{}", n); + wlogln!("{}", n); } } if !cnf_devices.is_empty() { @@ -153,13 +153,13 @@ pub fn initialize_opencl( let platforms = crate::opencl_diag::platform_list(); let Some(platform) = platforms.get(resolved_platform as usize).cloned() else { if platforms.is_empty() { - eprintln!( + wlogerr!( "[OpenCL] No OpenCL platform is available on this machine. Install the GPU \ driver's OpenCL runtime, then run ./list_opencl to confirm it. Mining \ continues on the CPU." ); } else { - eprintln!( + wlogerr!( "[OpenCL] Platform {} is unavailable ({} platform(s) detected)", resolved_platform, platforms.len() @@ -172,9 +172,9 @@ pub fn initialize_opencl( let vendor = platform.vendor().unwrap_or_else(|_| "unknown".into()); let version: String = platform.version().unwrap_or_else(|_| "unknown".into()); if !quiet { - println!("Platform name: {}", name); - println!("Manufacturer: {}", vendor); - println!("Version: {}", version); + wlogln!("Platform name: {}", name); + wlogln!("Manufacturer: {}", vendor); + wlogln!("Version: {}", version); } // Resolve exact device indices. `Device::by_idx_wrap` is intentionally not @@ -182,7 +182,7 @@ pub fn initialize_opencl( let available_devices = match Device::list_all(&platform) { Ok(devices) => devices, Err(error) => { - eprintln!("[OpenCL] Cannot enumerate devices: {error}"); + wlogerr!("[OpenCL] Cannot enumerate devices: {error}"); return Vec::new(); } }; @@ -191,7 +191,7 @@ pub fn initialize_opencl( .ok() .is_some_and(|index| index < available_devices.len()); if !in_range { - eprintln!( + wlogerr!( "[OpenCL] Device {device_id} is unavailable ({} device(s) detected)", available_devices.len() ); @@ -219,7 +219,7 @@ pub fn initialize_opencl( let arch_limits = ArchLimits::for_slug(&gpu_arch::arch_slug(&device_name)); let mut wg = gpu_arch::tune_workgroups(*workgroups, compute_units, vendor, arch_limits); if !quiet && compute_units > 0 { - println!( + wlogln!( "[OpenCL] CU={} tuned work_groups={} (config {})", compute_units, wg, workgroups ); @@ -233,7 +233,7 @@ pub fn initialize_opencl( arch_limits.panel_min_wg, ); if clamped < wg { - println!( + wlogln!( "[efficiency] VRAM clamp: work_groups {} -> {} ({} MB available)", wg, clamped, @@ -247,9 +247,9 @@ pub fn initialize_opencl( let num_work_items = wg.saturating_mul(*localsize); let global_work_size = num_work_items; - println!("-----------------------------------------"); - println!("Device {}: {}", device_id, device_name); - println!("-----------------------------------------"); + wlogln!("-----------------------------------------"); + wlogln!("Device {}: {}", device_id, device_name); + wlogln!("-----------------------------------------"); // Create context let context = match Context::builder() @@ -259,7 +259,7 @@ pub fn initialize_opencl( { Ok(context) => context, Err(e) => { - eprintln!("[OpenCL] Cannot create context for {device_name}: {e}"); + wlogerr!("[OpenCL] Cannot create context for {device_name}: {e}"); continue; } }; @@ -268,7 +268,7 @@ pub fn initialize_opencl( let amd_plat_count = crate::opencl_diag::count_amd_platforms(&scan.platforms); let capped = arch_limits.workgroups_cap(wg, amd_plat_count); if capped < wg && !quiet { - println!( + wlogln!( "[OpenCL] {}: work_groups {} -> {} ({} AMD platform(s))", slug, wg, capped, amd_plat_count ); @@ -278,10 +278,10 @@ pub fn initialize_opencl( } let amd_fast = vendor == GpuVendor::Amd; if amd_fast { - println!("AMD fast-path: enabling OpenCL amd_bfe optimizations for this device"); + wlogln!("AMD fast-path: enabling OpenCL amd_bfe optimizations for this device"); } if vendor == GpuVendor::Nvidia { - println!("NVIDIA OpenCL path: arch={}", slug); + wlogln!("NVIDIA OpenCL path: arch={}", slug); } let safe_name = gpu_arch::safe_device_filename(&device_name); let diamond_tag = if diamond_mining { "_dia" } else { "" }; @@ -297,7 +297,7 @@ pub fn initialize_opencl( match opencl_cache_fingerprint(opencl_path, &kernel_path, &compile_identity) { Ok(fingerprint) => fingerprint, Err(error) => { - eprintln!("[OpenCL] Invalid kernel source tree: {error}"); + wlogerr!("[OpenCL] Invalid kernel source tree: {error}"); continue; } }; @@ -319,7 +319,7 @@ pub fn initialize_opencl( }; let compile = || { - println!("Compiling..."); + wlogln!("Compiling..."); compile_program_from_source( &context, &device, @@ -333,7 +333,7 @@ pub fn initialize_opencl( }; let program = if !need_recompile { - println!("Loading OpenCL from the binary..."); + wlogln!("Loading OpenCL from the binary..."); let cached = read_cached_program_binary(&binary_path) .ok() .and_then(|binary_data| { @@ -346,24 +346,24 @@ pub fn initialize_opencl( .ok() }); if cached.is_none() { - eprintln!("[OpenCL] Cached binary is invalid; recompiling from source"); + wlogerr!("[OpenCL] Cached binary is invalid; recompiling from source"); } cached.or_else(|| compile()) } else { compile() }; let Some(program) = program else { - eprintln!("[OpenCL] Skipping {device_name}: program initialization failed"); + wlogerr!("[OpenCL] Skipping {device_name}: program initialization failed"); continue; }; if let Err(error) = prune_opencl_cache(opencl_path, &binary_path) { - eprintln!("[OpenCL] Cannot prune stale cache binaries: {error}"); + wlogerr!("[OpenCL] Cannot prune stale cache binaries: {error}"); } let (queue, out_of_order) = match create_command_queue(&context, &device) { Ok(queue) => queue, Err(e) => { - eprintln!("[OpenCL] Skipping {device_name}: {e}"); + wlogerr!("[OpenCL] Skipping {device_name}: {e}"); continue; } }; @@ -389,18 +389,18 @@ pub fn initialize_opencl( opencl_resource_devices.push(res); } Err(e) => { - // A Groestl integrity self-test failure is deterministic — it is NOT + // A Groestl integrity self-test failure is deterministic; it is NOT // a VRAM shortage, so shrinking work_groups just re-runs the same // failing test and prints a misleading "insufficient VRAM". Report it // as the integrity failure it is and skip the recovery loop. if e.contains("integrity self-test") { - eprintln!( - "[efficiency] Skipping device {} — GPU hash integrity self-test failed (not a VRAM issue): {}", + wlogerr!( + "[efficiency] Skipping device {}: GPU hash integrity self-test failed (not a VRAM issue): {}", device_id, e ); continue; } - eprintln!( + wlogerr!( "[efficiency] OpenCL buffer init failed at work_groups={}: {}", wg, e ); @@ -423,7 +423,7 @@ pub fn initialize_opencl( &slug, ) { if !quiet { - println!("[efficiency] Recovered with work_groups={candidate}"); + wlogln!("[efficiency] Recovered with work_groups={candidate}"); } res.platform_index = resolved_platform; res.device_index = device_id; @@ -434,8 +434,8 @@ pub fn initialize_opencl( reduced = next_recovery_workgroups(candidate, wg_floor); } if !built { - eprintln!( - "[efficiency] Skipping device {} — insufficient VRAM", + wlogerr!( + "[efficiency] Skipping device {}: insufficient VRAM", device_id ); } diff --git a/app/src/opencl_gpu/resources.rs b/app/src/opencl_gpu/resources.rs index 075fc1d4..1638dc73 100644 --- a/app/src/opencl_gpu/resources.rs +++ b/app/src/opencl_gpu/resources.rs @@ -62,12 +62,12 @@ pub(crate) fn create_command_queue( let ooo = CommandQueueProperties::new().out_of_order(); match Queue::new(context, device.clone(), Some(ooo)) { Ok(queue) => { - println!("[OpenCL] Out-of-order command queue enabled"); + wlogln!("[OpenCL] Out-of-order command queue enabled"); Ok((queue, true)) } Err(ooo_error) => Queue::new(context, device.clone(), None) .map(|queue| { - println!("[OpenCL] In-order command queue (OOO not supported)"); + wlogln!("[OpenCL] In-order command queue (OOO not supported)"); (queue, false) }) .map_err(|e| format!("cannot create command queue: {e}; OOO attempt: {ooo_error}")), @@ -125,9 +125,9 @@ pub struct OpenCLResources { buffer_share_found: Buffer, buffer_share_nonces: Buffer, buffer_share_hashes: Buffer, - /// Reused input buffer — avoids per-kernel GPU allocation. + /// Reused input buffer, avoids per-kernel GPU allocation. buffer_stuff: Buffer, - /// Cached OpenCL kernel — rebuilt only when `unit_size` changes. + /// Cached OpenCL kernel, rebuilt only when `unit_size` changes. kernel_slot: Mutex, } @@ -631,7 +631,7 @@ pub(crate) fn build_opencl_resources( .build() .map_err(|e| format!("buffer_share_hashes: {}", e))?; if out_of_order { - println!("[OpenCL] Pinned host buffers enabled for stuff + readback"); + wlogln!("[OpenCL] Pinned host buffers enabled for stuff + readback"); } let resources = OpenCLResources { workgroups, @@ -662,7 +662,7 @@ pub(crate) fn build_opencl_resources( }; if arch_slug == "gfx1201" && !diamond { run_gfx1201_groestl_self_test(&resources)?; - println!("[OpenCL] gfx1201 Groestl integrity self-test passed"); + wlogln!("[OpenCL] gfx1201 Groestl integrity self-test passed"); } Ok(resources) } diff --git a/app/src/opencl_list.rs b/app/src/opencl_list.rs index c9e354df..7ed5336d 100644 --- a/app/src/opencl_list.rs +++ b/app/src/opencl_list.rs @@ -1,4 +1,4 @@ -//! List OpenCL platforms/devices — used by `list_opencl` / `diagnose_opencl` binaries. +//! List OpenCL platforms/devices, used by `list_opencl` / `diagnose_opencl` binaries. #[cfg(feature = "ocl")] fn select_opencl_dir_hint( @@ -32,20 +32,20 @@ fn opencl_dir_hint() -> (&'static str, &'static str) { pub fn list_opencl_devices() -> bool { let scan = crate::opencl_diag::scan_opencl(); crate::opencl_diag::print_scan_report(&scan); - println!("\nConfig hints (HAC poworker.config.ini only; HACD is CPU-only):"); + wlogln!("\nConfig hints (HAC poworker.config.ini only; HACD is CPU-only):"); if let Some(rec) = &scan.recommended { - println!(" [gpu]"); - println!(" use_opencl = true"); - println!(" platform_id = {}", rec.platform_id); - println!(" device_ids = {}", rec.device_id); + wlogln!(" [gpu]"); + wlogln!(" use_opencl = true"); + wlogln!(" platform_id = {}", rec.platform_id); + wlogln!(" device_ids = {}", rec.device_id); } else { - println!(" [gpu]"); - println!(" use_opencl = true"); - println!(" platform_id = "); - println!(" device_ids = "); + wlogln!(" [gpu]"); + wlogln!(" use_opencl = true"); + wlogln!(" platform_id = "); + wlogln!(" device_ids = "); } let (opencl_dir, layout) = opencl_dir_hint(); - println!(" opencl_dir = {opencl_dir} {layout}"); + wlogln!(" opencl_dir = {opencl_dir} {layout}"); scan.recommended.is_some() } diff --git a/app/src/poworker.rs b/app/src/poworker.rs index f3961ef9..2cb2e5e8 100644 --- a/app/src/poworker.rs +++ b/app/src/poworker.rs @@ -118,7 +118,7 @@ impl PoWorkConf { efficiency, runtime, }; - println!( + wlogln!( "[efficiency] mode={} profile={} work_groups={} unit_size={} dynamic_supervene={}", cnf.efficiency.mode.label(), cnf.gpu_profile, @@ -180,7 +180,7 @@ pub fn poworker_with_conf(cnf: PoWorkConf) { pub fn poworker_with_stop(cnf: PoWorkConf, stop_flag: Option>) { if !block_mining_runtime::start_block_mining_workers(&cnf, stop_flag.clone()) { - eprintln!("[Fatal] Mining worker startup failed."); + wlogerr!("[Fatal] Mining worker startup failed."); return; } @@ -257,7 +257,7 @@ fn upstream_stale_reason(res: &JV) -> Option<&str> { /// continuing to hash it only burns power. fn enter_upstream_stale(reason: &str, source: &str) { if block_mining_runtime::set_upstream_stale(true) { - println!( + wlogln!( "\n[Mining] PAUSED: {source} reports stale work ({reason}). The template can no longer win anything, so hashing is idle until fresh work arrives. Check the pool or node this miner connects to." ); } @@ -266,7 +266,7 @@ fn enter_upstream_stale(reason: &str, source: &str) { /// Clear the stale-work pause once real work is available again. fn leave_upstream_stale() { if block_mining_runtime::set_upstream_stale(false) { - println!("\n[Mining] Fresh work received, mining resumes."); + wlogln!("\n[Mining] Fresh work received, mining resumes."); } } @@ -293,7 +293,7 @@ fn pull_pending_block_stuff(cnf: &PoWorkConf, stop_flag: &Option match crate::rpc_http::get_text(&HTTP_CLIENT, &urlapi_pending, &cnf.api_token, None) { Ok(t) => t, Err(e) => { - println!( + wlogln!( "Error: cannot get block data at {}: {}\n", &urlapi_pending, e ); @@ -301,7 +301,7 @@ fn pull_pending_block_stuff(cnf: &PoWorkConf, stop_flag: &Option } }; let Ok(res) = serde_json::from_str::(&jsdata) else { - println!( + wlogln!( "Error: invalid block data json at {} (body len {})\n", &urlapi_pending, jsdata.len() @@ -315,7 +315,7 @@ fn pull_pending_block_stuff(cnf: &PoWorkConf, stop_flag: &Option enter_upstream_stale(reason, "pending work"); delay_return!(STALE_UPSTREAM_BACKOFF_SECS); } - println!("Error: get block stuff error: {}", jstr("err")); + wlogln!("Error: get block stuff error: {}", jstr("err")); delay_return!(15); }; let pending_height = jnum("height"); @@ -329,7 +329,7 @@ fn pull_pending_block_stuff(cnf: &PoWorkConf, stop_flag: &Option // job's identity (height + parent hash), which that re-serialization leaves // untouched. A same-height reorg still changes the parent and is still detected. if let Err(e) = block_mining_runtime::set_pending_block_stuff(pending_height, res) { - println!("Error: invalid block data from {urlapi_pending}: {e}"); + wlogln!("Error: invalid block data from {urlapi_pending}: {e}"); delay_return!(10); } // Real work was served and installed, so any stale-work pause lifts. @@ -355,7 +355,7 @@ fn pull_pending_block_stuff(cnf: &PoWorkConf, stop_flag: &Option return; } if let Err(e) = getrandom::fill(&mut rpid) { - println!("Error: cannot generate request id: {e}"); + wlogln!("Error: cannot generate request id: {e}"); delay_return!(1); } let urlapi_notice = format!( @@ -365,7 +365,7 @@ fn pull_pending_block_stuff(cnf: &PoWorkConf, stop_flag: &Option pending_height, &hex::encode(&rpid) ); - // println!("\n-------- {} -------- {}\n", &ctshow(), &urlapi_notice); + // wlogln!("\n-------- {} -------- {}\n", &ctshow(), &urlapi_notice); let jsdata = match crate::rpc_http::get_text( &HTTP_CLIENT, &urlapi_notice, @@ -374,7 +374,7 @@ fn pull_pending_block_stuff(cnf: &PoWorkConf, stop_flag: &Option ) { Ok(t) => t, Err(e) => { - println!( + wlogln!( "Error: cannot get miner notice at {}: {}\n", &urlapi_notice, e ); @@ -382,7 +382,7 @@ fn pull_pending_block_stuff(cnf: &PoWorkConf, stop_flag: &Option } }; let Ok(res2) = serde_json::from_str::(&jsdata) else { - println!("Error: invalid miner notice JSON at {urlapi_notice}"); + wlogln!("Error: invalid miner notice JSON at {urlapi_notice}"); delay_return!(1); }; // The notice body carries the LAST known height alongside the error, so it @@ -449,7 +449,7 @@ fn push_block_mining_success(cnf: &PoWorkConf, success: &block_mining_runtime::B .as_ref() .and_then(|j| j["err"].as_str()) .unwrap_or(""); - println!("[submit] node rejected height {}: {}", success.height, err); + wlogln!("[submit] node rejected height {}: {}", success.height, err); break; } None => { @@ -458,7 +458,7 @@ fn push_block_mining_success(cnf: &PoWorkConf, success: &block_mining_runtime::B // decision, so treat it as transient and retry: a winning // block is not discarded on a front-end hiccup. let snippet: String = body.chars().take(120).collect(); - println!( + wlogln!( "[submit] attempt {}/{} unrecognized response, retrying: {}", attempt, MAX_SUBMIT_ATTEMPTS, snippet ); @@ -470,7 +470,7 @@ fn push_block_mining_success(cnf: &PoWorkConf, success: &block_mining_runtime::B } Err(e) => { last = format!("transport error: {e}"); - println!( + wlogln!( "[submit] attempt {}/{} failed: {e}", attempt, MAX_SUBMIT_ATTEMPTS ); @@ -480,22 +480,22 @@ fn push_block_mining_success(cnf: &PoWorkConf, success: &block_mining_runtime::B } } } - println!("{} {}", &urlapi_success, last); + wlogln!("{} {}", &urlapi_success, last); if accepted { - println!( + wlogln!( "\n\n████████████████ [MINING SUCCESS] Find a block height {},\n██ hash {} to submit.", success.height, success.result_hash.to_hex() ); } else { - println!( + wlogln!( "\n\n████████████████ [MINING SUBMIT FAILED] block height {} was NOT confirmed accepted\n██ after {} attempts (hash {}). Check the node/connection.", success.height, MAX_SUBMIT_ATTEMPTS, success.result_hash.to_hex() ); } - println!("▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔") + wlogln!("▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔") } #[cfg(feature = "ocl")] const AUTOTUNE_WARMUP_BATCHES: u32 = 3; @@ -773,19 +773,19 @@ fn run_block_mining_benchmark(cnf: &PoWorkConf, config_path: &str) { #[cfg(not(feature = "ocl"))] { let _ = (cnf, config_path); - println!("[benchmark] Rebuild with --features ocl and use_opencl=true"); + wlogln!("[benchmark] Rebuild with --features ocl and use_opencl=true"); return; } #[cfg(feature = "ocl")] { if !cnf.useopencl { - println!("[benchmark] Set use_opencl=true in [gpu]"); + wlogln!("[benchmark] Set use_opencl=true in [gpu]"); return; } - println!( + wlogln!( "[benchmark] Power and kH/J figures are estimates derived from configured board power; they are not hardware telemetry." ); - println!( + wlogln!( "[benchmark] NOTE: the MH/s below are raw X16RS repeat=1 tuning rates (relative comparison only). The live mainnet runs 16 rounds, so real block-hash throughput is roughly 1/11-1/16 of these numbers. For the honest mainnet figure run: set HACASH_REPEAT16_BENCH_SECONDS=30 and run poworker." ); let total_secs = cnf.efficiency.benchmark_seconds.max(15) as u64; @@ -816,11 +816,11 @@ fn run_block_mining_benchmark(cnf: &PoWorkConf, config_path: &str) { false, ); if opencl_resources.is_empty() { - println!("[benchmark] No OpenCL devices"); + wlogln!("[benchmark] No OpenCL devices"); return; } if !autotune_device_count_is_supported(opencl_resources.len()) { - println!( + wlogln!( "[benchmark] Auto Tune requires exactly one OpenCL device, but detected {}. The current config has one shared work_groups/unit_size pair, so multi-GPU tuning would be ambiguous. Set [gpu] device_ids to one device and tune each GPU separately; config unchanged.", opencl_resources.len() ); @@ -845,14 +845,14 @@ fn run_block_mining_benchmark(cnf: &PoWorkConf, config_path: &str) { max_us, ); if candidates.is_empty() { - println!( + wlogln!( "[benchmark] No safe tuning candidates for device #{}", dev_i ); continue; } let per = (profile_secs / candidates.len() as u64).max(4); - println!( + wlogln!( "[benchmark] Device #{}: {}s x {} exact tuning points{}", dev_i, per, @@ -864,7 +864,7 @@ fn run_block_mining_benchmark(cnf: &PoWorkConf, config_path: &str) { for pick in candidates { match benchmark_candidate(opencl, cnf, pick.clone(), per, max_wg, max_us) { Ok(result) => { - println!( + wlogln!( "[benchmark] dev{} {}: {} (estimated {:.1} kH/J @ {:.0}W, {} samples, wg={}, unit_size={})", dev_i, result.pick.profile, @@ -878,7 +878,7 @@ fn run_block_mining_benchmark(cnf: &PoWorkConf, config_path: &str) { bench_results.push(result); } Err(error) => { - println!( + wlogln!( "[benchmark] dev{} {}: REJECTED ({error}, wg={}, unit_size={})", dev_i, pick.profile, pick.workgroups, pick.unitsize ); @@ -887,7 +887,7 @@ fn run_block_mining_benchmark(cnf: &PoWorkConf, config_path: &str) { } let Some(base) = pick_benchmark_result(&bench_results, cnf.efficiency.mode) else { - println!( + wlogln!( "[benchmark] No successful tuning points; config unchanged (check OpenCL driver)." ); continue; @@ -906,7 +906,7 @@ fn run_block_mining_benchmark(cnf: &PoWorkConf, config_path: &str) { max_wg, ); let per_wg = (wg_sweep_secs / wg_candidates.len().max(1) as u64).max(3); - println!( + wlogln!( "[benchmark] dev{} bounded wg sweep: {:?} x {}s", dev_i, wg_candidates, per_wg ); @@ -919,7 +919,7 @@ fn run_block_mining_benchmark(cnf: &PoWorkConf, config_path: &str) { }; match benchmark_candidate(opencl, cnf, candidate, per_wg, max_wg, max_us) { Ok(result) => { - println!( + wlogln!( "[benchmark] dev{} wg={}: {} (estimated {:.1} kH/J @ {:.0}W, {} samples)", dev_i, wg_try, @@ -931,7 +931,7 @@ fn run_block_mining_benchmark(cnf: &PoWorkConf, config_path: &str) { wg_results.push(result); } Err(error) => { - println!("[benchmark] dev{} wg={}: REJECTED ({error})", dev_i, wg_try); + wlogln!("[benchmark] dev{} wg={}: REJECTED ({error})", dev_i, wg_try); } } } @@ -941,7 +941,7 @@ fn run_block_mining_benchmark(cnf: &PoWorkConf, config_path: &str) { let us_candidates = sweep_unitsize_candidates(selected.pick.unitsize, max_us); let per_us = (us_sweep_secs / us_candidates.len().max(1) as u64).max(3); - println!( + wlogln!( "[benchmark] dev{} bounded unit_size sweep: {:?} x {}s", dev_i, us_candidates, per_us ); @@ -954,7 +954,7 @@ fn run_block_mining_benchmark(cnf: &PoWorkConf, config_path: &str) { }; match benchmark_candidate(opencl, cnf, candidate, per_us, max_wg, max_us) { Ok(result) => { - println!( + wlogln!( "[benchmark] dev{} unit_size={}: {} (estimated {:.1} kH/J @ {:.0}W, {} samples)", dev_i, us_try, @@ -966,7 +966,7 @@ fn run_block_mining_benchmark(cnf: &PoWorkConf, config_path: &str) { us_results.push(result); } Err(error) => { - println!( + wlogln!( "[benchmark] dev{} unit_size={}: REJECTED ({error})", dev_i, us_try ); @@ -979,7 +979,7 @@ fn run_block_mining_benchmark(cnf: &PoWorkConf, config_path: &str) { } let verify_secs = verification_seconds(total_secs); - println!( + wlogln!( "[benchmark] dev{} final verification soak: {}s at wg={} unit_size={}", dev_i, verify_secs, selected.pick.workgroups, selected.pick.unitsize ); @@ -993,7 +993,7 @@ fn run_block_mining_benchmark(cnf: &PoWorkConf, config_path: &str) { ) { Ok(result) => result, Err(error) => { - println!( + wlogln!( "[benchmark] dev{} final verification REJECTED ({error}) - config unchanged.", dev_i ); @@ -1001,7 +1001,7 @@ fn run_block_mining_benchmark(cnf: &PoWorkConf, config_path: &str) { } }; if !verification_is_stable(selected.hps, verified.hps) { - println!( + wlogln!( "[benchmark] dev{} final verification REJECTED: {} is below {:.0}% of measured {} - config unchanged.", dev_i, rates_to_show(verified.hps), @@ -1010,7 +1010,7 @@ fn run_block_mining_benchmark(cnf: &PoWorkConf, config_path: &str) { ); continue; } - println!( + wlogln!( "[benchmark] dev{} verified: profile={} work_groups={} unit_size={} {} (estimated {:.1} kH/J @ {:.0}W, {} samples, mode={})", dev_i, verified.pick.profile, @@ -1025,11 +1025,11 @@ fn run_block_mining_benchmark(cnf: &PoWorkConf, config_path: &str) { if dev_i == 0 { match apply_benchmark_pick(config_path, &verified.pick) { Ok(()) => { - println!( + wlogln!( "[benchmark] Config updated only after successful final verification." ) } - Err(e) => println!("[benchmark] Could not patch ini: {}", e), + Err(e) => wlogln!("[benchmark] Could not patch ini: {}", e), } } } @@ -1212,7 +1212,7 @@ mod tests { .try_into() .unwrap(); - println!( + wlogln!( "nonce={result_nonce} pre_x16rs={} algorithm={} expected_x16rs={} bad_gpu={}", hex::encode(pre_x16rs), x16rs_algorithm_id(&pre_x16rs), @@ -1255,7 +1255,9 @@ mod tests { assert_eq!(upstream_stale_reason(&serde_json::json!({})), None); assert_eq!(upstream_stale_reason(&serde_json::json!({"err": ""})), None); assert_eq!( - upstream_stale_reason(&serde_json::json!({"err": "no job yet; wait for upstream fullnode"})), + upstream_stale_reason( + &serde_json::json!({"err": "no job yet; wait for upstream fullnode"}) + ), None ); assert_eq!(upstream_stale_reason(&serde_json::json!({"err": 7})), None); diff --git a/app/src/worker_log.rs b/app/src/worker_log.rs new file mode 100644 index 00000000..39dbef6f --- /dev/null +++ b/app/src/worker_log.rs @@ -0,0 +1,363 @@ +//! A rolling log file the miners write beside their config, plus the two macros +//! that fill it. +//! +//! Why it exists. On Windows the panel now gives the node and every worker their +//! OWN console window. That was the right call for the operator, who can finally +//! watch the work, but a child with its own console pipes nothing back: the log +//! channel the panel gets from `spawn_worker_with_logs` stays empty there, so the +//! event feed and the log view had nothing to show at all. +//! +//! The answer is both. The console output is untouched, byte for byte; this is a +//! second copy of the same lines, written where the panel can read them. +//! +//! Bounded on purpose. A miner left running for a month must not fill a disk, so +//! the live file is capped and rotated exactly once. At most `2 * max_bytes` of +//! log exists at any moment, and that is a ceiling, not an average. +//! +//! Nothing here may take a miner down. Every failure is reported once on stderr +//! and then swallowed until it recovers: losing the log is never a reason to stop +//! hashing. + +use std::fs::{self, File, OpenOptions}; +use std::io::{self, Write}; +use std::path::{Path, PathBuf}; +use std::sync::atomic::{AtomicBool, Ordering::Relaxed}; +use std::sync::{Mutex, MutexGuard}; + +/// Ceiling for the live file. One rotation is kept beside it, so the worst case +/// on disk is twice this. +pub const DEFAULT_MAX_BYTES: u64 = 4 * 1024 * 1024; + +/// Refuse to run with a cap so small that rotation would thrash. Also what makes +/// the bound meaningful for tests, which set a tiny cap on purpose. +const MIN_MAX_BYTES: u64 = 4 * 1024; + +/// Longest single entry kept. One enormous line cannot overshoot the file bound +/// by more than this. +const MAX_ENTRY_BYTES: usize = 8 * 1024; + +/// Set while a log file is open. Read on every printed line, so it is an atomic +/// rather than a lock: a worker that never called `init` pays one relaxed load. +static ACTIVE: AtomicBool = AtomicBool::new(false); +static LOG: Mutex> = Mutex::new(None); +/// True while a write failure is being suppressed, so a failing disk produces +/// one line on stderr instead of one per print. +static WARNED: AtomicBool = AtomicBool::new(false); + +/// Recover from a poisoned lock rather than propagate the panic. The worst case +/// is one lost log line. +fn lock() -> MutexGuard<'static, Option> { + LOG.lock().unwrap_or_else(|e| e.into_inner()) +} + +/// The log that belongs to a worker: `/.log`. +/// +/// Shared with the panel (`miner-panel` tails exactly this path) so the writer +/// and the reader can never disagree about the name. +pub fn log_path(dir: &Path, stem: &str) -> PathBuf { + dir.join(format!("{stem}.log")) +} + +/// The path a rotation moves the live file to. +pub fn rotated_path(path: &Path) -> PathBuf { + let mut name = path.file_name().unwrap_or_default().to_os_string(); + name.push(".1"); + path.with_file_name(name) +} + +/// Start logging to `/.log`. Returns the path so the caller +/// can say where it went. +pub fn init_beside_config(config_path: &Path, stem: &str) -> Result { + let dir = config_path.parent().unwrap_or_else(|| Path::new(".")); + let path = log_path(dir, stem); + init_with_limit(&path, DEFAULT_MAX_BYTES)?; + Ok(path) +} + +/// Start logging to an exact path with an exact cap. The whole file is one +/// global because a worker has one log; calling this again replaces it. +pub fn init_with_limit(path: &Path, max_bytes: u64) -> Result<(), String> { + let rolling = Rolling::open(path, max_bytes.max(MIN_MAX_BYTES)) + .map_err(|e| format!("cannot open {}: {e}", path.display()))?; + *lock() = Some(rolling); + WARNED.store(false, Relaxed); + ACTIVE.store(true, Relaxed); + Ok(()) +} + +/// Stop logging and close the file. Used by the tests; a worker just exits. +pub fn shutdown() { + ACTIVE.store(false, Relaxed); + *lock() = None; +} + +pub fn is_active() -> bool { + ACTIVE.load(Relaxed) +} + +/// Append one printed line. A multi-line argument becomes one entry per line, so +/// the panel never has to reassemble anything. +pub fn record(text: &str) { + if !ACTIVE.load(Relaxed) { + return; + } + let stamp = sys::ctshow(); + let mut guard = lock(); + let Some(log) = guard.as_mut() else { + return; + }; + for part in text.split('\n') { + if let Err(error) = log.write_entry(&stamp, part.trim_end_matches('\r')) { + drop(guard); + warn_once(&error.to_string()); + return; + } + } + WARNED.store(false, Relaxed); +} + +fn warn_once(error: &str) { + if !WARNED.swap(true, Relaxed) { + eprintln!("[log] cannot write the worker log ({error}); suppressing until it recovers"); + } +} + +struct Rolling { + path: PathBuf, + rotated: PathBuf, + /// `None` only for the instant a rotation holds, so the old file is closed + /// before it is renamed. + file: Option, + written: u64, + max_bytes: u64, +} + +impl Rolling { + fn open(path: &Path, max_bytes: u64) -> io::Result { + let file = open_append(path)?; + let written = file.metadata().map(|m| m.len()).unwrap_or(0); + let mut rolling = Rolling { + path: path.to_path_buf(), + rotated: rotated_path(path), + file: Some(file), + written, + max_bytes, + }; + // A previous run may have left the file at the cap. Roll it now rather + // than resume above the bound. + if rolling.written >= rolling.max_bytes { + rolling.rotate()?; + } + Ok(rolling) + } + + fn write_entry(&mut self, stamp: &str, text: &str) -> io::Result<()> { + let mut entry = String::with_capacity(stamp.len() + text.len() + 2); + entry.push_str(stamp); + entry.push(' '); + push_bounded(&mut entry, text); + entry.push('\n'); + + let len = entry.len() as u64; + if self.written > 0 && self.written + len > self.max_bytes { + self.rotate()?; + } + let file = self + .file + .as_mut() + .ok_or_else(|| io::Error::other("the worker log file is closed"))?; + // `std::fs::File` is unbuffered, so this reaches the OS on the spot. + // That is what lets the panel see a line within one poll instead of + // whenever a buffer happened to fill. + file.write_all(entry.as_bytes())?; + self.written += len; + Ok(()) + } + + /// Move the live file aside and start a new one. The cap is what matters: if + /// the rename cannot be done, the old file is dropped rather than allowed to + /// keep growing. + fn rotate(&mut self) -> io::Result<()> { + self.file = None; + let _ = fs::rename(&self.path, &self.rotated); + // Truncating, not appending: after a failed rename the file is still + // there, and reopening it in append mode would defeat the bound. + self.file = Some( + OpenOptions::new() + .create(true) + .write(true) + .truncate(true) + .open(&self.path)?, + ); + self.written = 0; + Ok(()) + } +} + +fn open_append(path: &Path) -> io::Result { + OpenOptions::new().create(true).append(true).open(path) +} + +/// Append `text`, cut to `MAX_ENTRY_BYTES` on a character boundary. +fn push_bounded(out: &mut String, text: &str) { + if text.len() <= MAX_ENTRY_BYTES { + out.push_str(text); + return; + } + let mut end = MAX_ENTRY_BYTES; + while end > 0 && !text.is_char_boundary(end) { + end -= 1; + } + out.push_str(&text[..end]); + out.push_str("..."); +} + +/// `println!`, plus a copy in the worker's rolling log. +/// +/// The console output is identical to what `println!` produced before, which is +/// the whole point: the operator's console window is not being changed, it is +/// being mirrored. +#[macro_export] +macro_rules! wlogln { + () => {{ + std::println!(); + $crate::worker_log::record(""); + }}; + ($($arg:tt)*) => {{ + let line = std::format!($($arg)*); + std::println!("{}", line); + $crate::worker_log::record(&line); + }}; +} + +/// `eprintln!`, plus a copy in the worker's rolling log. +#[macro_export] +macro_rules! wlogerr { + ($($arg:tt)*) => {{ + let line = std::format!($($arg)*); + std::eprintln!("{}", line); + $crate::worker_log::record(&line); + }}; +} + +#[cfg(test)] +mod tests { + use super::*; + + /// The log is one process-wide file, so these tests take turns. + static SERIAL: Mutex<()> = Mutex::new(()); + + fn serial() -> MutexGuard<'static, ()> { + SERIAL.lock().unwrap_or_else(|e| e.into_inner()) + } + + fn scratch(name: &str) -> PathBuf { + let unique = std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .unwrap_or_default() + .as_nanos(); + let dir = std::env::temp_dir().join(format!( + "hacash-worker-log-{}-{name}-{unique}", + std::process::id() + )); + fs::create_dir_all(&dir).expect("scratch dir"); + dir + } + + #[test] + fn the_log_lands_beside_the_config_under_the_worker_name() { + // The panel computes the same two paths to tail them, so they are fixed + // here rather than left to agree by accident. + let dir = Path::new("C:/miner"); + assert_eq!(log_path(dir, "poworker"), dir.join("poworker.log")); + assert_eq!( + rotated_path(&log_path(dir, "diaworker")), + dir.join("diaworker.log.1") + ); + } + + #[test] + fn every_printed_line_reaches_the_file_without_waiting_for_a_buffer() { + let _serial = serial(); + let dir = scratch("write"); + let path = log_path(&dir, "poworker"); + init_with_limit(&path, DEFAULT_MAX_BYTES).expect("open the log"); + + record("[Mining] first"); + // A multi-line print becomes one entry per line: the panel splits on + // newlines and must never see a half entry. + record("[Mining] second\r\n[Mining] third"); + + let text = fs::read_to_string(&path).expect("read back"); + let lines: Vec<&str> = text.lines().collect(); + assert_eq!(lines.len(), 3, "got {text:?}"); + assert!(lines[0].ends_with(" [Mining] first"), "{}", lines[0]); + assert!(lines[1].ends_with(" [Mining] second"), "{}", lines[1]); + assert!(lines[2].ends_with(" [Mining] third"), "{}", lines[2]); + // Every entry carries the time it was printed, so the panel shows a + // measured timestamp instead of the moment it happened to read the file. + assert_eq!(lines[0].chars().nth(4), Some('-'), "{}", lines[0]); + + shutdown(); + let _ = fs::remove_dir_all(&dir); + } + + #[test] + fn a_miner_left_running_cannot_grow_the_log_past_two_files() { + let _serial = serial(); + let dir = scratch("rotate"); + let path = log_path(&dir, "poworker"); + let cap = MIN_MAX_BYTES; + init_with_limit(&path, cap).expect("open the log"); + + // Far more than the cap, in small lines, the way a worker really writes. + for i in 0..4_000 { + record(&format!("[Mining] batch {i} of a very long night")); + } + + let live = fs::metadata(&path).expect("live file").len(); + let rolled = fs::metadata(rotated_path(&path)) + .expect("one rotation is kept") + .len(); + assert!(live <= cap, "live file {live} is over the {cap} byte cap"); + assert!(rolled <= cap, "rotated file {rolled} is over the cap"); + // And nothing else: exactly two files, whatever the miner printed. + let files = fs::read_dir(&dir).expect("dir").count(); + assert_eq!(files, 2, "only the live log and one rotation may exist"); + // The newest lines survived the rotation. + let text = fs::read_to_string(&path).expect("read back"); + assert!(text.contains("batch 3999"), "the last line must be kept"); + + shutdown(); + let _ = fs::remove_dir_all(&dir); + } + + #[test] + fn a_worker_that_never_opened_a_log_just_prints() { + let _serial = serial(); + shutdown(); + assert!(!is_active()); + record("[Mining] nowhere to go"); + // Nothing to assert but the absence of a panic: this is the path every + // test binary and the benchmark tools take. + } + + #[test] + fn one_enormous_line_cannot_overshoot_the_bound() { + let _serial = serial(); + let dir = scratch("huge"); + let path = log_path(&dir, "poworker"); + init_with_limit(&path, MIN_MAX_BYTES).expect("open the log"); + + record(&"x".repeat(MAX_ENTRY_BYTES * 4)); + + let live = fs::metadata(&path).expect("live file").len(); + assert!( + live <= MIN_MAX_BYTES + MAX_ENTRY_BYTES as u64 + 64, + "a single line wrote {live} bytes" + ); + + shutdown(); + let _ = fs::remove_dir_all(&dir); + } +} diff --git a/deploy/node/hacash.config.ini b/deploy/node/hacash.config.ini index ba171db6..e88b2ba9 100644 --- a/deploy/node/hacash.config.ini +++ b/deploy/node/hacash.config.ini @@ -14,13 +14,19 @@ listen = 3337 boots = 54.193.49.59:3337, 182.92.163.225:3337, 54.219.80.127:3337 not_find_nodes = false fast_sync = false -; NOT true. Measured 2026-07-27: a chain synced with fast_sync = true -; stops dead at a block whose state it never wrote, with -; [Block Sync Warning] insert N failed: diamond status HTAKES not found -; and nothing retries, so the node sits there forever looking healthy. -; Turning the flag off afterwards does not repair it: the state was never -; written. A clean sync with it OFF reached the tip in seven minutes with -; no errors, which is the only reason this is not still true. +; Deliberately false, and NOT for the reason first written here. An earlier +; version of this comment claimed fast_sync corrupts the chain. That was wrong: +; a controlled test on 2026-07-28 synced this chain from zero with fast_sync = true +; and reached the tip with no errors at all. The corruption seen on 2026-07-27 was +; repaired by the clean RESYNC, not by the flag, and concluding otherwise was +; reading causation out of one sample with no control. +; +; The real reason to leave it off is what it actually relaxes. fast_sync reaches +; execution through ChainInfo and skips checks: protocol/src/context/context.rs +; trusts a Type3 transaction's declared signers instead of verifying signatures, +; and protocol/src/action/macro.rs runs a lighter action precheck. For a node that +; will serve a pool paying other people, validating every signature is worth more +; than a faster first sync. ; No [mint] section. chain_id defaults to 0, which is mainnet. diff --git a/docs/MINER-POOL-DELTA-REVIEW.md b/docs/MINER-POOL-DELTA-REVIEW.md new file mode 100644 index 00000000..b0fd7002 --- /dev/null +++ b/docs/MINER-POOL-DELTA-REVIEW.md @@ -0,0 +1,138 @@ +# Delta Bug Review Report + +**Repo:** `C:/Users/KQHEX/Documents/hacash-fullnodedev` +**Scope:** Block miner, diamond GPU, hacd-pool / miner-pool +**Date:** 2026-07-24 + +--- + +## 1. Summary + +| Category | Count | +|----------|------:| +| **Fixed** | 5 | +| **Still open** | 14 | +| **Unclear** | 0 | +| **New** | 11 | + +**Net residual risk:** 25 open issues (14 prior + 11 new). Critical consensus/testnet hacks (**C1**, **C2**) and the major block-mining correctness suite (**C3**, **C4**, **P11**) are fixed. Remaining work clusters on **template/reorg lifecycle**, **OpenCL diamond kernel correctness**, and **submit durability**. + +--- + +## 2. Fixed + +| ID | Severity | File | Note | +|----|----------|------|------| +| **P11** | medium | `x16rs-cuda` batch launch | Batch path calls `clamped_block_size` at `cuda_init_miner` and refuses devices that cannot launch 256-thread blocks. | +| **C1** | critical | `x16rs/src/diamond.rs` | Mainnet `DMD_L=10` / `DMD_M=16` hardcoded; no testnet 4/10 override. | +| **C2** | critical | `mint/src/action/diamond_mint.rs` | Height `% 5` diamond mint rule enforced; no `if false` bypass. | +| **C3** | high | `app/src` mining winners / equal target / GPU fatal | Multi-winner submit, equal-inclusive target, logged GPU fail, CUDA verify, capped CPU recovery — implemented and unit-tested. | +| **C4** | high | `miner-pool` `rpc_proxy` / `stratum` | Non-JSON upstream → `ret:1`; stratum accepts only parseable JSON with `ret==0`. | + +--- + +## 3. Still open + +| ID | Severity | File | Issue | +|----|----------|------|-------| +| **P1** | high | `app/src/block_mining_runtime.rs` | Winners coalesced without epoch/template id; submit/drain skip live-template revalidation; same-height reorg stale result can out-rank live work. | +| **P2** | high | `app/src/poworker.rs` | Template install only on height advance or same-height `intro_changed`; reorg that **lowers** pending height is ignored. | +| **P3** | high | `x16rs/opencl/x16rs_diamond.cl` | Diamond OpenCL reduction seeds `best_hash=0` with uninitialized `best_name`; scans `i=1..` (unlike block kernel seed-from-index). | +| **P4** | high | `x16rs/opencl/x16rs_diamond.cl` | Diamond barriers use only `CLK_LOCAL_MEM_FENCE` while hashes live in global memory; block kernel uses `LOCAL\|GLOBAL`. | +| **P5** | high | `app/src/diaworker.rs` | `push_diamond_mining_success` treats any HTTP Ok as final; non-JSON / missing `tx_hash` fails permanently; no requeue after drain. | +| **P6** | medium | `app/src/poworker.rs` | `LAST_PENDING_INTRO` written **before** `set_pending_block_stuff` succeeds; failed same-height install never retried. | +| **P7** | medium | `app/src/block_mining_runtime.rs` | Job-switch height/epoch checked only **after** full batch + `send()`; drain does not filter by live epoch. | +| **P8** | medium | `app/src/block_mining_runtime.rs` | Worker rollover is advance-only (`check_hei > mining_hei`) + epoch; height decrease without epoch may not stop workers. | +| **P9** | medium | `x16rs/opencl/sha3_256.cl` | Host builds 61-byte diamond stuff for number ≤ 20000; GPU always pads as 93-byte. | +| **P10** | medium | `app/src/opencl_dia.rs` | Diamond GPU success trusts medium hash; no `x16rs_hash` recompute + byte-compare (unlike block `verify_gpu_best_result`). | +| **P12** | medium | `app/src/diaworker.rs` | After `MAX_SUBMIT_ATTEMPTS` or soft parse failure, drained `DiamondMint` is logged and dropped — no durable save/requeue. | +| **P13** | medium | `x16rs/opencl/x16rs_diamond.cl` | Per-work-item reduction never seeds `best_name` from unit index 0 before `i=1..` comparisons. | +| **P14** | low | `app/src/block_mining_runtime.rs` | Drain aggregates planned `res.nonce_space` even when recovery only mined a partial window — telemetry overstated. | +| **P15** | low | `app/src/opencl_gpu/block.rs` | OpenCL accepts stuff length ≤ 512 without enforcing the 89-byte block intro CUDA requires. | + +--- + +## 4. New bugs + +| Severity | File | Issue | Impact | +|----------|------|-------|--------| +| **high** | `app/src/poworker.rs` | Notice long-poll only breaks when notice height ≥ `pending_height`; fullnode notice reports chain tip (typically pending−1), so timeouts and same-height tip reorgs never re-fetch pending. `intro_changed` path is effectively dead on fullnode. | After tip reorg, workers hash orphaned parent for up to a full block interval; PoW cannot land on main chain. | +| **high** | `app/src/poworker.rs` | `leave_upstream_stale()` runs as soon as `block_intro` is present, before `set_pending_block_stuff` succeeds and even when install is skipped by the height/intro gate. | After upstream-stale outage, workers can resume grinding a previous dead template if recovery install fails/skips. | +| **high** | `app/src/diaworker.rs` | `pull_and_push_diamond` only advances when `next_num > mining_num`; never rolls number backward or refreshes `prev_hash` / `born.hash` after diamond reorg. | Workers mine invalid `(number, prev_hash)` until chain mints past stale number; successes fail node validation. | +| **high** | `miner-pool/src/job.rs` | `JobHub::update` always sets `job_id=h{height}` only; stratum dedup keys solely on `job_id`, so same-height template changes never emit `mining.notify`. | Stratum miners stay on orphaned/obsolete template while HTTP path can serve new job; shares miss live tip. | +| **medium** | `app/src/mining_batch.rs` | OpenCL batch/integrity failures only trigger `on_batch_error` + bounded CPU recovery; no consecutive-failure budget or session GPU-disable (CUDA has both at 20). | Permanently failing OpenCL device never fail-closes; miner stuck on tiny CPU recovery, masking hardware death. | +| **medium** | `app/src/block_mining_runtime.rs` | `set_pending_block_stuff` does not require JSON height == `block_intro.height()`; mining uses JSON height for x16rs repeat while consensus uses intro-embedded height. | Mismatched upstream height → wrong-repeat mining and/or submit rejection. | +| **medium** | `app/src/block_mining_runtime.rs` | Result drain thread returns immediately on `stop_flag` without draining the result channel; only submit queue is wound down. | Clean shutdown/restart can discard target-meeting winners still in the result channel (lost shares/blocks). | +| **medium** | `x16rs/opencl/x16rs_diamond.cl` | Kernel reduction ranks solely by `diamond_more_power` (more leading zeros); consensus requires **exactly** `DMD_L` zeros — overshoots are invalid. | Valid 10-zero diamond lost when an invalid 11+ overshoot is in the same unit/WG. | +| **medium** | `app/src/opencl_dia.rs` | Diamond GPU post-process never bounds-checks nonces against `[nonce_start, nonce_start+nonce_space)` (block path does). | Corrupted/out-of-window nonces can pass partial checks → false success or wasted submits. | +| **medium** | `app/src/opencl_dia.rs` | On `check_diamer_success`, function returns before `needs_queue_finish` / `queue.finish()`; block OpenCL always finishes on RDNA4/duplicate ICD. | After diamond success, AMD queue may not drain → stale/out-of-order batches. | +| **low** | `x16rs/opencl/util.cl` | `block_t` is 88 bytes; GPU SHA3 hardcodes pad lane forcing `intro[88]==0`; CPU/consensus use full 89-byte intro (`witness_stage`). | Latent while `witness_stage` is zero; non-zero 89th byte makes GPU SHA3 diverge from consensus. | + +--- + +## 5. Unclear + +None. Prior rechecks produced no unclear outcomes. + +--- + +## 6. Recommended next fix order + +Priority groups by **payout / correctness risk**, then **cluster affinity** (fix one area together). + +### Tier 0 — Template / reorg correctness (blocks payout path) + +1. **New: notice long-poll / tip vs pending** (`poworker.rs` + fullnode `miner_notice`) — unlocks same-height reorg path; without this, **P2**/`intro_changed` barely matter on fullnode. +2. **New: `leave_upstream_stale` before install** (`poworker.rs`) — stop grinding dead templates after outage recovery. +3. **P2** — install on height decrease (reorg to lower pending). +4. **P6** — write `LAST_PENDING_INTRO` only after successful `set_pending_block_stuff`. +5. **P1 + P7 + P8** together — epoch/template id on `BlockMiningResult`; filter drain; stop workers on height decrease; revalidate before submit. +6. **New: height vs intro.height mismatch** (`set_pending_block_stuff`) — cheap invariant, prevents wrong-repeat mining. + +### Tier 1 — Pool / diamond job identity + +7. **New: stratum `job_id=h{height}` only** (`miner-pool/job.rs` + stratum notify) — include intro/content hash so same-height reorgs notify. +8. **New: diamond number/prev_hash never roll back** (`diaworker.rs` `pull_and_push_diamond`) — reorg-safe diamond job refresh. + +### Tier 2 — OpenCL diamond correctness (find loss) + +9. **P3 + P13** — seed reduction from unit index 0 / current work-item (match block kernel pattern). +10. **P4** — `CLK_LOCAL_MEM_FENCE | CLK_GLOBAL_MEM_FENCE` on diamond barriers. +11. **New: diamond_more_power overshoot** — prefer valid exact-`DMD_L` names over stronger invalid overshoots (or validate before reduce). +12. **P9** — 61-byte vs 93-byte diamond SHA3 padding for number ≤ 20000. +13. **P10 + New: nonce window + finish-on-success** (`opencl_dia.rs`) — recompute medium hash, bounds-check nonces, always `queue.finish` when required. + +### Tier 3 — Submit durability & fail-close + +14. **P5 + P12** — durable diamond submit requeue / save on network and soft parse failure. +15. **New: result channel abandon-on-stop** — final drain before result thread exit. +16. **New: OpenCL consecutive-failure GPU disable** — parity with CUDA session latch. + +### Tier 4 — Low / latent + +17. **P14** — report actual mined nonce space, not planned. +18. **P15** — enforce 89-byte block intro on OpenCL upload. +19. **New: block_t 88-byte / intro[88]** — load 89th byte into SHA3 when `witness_stage` can be non-zero. + +--- + +### Cluster map (for parallel workstreams) + +| Stream | Items | +|--------|--------| +| **A. Block job lifecycle** | Notice long-poll, leave_upstream_stale, P2, P6, P1, P7, P8, height==intro.height | +| **B. Pool / diamond jobs** | Stratum job_id, diamond number rollback | +| **C. Diamond GPU kernel** | P3, P4, P13, overshoot, P9 | +| **D. Diamond host post** | P10, nonce window, finish-on-success, P5, P12 | +| **E. Ops / telemetry** | OpenCL GPU disable, shutdown drain, P14, P15, block_t 89th byte | + +--- + +### Severity rollup (open only) + +| Severity | Still open | New | Total | +|----------|----------:|----:|------:| +| high | 5 | 4 | **9** | +| medium | 7 | 6 | **13** | +| low | 2 | 1 | **3** | +| **Total** | **14** | **11** | **25** | diff --git a/docs/MINER-POOL-WORKFLOW-REVIEW.md b/docs/MINER-POOL-WORKFLOW-REVIEW.md new file mode 100644 index 00000000..49dc3306 --- /dev/null +++ b/docs/MINER-POOL-WORKFLOW-REVIEW.md @@ -0,0 +1,122 @@ +# Miner + Pool Bug Review Report + +**Repository:** `C:/Users/KQHEX/Documents/hacash-fullnodedev` +**Scope:** block miner runtime / poworker, GPU OpenCL+CUDA backends, diamond worker + consensus submit +**Status:** 15 confirmed open bugs (adversarially verified) + +--- + +## 1. Summary + +Review of the miner and related GPU/diamond paths confirmed **15 open bugs**. The most severe cluster is in **block mining job lifecycle and winner selection**: same-height reorgs can allow stale in-flight results to out-rank and replace a currently valid solution; depth-reducing reorgs never install a lower pending height; and workers only stop on height *advance* or epoch bump. A second high-severity cluster is in **diamond OpenCL** (uninitialized/wrong reduction seed + LOCAL-only fences on global-backed hashes) and **diamond submit durability** (HTTP-200 noise treated as final; no requeue/save after drain). Medium issues include failed same-height install not retried, late job-switch checks feeding the result channel, SHA3 length mismatch for low diamond numbers, missing GPU medium-hash re-verify, and CUDA batch launch without `clamped_block_size`. Lower-severity items cover nonce-span telemetry after recovery and OpenCL block stuff length acceptance. + +--- + +## 2. Confirmed open bugs + +| Severity | File | Issue | Impact | +|---|---|---|---| +| **high** | `app/src/block_mining_runtime.rs:709` | Winners coalesced only by height (keep strongest `result_hash`). After same-height reorg/epoch change, an in-flight stale-template result can still meet its own `target_hash` and out-rank a weaker but currently valid solution; only the stale entry is submitted. No epoch/template id on `BlockMiningResult`; `push_block_mining_success` submits the coalesced winner without live-template revalidation. | A real block acceptable for the current template can be silently discarded; miner loses payout while the node rejects the orphaned solution. | +| **high** | `app/src/poworker.rs:264` | Template install only when `pending_height > curr_hei` or `(same height && intro_changed)`. A reorg that **lowers** pending height is ignored, so `MINING_BLOCK_HEIGHT`/epoch never update and workers never job-switch. | After a depth-reducing reorg the miner can grind a non-existent height indefinitely, producing only rejected work and missing the new tip. | +| **high** | `x16rs/opencl/x16rs_diamond.cl:122` | Per-thread diamond reduction initializes `best_hash = 0` and leaves `best_name` uninitialized, then scans only `i=1..unit_size-1`. Block/CUDA paths already seed from `index` (see `x16rs_main.cl:84`, `block_miner.cu:71-76`). | For `local_id>0`, `best_hash` can point into another thread’s slots; work-group reduction propagates wrong diamond winners—missed finds and inconsistent nonce/hash pairs. | +| **high** | `x16rs/opencl/x16rs_diamond.cl:96` | Diamond kernel stores hashes in global memory (`local_hashes = global_hashes + …`) but post-SHA3 / reduction barriers use only `CLK_LOCAL_MEM_FENCE`. Block kernel uses `CLK_LOCAL_MEM_FENCE \| CLK_GLOBAL_MEM_FENCE` at matching points. | No guaranteed cross-item visibility of global hash writes before reduction; intermittent wrong diamond reductions (lost finds / corrupted best nonce-hash) on some devices. | +| **high** | `app/src/diaworker.rs:890` | `push_diamond_mining_success` treats any HTTP transport `Ok` body as final: breaks immediately, then fails permanently on non-JSON / missing `tx_hash` without retry. Block path (`poworker`) retries unrecognized HTTP-200 bodies; diamond only retries `reqwest` `Err`. | A rare mined diamond can be discarded after proxy/HTML 200, truncated body, or gateway noise within the timeout window. No durable requeue after drain. | +| **medium** | `app/src/poworker.rs:255` | `LAST_PENDING_INTRO` is overwritten **before** `set_pending_block_stuff` succeeds. On same-height install failure (bad target/coinbase/mkrl), the new intro is already remembered → `intro_changed` stays false and install is never retried while height is unchanged. | Miner can remain stuck on an orphaned same-height template until a later height advance, wasting hashrate and missing blocks after reorg/partial RPC payload. | +| **medium** | `app/src/block_mining_runtime.rs:617` | Job-switch (height/epoch) is checked only **after** a full batch and after `send()`. Stale `BlockMiningResult` values always enter the result channel; `deal_block_mining_results` does not filter by current epoch/target before `push_block_mining_success`. Stale submits block the single result thread (HTTP timeouts × attempts). | Feeds same-height winner coalescing failures; delays fresher queue items behind useless submit attempts. | +| **medium** | `app/src/block_mining_runtime.rs:643` | Worker rollover uses `check_hei > mining_hei` (advance-only) plus epoch; correctness for non-monotonic height depends entirely on epoch bumps from `set_pending`, which `pull_pending` may never call on height decrease. | Defense-in-depth gap: if height goes down without an epoch publish path, workers do not stop; compounds the reorg job-switch bug. | +| **medium** | `x16rs/opencl/sha3_256.cl:182` | Host builds 61-byte stuff when custom message is gated empty for diamond numbers ≤ 20000 (`opencl_dia.rs:31-54`), but `sha3_256_hash_diamond` always applies fixed 93-byte SHA3 padding. Consensus/CPU hash true length. | GPU SHA3 diverges from CPU/consensus for low diamond numbers; finds cannot verify—OpenCL diamond mining useless on that range (testnets / early numbers). | +| **medium** | `app/src/opencl_dia.rs:107` | Diamond GPU success path only runs `calculate_hash(stuff)` + `check_diamer_success`; never recomputes `x16rs_hash(repeat, ssshash)` and byte-compares to the GPU medium hash (unlike block `verify_gpu_best_result`). | GPU medium hash can pass independent name/difficulty checks without being the x16rs of the claimed nonce’s SHA3; false local successes under reduction bugs; consensus re-mines and rejects. | +| **medium** | `x16rs-cuda/src/lib.rs:472` | Batch mining launches `x16rs_cuda_main` with fixed `miner.local_size` (256) and never calls `clamped_block_size`. Single-hash path clamps to `maxThreadsPerBlock` to avoid `cudaErrorInvalidConfiguration`. Failures hit 100k-nonce CPU recovery. | If batch kernel `maxThreadsPerBlock` < 256, every CUDA batch fails; hashrate collapses despite a usable GPU. | +| **medium** | `app/src/diaworker.rs:905` | After `MAX_SUBMIT_ATTEMPTS` (5, ~7.5s backoff) network failure, or after a soft parse failure, the drained `DiamondMint` is dropped with only a log—no local save or later resubmit. Recovery curl is commented out. | Node restart, brief RPC outage, or auth blip during submit permanently loses the find even though PoW work completed. | +| **medium** | `x16rs/opencl/x16rs_diamond.cl:122` | Per-work-item reduction leaves `diamond_t best_name` uninitialized and starts comparison at `i=1`, so unit index 0 is never `diamond_hash`’d into `best_name` before comparisons. (Host re-checks candidates; `DiaWorkConf` currently forces `useopencl=false` for HACD.) | If HACD OpenCL is re-enabled, valid nonces can be discarded and hashrate/find rate understated. No false mint while host revalidates and OpenCL is disabled. | +| **low** | `app/src/block_mining_runtime.rs:691` | Drain aggregate adds planned `res.nonce_space` into `total_nonce_space` for the status line even when GPU/CPU recovery reports only a partial window; hashrate EWMA uses partial counts. | Misleading nonce-span / efficiency telemetry after OOM/integrity recovery; can skew operator decisions (not consensus). | +| **low** | `app/src/opencl_gpu/block.rs:20` | OpenCL block upload accepts any stuff length ≤ 512 (`write_stuff_to_gpu`) and never enforces the 89-byte block intro that CUDA requires (`STUFF_BYTES`). Kernel SHA3 assumes fixed 89-byte padded layout. | Short/long intro desyncs GPU vs CPU hashes; integrity verify fails every batch → 100k-nonce CPU recovery until template fixed. Wrong solutions not submitted, but GPU work wasted. | + +### Count by severity + +| Severity | Count | +|---|---| +| high | 5 | +| medium | 8 | +| low | 2 | +| **total open** | **15** | + +### Count by area + +| Area | Count | +|---|---| +| block-miner | 6 | +| gpu-backends | 6 | +| diamond-consensus | 3 | + +--- + +## 3. Notes on already-fixed / rejected claims + +**Rejected (do not treat as open bugs):** + +- **`MINING_BLOCK_HEIGHT` / `EPOCH` Relaxed atomics without Acquire on worker job-switch** (`app/src/block_mining_runtime.rs:316`) — Height/epoch act as Relaxed cancel flags only; template payload is correctly synced via `RwLock`. Missing Acquire/Release does **not** establish an extra real stale-batch bug beyond the confirmed job-switch and coalesce issues above. + +**Related “already fixed elsewhere” notes (still open on diamond path):** + +- Block OpenCL (`x16rs_main.cl`) and CUDA block miner already seed reduction from `index` and use LOCAL\|GLOBAL fences; diamond OpenCL still has the old reduction/fence pattern. +- Block mining has full `verify_gpu_best_result` recompute; diamond OpenCL success does not recompute/equality-check the medium hash. +- Block submit retries unrecognized HTTP-200 bodies; diamond submit does not. + +--- + +## 4. Recommended fix order + +1. **Block reorg job install + worker stop (high, root cause of grinding dead height)** + - In `poworker.rs`, install templates when `pending_height != curr_hei` (or explicitly handle `pending_height < curr_hei`), not only advance / same-height intro change. + - Always bump `MINING_BLOCK_EPOCH` on any template change including height decrease. + - Worker rollover: stop on **any** height change (`check_hei != mining_hei`) or epoch change, not advance-only. + +2. **Winner selection / submit revalidation (high, silent payout loss)** + - Tag `BlockMiningResult` with epoch (and/or template id / stuff hash). + - Coalesce or accept winners only for the **live** epoch/template; re-check result against live `MINING_BLOCK_STUFF` target before `push_block_mining_success`. + - Prefer filtering **before** `send()` (or drop in drain) so stale same-height results cannot suppress a live win. + +3. **Diamond OpenCL reduction + fences (high, correctness if/when GPU HACD is used)** + - Seed `best_hash = index`, hash slot 0 into `best_name` before the loop (mirror `x16rs_main.cl` / CUDA). + - Use `CLK_LOCAL_MEM_FENCE \| CLK_GLOBAL_MEM_FENCE` after global hash writes and during reduction. + - Keep host revalidation; re-enable only after kernel + SHA3 length fixes. + +4. **Diamond submit durability (high → medium)** + - Align with block path: retry unrecognized/non-JSON HTTP-200 bodies. + - On exhausted attempts or soft parse failure: **local save + requeue** the drained `DiamondMint` (do not only log). + - Do not treat every transport `Ok` as terminal success. + +5. **Same-height install atomicity (medium)** + - Update `LAST_PENDING_INTRO` **only after** successful `set_pending_block_stuff`, or roll back on `Err` so `intro_changed` can retry. + +6. **Stale result pipeline (medium)** + - Check height/epoch **before** building/sending `BlockMiningResult`. + - Drain path: drop winners whose epoch/target no longer match live template before blocking HTTP submit. + +7. **Diamond SHA3 length + host verify (medium)** + - Make `sha3_256_hash_diamond` length-aware (61 vs 93) consistent with consensus/CPU for numbers ≤ 20000. + - Recompute `x16rs_hash` and require equality to GPU medium hash (parity with block `verify_gpu_best_result`). + +8. **CUDA batch `clamped_block_size` (medium)** + - Launch batch kernel with clamped block size like the single-hash path to avoid permanent invalid-config → 100k CPU recovery collapse. + +9. **OpenCL block stuff length (low)** + - Enforce 89-byte intro (or reject) on OpenCL upload to match CUDA/`STUFF_BYTES` and fixed kernel pad layout. + +10. **Nonce-space telemetry (low)** + - Aggregate status `total_nonce_space` from actual recovered `gpu_nonce_space`/`cpu_nonce_space` when partial recovery is reported, not planned `res.nonce_space` alone. + +### Suggested patch grouping + +| PR | Scope | Severity | +|---|---|---| +| A | Reorg install gate + epoch bump + worker `!=` height stop | high | +| B | Result epoch tag + live-template winner filter + pre-send job check | high | +| C | Diamond CL reduction seed + GLOBAL fences + SHA3 length | high/medium | +| D | Diamond submit retry parity + durable requeue/save | high/medium | +| E | `LAST_PENDING_INTRO` only-after-success; CUDA clamp; OpenCL stuff len; telemetry | medium/low | + +--- + +*End of report — 15 confirmed open bugs; 1 rejected claim.* \ No newline at end of file diff --git a/hbit-pool/examples/fee_probe.rs b/hbit-pool/examples/fee_probe.rs new file mode 100644 index 00000000..e4136b7d --- /dev/null +++ b/hbit-pool/examples/fee_probe.rs @@ -0,0 +1,42 @@ +//! Ask a LIVE node what one of the pool's own blocks paid it in transaction +//! fees, using the exact function the settlement path uses. +//! +//! This exists to make the fee hold-back falsifiable. `block_fees` is the one +//! step of the money path that cannot be exercised without a node: it reads +//! `/query/block/intro?tx_hash_list=true` and then one `/query/transaction` per +//! packed transaction, and every one of those field names is a promise about a +//! node this crate does not build. Run it against a height the pool really won +//! and the answer is either the fee total in units of 0.1 HAC, or the refusal +//! the settlement turns into "settle nothing this cycle". +//! +//! usage: fee_probe + +use hbit_pool::{BlockFees, block_fees, http_client}; + +fn main() { + let a: Vec = std::env::args().collect(); + if a.len() < 4 { + eprintln!("usage: fee_probe "); + std::process::exit(2); + } + let node = a[1].trim_end_matches('/').to_string(); + let height: u64 = a[2].parse().expect("height"); + let hash = a[3].to_lowercase(); + + let client = http_client(); + match block_fees(&client, &node, height, &hash) { + BlockFees::Counted(u) => { + println!("Counted({u}) units of 0.1 HAC"); + } + BlockFees::NotOnChain => { + println!("NotOnChain (the chain does not hold our block at that height)"); + std::process::exit(1); + } + BlockFees::Unknown(why) => { + println!("Unknown: {why}"); + // The settlement turns this into "nothing is settled this cycle", + // so a probe that answered 0 here would be worse than useless. + std::process::exit(1); + } + } +} diff --git a/hbit-pool/examples/local_chain_watch.rs b/hbit-pool/examples/local_chain_watch.rs new file mode 100644 index 00000000..680a36eb --- /dev/null +++ b/hbit-pool/examples/local_chain_watch.rs @@ -0,0 +1,162 @@ +//! Watch a local (non-mainnet) chain climb to a difficulty the HBIT pool will serve. +//! +//! This is the MEASURING instrument for the rig in +//! `scripts/hbit-local-chain-rig/`. It does not model anything: it polls the +//! node and reports the difficulty the chain actually stored, converted to +//! leading zero bits by the SAME function the pool's own admission check uses +//! (`pool_core::share_cost_bits` over `pool_core::network_target_hash`), so the +//! number printed here is the number `check_share_target` will see. +//! +//! Exit code is the result, because that is the only part a script may trust: +//! 0 the chain reached `--want` bits inside the deadline +//! 1 the deadline passed first (prints the best it managed) +//! 2 bad arguments / the node could not be read +//! +//! Usage: +//! cargo run --release -p hbit-pool --example local_chain_watch -- \ +//! [poll_secs] + +use hbit_pool::pool_core::{ + achieved_share_factor, network_target_hash, share_cost_bits, share_target_hash, +}; +use hbit_pool::{find_u64, get_json, http_client}; +use std::time::{Duration, Instant}; + +/// server.rs MIN_SHARE_FACTOR: how much easier a share may be than a block. +const MIN_SHARE_FACTOR: u32 = 18; +/// server.rs MIN_SHARE_COST_BITS: what the share itself must cost. +const MIN_SHARE_COST_BITS: u32 = 16; + +/// Reproduce the pool's own admission gate for a given network difficulty. +/// +/// server.rs derives exactly these two numbers from exactly these two calls and +/// hands them to `check_share_target`, so if this says yes the pool says yes. +/// `check_share_target` is private to that binary, which is why the inputs are +/// recomputed here rather than the decision being imported. +fn pool_would_serve(difficulty: u32, share_bits: u32) -> (bool, u32, u32) { + let network = network_target_hash(difficulty); + let share = share_target_hash(difficulty, share_bits); + let achieved = achieved_share_factor(&network, &share); + let cost = share_cost_bits(&share); + ( + achieved >= MIN_SHARE_FACTOR && cost >= MIN_SHARE_COST_BITS, + achieved, + cost, + ) +} + +fn main() { + let a: Vec = std::env::args().collect(); + let node = a + .get(1) + .cloned() + .unwrap_or_else(|| "http://127.0.0.1:8080".to_string()); + let node = node.trim_end_matches('/').to_string(); + let want: u32 = a.get(2).and_then(|s| s.parse().ok()).unwrap_or(34); + let deadline: u64 = a.get(3).and_then(|s| s.parse().ok()).unwrap_or(3600); + let poll: u64 = a.get(4).and_then(|s| s.parse().ok()).unwrap_or(5); + + let client = http_client(); + let started = Instant::now(); + + // Read the tip's height, timestamp and stored difficulty. + let tip = |c: &reqwest::blocking::Client| -> Option<(u64, u64, u32)> { + let h = find_u64(&get_json(c, &format!("{node}/query/latest")), "height")?; + let b = get_json(c, &format!("{node}/query/block/intro?height={h}")); + Some(( + h, + find_u64(&b, "timestamp")?, + find_u64(&b, "difficulty")? as u32, + )) + }; + + let Some((h0, _, d0)) = tip(&client) else { + eprintln!("could not read the chain tip from {node}"); + std::process::exit(2); + }; + println!("== local chain watch =="); + println!("node = {node}"); + println!("want >= {want} network leading-zero bits"); + println!("deadline = {deadline}s, poll every {poll}s"); + println!( + "start = height {h0}, difficulty {d0}, {} bits\n", + share_cost_bits(&network_target_hash(d0)) + ); + println!( + "{:>8} {:>8} {:>6} {:>12} {:>9} {:>10}", + "elapsed", "height", "bits", "difficulty", "blk_secs", "blocks" + ); + + let mut last_height = h0; + let mut last_ts: Option = None; + let mut best_bits = share_cost_bits(&network_target_hash(d0)); + // First time each milestone was seen, as (bits, elapsed_secs, height). + let mut milestones: Vec<(u32, u64, u64)> = Vec::new(); + + loop { + let elapsed = started.elapsed().as_secs(); + if let Some((h, ts, d)) = tip(&client) { + let bits = share_cost_bits(&network_target_hash(d)); + let blk_secs = match last_ts { + Some(prev) if h > last_height => { + format!("{}", ts.saturating_sub(prev) / (h - last_height).max(1)) + } + _ => "-".to_string(), + }; + if h != last_height || bits != best_bits { + println!( + "{:>7}s {:>8} {:>6} {:>12} {:>9} {:>10}", + elapsed, + h, + bits, + d, + blk_secs, + h.saturating_sub(h0) + ); + } + if bits > best_bits { + best_bits = bits; + milestones.push((bits, elapsed, h)); + } + last_height = h; + last_ts = Some(ts); + if bits >= want { + println!("\nREACHED {want} bits at height {h} after {elapsed}s."); + print_milestones(&milestones, want); + println!( + "\nA block at {bits} bits costs 2^{bits} hashes on average.\n\ + The pool's own admission gate at this difficulty ({d}):" + ); + for sb in [MIN_SHARE_FACTOR, 20, 24] { + let (ok, achieved, cost) = pool_would_serve(d, sb); + let verdict = if ok { "SERVES" } else { "REFUSES" }; + println!( + " share_bits {sb:>2}: achieved {achieved:>2} (min {MIN_SHARE_FACTOR}), \ + cost bits {cost:>2} (min {MIN_SHARE_COST_BITS}) -> {verdict}" + ); + } + std::process::exit(0); + } + } else { + println!("{:>7}s (node not answering yet)", elapsed); + } + if elapsed >= deadline { + println!("\nDEADLINE: only reached {best_bits} bits (wanted {want}) in {elapsed}s."); + print_milestones(&milestones, want); + std::process::exit(1); + } + std::thread::sleep(Duration::from_secs(poll.max(1))); + } +} + +fn print_milestones(ms: &[(u32, u64, u64)], want: u32) { + if ms.is_empty() { + println!("(difficulty never moved)"); + return; + } + println!("\nfirst sighting of each bit level:"); + for (bits, secs, hei) in ms { + let mark = if *bits >= want { " <- pool servable" } else { "" }; + println!(" {bits:>2} bits at {secs:>6}s height {hei}{mark}"); + } +} diff --git a/hbit-pool/examples/rig_tx.rs b/hbit-pool/examples/rig_tx.rs new file mode 100644 index 00000000..35ce26e7 --- /dev/null +++ b/hbit-pool/examples/rig_tx.rs @@ -0,0 +1,100 @@ +//! Rig helper: derive an address from a throwaway secret, and push real +//! transfers into a node's mempool so a mined block CONTAINS TRANSACTIONS. +//! +//! This exists because the pool's transaction-fee hold-back can only be +//! exercised by a block that actually carries fee-paying transactions, and a +//! private rig chain has no other traffic on it. +//! +//! TESTNET ONLY. It takes a raw private key on the command line. +//! +//! usage: +//! rig_tx addr +//! rig_tx send [count] +//! +//! `addr` prints the readable address for a secret so it can be funded. +//! `send` builds, signs and submits `count` transfers (each with a distinct +//! timestamp so the hashes differ), printing every node answer. + +use basis::interface::*; +use field::*; +use protocol::action::HacToTrs; +use protocol::transaction::TransactionType2; +use sys::*; + +use hbit_pool::{get_json, http_client, post_hex}; + +fn secret_from_hex(s: &str) -> [u8; 32] { + let b = hex::decode(s).expect("secret must be 64 hex chars"); + assert_eq!(b.len(), 32, "secret must be 32 bytes"); + let mut k = [0u8; 32]; + k.copy_from_slice(&b); + k +} + +fn main() { + let a: Vec = std::env::args().collect(); + match a.get(1).map(|s| s.as_str()) { + Some("addr") => { + let acc = Account::create_by_secret_key_value(secret_from_hex(&a[2])).expect("account"); + println!("{}", acc.readable()); + } + Some("send") => { + let base = a[2].trim_end_matches('/').to_string(); + let acc = Account::create_by_secret_key_value(secret_from_hex(&a[3])).expect("account"); + let to = Address::from_readable(&a[4]).expect("to_address"); + let amt = Amount::from(&a[5]).expect("hac amount"); + let fee = Amount::from(&a[6]).expect("fee amount"); + let count: u64 = a.get(7).and_then(|s| s.parse().ok()).unwrap_or(1); + + let client = http_client(); + let main = Address::from(*acc.address()); + println!("from = {}", acc.readable()); + println!("to = {}", a[4]); + println!( + "amount = {} each, fee = {} each, count = {count}", + a[5], a[6] + ); + + let base_ts = curtimes(); + for i in 0..count { + // Distinct timestamps make the hashes differ. They go BACKWARDS: + // the node refuses a transaction stamped later than its own + // clock, so counting up rejects everything after the first. + let mut tx = + TransactionType2::new_by(main.clone(), fee.clone(), base_ts.saturating_sub(i)); + let mut act = HacToTrs::new(); + act.to = AddrOrPtr::from_addr(to.clone()); + act.hacash = amt.clone(); + tx.push_action(Box::new(act)).expect("push action"); + tx.fill_sign(&acc).expect("fill_sign"); + let hash = hex::encode(tx.hash().serialize()); + let body = hex::encode(tx.serialize()); + let resp = post_hex( + &client, + &format!("{base}/submit/transaction?hexbody=true"), + &body, + ); + println!("submit {hash} -> {resp}"); + } + // Report what the node thinks it now holds, so "submitted" is never + // confused with "in the mempool". + std::thread::sleep(std::time::Duration::from_millis(800)); + let pend = get_json( + &client, + &format!("{base}/query/miner/pending?detail=true&transaction=true&stuff=true"), + ); + let n = pend + .get("data") + .and_then(|d| d.get("transactions")) + .and_then(|v| v.as_array()) + .map(|a| a.len()); + println!("node template now carries transactions: {n:?}"); + } + _ => { + eprintln!( + "usage:\n rig_tx addr \n rig_tx send [count]" + ); + std::process::exit(2); + } + } +} diff --git a/hbit-pool/examples/testnet_rig_plan.rs b/hbit-pool/examples/testnet_rig_plan.rs new file mode 100644 index 00000000..d317797c --- /dev/null +++ b/hbit-pool/examples/testnet_rig_plan.rs @@ -0,0 +1,224 @@ +//! Pick (and prove) the `[mint]` settings for a local chain the HBIT pool will +//! actually serve. +//! +//! WHY THIS EXISTS +//! +//! The pool refuses to hand out work when a share is not worth counting: +//! `share_cost_bits >= 16` AND `achieved_share_factor >= 18`, and those two add +//! up to the leading zero bits of the NETWORK target. So the pool needs a chain +//! sitting at >= 34 leading zero bits. A freshly started non-mainnet chain does +//! not: off mainnet ASERT anchors at `difficulty_adjust_blocks + 2` with the +//! fixed constant 0xe9cfffff, which is exactly 22 leading zero bits. +//! +//! The only knob that moves where the chain SETTLES is `each_block_target_time`. +//! ASERT is an equilibrium controller: it drives the target until a block takes +//! `each_block_target_time` seconds, so the resting difficulty of a private +//! chain is whatever your own hashrate can do in that time. Pick the target time +//! and you pick the difficulty. That is the whole trick, and it needs no code +//! change at all, so mainnet consensus is untouched by construction. +//! +//! This example does not model that with a formula. It runs the REAL difficulty +//! function (`hbit_pool::difficulty::next_difficulty`, which `hbit-asert-check` +//! has already proven byte-identical to the node against real mainnet history) +//! block by block, and reports the wall clock. Compare its answer against a real +//! run; if they disagree, believe the run. +//! +//! USAGE +//! cargo run -p hbit-pool --example testnet_rig_plan -- \ +//! [adjust_blocks] [target_time_secs] [max_sim_secs] [min_block_secs] +//! +//! With no target time it SEARCHES for one and prints a table, which is how you +//! choose the number to put in the ini. +//! +//! Note on hashrate: x16rs repeats `height/50000 + 1` times (capped at 16), so a +//! private chain below height 50000 runs at repeat=1. Use your repeat=1 figure +//! here, not the mainnet repeat=16 one. They differ by more than 10x. + +use hbit_pool::difficulty::{ChainParams, next_difficulty}; +use hbit_pool::pool_core::share_cost_bits; + +/// What the pool demands of the NETWORK target before it will serve work. +/// Mirrors MIN_SHARE_FACTOR (18) + MIN_SHARE_COST_BITS (16) in server.rs. +const POOL_MIN_NETWORK_BITS: u32 = 34; +/// Above this a block stops being winnable in a sitting; not a hard rule, just +/// the top of the band this rig aims for. +const BAND_MAX_BITS: u32 = 38; + +struct Outcome { + /// Seconds of wall clock until the chain first reaches POOL_MIN_NETWORK_BITS. + reach_secs: Option, + /// Height at that moment. + reach_height: Option, + /// Seconds spent inside [POOL_MIN_NETWORK_BITS, BAND_MAX_BITS] within the sim. + band_secs: u64, + /// Bits at the end of the simulated window. + final_bits: u32, + /// Expected seconds for one block at the final difficulty. + final_block_secs: f64, + /// True if the sim ran out of time before reaching the band. + timed_out: bool, +} + +/// Walk the chain forward through the real difficulty rule. +/// +/// The model has exactly one assumption: a block at B leading zero bits takes +/// 2^B / hashrate seconds, floored at 1 because `chain/src/verify.rs` rejects +/// `blk_time <= prev_blk_time` in a release build. Everything else is the +/// shipped consensus code. +fn simulate( + hashrate: f64, + adjust_blocks: u64, + target_time: u64, + max_secs: u64, + min_block_secs: u64, +) -> Outcome { + let p = ChainParams::testnet(adjust_blocks, target_time); + // Bootstrap heights are LOWEST_DIFFICULTY (every hash wins) and the anchor + // block itself is the fixed start target; both are one second each because + // of the timestamp floor. + let anchor_time: u64 = p.asert_height; // one second per bootstrap block from t=0 + let mut clock = anchor_time; + let mut height = p.asert_height; + let mut prev_diff = next_difficulty(&p, p.asert_height, anchor_time, 0, 0).0; + + let mut reach_secs = None; + let mut reach_height = None; + let mut band_secs = 0u64; + let mut bits = share_cost_bits(&next_difficulty(&p, p.asert_height, anchor_time, 0, 0).1); + let mut block_secs = 2f64.powi(bits as i32) / hashrate; + + while clock - anchor_time < max_secs { + // How long this block takes at the difficulty now in force. + block_secs = 2f64.powi(bits as i32) / hashrate; + let step = (block_secs.round() as u64).max(min_block_secs); + if (POOL_MIN_NETWORK_BITS..=BAND_MAX_BITS).contains(&bits) { + band_secs += step; + } + clock += step; + height += 1; + let (num, hash) = next_difficulty(&p, height, clock, prev_diff, anchor_time); + prev_diff = num; + bits = share_cost_bits(&hash); + if bits >= POOL_MIN_NETWORK_BITS && reach_secs.is_none() { + reach_secs = Some(clock - anchor_time); + reach_height = Some(height); + } + } + Outcome { + reach_secs, + reach_height, + band_secs, + final_bits: bits, + final_block_secs: block_secs, + timed_out: reach_secs.is_none(), + } +} + +fn hms(s: u64) -> String { + format!("{:02}h{:02}m{:02}s", s / 3600, (s % 3600) / 60, s % 60) +} + +fn main() { + let a: Vec = std::env::args().collect(); + let mhs: f64 = a + .get(1) + .and_then(|s| s.parse().ok()) + .unwrap_or_else(|| { + eprintln!( + "usage: testnet_rig_plan [adjust_blocks] [target_time_secs] [max_sim_secs] [min_block_secs]" + ); + std::process::exit(2) + }); + let hashrate = mhs * 1e6; + let adjust_blocks: u64 = a.get(2).and_then(|s| s.parse().ok()).unwrap_or(8); + let explicit_tt: Option = a.get(3).and_then(|s| s.parse().ok()); + let max_secs: u64 = a.get(4).and_then(|s| s.parse().ok()).unwrap_or(6 * 3600); + // Floor on how long a block takes in practice. The consensus floor is 1 + // second (block_build.rs nextts = max(now, prev_ts+1)), but a real worker + // also spends time polling for a template and posting the solution, so the + // cheap early blocks land slower than the pure hash cost suggests. A run + // measured on this rig arrived at 34 bits about 20% later than the 1-second + // model; pass 2 here to see that bracket. + let min_block_secs: u64 = a.get(5).and_then(|s| s.parse().ok()).unwrap_or(1).max(1); + + let p = ChainParams::testnet(adjust_blocks, explicit_tt.unwrap_or(300)); + let anchor_bits = share_cost_bits(&next_difficulty(&p, p.asert_height, 0, 0, 0).1); + + println!("== HBIT local-chain rig plan =="); + println!("hashrate = {mhs} MH/s (x16rs repeat=1)"); + println!("difficulty_adjust_blocks = {adjust_blocks} -> ASERT anchors at height {}", p.asert_height); + println!("anchor difficulty = {anchor_bits} leading zero bits (fixed constant 0xe9cfffff)"); + println!("pool needs >= {POOL_MIN_NETWORK_BITS} network bits (share_bits 18 + share cost 16)"); + println!("simulation window = {}\n", hms(max_secs)); + + // Equilibrium: a block at B bits takes 2^B/hashrate seconds, so the chain + // rests where that equals each_block_target_time. + let eq_bits = |tt: u64| (hashrate * tt as f64).log2(); + let tt_for = |bits: f64| (2f64.powf(bits) / hashrate).round() as u64; + + match explicit_tt { + Some(tt) => { + let o = simulate(hashrate, adjust_blocks, tt, max_secs, min_block_secs); + report(tt, eq_bits(tt), &o); + // The verdict is the exit code, not the text. A script must be able + // to reject an unviable target time without reading English. + if o.reach_secs.is_none() { + std::process::exit(1); + } + } + None => { + println!( + "{:>9} {:>8} {:>12} {:>8} {:>12} {:>10}", + "target_t", "eq_bits", "reach>=34", "at_hei", "in_band", "blk_at_end" + ); + println!("{}", "-".repeat(70)); + // Candidate target times that put equilibrium at 34..39 bits. + for b in POOL_MIN_NETWORK_BITS..=(BAND_MAX_BITS + 1) { + let tt = tt_for(b as f64).max(1); + let o = simulate(hashrate, adjust_blocks, tt, max_secs, min_block_secs); + println!( + "{:>9} {:>8.2} {:>12} {:>8} {:>12} {:>9.0}s", + tt, + eq_bits(tt), + o.reach_secs.map(hms).unwrap_or_else(|| "NEVER".into()), + o.reach_height + .map(|h| h.to_string()) + .unwrap_or_else(|| "-".into()), + hms(o.band_secs), + o.final_block_secs + ); + } + println!( + "\nPick the row with the smallest `reach>=34` that still leaves a long `in_band`,\n\ + put that target_t in [mint].each_block_target_time, then re-run this with it\n\ + as argument 3 for the full report." + ); + } + } +} + +fn report(tt: u64, eq: f64, o: &Outcome) { + println!("each_block_target_time = {tt}s (equilibrium ~{eq:.2} bits)"); + match (o.reach_secs, o.reach_height) { + (Some(s), Some(h)) => { + println!("reaches {POOL_MIN_NETWORK_BITS} bits after {} of mining, at height {h}", hms(s)); + println!( + "stays in the {POOL_MIN_NETWORK_BITS}..{BAND_MAX_BITS} band for {} of the simulated window", + hms(o.band_secs) + ); + println!( + "at the end of the window: {} bits, ~{:.0}s per block", + o.final_bits, o.final_block_secs + ); + println!("\nPLAN IS VIABLE."); + } + _ => { + println!( + "NEVER reaches {POOL_MIN_NETWORK_BITS} bits in the window (ends at {} bits, ~{:.0}s per block).", + o.final_bits, o.final_block_secs + ); + println!("timed_out={}", o.timed_out); + println!("\nPLAN IS NOT VIABLE at this target time."); + } + } +} diff --git a/hbit-pool/src/asert_check.rs b/hbit-pool/src/asert_check.rs index 0ba2def4..9681b72a 100644 --- a/hbit-pool/src/asert_check.rs +++ b/hbit-pool/src/asert_check.rs @@ -57,7 +57,10 @@ fn main() { println!("== HBIT ASERT check =="); println!("node = {node}"); - println!("chain = {chain} (ASERT anchor at height {})", params.asert_height); + println!( + "chain = {chain} (ASERT anchor at height {})", + params.asert_height + ); println!("tip = {tip}"); let anchor_time = find_u64( @@ -80,7 +83,10 @@ fn main() { println!("h={h} (missing block data, skipped)"); continue; }; - let pb = get_json(&client, &format!("{node}/query/block/intro?height={}", h - 1)); + let pb = get_json( + &client, + &format!("{node}/query/block/intro?height={}", h - 1), + ); let Some(prev_diff) = find_u64(&pb, "difficulty") else { println!("h={h} (missing parent, skipped)"); continue; @@ -107,13 +113,18 @@ fn main() { } Some(_) => { ok += 1; - println!("h={h} OK difficulty={stored} target={}", hex::encode(target)); + println!( + "h={h} OK difficulty={stored} target={}", + hex::encode(target) + ); } None => { // Without the block's hash only half the check ran; do not report // that as a pass. bad += 1; - println!("h={h} NO-HASH could not read the block's own hash to verify the target"); + println!( + "h={h} NO-HASH could not read the block's own hash to verify the target" + ); } } } diff --git a/hbit-pool/src/difficulty.rs b/hbit-pool/src/difficulty.rs index 8918b537..c55910fb 100644 --- a/hbit-pool/src/difficulty.rs +++ b/hbit-pool/src/difficulty.rs @@ -176,7 +176,10 @@ mod tests { // Documented defaults and mainnet still parse. let d = ChainParams::parse("testnet").expect("bare testnet"); assert_eq!((d.asert_height, d.target_time), (290, 10)); - assert_eq!(ChainParams::parse("mainnet").expect("mainnet").asert_height, 738654); + assert_eq!( + ChainParams::parse("mainnet").expect("mainnet").asert_height, + 738654 + ); // Anything we cannot mine is refused rather than silently guessed. assert!(ChainParams::parse("regtest").is_none()); assert!(ChainParams::parse("testnet:8").is_none()); @@ -201,7 +204,10 @@ mod tests { let p = ChainParams::testnet(288, 10); let (num, hash) = next_difficulty(&p, 290, 9_999, LOWEST_DIFFICULTY, 0); assert_eq!(num, ASERT_START_TARGET_NUM); - assert_eq!(hash, DifficultyTarget::from_num(ASERT_START_TARGET_NUM).hash); + assert_eq!( + hash, + DifficultyTarget::from_num(ASERT_START_TARGET_NUM).hash + ); // mainnet anchors at 738654 let m = ChainParams::mainnet(); assert_eq!( @@ -232,11 +238,25 @@ mod tests { ); // ahead of schedule (mined too fast) -> smaller target (harder) let fast = DifficultyTarget::from_num( - next_difficulty(&p, height, on_time - 600, ASERT_START_TARGET_NUM, anchor_time).0, + next_difficulty( + &p, + height, + on_time - 600, + ASERT_START_TARGET_NUM, + anchor_time, + ) + .0, ); // behind schedule -> larger target (easier), capped at 2x the parent let slow = DifficultyTarget::from_num( - next_difficulty(&p, height, on_time + 600, ASERT_START_TARGET_NUM, anchor_time).0, + next_difficulty( + &p, + height, + on_time + 600, + ASERT_START_TARGET_NUM, + anchor_time, + ) + .0, ); assert!(fast.big < base.big, "faster blocks must tighten the target"); assert!(slow.big > base.big, "slower blocks must ease the target"); diff --git a/hbit-pool/src/lib.rs b/hbit-pool/src/lib.rs index b55f0f24..09f699b9 100644 --- a/hbit-pool/src/lib.rs +++ b/hbit-pool/src/lib.rs @@ -68,10 +68,98 @@ pub fn find_value<'a>(v: &'a Value, key: &str) -> Option<&'a Value> { } } -/// The recipient's "hacash" balance string (e.g. "1:248"), or "" if none. -pub fn balance(client: &reqwest::blocking::Client, base: &str, addr: &str) -> String { - let j = get_json(client, &format!("{base}/query/balance?address={addr}")); - find_str(&j, "hacash").unwrap_or_default() +/// What the node said when it was asked for an address's balance. +/// +/// Three states, not two. This used to be a bare `String` that folded every +/// failure into "", which `balance_units` then valued as a confident zero. A +/// node that was down, restarting, or answering with its own error object read +/// exactly like a wallet holding nothing: settlement published "matured = 0" and +/// flagged it CURRENT, so every miner polling `/earnings` was told it was owed +/// nothing for as long as the outage lasted, and the template loop re-poisoned +/// the same figure every 30 seconds. Miners whose shares rolled out of the PPLNS +/// window during the outage were never paid for that work. +/// +/// A wallet holding nothing is NOT this state: the node always emits the +/// `hacash` field and renders an empty wallet as "0:0", which is +/// [`Reported`](BalanceAnswer::Reported) and values as a real zero. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum BalanceAnswer { + /// The node gave a balance string for the address, e.g. "1:248" or "0:0". + Reported(String), + /// The node answered, but not with a balance: its own `{"ret":1,...}` error + /// object, or a body carrying no `hacash` field at all. + Refused(String), + /// Nothing usable came back: connection refused, a timeout, a proxy's error + /// page. The wallet is UNKNOWN, not empty. + NoAnswer(String), +} + +impl BalanceAnswer { + /// The balance in whole units of 0.1 HAC, or `None` when the pool must not + /// act on this answer at all. Anything but a reported balance is `None`: + /// callers already treat `None` as "skip this cycle, keep the last good + /// figure and mark it stale", which is exactly right for a silent node. + pub fn units(&self) -> Option { + match self { + BalanceAnswer::Reported(s) => balance_units(s), + BalanceAnswer::Refused(_) | BalanceAnswer::NoAnswer(_) => None, + } + } +} + +impl std::fmt::Display for BalanceAnswer { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + BalanceAnswer::Reported(s) => write!(f, "{s}"), + BalanceAnswer::Refused(s) => write!(f, "the node refused to report a balance: {s}"), + BalanceAnswer::NoAnswer(s) => write!(f, "no answer from the node: {s}"), + } + } +} + +/// How much of an unusable answer goes into a log line. A non-JSON body can be a +/// whole HTML error page, and a log that scrolls the real message away is a log +/// nobody can read during the outage it is describing. +const ANSWER_EXCERPT_CHARS: usize = 200; + +/// Truncate on a CHARACTER boundary: this text comes off the wire and slicing it +/// by bytes would panic the caller on any multi-byte error message. +fn excerpt(s: &str) -> String { + s.chars().take(ANSWER_EXCERPT_CHARS).collect() +} + +/// Classify a `/query/balance` response. Fails SAFE: only an answer that really +/// carries a balance becomes [`BalanceAnswer::Reported`], and everything else is +/// a state the caller must refuse to pay on. +pub fn balance_answer(j: &Value) -> BalanceAnswer { + // get_json encodes a transport failure as {"http_error": "..."} and a + // non-JSON body as a bare string. Neither is the node speaking. + if let Some(e) = j.get("http_error").and_then(|v| v.as_str()) { + return BalanceAnswer::NoAnswer(excerpt(e)); + } + if !j.is_object() { + return BalanceAnswer::NoAnswer(excerpt(&j.to_string())); + } + // The node answered, and its answer is "no": a bad address, too many + // addresses, an unreadable state. There is no balance in it to pay on. + if find_u64(j, "ret").is_some_and(|r| r != 0) { + return BalanceAnswer::Refused(excerpt(&j.to_string())); + } + match find_str(j, "hacash") { + Some(s) if !s.trim().is_empty() => BalanceAnswer::Reported(s), + // ret=0 with no `hacash` is a shape this pool does not recognise. The + // node always emits the field, so its absence means we are not talking + // to one - never that the wallet is empty. + _ => BalanceAnswer::Refused(excerpt(&j.to_string())), + } +} + +/// The address's "hacash" balance as the node reported it, or why it did not. +pub fn balance(client: &reqwest::blocking::Client, base: &str, addr: &str) -> BalanceAnswer { + balance_answer(&get_json( + client, + &format!("{base}/query/balance?address={addr}"), + )) } /// The largest balance the pool will act on, in units of 0.1 HAC. Hacash's whole @@ -92,12 +180,15 @@ pub const MAX_PLAUSIBLE_UNITS: u64 = 1_000_000_000_000; /// larger than any real wallet: the caller must SKIP settlement rather than pay /// out on it. Saturating to u64::MAX here (as this used to) means "infinite /// money" to `distributable_units` and `split_payout`, which then plan a payout -/// of the whole u64 range off one malformed response. An EMPTY string is not an -/// error: the node simply omits the field for an address holding nothing. +/// of the whole u64 range off one malformed response. +/// +/// An EMPTY string is one of those refusals, and it is the important one. The +/// node always emits the `hacash` field and renders a wallet holding nothing as +/// "0:0", so "" is never something it reported: it is what the old reader +/// produced when there was no answer at all. Valuing it as `Some(0)` told the +/// settlement that a wallet it could not see was empty, and told every miner +/// polling `/earnings` that it was owed nothing for the length of the outage. pub fn balance_units(bal: &str) -> Option { - if bal.trim().is_empty() { - return Some(0); - } let (m, u) = bal.split_once(':')?; let (Ok(m), Ok(u)) = (m.trim().parse::(), u.trim().parse::()) else { return None; @@ -118,13 +209,204 @@ pub fn balance_units(bal: &str) -> Option { (units <= MAX_PLAUSIBLE_UNITS).then_some(units) } -/// The coinbase subsidy of the block at `height`, in units of 0.1 HAC. The pool -/// mines coinbase-only blocks, so this is the entire income a found block brings -/// into the wallet (`block_reward` is a whole number of HAC = unit 248). +/// The coinbase subsidy of the block at `height`, in units of 0.1 HAC +/// (`block_reward` is a whole number of HAC = unit 248). +/// +/// This is NOT the whole income a found block brings in. The pool packs the +/// node's transactions, and the chain credits the sum of their fees to the +/// coinbase address as well - the same wallet the pool settles from. See +/// [`block_fees`] for that half of it. pub fn block_reward_units(height: u64) -> u64 { mint::genesis::block_reward_number(height) as u64 * 10 } +/// Fine steps in one payout unit of 0.1 HAC; one step is 10^-9 HAC. +/// +/// A transaction fee is routinely a thousandth of a payout unit, so a block's +/// fees are summed on this finer scale and rounded up to whole units only once, +/// at the end. Rounding each fee up on its own would hold back a whole unit per +/// transaction and freeze real money for a whole maturity window. +const FEE_FINE_PER_UNIT: u128 = 100_000_000; + +/// The largest fee total this pool will believe, on the fine scale. Past the +/// whole coin supply the answer is corrupt or hostile, not a rich block, and +/// turning it into a hold-back would stop every payout the pool ever makes. +const MAX_FEE_FINE: u128 = MAX_PLAUSIBLE_UNITS as u128 * FEE_FINE_PER_UNIT; + +/// A node "mantissa:unit" amount on the fine scale, ROUNDED UP. +/// +/// Rounds UP because this number becomes money the pool refuses to pay out yet. +/// Rounding a fee down to nothing is exactly how the fee ends up distributed at +/// zero confirmations, which is the failure this exists to stop. +/// +/// `None` is "this is not an amount I can value" - a negative mantissa, a +/// missing separator, an exponent no wallet could hold - and the caller must +/// then refuse to settle rather than read it as a zero fee. +pub fn fin_fine_ceil(amount: &str) -> Option { + let (m, u) = amount.split_once(':')?; + let (Ok(m), Ok(u)) = (m.trim().parse::(), u.trim().parse::()) else { + return None; + }; + if !(0..=255).contains(&u) { + return None; // the chain's unit is a u8; anything else is not its answer + } + if m == 0 { + return Some(0); // a real zero, at any unit + } + // value = m * 10^(u-248) HAC, and one fine step is 10^-9 HAC. + let exp = u - 239; + let fine = if exp >= 0 { + let scale = u32::try_from(exp) + .ok() + .and_then(|e| 10u128.checked_pow(e))?; + m.checked_mul(scale)? + } else { + match u32::try_from(-exp).ok().and_then(|e| 10u128.checked_pow(e)) { + Some(d) => m.div_ceil(d), + // Finer than a fine step by more orders of magnitude than a u128 can + // express. It is still money, so it still counts as one step. + None => 1, + } + }; + (fine <= MAX_FEE_FINE).then_some(fine) +} + +/// Fine steps as whole payout units of 0.1 HAC, rounded UP. +/// +/// The wallet balance the pool settles against is itself floored to whole units, +/// and a fee that straddles a unit boundary can push that floor up by one. The +/// ceiling is what makes the hold-back cover that case instead of leaving one +/// unit payable out of income a reorg can still revoke. +pub fn fine_to_units_ceil(fine: u128) -> u64 { + fine.div_ceil(FEE_FINE_PER_UNIT) + .min(MAX_PLAUSIBLE_UNITS as u128) as u64 +} + +/// What the node says about the transaction fees one of OUR blocks credited. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum BlockFees { + /// The chain holds our block at that height, and it credited this much fee + /// income to the pool wallet, in units of 0.1 HAC rounded up. + Counted(u64), + /// The node answered, and the chain does NOT hold our block at that height. + /// It credited nothing there - no subsidy and no fee - so there is no fee + /// income to hold back. + NotOnChain, + /// No usable answer. This is NOT a zero fee: the wallet may be holding fee + /// income the pool cannot value, so the caller must refuse to settle. + Unknown(String), +} + +/// Which transactions the chain says are in our block, or why it cannot say. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum BlockTxs { + /// The chain holds OUR block at that height, and these are the hashes of the + /// transactions in it. The coinbase is not among them: it pays no fee. + Ours(Vec), + /// The node answered, and the chain does not hold our block there. + NotOnChain, + /// No usable answer. + Unknown(String), +} + +/// Read a `/query/block/intro?tx_hash_list=true` answer for one of OUR blocks. +/// +/// Split out from [`block_fees`] so the decision - price it, ignore it, or stop +/// settling - is testable without a node. Fails SAFE: only an answer that really +/// carries our block's transaction list is [`BlockTxs::Ours`]. +pub fn block_txs_of(j: &Value, our_hash_hex: &str) -> BlockTxs { + // get_json encodes a transport failure as {"http_error": "..."} and a + // non-JSON body as a bare string. Neither is the node speaking. + if !j.is_object() || j.get("http_error").is_some() { + return BlockTxs::Unknown(excerpt(&j.to_string())); + } + let Some(ret) = find_u64(j, "ret") else { + return BlockTxs::Unknown(excerpt(&j.to_string())); + }; + if ret != 0 { + // The node is up and has no block at that height: ours was refused, or + // has not been inserted yet. Either way it has credited nothing. + return BlockTxs::NotOnChain; + } + let Some(hash) = find_str(j, "hash") else { + return BlockTxs::Unknown(excerpt(&j.to_string())); + }; + if !hash.eq_ignore_ascii_case(our_hash_hex) { + return BlockTxs::NotOnChain; // another block won that height + } + // ret=0 for our block but no list at all is an answer this pool does not + // recognise - never "the block had no transactions". A node that quietly + // dropped the field would otherwise read as a zero fee on every block. + let Some(list) = find_value(j, "tx_hash_list").and_then(|v| v.as_array()) else { + return BlockTxs::Unknown(excerpt(&j.to_string())); + }; + match list + .iter() + .map(|h| h.as_str().map(|s| s.to_string())) + .collect::>>() + { + Some(hs) => BlockTxs::Ours(hs), + None => BlockTxs::Unknown(excerpt(&j.to_string())), + } +} + +/// The `fee_got` in an answer to `/query/transaction`, on the fine scale. +/// +/// `fee_got` and not `fee`: what the chain adds to the coinbase address is the +/// fee the transaction actually PAID for its place in the block, which a +/// fee-raise or a gas refund can make smaller than the fee it declared. +pub fn fee_got_fine(j: &Value) -> Option { + if !j.is_object() || j.get("http_error").is_some() { + return None; + } + if find_u64(j, "ret") != Some(0) { + return None; + } + fin_fine_ceil(&find_str(j, "fee_got")?) +} + +/// What the chain credited the pool wallet in TRANSACTION FEES for our block at +/// `height`, in units of 0.1 HAC rounded up. +/// +/// The figure cannot be taken from the block the pool built: its transaction +/// bodies are raw bytes the pool deliberately has no codec for, and +/// `/submit/block` answers only `{"ok":true}`. So it is read back off the node +/// once the block exists, one `/query/transaction` per packed transaction. +/// +/// Answers `Unknown` on anything short of a definitive reply, because the caller +/// turns that into "settle nothing this cycle". The alternative - treating an +/// unreachable node as a zero fee - pays that fee income out at zero +/// confirmations, and an orphan then leaves the operator funding a payout out of +/// a block the chain no longer has. +pub fn block_fees( + client: &reqwest::blocking::Client, + node: &str, + height: u64, + our_hash_hex: &str, +) -> BlockFees { + let j = get_json( + client, + &format!("{node}/query/block/intro?height={height}&tx_hash_list=true"), + ); + let hashes = match block_txs_of(&j, our_hash_hex) { + BlockTxs::Ours(hs) => hs, + BlockTxs::NotOnChain => return BlockFees::NotOnChain, + BlockTxs::Unknown(why) => return BlockFees::Unknown(why), + }; + let mut fine: u128 = 0; + for h in &hashes { + let t = get_json(client, &format!("{node}/query/transaction?hash={h}")); + let Some(f) = fee_got_fine(&t) else { + return BlockFees::Unknown(format!("transaction {h}: {}", excerpt(&t.to_string()))); + }; + fine = fine.saturating_add(f); + if fine > MAX_FEE_FINE { + return BlockFees::Unknown(format!("fees at height {height} exceed any real block")); + } + } + BlockFees::Counted(fine_to_units_ceil(fine)) +} + /// How deep a payout transaction must be buried before the pool stops tracking /// it. The node keeps up to `unstable_block` (4) blocks reorg-able, so a payout /// that is only 1-3 confirmations deep can still come back to the mempool; @@ -227,11 +509,7 @@ pub fn admission_of(j: &Value) -> Admission { /// Ask the node whether it really holds `txhash`, retrying while it has not made /// up its mind. The insert runs on a background task, so an immediate "not /// found" only becomes a verdict once the node has had time to do it. -pub fn verify_admitted( - client: &reqwest::blocking::Client, - node: &str, - txhash: &str, -) -> Admission { +pub fn verify_admitted(client: &reqwest::blocking::Client, node: &str, txhash: &str) -> Admission { let mut last = Admission::Unresolved; for attempt in 0..ADMIT_POLL_TRIES { let j = get_json(client, &format!("{node}/query/transaction?hash={txhash}")); @@ -250,10 +528,14 @@ pub fn verify_admitted( /// reorg could still take back, MINUS the fee reserve. `None` means "nothing /// spendable, do not settle this cycle". /// -/// `immature_units` is the coinbase of blocks the pool found that are not yet -/// buried deep enough to be final. Distributing that and then losing the block -/// to a reorg is an unrecoverable operator loss: the income disappears from the -/// canonical chain while the payout transaction that spent it stays valid. +/// `immature_units` is the WHOLE income of blocks the pool found that are not +/// yet buried deep enough to be final. Whole, because the chain credits the +/// coinbase address both the subsidy and the sum of the fees of every +/// transaction in the block, and this pool packs the node's transactions: a +/// hold-back of the subsidy alone leaves the fees payable here. Distributing +/// either and then losing the block to a reorg is an unrecoverable operator +/// loss: the income disappears from the canonical chain while the payout +/// transaction that spent it stays valid. /// /// All arithmetic saturates, so an out-of-range reserve can never wrap the /// guard open the way `reserve + 1` used to. @@ -271,7 +553,8 @@ pub fn distributable_units( /// Atomic file write (temp + optional fsync + rename) so a crash or a full disk /// mid-write can never leave a truncated or corrupt file behind. `durable` -/// fsyncs before the rename. +/// fsyncs the bytes before the rename and the directory after it, and FAILS if +/// either fsync fails. pub fn atomic_write(path: &str, body: &[u8], durable: bool) -> std::io::Result<()> { use std::io::Write; let tmp = format!("{path}.tmp.{}", std::process::id()); @@ -279,10 +562,53 @@ pub fn atomic_write(path: &str, body: &[u8], durable: bool) -> std::io::Result<( let mut f = std::fs::File::create(&tmp)?; f.write_all(body)?; if durable { - let _ = f.sync_all(); + // This used to be `let _ = f.sync_all()`. `durable` is the promise + // the settlement path broadcasts a payout on, and a discarded error + // here answers "recorded" for bytes that only ever reached the page + // cache - a full disk, an I/O error, or a network mount that went + // away all report themselves at flush time and nowhere else. Lose + // power in the seconds that follow and the pool restarts with no + // memory of the transaction it signed, so the next cycle signs a + // SECOND payout for the same PPLNS window and the operator funds + // the difference out of their own wallet. + f.sync_all()?; } } - std::fs::rename(&tmp, path) + std::fs::rename(&tmp, path)?; + if durable { + // The bytes can be on the platter while the directory entry pointing at + // them is not: the rename is its own metadata change and is lost on its + // own. That leaves the PREVIOUS state file in place - the one without + // the payout hash or without the immature hold-back - which costs + // exactly what the paragraph above costs. + fsync_parent_dir(path)?; + } + Ok(()) +} + +/// fsync the directory that holds `path`, so a rename into it survives a power +/// cut rather than only the bytes it points at. +/// +/// A bare filename (a relative `wallet_file` in the config) has no directory +/// component and must fall back to the working directory: treating that as a +/// failure would make every durable write fail and stop the pool paying anyone. +#[cfg(unix)] +fn fsync_parent_dir(path: &str) -> std::io::Result<()> { + let dir = match std::path::Path::new(path).parent() { + Some(d) if !d.as_os_str().is_empty() => d.to_path_buf(), + _ => std::path::PathBuf::from("."), + }; + std::fs::File::open(dir)?.sync_all() +} + +/// Windows has no directory fsync: a directory handle cannot be opened for +/// `FlushFileBuffers`, so there is nothing to call and the durability of the +/// rename is NTFS's own metadata journal. Reporting that as a failure would +/// refuse every settlement the pool ever tried, which is worse than the gap it +/// would be reporting. The data fsync above still holds on this platform. +#[cfg(not(unix))] +fn fsync_parent_dir(_path: &str) -> std::io::Result<()> { + Ok(()) } /// The pool's accounting file for `wallet_file`. The auto-settle server and the @@ -317,13 +643,100 @@ pub fn load_pending_payout_txs(state_file: &str) -> Vec { /// Rolling PPLNS window: the last N accepted shares decide the payout split. pub const PPLNS_WINDOW: usize = 4096; -/// Rebuild the PPLNS share counts from the pool's own accounting file. +/// How long a share may go on earning credit, and how long the credit an evicted +/// share already earned survives, expressed in settlement intervals. +/// +/// One interval is the unit that matters because that is the longest a miner can +/// usefully sit on shares: the pool pins a template for a block interval, and it +/// publishes the settlement interval in `/terms`. Anything shorter would let a +/// hoarder time its dump; much longer and the split stops tracking who is mining +/// now. +const PPLNS_HORIZON_INTERVALS: u64 = 1; + +/// The documented default settlement interval. `hbit-pool-server`'s `usage()` +/// quotes it and `hbit-pool-payout` falls back to it when it has to read an +/// accounting file that does not record the interval the server was running. +pub const DEFAULT_SETTLE_SECS: u64 = 300; + +/// The credit horizon in milliseconds for a pool settling every `settle_secs`. +pub fn pplns_horizon_ms(settle_secs: u64) -> u64 { + settle_secs + .saturating_mul(PPLNS_HORIZON_INTERVALS) + .saturating_mul(1_000) + .max(1_000) +} + +/// Read the persisted share window, accepting BOTH the timestamped form this +/// pool writes now and the bare list of worker ids older builds wrote. +/// +/// An older file carries no arrival times at all. Every share in it is given the +/// SAME stamp, `fallback_ms`, because that is the only assumption that treats +/// every miner alike: credit is proportional, so one common start time preserves +/// the split exactly, while inventing different ages would silently move money +/// between miners on a restart. +pub fn parse_share_order(j: &Value, fallback_ms: u64) -> Vec<(String, u64)> { + j.get("order") + .and_then(|v| v.as_array()) + .map(|a| { + a.iter() + .filter_map(|x| { + if let Some(s) = x.as_str() { + return Some((s.to_string(), fallback_ms)); + } + let row = x.as_array()?; + let w = row.first()?.as_str()?.to_string(); + let at = row.get(1)?.as_u64().unwrap_or(fallback_ms); + Some((w, at)) + }) + .collect() + }) + .unwrap_or_default() +} + +/// Read the banked credit of shares that have already left the window. Absent in +/// a file written before shares were timed, which reads as "none banked" rather +/// than as a corrupt file. +pub fn parse_banked_credit(j: &Value) -> Vec<(u64, Vec<(String, u64)>)> { + j.get("banked") + .and_then(|v| v.as_array()) + .map(|a| { + a.iter() + .filter_map(|x| { + let at = x.get("at").and_then(|v| v.as_u64())?; + let rows = x + .get("rows") + .and_then(|v| v.as_array()) + .map(|r| { + r.iter() + .filter_map(|e| { + let row = e.as_array()?; + Some(( + row.first()?.as_str()?.to_string(), + row.get(1)?.as_u64()?, + )) + }) + .collect() + }) + .unwrap_or_default(); + Some((at, rows)) + }) + .collect() + }) + .unwrap_or_default() +} + +/// Rebuild the PPLNS payout credit from the pool's own accounting file. /// /// The manual payout tool needs this because the server holds the wallet's /// settlement lock for its whole run: if the tool is able to settle at all then /// the server is stopped, so its `/stats` endpoint cannot answer and the file it /// left behind is the authority on who is owed what. -pub fn load_pplns_counts(state_file: &str) -> Vec<(String, u64)> { +/// +/// It returns CREDIT, not share counts, for the same reason the server settles on +/// credit: a headcount taken at the instant of a payout is a number one miner can +/// own outright by dumping a window's worth of withheld shares, and this tool +/// signs the same money. +pub fn load_pplns_credit(state_file: &str) -> Vec<(String, u64)> { let Some(j) = read_state_json(state_file) else { return Vec::new(); }; @@ -331,36 +744,119 @@ pub fn load_pplns_counts(state_file: &str) -> Vec<(String, u64)> { .get("window") .and_then(|v| v.as_u64()) .unwrap_or(PPLNS_WINDOW as u64) as usize; - let order: Vec = j - .get("order") - .and_then(|v| v.as_array()) - .map(|a| { - a.iter() - .filter_map(|x| x.as_str().map(|s| s.to_string())) - .collect() - }) - .unwrap_or_default(); - if order.is_empty() { + // The horizon the SERVER was running, so the manual tool splits money the + // same way the automatic settlement would have. + let horizon = j + .get("credit_horizon_ms") + .and_then(|v| v.as_u64()) + .filter(|h| *h > 0) + .unwrap_or_else(|| pplns_horizon_ms(DEFAULT_SETTLE_SECS)); + let at = credit_anchor_ms(&j, pool_core::now_ms()); + // A file with no arrival times is read as if every share landed one horizon + // before that instant: they are all treated alike, so the split is the one + // the old build would have made, and no share reads as newer than it is. + let order = parse_share_order(&j, at.saturating_sub(horizon)); + let banked = parse_banked_credit(&j); + if order.is_empty() && banked.is_empty() { return Vec::new(); } - pool_core::Pplns::restore(window, order).counts() + pool_core::Pplns::restore(window, horizon, order, banked).credit(at) +} + +/// The instant a stored share window is worth valuing at: the last moment the +/// pool that wrote the file was actually accounting. +/// +/// NOT the wall clock. Credit is residence in the window, and nothing enters or +/// leaves that window while the server is stopped - which it always is when this +/// tool runs, because the tool can only get the settlement lock if it is. Valuing +/// at the wall clock let the whole window age together while nothing happened, +/// and past one horizon that undoes the fix this file exists to carry: every +/// share caps at the horizon, so the split flattens back to a HEADCOUNT, and the +/// banked credit of the miners a dump evicted expires entirely. A miner that +/// withheld a window's worth and dumped it before the server stopped would then +/// take the lot - the exact attack, back again, on the settler an operator +/// reaches for when the server is down. Five minutes between stopping the pool +/// and running the payout is all it took. +/// +/// Anchoring here also makes the payout DETERMINISTIC: the same file settles the +/// same way whether the operator runs the tool immediately or an hour later. +/// +/// The tool's own clock is deliberately not consulted when the file has times of +/// its own. Every credit figure is a DIFFERENCE against this instant, so a +/// consistent anchor taken from the file makes the split independent of what the +/// machine running the settlement thinks the time is. +/// +/// Falls back to `now_ms` when the file carries no times at all (a window written +/// before shares were stamped), because there is nothing to anchor to and the +/// fallback stamp one horizon back then weighs every share alike, as it must. +pub fn credit_anchor_ms(j: &Value, now_ms: u64) -> u64 { + let newest_share = j + .get("order") + .and_then(|v| v.as_array()) + .and_then(|a| a.iter().filter_map(|x| x.as_array()?.get(1)?.as_u64()).max()); + let newest_bank = j + .get("banked") + .and_then(|v| v.as_array()) + .and_then(|a| a.iter().filter_map(|x| x.get("at")?.as_u64()).max()); + newest_share + .into_iter() + .chain(newest_bank) + .max() + .unwrap_or(now_ms) } -/// Total held-back (not yet final) block income recorded by the pool server, in -/// units of 0.1 HAC. The manual payout tool reads it so it applies the SAME -/// maturity gate as the automatic settlement instead of paying at the tip. -pub fn load_immature_units(state_file: &str) -> u64 { +/// One block of not-yet-final income the pool server is holding back. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct ImmatureBlock { + pub height: u64, + /// OUR block's hash at that height, hex. The income is only real while the + /// chain still holds this hash there. + pub hash: String, + /// What it put into the pool wallet so far, in units of 0.1 HAC. + pub units: u64, + /// Are that block's TRANSACTION FEES already inside `units`? + /// + /// False means `units` is the coinbase subsidy alone, and the block's fees - + /// which the chain credits to the very same wallet - are still sitting in + /// the balance unaccounted for. Paying against that balance hands those fees + /// out at zero confirmations. + pub fees_counted: bool, +} + +/// Every block of not-yet-final income the pool server recorded. The manual +/// payout tool reads it so it applies the SAME maturity gate as the automatic +/// settlement instead of paying at the tip. +/// +/// `fees_counted` defaults to FALSE when the field is absent, because a file +/// written by a build that held back only the subsidy really does carry the +/// subsidy alone. Defaulting the other way would silently distribute those +/// blocks' fees on the first settlement after an upgrade. +pub fn load_immature_blocks(state_file: &str) -> Vec { let Some(j) = read_state_json(state_file) else { - return 0; + return Vec::new(); }; j.get("immature") .and_then(|v| v.as_array()) .map(|a| { a.iter() - .filter_map(|x| x.get("units").and_then(|v| v.as_u64())) - .sum() + .filter_map(|x| { + Some(ImmatureBlock { + height: x.get("height").and_then(|v| v.as_u64())?, + hash: x + .get("hash") + .and_then(|v| v.as_str()) + .unwrap_or_default() + .to_string(), + units: x.get("units").and_then(|v| v.as_u64())?, + fees_counted: x + .get("fees_counted") + .and_then(|v| v.as_bool()) + .unwrap_or(false), + }) + }) + .collect() }) - .unwrap_or(0) + .unwrap_or_default() } /// Replace `settle_pending_txs` in the pool state file, preserving every other @@ -436,6 +932,17 @@ pub struct PayoutRecord { /// submitted but the node's verdict could not be read: it may well be in /// flight, so it stays tracked, but nothing about it is claimed. pub node_holds: bool, + /// The exact signed bytes that were submitted, hex-encoded. + /// + /// Kept because a node that once HELD this transaction also relayed it, and + /// the mempool is memory-only: a routine node restart empties it, and the + /// pool then asks about a hash the node no longer knows. Re-splitting and + /// re-signing that window makes a DIFFERENT transaction (fresh timestamp, + /// so a different hash); replay protection on this chain is by hash alone, + /// so both can be mined and the operator pays the same miners twice out of + /// its own wallet. With the bytes here the pool re-broadcasts the identical + /// transaction, which can only ever be included once. + pub body_hex: String, /// (worker address, units of 0.1 HAC) exactly as the transaction pays them. pub rows: Vec<(String, u64)>, } @@ -443,7 +950,10 @@ pub struct PayoutRecord { impl PayoutRecord { /// Total this transaction pays, in units of 0.1 HAC. pub fn units(&self) -> u64 { - self.rows.iter().map(|(_, u)| *u).fold(0u64, |a, b| a.saturating_add(b)) + self.rows + .iter() + .map(|(_, u)| *u) + .fold(0u64, |a, b| a.saturating_add(b)) } /// What this transaction pays ONE worker. @@ -460,6 +970,7 @@ impl PayoutRecord { "hash": self.hash, "at": self.at, "node_holds": self.node_holds, + "body_hex": self.body_hex, "rows": self.rows.iter() .map(|(w, u)| serde_json::json!([w, u])) .collect::>(), @@ -490,6 +1001,15 @@ impl PayoutRecord { .get("node_holds") .and_then(|x| x.as_bool()) .unwrap_or(false), + // A record written before the bytes were kept reads as "no bytes", + // which `gone_action` treats as un-rebroadcastable rather than as + // safe to re-issue. Missing evidence must never become permission + // to sign the same window again. + body_hex: v + .get("body_hex") + .and_then(|x| x.as_str()) + .unwrap_or_default() + .to_string(), rows, }) } @@ -632,6 +1152,232 @@ pub fn drop_payout(records: &mut Vec, hash: &str) -> Option) -> GoneAction { + match rec { + Some(r) if !r.body_hex.is_empty() => GoneAction::Rebroadcast, + Some(r) if r.node_holds => GoneAction::Stuck, + _ => GoneAction::Forget, + } +} + +/// What `/submit/transaction` said about a payout we just posted. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum SubmitVerdict { + /// `ret=0`: the API took the bytes. Still not proof the node holds it - only + /// [`verify_admitted`] is that. + Accepted, + /// The node itself answered with a non-zero `ret`: it refused the + /// transaction during synchronous validation, so it never inserted it into + /// the mempool and never relayed it. + Rejected, + /// No verdict at all: [`post_hex`] returns the plain string + /// `"http_error: ..."` when the request times out or the connection drops, + /// and a timeout happens AFTER the node may have already taken and relayed + /// the transaction. Reading that as a rejection and forgetting the hash is + /// how a payout gets issued a second time. + Unresolved, +} + +/// Classify a `/submit/transaction` response body. +/// +/// Fails SAFE: anything that is not the node speaking a `ret` we can read is +/// `Unresolved`, which keeps the payout tracked for the next cycle's poll. +pub fn submit_verdict(resp: &str) -> SubmitVerdict { + let Ok(j) = serde_json::from_str::(resp) else { + return SubmitVerdict::Unresolved; // "http_error: ..." lands here + }; + if j.get("http_error").is_some() { + return SubmitVerdict::Unresolved; + } + match find_u64(&j, "ret") { + Some(0) => SubmitVerdict::Accepted, + Some(_) => SubmitVerdict::Rejected, + None => SubmitVerdict::Unresolved, + } +} + +/* --------------------------------------------------------------------------- + * The owed ledger. + * + * When a settlement chunk definitively does not happen, the money it carried + * does not simply return to the pot: it is owed to the exact miners that chunk + * named. Dropping the rows and letting the next cycle re-split the whole balance + * over the live PPLNS window hands that money to whoever is mining now - + * including the miners whose chunks DID go through, who are then paid twice for + * the same window while the miners in the failed chunk are paid once, or never. + * + * These rows are therefore persisted and paid FIRST, before a single unit of + * fresh income is split. + * ------------------------------------------------------------------------- */ + +/// Add the rows of a payout that definitively did not happen to the owed ledger. +pub fn owe_rows(owed: &mut Vec<(String, u64)>, rows: &[(String, u64)]) { + for (w, u) in rows { + if *u == 0 { + continue; + } + match owed.iter_mut().find(|(x, _)| x == w) { + Some(e) => e.1 = e.1.saturating_add(*u), + None => owed.push((w.clone(), *u)), + } + } +} + +/// Take off the owed ledger what a payout the pool has now RECORDED carries. +/// +/// Saturating, and only ever downward: a row paying more than is owed (an owed +/// row and a fresh share to the same miner, merged into one action) clears the +/// debt and no more. +pub fn deduct_owed(owed: &mut Vec<(String, u64)>, rows: &[(String, u64)]) { + for (w, u) in rows { + if let Some(e) = owed.iter_mut().find(|(x, _)| x == w) { + e.1 = e.1.saturating_sub(*u); + } + } + owed.retain(|(_, u)| *u > 0); +} + +/// The rows this cycle must pay BEFORE it splits any fresh income, and what is +/// left to split after them. +/// +/// Owed rows are taken in order and partially where the balance runs out, so one +/// large debt cannot starve while smaller ones keep being paid around it. What is +/// not taken stays on the ledger for the next cycle. +pub fn take_owed(owed: &[(String, u64)], distributable: u64) -> (Vec<(String, u64)>, u64) { + let mut left = distributable; + let mut rows: Vec<(String, u64)> = Vec::new(); + for (w, u) in owed { + if left == 0 { + break; + } + let pay = (*u).min(left); + if pay == 0 { + continue; + } + left -= pay; + rows.push((w.clone(), pay)); + } + (rows, left) +} + +/// Fold rows paying the same address into one action, keeping first-seen order. +/// +/// An owed row and a fresh share for the same miner would otherwise be two +/// actions in one transaction, and each action costs against the node's +/// TX_ACTIONS_MAX limit that `PAYOUT_CHUNK` is sized against. +pub fn merge_payout_rows(rows: &mut Vec<(String, u64)>) { + let mut at: HashMap = HashMap::with_capacity(rows.len()); + let mut out: Vec<(String, u64)> = Vec::with_capacity(rows.len()); + for (w, u) in rows.drain(..) { + match at.get(&w) { + Some(i) => out[*i].1 = out[*i].1.saturating_add(u), + None => { + at.insert(w.clone(), out.len()); + out.push((w, u)); + } + } + } + *rows = out; +} + +/// The owed ledger as it is stored in the pool state file. +pub fn owed_to_json(owed: &[(String, u64)]) -> Value { + Value::Array( + owed.iter() + .map(|(w, u)| serde_json::json!([w, u])) + .collect(), + ) +} + +/// Read the owed ledger out of an already-parsed state document. A row the file +/// cannot describe is dropped rather than guessed at: an unreadable amount must +/// never become a payment. +pub fn parse_owed(j: &Value) -> Vec<(String, u64)> { + j.get("owed") + .and_then(|v| v.as_array()) + .map(|a| { + a.iter() + .filter_map(|r| { + let x = r.as_array()?; + let w = x.first()?.as_str()?.to_string(); + let u = x.get(1)?.as_u64()?; + (!w.is_empty() && u > 0).then_some((w, u)) + }) + .collect() + }) + .unwrap_or_default() +} + +/// The owed ledger. Losing it means the miners in a failed chunk are never paid +/// for that window, so it is persisted with everything else. +pub fn load_owed(state_file: &str) -> Vec<(String, u64)> { + let Some(j) = read_state_json(state_file) else { + return Vec::new(); + }; + parse_owed(&j) +} + /// The per-transaction rows of every payout this pool has in flight. pub fn load_payout_records(state_file: &str) -> Vec { let Some(j) = read_state_json(state_file) else { @@ -661,21 +1407,25 @@ pub fn parse_paid_ledger(j: &Value) -> PaidLedger { j.get("paid").map(PaidLedger::from_json).unwrap_or_default() } -/// Replace the WHOLE settlement ledger (pending hashes, per-transaction rows and -/// confirmed totals) in the pool state file, preserving every other field. +/// Replace the WHOLE settlement ledger (pending hashes, per-transaction rows, +/// what is still owed, and confirmed totals) in the pool state file, preserving +/// every other field. /// -/// One write, because the three move together: a payout leaves the in-flight -/// rows at the same instant it enters the paid totals, and a crash between the -/// two would either lose a payment or count it twice. +/// One write, because the four move together: a payout leaves the in-flight rows +/// at the same instant it enters the paid totals, and a chunk that failed leaves +/// them at the same instant its rows become owed. A crash between any two would +/// either lose a payment or count it twice. pub fn save_settlement_ledger( state_file: &str, hashes: &[String], records: &[PayoutRecord], + owed: &[(String, u64)], paid: &PaidLedger, ) -> std::io::Result<()> { let mut j = read_state_json(state_file).unwrap_or_else(|| serde_json::json!({})); j["settle_pending_txs"] = serde_json::json!(hashes); j["payouts_inflight"] = Value::Array(records.iter().map(|r| r.to_json()).collect()); + j["owed"] = owed_to_json(owed); j["paid"] = paid.to_json(); atomic_write(state_file, j.to_string().as_bytes(), true) } @@ -903,9 +1653,14 @@ fn encrypt_key_hex(key_hex: &str, pass: &str) -> Result { } fn envelope_u32(j: &Value, key: &str, default: u32, max: u32) -> Result { - let v = j.get(key).and_then(|v| v.as_u64()).unwrap_or(default as u64); + let v = j + .get(key) + .and_then(|v| v.as_u64()) + .unwrap_or(default as u64); if v == 0 || v > max as u64 { - return Err(format!("its `{key}` is outside the range this build accepts")); + return Err(format!( + "its `{key}` is outside the range this build accepts" + )); } Ok(v as u32) } @@ -970,14 +1725,23 @@ fn decrypt_key_hex(body: &str, pass: &str) -> Result, Envelope let key = wallet_derive_key( pass, &salt, - envelope_u32(&j, "kdf_m_cost_kb", WALLET_KDF_M_COST_KB, WALLET_KDF_MAX_M_COST_KB) - .map_err(EnvelopeError::Shape)?, + envelope_u32( + &j, + "kdf_m_cost_kb", + WALLET_KDF_M_COST_KB, + WALLET_KDF_MAX_M_COST_KB, + ) + .map_err(EnvelopeError::Shape)?, envelope_u32(&j, "kdf_t_cost", WALLET_KDF_T_COST, WALLET_KDF_MAX_T_COST) .map_err(EnvelopeError::Shape)?, envelope_u32(&j, "kdf_p_cost", WALLET_KDF_P_COST, WALLET_KDF_MAX_P_COST) .map_err(EnvelopeError::Shape)?, ) - .map_err(|e| shape(format!("its key-derivation settings cannot be used here ({e})")))?; + .map_err(|e| { + shape(format!( + "its key-derivation settings cannot be used here ({e})" + )) + })?; let cipher = Aes256Gcm::new_from_slice(&*key) .map_err(|e| shape(format!("its cipher key could not be set up ({e})")))?; // The one check that cannot say WHY it failed. @@ -1372,7 +2136,9 @@ fn migrate_key_file_to_encrypted(path: &str, key_hex: &str, pass: &str) -> Resul match decrypt_key_hex(&body, pass) { Ok(back) if back.trim().eq_ignore_ascii_case(key_hex) => {} _ => { - eprintln!("[wallet] WARNING: the encrypted form of {path} did not verify; leaving it as-is."); + eprintln!( + "[wallet] WARNING: the encrypted form of {path} did not verify; leaving it as-is." + ); return Ok(()); } } @@ -1587,7 +2353,9 @@ fn windows_sid_of(principal: &str) -> Option { #[cfg(windows)] fn windows_verify_owner_only(path: &str, name: &str, sid: &str) -> std::io::Result<()> { let unverified = |why: &str| { - eprintln!("[wallet] WARNING: could not verify the ACL of {path} ({why}); check it manually."); + eprintln!( + "[wallet] WARNING: could not verify the ACL of {path} ({why}); check it manually." + ); }; let Ok(out) = std::process::Command::new("icacls").arg(path).output() else { unverified("icacls did not run"); @@ -1701,6 +2469,61 @@ pub struct Template { pub txs: Arc, } +/// The header timestamp a pool has ALREADY handed out for a height, so a restart +/// can reproduce the same header bytes instead of inventing new ones. +/// +/// The stamp is `max(now, prev_ts + 1)` at the moment the template is first +/// fetched, and it lives in the 89-byte header every worker hashes. A pool pins +/// one template per height, so a restart part-way through a height would +/// otherwise serve a DIFFERENT header for the SAME height: measured on a rig, a +/// restart at height 350 served a stamp 68 seconds later than the one already in +/// flight. `/query/miner/notice` signals only a HEIGHT change, so nothing tells a +/// worker to reload - it goes on hashing the dead header until its current scan +/// pass ends, and every share it finds in the meantime is thrown away. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct StampPin { + pub height: u64, + pub prevhash: Hash, + pub timestamp: u64, +} + +/// The header timestamp to serve for `height` on `prevhash`. +/// +/// `pin` is honoured ONLY when it describes this exact block and would produce a +/// header the node still accepts. Both guards cost a whole block reward when they +/// are wrong: `chain::verify::block_verify` rejects a block whose timestamp is +/// `<= prev_blk_time` or `> curtimes()`, and a rejected block is reported +/// asynchronously, so the pool would mine a full round into nothing and only +/// notice because the tip never reached its height. +/// +/// The upper guard is `fresh`, not `now`: a pin may never move the stamp FORWARD +/// past what a fresh fetch would produce, so a stale or tampered state file +/// cannot talk the pool into mining a future-stamped block. Moving it BACKWARD to +/// a stamp this pool already served is exactly what a pool that never restarted +/// would be serving. +pub fn template_timestamp( + pin: Option<&StampPin>, + height: u64, + prevhash: &Hash, + prev_ts: u64, + now: u64, +) -> u64 { + let fresh = std::cmp::max(now, prev_ts.saturating_add(1)); + let Some(pin) = pin else { + return fresh; + }; + // A pin from another block says nothing about this one. Height alone is not + // enough: after a same-height reorg the parent differs, and the old stamp was + // computed against a parent whose timestamp this block no longer follows. + if pin.height != height || pin.prevhash != *prevhash { + return fresh; + } + if pin.timestamp <= prev_ts || pin.timestamp > fresh { + return fresh; + } + pin.timestamp +} + /// Read the chain tip and build a template for the next block, computing the /// next difficulty off-node with the same rule the node will validate against. /// @@ -1712,6 +2535,18 @@ pub fn fetch_template( base: &str, coinbase_addr: &str, params: &ChainParams, +) -> Option