diff --git a/.dockerignore b/.dockerignore new file mode 100644 index 00000000..9d01bcb2 --- /dev/null +++ b/.dockerignore @@ -0,0 +1,61 @@ +# Without this file `docker build .` sends the whole working tree to the daemon. +# Measured here: 113 GB of build context, which took nearly two hours and then +# killed the daemon. The image needs the sources and nothing else; every build +# artifact it needs is produced inside the builder stage. + +# Rust build output. This is the one that matters: it is almost all of the 113 GB. +# target2 is a second build directory that exists here and is another 188 MB. +target/ +target2/ +*/target/ +**/target/ + +# Build directories from tooling that names them something other than "target". +# The patterns above match a directory CALLED target, so they missed +# .codex-target-security-20260719-reaudit entirely: 3.5 GB of compiler output +# that went on being uploaded to the daemon on every build, 198 seconds of it. +# Anything with "target" in the name is compiler output, wherever it sits. +.codex-*/ +*target*/ +**/*target*/ + +# Packaged releases and their staging directories. Deliberately broad: this tree +# has accumulated dist, dist-final-*, dist-opencl-upgrade-* and will accumulate +# more, and each is tens of megabytes of already-built output the image rebuilds +# from source anyway. +dist*/ + +# Vendored Colab upload snapshots: a whole second copy of the app tree plus +# prebuilt artifacts, 36 MB, none of which the build reads. +scripts/mining-nvidia/colab-upload/ + +# Git history. The build reads no VCS metadata. +.git/ +.github/ + +# Local test rigs, chain data and wallets. None of this belongs in an image, and +# a wallet key reaching a layer would be a leak that survives every later layer. +**/hacash_mainnet_data/ +**/hacash_testnet_data/ +**/*.key +**/*.key.* +deploy/secrets/ +*.log +*.err +*.out + +# Editor, OS and tooling noise. +.vscode/ +.idea/ +.grok/ +.claude/ +Thumbs.db +.DS_Store + +# The prebuilt Windows toolchain vendored for local builds. The image is Linux. +.tools/ +osxcross/ + +# Docker's own files: changing the compose file must not invalidate the build +# cache for the source layers. +deploy/docker-compose.yml diff --git a/.github/dependabot.yml b/.github/dependabot.yml new file mode 100644 index 00000000..cc2fe201 --- /dev/null +++ b/.github/dependabot.yml @@ -0,0 +1,9 @@ +# Every action in .github/workflows is pinned to a 40-hex commit SHA. A pin does +# not follow upstream security fixes, so let Dependabot raise the pin instead of +# letting it rot. +version: 2 +updates: + - package-ecosystem: "github-actions" + directory: "/" + schedule: + interval: "weekly" diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 918f2325..2acbdc1f 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -23,21 +23,33 @@ jobs: contents: read steps: - name: Checkout - uses: actions/checkout@v6 + uses: actions/checkout@d23441a48e516b6c34aea4fa41551a30e30af803 # v6 - name: Install Rust - uses: dtolnay/rust-toolchain@stable + uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable branch - name: Cache cargo - uses: Swatinem/rust-cache@v2 + uses: Swatinem/rust-cache@e18b497796c12c097a38f9edb9d0641fb99eee32 # v2 with: shared-key: release-windows-ocl - name: Install OpenCL.lib (vcpkg) shell: pwsh run: | + # Pinned vcpkg release: the OpenCL import library is linked into every + # shipped GPU binary, so it must not come from a moving upstream head. + $vcpkgCommit = "cd61e1e26a038e82d6550a3ebbe0fbbfe7da78e3" # tag 2026.06.24 $vcpkgRoot = "$env:GITHUB_WORKSPACE\vcpkg-ci" - git clone --depth 1 https://github.com/microsoft/vcpkg.git $vcpkgRoot + git init $vcpkgRoot + if ($LASTEXITCODE -ne 0) { throw "git init failed for $vcpkgRoot" } + git -C $vcpkgRoot remote add origin https://github.com/microsoft/vcpkg.git + if ($LASTEXITCODE -ne 0) { throw "git remote add failed for $vcpkgRoot" } + git -C $vcpkgRoot fetch --depth 1 origin $vcpkgCommit + if ($LASTEXITCODE -ne 0) { throw "git fetch of pinned vcpkg commit $vcpkgCommit failed" } + git -C $vcpkgRoot checkout --detach FETCH_HEAD + if ($LASTEXITCODE -ne 0) { throw "git checkout of pinned vcpkg commit $vcpkgCommit failed" } + $head = (git -C $vcpkgRoot rev-parse HEAD).Trim() + if ($head -ne $vcpkgCommit) { throw "vcpkg is at $head, expected pinned $vcpkgCommit" } & "$vcpkgRoot\bootstrap-vcpkg.bat" -disableMetrics & "$vcpkgRoot\vcpkg.exe" install opencl:x64-windows $lib = "$vcpkgRoot\installed\x64-windows\lib" @@ -53,9 +65,36 @@ jobs: - name: Verify Windows HACD binaries are CPU-only shell: pwsh run: | - $dumpbin = Get-ChildItem -Path "$env:ProgramFiles\Microsoft Visual Studio\2022\*\VC\Tools\MSVC\*\bin\Hostx64\x64\dumpbin.exe" -File | - Sort-Object FullName -Descending | Select-Object -First 1 - if (-not $dumpbin) { throw "dumpbin.exe not found" } + # Do NOT hardcode the Visual Studio year. This step searched + # "Microsoft Visual Studio\2022\..." and started failing every release + # the moment the hosted runner moved on, and it failed loudly in the + # wrong place: GitHub runs pwsh with ErrorActionPreference = Stop, so + # Get-ChildItem threw on the missing directory before the script's own + # "dumpbin.exe not found" check could report anything useful. + # + # Ask the installer where Visual Studio is, then fall back to globbing + # every year and edition under both Program Files roots. + $dumpbin = $null + $vswhere = "${env:ProgramFiles(x86)}\Microsoft Visual Studio\Installer\vswhere.exe" + if (Test-Path $vswhere) { + $vsRoot = & $vswhere -latest -products * -property installationPath 2>$null + if ($vsRoot) { + $dumpbin = Get-ChildItem -Path "$vsRoot\VC\Tools\MSVC\*\bin\Hostx64\x64\dumpbin.exe" ` + -File -ErrorAction SilentlyContinue | Sort-Object FullName -Descending | Select-Object -First 1 + } + } + if (-not $dumpbin) { + $globs = @( + "$env:ProgramFiles\Microsoft Visual Studio\*\*\VC\Tools\MSVC\*\bin\Hostx64\x64\dumpbin.exe", + "${env:ProgramFiles(x86)}\Microsoft Visual Studio\*\*\VC\Tools\MSVC\*\bin\Hostx64\x64\dumpbin.exe" + ) + $dumpbin = $globs | ForEach-Object { Get-ChildItem -Path $_ -File -ErrorAction SilentlyContinue } | + Sort-Object FullName -Descending | Select-Object -First 1 + } + if (-not $dumpbin) { + throw "dumpbin.exe not found. HACD must be proven CPU-only before release, so this build stops rather than shipping unverified binaries." + } + Write-Host "using $($dumpbin.FullName)" foreach ($binary in @("hacash.exe", "diaworker.exe")) { $binaryPath = Join-Path "$env:GITHUB_WORKSPACE\target\release" $binary $dependencies = (& $dumpbin.FullName /DEPENDENTS $binaryPath | Out-String) @@ -68,6 +107,16 @@ jobs: - name: Build miner panel run: cargo build --locked --release -p miner-panel + - name: Build public free-IP pool (hac-pool) + run: cargo build --locked --release -p miner-pool + + # HBIT payout pool: hbit-pool-server, hbit-settle-spike and + # hbit-pool-payout hold the pool wallet and compute real payouts, so they + # must compile in the release pipeline even though they are not packaged + # (operators self-build). + - name: Build HBIT payout pool server (hbit-pool) + run: cargo build --locked --release -p hbit-pool + - name: Run Windows release tests run: | cargo test --locked -p x16rs @@ -76,6 +125,7 @@ jobs: cargo test --locked -p app --features ocl cargo test --locked -p miner-panel cargo test --locked -p basis --lib + cargo test --locked -p hbit-pool - name: Package release ZIPs shell: pwsh @@ -90,11 +140,46 @@ jobs: } & "$env:GITHUB_WORKSPACE\scripts\pack-release.ps1" -Version $version + - name: Package HBIT pool (Windows) + shell: pwsh + run: | + # Packaged APART from the miner on purpose. This is server software an + # operator runs beside their own full node; bundling a wallet-holding + # binary into the miner archive would put it on every gaming PC that + # downloads this. + # + # Ships with NO wallet, NO config and NO address. The operator supplies + # all three and the pool refuses to start without them, so nothing here + # can pay a stranger by default. + $version = "manual" + if ($env:GITHUB_REF_TYPE -eq "tag") { $version = $env:RELEASE_REF_NAME } + $ws = $env:GITHUB_WORKSPACE + $pooldir = Join-Path $ws "dist/hbit-pool-windows-x64" + if (Test-Path $pooldir) { Remove-Item $pooldir -Recurse -Force } + New-Item -ItemType Directory -Force -Path $pooldir | Out-Null + foreach ($f in @("hbit-pool-server.exe", "hbit-pool-payout.exe")) { + Copy-Item (Join-Path $ws "target/release/$f") $pooldir + } + foreach ($f in @("POOL-OPERATOR.md", "POOL-README.md")) { + Copy-Item (Join-Path $ws "docs/$f") $pooldir + } + [IO.File]::WriteAllText((Join-Path $pooldir "VERSION.txt"), $version) + $name = if ($version -like "v*") { "hbit-pool-windows-x64-$version.zip" } else { "hbit-pool-windows-x64.zip" } + $poolzip = Join-Path $ws "dist/$name" + Compress-Archive -Path $pooldir -DestinationPath $poolzip -Force + $h = (Get-FileHash -Algorithm SHA256 -LiteralPath $poolzip).Hash.ToLowerInvariant() + [IO.File]::WriteAllText("$poolzip.sha256", "$h $name") + if ((Get-Item $poolzip).Length -lt 1000) { throw "pool zip looks empty" } + Write-Host "packaged $name" + - name: Validate Windows release ZIPs shell: pwsh run: | - $zips = @(Get-ChildItem "$env:GITHUB_WORKSPACE\dist" -Filter "*.zip" -File) - if ($zips.Count -ne 2) { throw "Expected 2 Windows ZIPs, found $($zips.Count)" } + # Miner ZIPs only. The pool ships its own archive beside these, and + # counting every zip would fail this build for the wrong reason the + # moment another package is added. + $zips = @(Get-ChildItem "$env:GITHUB_WORKSPACE\dist" -Filter "hacash-miner-*.zip" -File) + if ($zips.Count -ne 2) { throw "Expected 2 Windows miner ZIPs, found $($zips.Count)" } foreach ($zip in $zips) { $checksumPath = "$($zip.FullName).sha256" if (-not (Test-Path -LiteralPath $checksumPath -PathType Leaf)) { @@ -138,7 +223,7 @@ jobs: } - name: Upload Windows artifacts - uses: actions/upload-artifact@v7 + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 with: name: hacash-miner-windows-x64 path: | @@ -154,13 +239,13 @@ jobs: contents: read steps: - name: Checkout - uses: actions/checkout@v6 + uses: actions/checkout@d23441a48e516b6c34aea4fa41551a30e30af803 # v6 - name: Install Rust - uses: dtolnay/rust-toolchain@stable + uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable branch - name: Cache cargo - uses: Swatinem/rust-cache@v2 + uses: Swatinem/rust-cache@e18b497796c12c097a38f9edb9d0641fb99eee32 # v2 with: shared-key: release-linux-ocl @@ -180,15 +265,21 @@ jobs: --bin poworker --bin list_opencl --bin diagnose_opencl cargo build --locked --release --bin hacash --bin diaworker - - name: Build and test miner panel + - name: Build and test miner panel, public pool and payout pool run: | cargo build --locked --release -p miner-panel + cargo build --locked --release -p miner-pool + # hbit-pool holds the payout pool wallet and the money math + # (split_payout, balance_units, off-node ASERT difficulty). + cargo build --locked --release -p hbit-pool + cargo test --locked -p miner-pool cargo test --locked -p x16rs cargo test --locked --release -p x16rs cargo test --locked -p sys cargo test --locked -p app --features ocl cargo test --locked -p miner-panel cargo test --locked -p basis --lib + cargo test --locked -p hbit-pool - name: Verify HACD CPU-only and release preset completeness run: bash scripts/verify-mining-presets.sh @@ -232,6 +323,39 @@ jobs: version="$candidate" fi bash scripts/pack-release-linux.sh "$version" + + # HBIT pool, packaged apart from the miner for the same reason as on + # Windows: it is server software that holds a wallet, and it has no + # business inside an archive aimed at people's gaming PCs. Ships with + # no wallet, no config and no address; the pool refuses to start + # without them. + pooldir="dist/hbit-pool-linux-x86_64" + rm -rf "$pooldir" && mkdir -p "$pooldir/systemd" + cp target/release/hbit-pool-server target/release/hbit-pool-payout "$pooldir/" + # The full node ships WITH the pool. Without it the archive is not + # deployable at all: the pool gets its templates from a node and submits + # blocks to one, so an operator who downloaded only this would install + # something that cannot start. The first version of this package left it + # out, which is a packaging bug and not a documentation problem. + cp target/release/hacash "$pooldir/" + cp deploy/node/hacash.config.ini "$pooldir/hacash.config.ini.example" + cp deploy/systemd/hacash-node.service deploy/systemd/hbit-pool.service "$pooldir/systemd/" + cp deploy/hbit-wait-for-node.sh "$pooldir/" + cp deploy/README.md "$pooldir/DEPLOY.md" + cp docs/POOL-OPERATOR.md docs/POOL-README.md "$pooldir/" + cp scripts/hbit-vps-setup.sh "$pooldir/SETUP-POOL.sh" + chmod u+x "$pooldir"/hbit-pool-server "$pooldir"/hbit-pool-payout \ + "$pooldir"/hacash "$pooldir"/hbit-wait-for-node.sh "$pooldir"/SETUP-POOL.sh + printf '%s' "$version" > "$pooldir/VERSION.txt" + if [[ "$version" == v* ]]; then + poolname="hbit-pool-linux-x86_64-$version.tar.gz" + else + poolname="hbit-pool-linux-x86_64.tar.gz" + fi + tar -czf "dist/$poolname" -C dist hbit-pool-linux-x86_64 + (cd dist && sha256sum "$poolname" > "$poolname.sha256") + test -s "dist/$poolname" + archives=(dist/hacash-miner-*-linux-x86_64*.tar.gz) test "${#archives[@]}" -eq 2 for archive in "${archives[@]}"; do @@ -265,7 +389,7 @@ jobs: grep -Fxq 'connect = pool.example.test:9000' "$full/poworker.config.ini" - name: Upload Linux artifacts - uses: actions/upload-artifact@v7 + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 with: name: hacash-miner-linux-x86_64 path: | @@ -281,6 +405,10 @@ jobs: runs-on: ubuntu-22.04 permissions: contents: write + # Build provenance is signed through GitHub's OIDC identity, so no + # maintainer key is stored in the repository. + id-token: write + attestations: write steps: - name: Validate release tag run: | @@ -291,13 +419,25 @@ jobs: fi - name: Download all release artifacts - uses: actions/download-artifact@v5 + uses: actions/download-artifact@634f93cb2916e3fdff6788551b99b062d0335ce0 # v5 with: path: release-assets merge-multiple: true + # Runs BEFORE the release is created: a package that cannot be attested + # must never reach a user. The .sha256 files ship next to the archives and + # therefore only detect corruption, not tampering; the attestation is the + # tamper-evident check because it is signed by GitHub, not by whoever + # serves the download. + - name: Attest release artifact provenance + uses: actions/attest-build-provenance@977bb373ede98d70efdf65b84cb5f73e068dcc2a # v3 + with: + subject-path: | + release-assets/*.zip + release-assets/*.tar.gz + - name: Publish release - uses: softprops/action-gh-release@v2 + uses: softprops/action-gh-release@3bb12739c298aeb8a4eeaf626c5b8d85266b0e65 # v2 with: tag_name: ${{ github.ref_name }} files: release-assets/* @@ -315,10 +455,30 @@ jobs: | `hacash-miner-only-linux-x86_64*.tar.gz` | Linux fullnode already exists | Windows: extract and run `SETUP.bat` or `SETUP-MINER.bat`. - Verify a ZIP with `Get-FileHash .\.zip -Algorithm SHA256` and compare it with `.zip.sha256`. - Linux: extract and run `./SETUP-LINUX.sh` (or `bash SETUP-LINUX.sh`). - Verify an archive with `sha256sum -c .tar.gz.sha256`. + + ### Check the download before you run it + + These binaries can hold mining rewards and wallet keys, so verify the + build provenance first. This is the only check that detects tampering, + because the signature is produced by GitHub and not by whoever serves + the file: + + ``` + gh attestation verify --repo ${{ github.repository }} + ``` + + (`gh` is the GitHub CLI, available for Windows and Linux.) + + The `.sha256` files detect a truncated or corrupted download only. + They are NOT tamper protection: they are published from the same place + as the archives, so anyone able to replace an archive can replace its + checksum too. Do not treat a matching checksum as proof the file is + genuine. + + Corruption check, Windows: `Get-FileHash .\.zip -Algorithm SHA256` + and compare with `.zip.sha256`. + Corruption check, Linux: `sha256sum -c .tar.gz.sha256`. HAC uses OpenCL GPU mining and supports automatic GPU detection/Auto Tune. Auto Tune supports one selected GPU per poworker config; use separate instances/configs for heterogeneous GPUs. diff --git a/.gitignore b/.gitignore index 000954c5..f690b0cb 100644 --- a/.gitignore +++ b/.gitignore @@ -95,6 +95,23 @@ hacash_* !scripts/mining-amd/presets/diaworker/ !scripts/mining-amd/presets/poworker/ !scripts/mining-amd/presets/**/*.ini +# Same rule for the shipped config TEMPLATES. The blanket *.ini above is meant to +# keep an operator's own filled-in config (with their address and passwords) out +# of the repo, but it was also swallowing the templates the docs tell people to +# copy, so those never reached a clone or a release. These carry no address, no +# password and no machine-specific path: every such key is left commented out on +# purpose, so the strict loader fails loudly and names the key instead of letting +# anyone mine to a value they did not choose. +!mainnet-configs/*.ini +!miner-pool/hac-pool.example.ini +!hbit-pool/hbit-pool.example.ini +!deploy/node/hacash.config.ini + +# The pool's wallet passphrase lives here at deploy time. It must NEVER be +# committed: it is one half of the wallet, and the key file is the other. This +# is listed AFTER the negations above on purpose, so nothing can re-include it. +deploy/secrets/ + *.exe *_ubuntu *_windows diff --git a/Cargo.lock b/Cargo.lock index 61e63bb8..899db985 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -135,6 +135,7 @@ dependencies = [ "cfg-if", "cipher", "cpufeatures 0.2.17", + "zeroize", ] [[package]] @@ -149,6 +150,7 @@ dependencies = [ "ctr", "ghash", "subtle", + "zeroize", ] [[package]] @@ -207,6 +209,56 @@ dependencies = [ "libc", ] +[[package]] +name = "anstream" +version = "1.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "824a212faf96e9acacdbd09febd34438f8f711fb84e09a8916013cd7815ca28d" +dependencies = [ + "anstyle", + "anstyle-parse", + "anstyle-query", + "anstyle-wincon", + "colorchoice", + "is_terminal_polyfill", + "utf8parse", +] + +[[package]] +name = "anstyle" +version = "1.0.14" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "940b3a0ca603d1eade50a4846a2afffd5ef57a9feac2c0e2ec2e14f9ead76000" + +[[package]] +name = "anstyle-parse" +version = "1.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "52ce7f38b242319f7cabaa6813055467063ecdc9d355bbb4ce0c68908cd8130e" +dependencies = [ + "utf8parse", +] + +[[package]] +name = "anstyle-query" +version = "1.1.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "40c48f72fd53cd289104fc64099abca73db4166ad86ea0b4341abe65af83dadc" +dependencies = [ + "windows-sys 0.61.2", +] + +[[package]] +name = "anstyle-wincon" +version = "3.0.11" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "291e6a250ff86cd4a820112fb8898808a366d8f9f58ce16d1f538353ad55747d" +dependencies = [ + "anstyle", + "once_cell_polyfill", + "windows-sys 0.61.2", +] + [[package]] name = "app" version = "0.1.0" @@ -390,7 +442,7 @@ checksum = "3b43422f69d8ff38f95f1b2bb76517c91589a924d1559a0e935d7c8ce0274c11" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.118", ] [[package]] @@ -425,7 +477,7 @@ checksum = "9035ad2d096bed7955a320ee7e2230574d28fd3c3a0f186cbea1ff3c7eed5dbb" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.118", ] [[package]] @@ -601,7 +653,7 @@ dependencies = [ "regex", "rustc-hash 2.1.2", "shlex 1.3.0", - "syn", + "syn 2.0.118", ] [[package]] @@ -745,7 +797,7 @@ checksum = "f9abbd1bc6865053c427f7198e6af43bfdedc55ab791faed4fbd361d789575ff" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.118", ] [[package]] @@ -941,6 +993,46 @@ dependencies = [ "libloading", ] +[[package]] +name = "clap" +version = "4.6.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d91e0c145792ef73a6ad36d27c75ac09f1832222a3c209689d90f534685ee5b7" +dependencies = [ + "clap_builder", + "clap_derive", +] + +[[package]] +name = "clap_builder" +version = "4.6.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f09628afdcc538b57f3c6341e9c8e9970f18e4a481690a64974d7023bd33548b" +dependencies = [ + "anstream", + "anstyle", + "clap_lex", + "strsim", +] + +[[package]] +name = "clap_derive" +version = "4.6.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d012d2b9d65aca7f18f4d9878a045bc17899bba951561ba5ec3c2ba1eed9a061" +dependencies = [ + "heck", + "proc-macro2", + "quote", + "syn 3.0.3", +] + +[[package]] +name = "clap_lex" +version = "1.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c8d4a3bb8b1e0c1050499d1815f5ab16d04f0959b233085fb31653fbfc9d98f9" + [[package]] name = "clipboard-win" version = "5.4.1" @@ -966,6 +1058,12 @@ dependencies = [ "unicode-width", ] +[[package]] +name = "colorchoice" +version = "1.0.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1d07550c9036bf2ae0c684c4297d503f838287c83c53686d05370d0e139ae570" + [[package]] name = "combine" version = "4.6.7" @@ -983,7 +1081,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "f76990911f2267d837d9d0ad060aa63aaad170af40904b29461734c339030d4d" dependencies = [ "quote", - "syn", + "syn 2.0.118", ] [[package]] @@ -1315,7 +1413,7 @@ checksum = "1ac70aa55017e108007fbaf5aa0f54b021c98f92ff8af59d42eda9da96e3dd4f" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.118", ] [[package]] @@ -1520,7 +1618,7 @@ checksum = "67c78a4d8fdf9953a5c9d458f9efe940fd97a0cab0941c075a813ac594733827" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.118", ] [[package]] @@ -1686,7 +1784,7 @@ checksum = "1a5c6c585bc94aaf2c7b51dd4c2ba22680844aba4c687be581871a6f518c5742" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.118", ] [[package]] @@ -1763,7 +1861,7 @@ checksum = "e835b70203e41293343137df5c0664546da5745f82ec9b84d40be8336958447b" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.118", ] [[package]] @@ -2031,7 +2129,7 @@ dependencies = [ [[package]] name = "hacash" -version = "0.4.0" +version = "0.5.2" dependencies = [ "app", "basis", @@ -2065,6 +2163,33 @@ version = "0.17.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "ed5909b6e89a2db4456e54cd5f673791d7eca6732202bbf2a9cc504fe2f9b84a" +[[package]] +name = "hbit-pool" +version = "0.1.0" +dependencies = [ + "aes-gcm", + "argon2", + "basis", + "field", + "fs2", + "getrandom 0.3.4", + "hex", + "mint", + "num-bigint", + "protocol", + "reqwest", + "serde_json", + "sys", + "x16rs", + "zeroize", +] + +[[package]] +name = "heck" +version = "0.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2304e00983f87ffb38b55b444b5e3b60a884b5d30c0fca7d82fe33449bbe55ea" + [[package]] name = "hermit-abi" version = "0.5.2" @@ -2157,6 +2282,7 @@ checksum = "9155a582abd142abc056962c29e3ce5ff2ad5469f4246b537ed42c5deba857da" dependencies = [ "ctutils", "typenum", + "zeroize", ] [[package]] @@ -2417,6 +2543,12 @@ version = "2.12.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "d98f6fed1fde3f8c21bc40a1abb88dd75e67924f9cffc3ef95607bad8017f8e2" +[[package]] +name = "is_terminal_polyfill" +version = "1.70.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a6cb138bb79a146c1bd460005623e142ef0181e3d0219cb493e02f7d08a35695" + [[package]] name = "itertools" version = "0.13.0" @@ -2459,7 +2591,7 @@ dependencies = [ "quote", "rustc_version", "simd_cesu8", - "syn", + "syn 2.0.118", ] [[package]] @@ -2487,7 +2619,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "38c0b942f458fe50cdac086d2f946512305e5631e720728f2a61aabcd47a6264" dependencies = [ "quote", - "syn", + "syn 2.0.118", ] [[package]] @@ -2546,6 +2678,12 @@ version = "3.1.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "e2db585e1d738fc771bf08a151420d3ed193d9d895a36df7f6f8a9456b911ddc" +[[package]] +name = "lazy_static" +version = "1.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bbd2bcb4c963f2ddae06a2efc7e9f3591312473c50c6685e1f298068316e66fe" + [[package]] name = "leveldb-sys" version = "2.0.4" @@ -2721,6 +2859,15 @@ dependencies = [ "libc", ] +[[package]] +name = "matchers" +version = "0.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d1525a2a28c7f4fa0fc98bb91ae755d1e2d1505079e05539e35bc876b5d65ae9" +dependencies = [ + "regex-automata", +] + [[package]] name = "matchit" version = "0.7.3" @@ -2788,6 +2935,21 @@ dependencies = [ "windows-sys 0.59.0", ] +[[package]] +name = "miner-pool" +version = "0.1.0" +dependencies = [ + "axum", + "clap", + "hex", + "reqwest", + "serde", + "serde_json", + "tokio", + "tracing", + "tracing-subscriber", +] + [[package]] name = "minimal-lexical" version = "0.2.1" @@ -2808,8 +2970,6 @@ dependencies = [ name = "mint" version = "0.1.0" dependencies = [ - "aes-gcm", - "argon2", "basis", "concat-idents", "field", @@ -2818,10 +2978,12 @@ dependencies = [ "num-bigint", "num-traits 0.2.19", "protocol", + "sdk", "serde_json", "sys", "tokio", "x16rs", + "zeroize", ] [[package]] @@ -2849,6 +3011,7 @@ dependencies = [ "pkcs8", "shake", "signature", + "zeroize", ] [[package]] @@ -2860,6 +3023,7 @@ dependencies = [ "ctutils", "hybrid-array", "num-traits 0.2.19", + "zeroize", ] [[package]] @@ -2993,6 +3157,15 @@ dependencies = [ "minimal-lexical", ] +[[package]] +name = "nu-ansi-term" +version = "0.50.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7957b9740744892f114936ab4a57b3f487491bbeafaf8083688b16841a4240e5" +dependencies = [ + "windows-sys 0.61.2", +] + [[package]] name = "num-bigint" version = "0.4.6" @@ -3058,7 +3231,7 @@ dependencies = [ "proc-macro-crate", "proc-macro2", "quote", - "syn", + "syn 2.0.118", ] [[package]] @@ -3386,6 +3559,12 @@ version = "1.21.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50" +[[package]] +name = "once_cell_polyfill" +version = "1.70.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "384b8ab6d37215f3c5301a95a4accb5d64aa607f1fcb26a11b5303878451b4fe" + [[package]] name = "opaque-debug" version = "0.2.3" @@ -3521,7 +3700,7 @@ checksum = "c96395f0a926bc13b1c17622aaddda1ecb55d49c8f1bf9777e4d877800a43f8b" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.118", ] [[package]] @@ -4120,6 +4299,7 @@ dependencies = [ name = "sdk" version = "0.1.1" dependencies = [ + "aes", "aes-gcm", "argon2", "basis", @@ -4132,6 +4312,7 @@ dependencies = [ "serde_json", "sys", "wasm-bindgen", + "zeroize", ] [[package]] @@ -4167,7 +4348,7 @@ checksum = "d540f220d3187173da220f885ab66608367b6574e925011a9353e4badda91d79" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.118", ] [[package]] @@ -4202,7 +4383,7 @@ checksum = "175ee3e80ae9982737ca543e96133087cbd9a485eecc3bc4de9c1a37b47ea59c" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.118", ] [[package]] @@ -4301,6 +4482,15 @@ dependencies = [ "sponge-cursor", ] +[[package]] +name = "sharded-slab" +version = "0.1.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f40ca3c46823713e0d4209592e8d6e826aa57e928f09752619fc696c499637f6" +dependencies = [ + "lazy_static", +] + [[package]] name = "shlex" version = "1.3.0" @@ -4523,6 +4713,12 @@ version = "0.1.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "6637bab7722d379c8b41ba849228d680cc12d0a45ba1fa2b48f2a30577a06731" +[[package]] +name = "strsim" +version = "0.11.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7da8b5736845d9f2fcb837ea5d9e2628564b3b043a70948a3f0b778838c5fb4f" + [[package]] name = "subtle" version = "2.6.1" @@ -4540,6 +4736,17 @@ dependencies = [ "unicode-ident", ] +[[package]] +name = "syn" +version = "3.0.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "53e9bae58849f64dfa4f5d5ae372c8341f7305f82a3868709269343628b659a3" +dependencies = [ + "proc-macro2", + "quote", + "unicode-ident", +] + [[package]] name = "sync_wrapper" version = "1.0.2" @@ -4557,7 +4764,7 @@ checksum = "728a70f3dbaf5bab7f0c4b1ac8d7ae5ea60a4b5549c8a5914361c99147a709d2" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.118", ] [[package]] @@ -4577,6 +4784,7 @@ dependencies = [ "ripemd", "sha2 0.10.9", "sha3", + "zeroize", ] [[package]] @@ -4645,7 +4853,7 @@ checksum = "4fee6c4efc90059e10f81e6d42c60a18f76588c3d74cb83a0b242a2b6c7504c1" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.118", ] [[package]] @@ -4656,7 +4864,16 @@ checksum = "ebc4ee7f67670e9b64d05fa4253e753e016c6c95ff35b89b7941d6b856dec1d5" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.118", +] + +[[package]] +name = "thread_local" +version = "1.1.10" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1ad99c4c6d32803332c548b1af0540b357b3f5fc0be8f6c6bfe8b2e6ae784070" +dependencies = [ + "cfg-if", ] [[package]] @@ -4719,6 +4936,7 @@ dependencies = [ "libc", "mio", "pin-project-lite", + "signal-hook-registry", "socket2", "tokio-macros", "windows-sys 0.61.2", @@ -4732,7 +4950,7 @@ checksum = "385a6cb71ab9ab790c5fe8d67f1645e6c450a7ce006a33de03daa956cf70a496" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.118", ] [[package]] @@ -4841,7 +5059,7 @@ checksum = "7490cfa5ec963746568740651ac6781f701c9c5ea257c58e057f3ba8cf69e8da" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.118", ] [[package]] @@ -4851,6 +5069,36 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "db97caf9d906fbde555dd62fa95ddba9eecfd14cb388e4f491a66d74cd5fb79a" dependencies = [ "once_cell", + "valuable", +] + +[[package]] +name = "tracing-log" +version = "0.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ee855f1f400bd0e5c02d150ae5de3840039a3f54b025156404e34c23c03f47c3" +dependencies = [ + "log", + "once_cell", + "tracing-core", +] + +[[package]] +name = "tracing-subscriber" +version = "0.3.23" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cb7f578e5945fb242538965c2d0b04418d38ec25c79d160cd279bf0731c8d319" +dependencies = [ + "matchers", + "nu-ansi-term", + "once_cell", + "regex-automata", + "sharded-slab", + "smallvec", + "thread_local", + "tracing", + "tracing-core", + "tracing-log", ] [[package]] @@ -4949,6 +5197,18 @@ version = "1.0.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "b6c140620e7ffbb22c2dee59cafe6084a59b5ffc27a8859a5f0d494b5d52b6be" +[[package]] +name = "utf8parse" +version = "0.2.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "06abde3611657adf66d383f00b093d7faecc7fa57071cce2578660c9f1010821" + +[[package]] +name = "valuable" +version = "0.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ba73ea9cf16a25df0c8caa16c51acb937d5712a8429db78a3ee29d5dcacd3a65" + [[package]] name = "vcpkg" version = "0.2.15" @@ -5037,7 +5297,7 @@ dependencies = [ "log", "proc-macro2", "quote", - "syn", + "syn 2.0.118", "wasm-bindgen-shared", ] @@ -5072,7 +5332,7 @@ checksum = "8ae87ea40c9f689fc23f209965b6fb8a99ad69aeeb0231408be24920604395de" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.118", "wasm-bindgen-backend", "wasm-bindgen-shared", ] @@ -5441,7 +5701,7 @@ checksum = "2bbd5b46c938e506ecbce286b6628a02171d56153ba733b6c741fc627ec9579b" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.118", ] [[package]] @@ -5452,7 +5712,7 @@ checksum = "053e2e040ab57b9dc951b72c264860db7eb3b0200ba345b4e4c3b14f67855ddf" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.118", ] [[package]] @@ -5463,7 +5723,7 @@ checksum = "053c4c462dc91d3b1504c6fe5a726dd15e216ba718e84a0e46a88fbe5ded3515" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.118", ] [[package]] @@ -5474,7 +5734,7 @@ checksum = "3f316c4a2570ba26bbec722032c4099d8c8bc095efccdc15688708623367e358" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.118", ] [[package]] @@ -5806,7 +6066,7 @@ checksum = "de844c262c8848816172cef550288e7dc6c7b7814b4ee56b3e1553f275f1858e" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.118", "synstructure", ] @@ -5866,7 +6126,7 @@ checksum = "709ab20fc57cb22af85be7b360239563209258430bccf38d8b979c5a2ae3ecce" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.118", "zbus-lockstep", "zbus_xml", "zvariant", @@ -5881,7 +6141,7 @@ dependencies = [ "proc-macro-crate", "proc-macro2", "quote", - "syn", + "syn 2.0.118", "zvariant_utils", ] @@ -5926,7 +6186,7 @@ checksum = "1ae7f38b72ec2a254e2b87ef277cf2cd4fb97cbebf944faa6f33354da0867930" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.118", ] [[package]] @@ -5946,7 +6206,7 @@ checksum = "11532158c46691caf0f2593ea8358fed6bbf68a0315e80aae9bd41fbade684a1" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.118", "synstructure", ] @@ -5986,7 +6246,7 @@ checksum = "625dc425cab0dca6dc3c3319506e6593dcb08a9f387ea3b284dbd52a92c40555" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.118", ] [[package]] @@ -6027,7 +6287,7 @@ dependencies = [ "proc-macro-crate", "proc-macro2", "quote", - "syn", + "syn 2.0.118", "zvariant_utils", ] @@ -6039,5 +6299,5 @@ checksum = "c51bcff7cc3dbb5055396bcf774748c3dab426b4b8659046963523cee4808340" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.118", ] diff --git a/Cargo.toml b/Cargo.toml index 9ccd542c..5e4a7ee3 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -1,7 +1,7 @@ [package] name = "hacash" default-run = "hacash" -version = "0.4.0" +version = "0.5.2" edition = "2024" [workspace] @@ -24,6 +24,8 @@ members = [ "sys", "x16rs", "x16rs-cuda", + "miner-pool", + "hbit-pool", ] exclude = [ "chainv1", @@ -69,11 +71,27 @@ db-rocksdb = ["db/db-rocksdb"] [profile.release] debug = false # for RUST_BACKTRACE=1 opt-level = 3 # Optimize for size. -lto = true # Enable Link Time Optimization +lto = "thin" # Thin LTO: nearly the perf of fat LTO for this workload + # (the hot path is x16rs / GPU kernels, not Rust glue) with far + # less memory and no fat-LTO rustc crashes under panic=unwind. codegen-units = 1 # Reduce number of codegen units to increase optimizations. -panic = 'abort' # Abort on panic +panic = 'unwind' # Unwind (not abort): a panic in one request/mining thread must + # become an error/skip, not kill the whole 24/7 pool or miner. + # Handlers and the block/diamond mining threads wrap panic-prone + # work in catch_unwind (app/src/mining_guard.rs); abort would make + # that impossible and turn any panic into full downtime. strip = true # Strip symbols from binary +# Money crate only: hbit-pool owns the share accounting, payout splitting and settlement. +# There a silent wrapping add or multiply would pay a miner the wrong amount and the pool +# would never notice, so arithmetic overflow must panic. The settlement work is already +# wrapped in catch_unwind, which turns such a panic into "skip this item and report it" +# instead of a wrong transfer. +# This is deliberately NOT workspace wide: overflow checks in the x16rs / CPU mining hot +# path cost real hashrate, and that code is consensus defined and already range safe. +[profile.release.package.hbit-pool] +overflow-checks = true + [profile.wasm-release] inherits = "release" opt-level = "z" # Optimize wasm for code size diff --git a/README-LINUX-MINER-ONLY.txt b/README-LINUX-MINER-ONLY.txt index 7a3b8874..cc1c20e8 100644 --- a/README-LINUX-MINER-ONLY.txt +++ b/README-LINUX-MINER-ONLY.txt @@ -5,6 +5,21 @@ By Mosky FOR: You already run the Hacash fullnode with miner RPC on 127.0.0.1:8080. This archive does not include the fullnode. +CHECK THE DOWNLOAD BEFORE YOU RUN IT +------------------------------------ +These binaries can hold mining rewards and wallet keys. Every release is signed +by GitHub with build provenance attestation, and verifying it is the only check +that detects tampering. With the GitHub CLI (gh) installed, run this in the +folder holding the downloaded hacash-miner-only-linux-x86_64 .tar.gz: + + gh attestation verify .tar.gz --repo Moskyera/fullnodedev + +If verification fails, delete the file and do not run it. + +The .sha256 files are NOT a signature. They only catch a truncated or corrupted +download, and they come from the same place as the archives, so a matching +checksum is not proof the file is genuine. The attestation is the real check. + QUICK START ----------- 1. Extract the .tar.gz archive. diff --git a/README-LINUX-RELEASE.txt b/README-LINUX-RELEASE.txt index c4311a81..c6537b50 100644 --- a/README-LINUX-RELEASE.txt +++ b/README-LINUX-RELEASE.txt @@ -4,6 +4,21 @@ By Mosky FOR: A new Ubuntu/Debian PC. Fullnode, miners and panel are included. +CHECK THE DOWNLOAD BEFORE YOU RUN IT +------------------------------------ +These binaries can hold mining rewards and wallet keys. Every release is signed +by GitHub with build provenance attestation, and verifying it is the only check +that detects tampering. With the GitHub CLI (gh) installed, run this in the +folder holding the downloaded hacash-miner-full-linux-x86_64 .tar.gz: + + gh attestation verify .tar.gz --repo Moskyera/fullnodedev + +If verification fails, delete the file and do not run it. + +The .sha256 files are NOT a signature. They only catch a truncated or corrupted +download, and they come from the same place as the archives, so a matching +checksum is not proof the file is genuine. The attestation is the real check. + QUICK START ----------- 1. Extract the .tar.gz archive. diff --git a/README-MINER-ONLY.txt b/README-MINER-ONLY.txt index a038abf1..db384b3b 100644 --- a/README-MINER-ONLY.txt +++ b/README-MINER-ONLY.txt @@ -1,9 +1,24 @@ -HAC Miner — WORKERS ONLY (Windows x64) +HAC Miner - WORKERS ONLY (Windows x64) By Mosky ====================================== FOR: You already run hacash.exe fullnode + hacash.config.ini +CHECK THE DOWNLOAD BEFORE YOU RUN IT +------------------------------------ +These binaries can hold mining rewards and wallet keys. Every release is signed +by GitHub with build provenance attestation, and verifying it is the only check +that detects tampering. With the GitHub CLI (gh) installed, run this in the +folder holding the downloaded hacash-miner-only-windows-x64 .zip: + + gh attestation verify .zip --repo Moskyera/fullnodedev + +If verification fails, delete the file and do not run it. + +The .sha256 files are NOT a signature. They only catch a truncated or corrupted +download, and they come from the same place as the archives, so a matching +checksum is not proof the file is genuine. The attestation is the real check. + QUICK START ----------- 1. Extract ZIP to a folder diff --git a/README-RELEASE.txt b/README-RELEASE.txt index aff0659e..93817b4c 100644 --- a/README-RELEASE.txt +++ b/README-RELEASE.txt @@ -1,9 +1,24 @@ -HAC Miner — FULL PACKAGE (Windows x64) +HAC Miner - FULL PACKAGE (Windows x64) By Mosky ====================================== FOR: Clean PC — everything in one ZIP +CHECK THE DOWNLOAD BEFORE YOU RUN IT +------------------------------------ +These binaries can hold mining rewards and wallet keys. Every release is signed +by GitHub with build provenance attestation, and verifying it is the only check +that detects tampering. With the GitHub CLI (gh) installed, run this in the +folder holding the downloaded hacash-miner-full-windows-x64 .zip: + + gh attestation verify .zip --repo Moskyera/fullnodedev + +If verification fails, delete the file and do not run it. + +The .sha256 files are NOT a signature. They only catch a truncated or corrupted +download, and they come from the same place as the archives, so a matching +checksum is not proof the file is genuine. The attestation is the real check. + QUICK START ----------- 1. Extract ZIP to a folder (e.g. C:\HacashMiner) diff --git a/README.md b/README.md index cac3d66e..a6ed6a47 100644 --- a/README.md +++ b/README.md @@ -16,15 +16,45 @@ After extract on Windows: **`SETUP.bat`** or **`SETUP-MINER.bat`** → **`miner- On Linux x86_64: extract the `.tar.gz` package → **`./SETUP-LINUX.sh`** → **`./START-MINER-PANEL.sh`** -Verify Windows downloads with `Get-FileHash .zip -Algorithm SHA256`; on Linux use `sha256sum -c .tar.gz.sha256`. +### Check the download before running it -### HAC OpenCL + HACD CPU mining +These binaries can hold mining rewards and wallet keys. Every release is signed +by GitHub with **build provenance attestation** (OIDC, no maintainer key in the +repo). Verifying it is the **only** check that detects tampering: -**HAC** uses OpenCL GPUs (AMD/NVIDIA/Intel). CUDA is intentionally not included in the release. **HACD** is CPU/fullnode mining and does not use OpenCL. +``` +gh attestation verify --repo / +``` + +`gh` is the GitHub CLI (Windows and Linux). Use the repo you downloaded from, for +example `--repo Moskyera/fullnodedev`. If verification fails, **delete the file +and do not run it.** + +The `.sha256` files are **not a signature**. They only catch a truncated or +corrupted download, and they are published from the same place as the archives, +so anyone able to replace an archive can replace its checksum too. A matching +checksum is **not** proof the file is genuine; the attestation is. + +| Check | Windows | Linux | +|---|---|---| +| Genuine (tamper) | `gh attestation verify .zip --repo /` | `gh attestation verify .tar.gz --repo /` | +| Corruption only | `Get-FileHash .zip -Algorithm SHA256` | `sha256sum -c .tar.gz.sha256` | + +### HAC OpenCL / CUDA + HACD CPU mining + +| Coin | Worker | Backend | +|------|--------|---------| +| **HAC** | `poworker` | OpenCL (AMD/NVIDIA/Intel) and/or **CUDA** (NVIDIA) | +| **HACD** | `diaworker` | CPU only (no OpenCL/CUDA) | -See **[docs/MINING-AMD.md](docs/MINING-AMD.md)** (Windows) and **[docs/MINING-LINUX.md](docs/MINING-LINUX.md)** (Linux) — `scripts/mining-amd/` build scripts, GPU configs, `list_opencl` device discovery. +- OpenCL: **[docs/MINING-AMD.md](docs/MINING-AMD.md)**, **[docs/MINING-LINUX.md](docs/MINING-LINUX.md)** +- CUDA: **[docs/MINING-NVIDIA-CUDA.md](docs/MINING-NVIDIA-CUDA.md)** (T4 Colab validated) +- Public free-IP pool: **`hac-pool`** - **[docs/PUBLIC-POOL.md](docs/PUBLIC-POOL.md)** +- Running a payout pool (wallet passphrase, payout procedure, 16-block payout lag): **[docs/POOL-OPERATOR.md](docs/POOL-OPERATOR.md)** (read this before mining for other people) +- Community requirements map: **[docs/COMMUNITY-REQUIREMENTS.md](docs/COMMUNITY-REQUIREMENTS.md)** +- Official rebuild notes: **[docs/JOJOIN-REBUILD.md](docs/JOJOIN-REBUILD.md)** -**Maintainers:** pushing a new SemVer tag such as `vX.Y.Z` runs `.github/workflows/release.yml` and builds both Windows ZIPs and both Linux `.tar.gz` archives. +**Maintainers:** tag `vX.Y.Z` runs `.github/workflows/release.yml` (OpenCL workers + panel). CUDA and `hac-pool` builds: see JoJoin rebuild doc. ### Module Architecture diff --git a/SETUP.bat b/SETUP.bat index 53e1f88f..1b6b9f3b 100644 --- a/SETUP.bat +++ b/SETUP.bat @@ -92,8 +92,14 @@ powershell -NoProfile -Command ^ " Set-Content -Path $p -Value $t -NoNewline}}" :: --- 4. Fullnode config template --- +:: Prefer the MAINNET template. The repo-root hacash.config.ini is a local +:: development config (not_find_nodes = true) and would build an isolated +:: chain from height 0, where the reported MH/s is about 16x the real rate. if not exist "%BIN%hacash.config.ini" ( - if exist "%~dp0hacash.config.ini" ( + if exist "%~dp0mainnet-configs\hacash.config.mainnet.ini" ( + copy /Y "%~dp0mainnet-configs\hacash.config.mainnet.ini" "%BIN%hacash.config.ini" >nul + echo [CREATED] hacash.config.ini - from the mainnet template + ) else if exist "%~dp0hacash.config.ini" ( copy /Y "%~dp0hacash.config.ini" "%BIN%hacash.config.ini" >nul echo [CREATED] hacash.config.ini - from template ) else ( @@ -167,7 +173,9 @@ echo 2. Settings - pick CPU/GPU, enter wallet, Save echo 3. Start mining (panel can auto-start fullnode) echo. echo Solo mining needs hacash.exe running with RPC on port 8080. -echo Edit hacash.config.ini - set [miner] reward wallet before first run. +echo hacash.config.ini ships with the miner OFF and no reward address. +echo Before the first run set [miner] reward to YOUR OWN 1... address and +echo then set [miner] enable = true, or let the panel do both when you Save. echo. set /p "LAUNCH= Open HAC Miner Panel now? [Y/N]: " @@ -229,6 +237,22 @@ exit /b 0 :write_default_hacash_ini ( + echo ; Mainnet fullnode config written by SETUP.bat. + echo ; Both miners start OFF with no reward address. An address you did not + echo ; type yourself would be paid every block reward, permanently. + echo ; To mine HAC: uncomment reward with YOUR OWN address, then set + echo ; enable = true in [miner]. Reward addresses are PRIVAKEY addresses, + echo ; the ordinary kind, which start with 1 - not with 3. + echo ; enable = true with no reward stops the node with a config error. + echo ; It never picks an address for you. + echo. + echo [node] + echo name = rust_node + echo listen = 3337 + echo boots = 54.193.49.59:3337, 182.92.163.225:3337, 54.219.80.127:3337 + echo not_find_nodes = false + echo fast_sync = true + echo. echo [server] echo enable = true echo listen = 8080 @@ -237,11 +261,16 @@ exit /b 0 echo. echo [miner] echo enable = false - echo reward = YOUR_HAC_WALLET_ADDRESS + echo ; reward = YOUR_HAC_PRIVAKEY_1x + echo message = hacashminer echo. echo [diamondminer] echo enable = false - echo reward = YOUR_HACD_PRIVAKEY_3x + echo ; reward = YOUR_HACD_PRIVAKEY_1x + echo ; bid_password = change-me + echo bid_min = 1 + echo bid_max = 31 + echo bid_step = 0.5 ) > "%BIN%hacash.config.ini" exit /b 0 diff --git a/app/src/block_mining_runtime.rs b/app/src/block_mining_runtime.rs index b8f156f7..53595954 100644 --- a/app/src/block_mining_runtime.rs +++ b/app/src/block_mining_runtime.rs @@ -2,7 +2,7 @@ use std::collections::HashMap; use std::sync::atomic::{AtomicU64, Ordering::Relaxed}; -use std::sync::{Arc, LazyLock, RwLock, mpsc}; +use std::sync::{Arc, LazyLock, Mutex, MutexGuard, RwLock, mpsc}; use std::thread::spawn; use std::time::*; @@ -11,6 +11,9 @@ use serde_json::Value as JV; use crate::efficiency::*; use crate::hash_util::{hash_left_zero_pad3, hash_more_power}; +// The panic firewall is shared with the diamond (HACD) worker, which has exactly +// the same "one result thread owns every submission" shape. +use crate::mining_guard::guard_mining_iteration; #[cfg(feature = "cuda")] use crate::mining_batch::CudaBlockBackend; #[cfg(feature = "ocl")] @@ -33,6 +36,48 @@ use super::CudaMiningResources; use crate::opencl_gpu::{OpenclGpuHandle, initialize_opencl, opencl_snapshot_from_resource}; const HASH_WIDTH: usize = 32; +/// A real difficulty target never comes anywhere near this many leading zero +/// bytes (that would leave under 2^32 of the hash space). Refuse such a template +/// instead of installing an unmineable target from a hostile or buggy upstream. +const MAX_TARGET_LEADING_ZERO_BYTES: usize = 28; +/// Bounded result channel: an unbounded queue grows without limit whenever the +/// drain thread stalls. Under backpressure a statistics-only batch may be +/// dropped, but a result meeting its target is real money and always waits. +const RESULT_CHANNEL_CAPACITY: usize = 1024; +/// Winning results queued for the dedicated submit thread. Deep enough that a +/// burst of pool shares never blocks the result drain. +const SUBMIT_QUEUE_CAPACITY: usize = 256; +/// Floor on how many results one drain tick takes. +/// +/// The old bound was four per worker per tick, which metered a GPU batch's pool +/// shares out at a few dozen a second no matter how many the card actually +/// found: the fix upstream would have been undone right here. It is deliberately +/// equal to `SUBMIT_QUEUE_CAPACITY`, so one tick can fill the submit queue and no +/// more, and raising it can never provoke the blocking inline-submit fallback on +/// the result thread. Still hard bounded, so a tick cannot run long. +const RESULT_DRAIN_MIN_PER_TICK: usize = SUBMIT_QUEUE_CAPACITY; +/// How often the miner may say it is undersampling. A pool serving a share +/// target far too easy for the card overflows the GPU share list on EVERY batch, +/// and an unthrottled warning there is the next log that fills a disk. +const SHARE_OVERFLOW_LOG_INTERVAL: Duration = Duration::from_secs(30); +/// How many recent heights the per-template submit gate remembers. A template +/// this far behind the tip can no longer be submitted anywhere, so its +/// bookkeeping is dropped: the gate must never grow for the life of the process. +const SUBMIT_GATE_HEIGHT_WINDOW: u64 = 8; +/// Hard ceiling on gate entries, so a burst of same-height reorgs cannot grow the +/// map even while the height itself stands still. +const SUBMIT_GATE_MAX_TEMPLATES: usize = 64; +/// How long one template is left alone after a pool answered `kind:"busy"`. +/// +/// "Busy" means the pool is shedding load (HBIT refuses a share once it is +/// already holding its cap for that height), so the very next winner would be +/// refused too and hammering it makes the overload worse. Two seconds is an order +/// of magnitude longer than the 123 ms result drain tick, so the drain cadence +/// cannot defeat the back-off, and it is a tiny fraction of the 300 s target block +/// time, so a full network block found while the pool was busy is retried long +/// before the height can be lost. The template stays ALIVE throughout: this only +/// spaces the retries out. +const POOL_BUSY_COOLDOWN: Duration = Duration::from_secs(2); const MINING_INTERVAL: f64 = 3.0; const WORKER_RATE_STALE_MS: u64 = 15_000; const HASHRATE_EWMA_NEW_WEIGHT: f64 = 0.25; @@ -40,8 +85,27 @@ const TARGET_BLOCK_TIME: f64 = 300.0; const ONEDAY_BLOCK_NUM: f64 = 288.0; static MINING_BLOCK_HEIGHT: AtomicU64 = AtomicU64::new(0); +/// Set while the upstream (pool bridge or fullnode) tells us the work it is +/// serving is no longer being refreshed. The installed template can no longer win +/// anything, so the workers idle instead of burning power on dead work. +static UPSTREAM_STALE: std::sync::atomic::AtomicBool = std::sync::atomic::AtomicBool::new(false); +/// Bumped every time the template actually changes, INCLUDING a same-height reorg +/// (a new block replacing the tip at the same height). Workers watch this in +/// addition to the height so they stop grinding an orphaned template promptly. +static MINING_BLOCK_EPOCH: AtomicU64 = AtomicU64::new(0); +/// Bumped on EVERY template install, including a mere re-serialization of the +/// same job. The submit gate uses it to tell "the upstream served us this exact +/// job again" from "nothing moved", which is the difference between a template +/// the upstream still treats as live work and one it has consumed. +static MINING_TEMPLATE_INSTALL_SEQ: AtomicU64 = AtomicU64::new(0); static MINING_BLOCK_STUFF: LazyLock>> = LazyLock::new(|| RwLock::default()); +/// Monotonic reference for the undersampling warning throttle. +static MINER_CLOCK: LazyLock = LazyLock::new(Instant::now); +/// Total share-list hits the GPU counted but could not hand back this session. +static SHARE_HITS_DROPPED: AtomicU64 = AtomicU64::new(0); +/// `MINER_CLOCK` millis at the last undersampling line (0 = never). +static SHARE_OVERFLOW_LAST_LOG_MS: AtomicU64 = AtomicU64::new(0); #[derive(Clone)] pub(crate) enum MinerBackend { @@ -57,6 +121,11 @@ pub(crate) enum MinerBackend { #[derive(Clone, Default)] struct BlockMiningStuff { height: u64, + /// Generation of this job. Bumped ONLY when the job really changed (a + /// different height, or the same height on a different parent). It is stored + /// inside the template so a worker reads height, epoch and the bytes it mines + /// from ONE snapshot and can never mislabel its own batch. + epoch: u64, target_hash: Hash, block_intro: BlockIntro, coinbase_tx: TransactionCoinbase, @@ -67,6 +136,10 @@ struct BlockMiningStuff { pub(crate) struct BlockMiningResult { worker_id: usize, pub height: u64, + /// Parent hash of the template this result was mined against. Only a template + /// replaced at THIS height with a different parent makes the result dead + /// work; the tip merely advancing does not (see `result_is_orphaned`). + prevhash: Vec, pub nonce_start: u32, nonce_space: u32, gpu_nonce_space: u32, @@ -176,12 +249,733 @@ impl HashrateTracker { } } +/// The threshold the GPU should build a pool share list against, or `None` for +/// solo mining. +/// +/// This is the entire guarantee that solo behaviour is unchanged. The decision +/// is `worker_param()`, the one place that already decides whether this miner +/// announces itself to a pool at all, so the two can never drift apart. `None` +/// makes the OpenCL path launch the kernel with share_capacity=0: the kernel +/// skips the whole appending block, the host writes and reads no share buffer, +/// and the batch carries the same single best result it always did. +/// +/// When there IS a pool, the served `target_hash` already IS the share target - +/// HBIT's `/query/miner/pending` puts its share target in that field - so there +/// is no second threshold to plumb and no way for the two to disagree. +fn pool_share_target(cnf: &PoWorkConf, stuff: &BlockMiningStuff) -> Option<[u8; HASH_WIDTH]> { + if cnf.worker_param().is_empty() { + return None; + } + stuff.target_hash.to_vec().try_into().ok() +} + +/// Say, at most twice a minute, that the card found more payable nonces than one +/// batch can hand back. +/// +/// Silence here would be the original defect wearing a different hat: the miner +/// would be paid for a fraction of what it mined and nothing would say so. +fn report_share_undersampling(dropped: u64) { + if dropped == 0 { + return; + } + let total = SHARE_HITS_DROPPED + .fetch_add(dropped, Relaxed) + .saturating_add(dropped); + let now_ms = MINER_CLOCK.elapsed().as_millis() as u64; + let last = SHARE_OVERFLOW_LAST_LOG_MS.load(Relaxed); + if last != 0 && now_ms.saturating_sub(last) < SHARE_OVERFLOW_LOG_INTERVAL.as_millis() as u64 { + return; + } + if SHARE_OVERFLOW_LAST_LOG_MS + .compare_exchange(last, now_ms, Relaxed, Relaxed) + .is_err() + { + return; + } + eprintln!( + "\n[Mining] UNDERSAMPLING: the GPU share list filled up, so {dropped} payable nonces from this batch were never submitted ({total} so far this session). This is lost income, and it means the pool's share target is far too easy for this card: ask the operator to LOWER share_bits, which makes each share harder. Raising it makes shares easier and loses more. Nothing is wrong with the GPU. This line is rate limited to one per 30 seconds, so it undercounts how often this happens; the session figure is the number that matters." + ); +} + +/// True when this result independently meets its own PoW target. The node +/// accepts a block whose hash <= target (equal-inclusive), so use the same +/// equal-inclusive test rather than the strict "more power" comparison, which +/// would drop a hash landing exactly on target. +fn result_meets_target(res: &BlockMiningResult) -> bool { + !res.target_hash.is_empty() && !hash_more_power(&res.target_hash, &res.result_hash) +} + +/// (height, parent hash) of the template currently installed, read from ONE +/// snapshot so the pair is always consistent. `None` when the state cannot be +/// read: nothing is treated as dead work in that case, because dropping a winner +/// costs a whole block reward while submitting a late one costs one HTTP +/// round-trip and a deterministic rejection. +fn live_template_identity() -> Option<(u64, Vec)> { + let stuff = MINING_BLOCK_STUFF.read().ok()?; + Some((stuff.height, stuff.block_intro.prevhash().to_vec())) +} + +/// A finished result is dead work ONLY when the node replaced the template at the +/// very height it was mined for with one built on a different parent (a +/// same-height reorg). That old template is evicted from the node's cache, so the +/// solution can no longer be reconstructed there. +/// +/// Every other case is still worth submitting, and the NODE is the authority on +/// staleness: `miner_success` looks the template up by the SUBMITTED height and +/// the node deliberately keeps several recent heights, so a solution found a +/// moment before the tip advanced is still accepted as a competing candidate. +fn result_is_orphaned(res: &BlockMiningResult, live: Option<&(u64, Vec)>) -> bool { + let Some((live_hei, live_prevhash)) = live else { + return false; + }; + res.height == *live_hei + && !res.prevhash.is_empty() + && !live_prevhash.is_empty() + && res.prevhash != *live_prevhash +} + +/// Keep EVERY distinct winning result. Against a pool each submission is an +/// independent PPLNS share keyed by (height, coinbase_nonce, block_nonce), so +/// collapsing winners by height alone would throw away earned shares (real +/// money). Only an exact repeat of that triple is a true duplicate. +fn record_winner(winners: &mut Vec>, res: &Arc) { + let already_queued = winners.iter().any(|w| { + w.height == res.height + && w.head_nonce == res.head_nonce + && w.coinbase_nonce == res.coinbase_nonce + }); + if !already_queued { + winners.push(res.clone()); + } +} + +/// Hand a batch result to the drain thread. Returns false only when the drain +/// side is gone, so the worker knows to exit. Under backpressure a +/// statistics-only result is dropped with a log, but a result meeting its own +/// target is a payout and waits for space instead. +fn send_mining_result( + result_ch_tx: &mpsc::SyncSender>, + res: Arc, +) -> bool { + match result_ch_tx.try_send(res) { + Ok(()) => true, + Err(mpsc::TrySendError::Full(res)) => { + if result_meets_target(&res) { + return result_ch_tx.send(res).is_ok(); + } + eprintln!( + "[Mining] Result queue full, dropped a statistics-only batch at height {}.", + res.height + ); + true + } + Err(mpsc::TrySendError::Disconnected(_)) => false, + } +} + +/// What the UPSTREAM said about one submitted winner. +/// +/// `poworker::push_block_mining_success` returns `()`, so the classification the +/// per-template gate needs (a definitive upstream verdict versus never having +/// reached the upstream at all) is made here, next to the gate that consumes it. +/// +/// Two upstreams speak here and they answer in different dialects. A FULLNODE +/// replies `{"ret":0,"mining":"success"}` or `{"ret":1,"err":""}`, so its +/// meaning is carried by the error TEXT. A POOL replies `{"ret":0|1,"kind":""}` +/// and carries no `err` at all, so its meaning is carried by the KIND. Both are +/// classified below; reading only the fullnode dialect made every pool answer look +/// like "a rejection about this one submission". +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +enum SubmitVerdict { + /// `ret:0` / `"mining":"success"` (fullnode), or `ret:0` / `"kind":"block"` + /// (pool: it relayed a full NETWORK block). The block was accepted, so the + /// height is filled and the node dropped its pending entry for it: no later + /// winner for this template can ever be taken. + Accepted, + /// `ret:0` / `"kind":"share"` (pool). A PPLNS share was CREDITED and the pool + /// wants every further share for the same template. On mainnet the share + /// target is orders of magnitude easier than the network target, so this is + /// the OVERWHELMINGLY common pool answer and each one of them is paid work. + /// It must never settle the template and must never count a later winner as + /// redundant, or the miner silently stops submitting the shares it is paid for. + ShareCredited, + /// The upstream answered, and the answer proves the TEMPLATE is gone or + /// unusable: the fullnode's "pending block height N not found", "pending block + /// not ready" or a difficulty check failure (it rebuilt the hash from a + /// template that is not the one we mined), or a pool's `"kind":"stale"`, which + /// says outright that it has moved to another job. Later winners are dead too. + TemplateGone, + /// `ret:1` / `"kind":"busy"` (pool). The pool is shedding load, NOT refusing + /// the template: it is still live work. Retry, but not immediately, so a miner + /// finding hundreds of nonces a second does not hammer an upstream that just + /// said it is overloaded (see `POOL_BUSY_COOLDOWN`). + UpstreamBusy, + /// `ret:1` / `"kind":"duplicate"` (pool). This exact (height, coinbase_nonce, + /// block_nonce) was already credited. Nothing is proved about the template and + /// a DIFFERENT nonce is still wanted, so it stays open; resending this same + /// submission is the one thing that cannot help. + DuplicateSubmission, + /// The upstream answered, but about THIS submission only (a fullnode error + /// text that is not one of the template rules, or a pool's `"kind":"invalid"`). + /// Nothing is proved about the template, so a later winner still deserves its + /// own attempt. + SubmissionRejected, + /// No upstream verdict at all: connection refused, timeout, or a body carrying + /// no `ret` (proxy error page, truncated reply). A network hiccup must never + /// suppress a height. + NoNodeVerdict, +} + +impl SubmitVerdict { + /// True when the verdict proves nothing more can be won on this template. + fn ends_template(self) -> bool { + matches!(self, SubmitVerdict::Accepted | SubmitVerdict::TemplateGone) + } + + /// True when the upstream PAID for this winner and wants the next one. The + /// gate stops suppressing redundant winners for such a template: on a pool + /// every further share is separately payable, so dropping one is lost income. + fn credits_share(self) -> bool { + matches!(self, SubmitVerdict::ShareCredited) + } + + /// How long to leave this template alone before submitting for it again. + /// Only an overloaded upstream asks for that; every other verdict is either + /// final or immediately retryable. + fn retry_cooldown(self) -> Option { + match self { + SubmitVerdict::UpstreamBusy => Some(POOL_BUSY_COOLDOWN), + _ => None, + } + } + + /// Wording for the single per-template operator line. + fn outcome_text(self) -> &'static str { + match self { + SubmitVerdict::Accepted => "settled (accepted)", + SubmitVerdict::ShareCredited => "retired with its last share credited", + SubmitVerdict::TemplateGone => "settled (template gone or stale)", + SubmitVerdict::UpstreamBusy => "retired while the upstream was busy", + SubmitVerdict::DuplicateSubmission => "retired after a duplicate submission", + SubmitVerdict::SubmissionRejected => "retired after a submission rejection", + SubmitVerdict::NoNodeVerdict => "retired without a node verdict", + } + } +} + +/// A node rejection that is about the TEMPLATE rather than about one submission. +/// Every text below means the node can no longer rebuild our block at all, so +/// every further winner for the same template would get the same answer. +fn node_error_ends_template(err: &str) -> bool { + let err = err.to_ascii_lowercase(); + err.contains("not found") + || err.contains("pending block not ready") + || err.contains("difficulty check failed") +} + +/// The POOL dialect of `/submit/miner/success`: `{"ret":0|1,"kind":""}`, +/// with no `err` text at all for most kinds (the pool's miner-API arm keeps only +/// `ret` and `kind`). `None` means no `kind` we know, so the fullnode rules below +/// decide instead. +/// +/// The kinds come straight from the HBIT pool's `handle_submission`: +/// `block` the pool relayed a full network block: the template is DEAD. +/// `share` a PPLNS share was credited: the template is ALIVE and paying. +/// `stale` the pool has moved to another job: DEAD. +/// `busy` too many shares held for this height: ALIVE, back off. +/// `duplicate` this exact (height, coinbase_nonce, block_nonce) was seen: ALIVE. +/// `invalid` this one submission is bad (bad nonce, above share target, or an +/// unknown worker): ALIVE. +fn pool_kind_verdict(ret: i64, kind: &str) -> Option { + match kind { + "block" if ret == 0 => Some(SubmitVerdict::Accepted), + "share" if ret == 0 => Some(SubmitVerdict::ShareCredited), + "stale" => Some(SubmitVerdict::TemplateGone), + "busy" => Some(SubmitVerdict::UpstreamBusy), + "duplicate" => Some(SubmitVerdict::DuplicateSubmission), + "invalid" => Some(SubmitVerdict::SubmissionRejected), + _ => None, + } +} + +/// Classify one `/submit/miner/success` reply, returning the verdict and the +/// upstream's reason text alongside it. +/// +/// `None` means the body carries no numeric `ret` at all (proxy error page, +/// truncated or non-JSON body). That is NOT an upstream decision, so the caller +/// retries instead of treating a front-end hiccup as a rejection. +fn classify_submit_response(body: &str) -> Option<(SubmitVerdict, String)> { + let parsed = serde_json::from_str::(body).ok()?; + let ret = parsed["ret"].as_i64()?; + // The fullnode answers with `err`; the pool relay wraps an unrecognized + // upstream body as `msg`. Both carry the reason. + let err = parsed["err"] + .as_str() + .or_else(|| parsed["msg"].as_str()) + .unwrap_or("") + .to_string(); + // A pool states its meaning in `kind` and sends no error text, so this is read + // first: falling through to the text rules classified every one of its answers + // as a rejection of one submission, which re-opened a template the pool had + // already declared stale and let every later winner pay for another round trip. + let kind = parsed["kind"].as_str().unwrap_or(""); + if let Some(verdict) = pool_kind_verdict(ret, kind) { + let reason = if err.is_empty() { + format!("kind \"{kind}\"") + } else { + format!("kind \"{kind}\": {err}") + }; + return Some((verdict, reason)); + } + if ret == 0 { + // An unmarked success keeps the meaning it has always had here: the + // fullnode's own success carries `"mining":"success"`, and older or + // proxied fullnode builds answer a bare `{"ret":0}` for the same thing. + // A wrong settle here is self-healing anyway, because `maintain` re-opens + // a settled template the upstream keeps serving. + return Some((SubmitVerdict::Accepted, err)); + } + let verdict = if node_error_ends_template(&err) { + SubmitVerdict::TemplateGone + } else { + SubmitVerdict::SubmissionRejected + }; + Some((verdict, err)) +} + +/// Submit one winning result and report what the node said about it. +/// +/// Same wire behaviour as `poworker::push_block_mining_success` (same URL, same +/// five attempts, same operator banners); the difference is the return value. +/// The gate below cannot be driven safely without knowing a definitive node +/// verdict from a transport failure, and that function returns `()`. +fn submit_block_mining_success(cnf: &PoWorkConf, success: &BlockMiningResult) -> SubmitVerdict { + let urlapi_success = format!( + "http://{}/submit/miner/success?height={}&block_nonce={}&coinbase_nonce={}&t={}{}", + &cnf.rpcaddr, + success.height, + success.head_nonce, + success.coinbase_nonce.to_hex(), + sys::curtimes(), + cnf.worker_param() + ); + // Submitting the winning block is the entire payoff of solo mining, and the + // result was already drained from the channel, so a single transient network + // error must not silently lose it. Retry transport failures with backoff, and + // only claim SUCCESS once the node confirms acceptance (ret == 0). + const MAX_SUBMIT_ATTEMPTS: u32 = 5; + let mut verdict = SubmitVerdict::NoNodeVerdict; + let mut last = String::new(); + for attempt in 1..=MAX_SUBMIT_ATTEMPTS { + match crate::rpc_http::get_text(&super::HTTP_CLIENT, &urlapi_success, &cnf.api_token, None) + { + Ok(body) => { + last = body.clone(); + match classify_submit_response(&body) { + Some((node_verdict, err)) => { + verdict = node_verdict; + match node_verdict { + // A taken block and a credited pool share are both + // normal PAID outcomes, not refusals. + SubmitVerdict::Accepted | SubmitVerdict::ShareCredited => {} + // Deterministic upstream rejection (stale template, + // duplicate, busy, invalid): resending this exact + // submission cannot help, so stop attempting. + _ => { + println!( + "[submit] node rejected height {}: {}", + success.height, err + ); + } + } + break; + } + None => { + // HTTP 200 but no parseable `ret` (proxy/load-balancer + // error page, truncated or non-JSON body). This is NOT a + // node decision, so treat it as transient and retry: a + // winning block is not discarded on a front-end hiccup. + let snippet: String = body.chars().take(120).collect(); + println!( + "[submit] attempt {}/{} unrecognized response, retrying: {}", + attempt, MAX_SUBMIT_ATTEMPTS, snippet + ); + if attempt < MAX_SUBMIT_ATTEMPTS { + std::thread::sleep(Duration::from_millis(500u64 * attempt as u64)); + } + } + } + } + Err(e) => { + last = format!("transport error: {e}"); + println!( + "[submit] attempt {}/{} failed: {e}", + attempt, MAX_SUBMIT_ATTEMPTS + ); + if attempt < MAX_SUBMIT_ATTEMPTS { + std::thread::sleep(Duration::from_millis(500u64 * attempt as u64)); + } + } + } + } + println!("{} {}", &urlapi_success, last); + match verdict { + SubmitVerdict::Accepted => { + println!( + "\n\n████████████████ [MINING SUCCESS] Find a block height {},\n██ hash {} to submit.", + success.height, + success.result_hash.to_hex() + ); + } + // Routine pool traffic, not a rig fault: a credited share is income and a + // busy or duplicate answer is the pool's own bookkeeping. Raising the + // "check the node/connection" alarm for these would cry wolf on every + // single share a pool miner earns. + SubmitVerdict::ShareCredited => { + println!( + "[submit] pool credited a share at height {} (hash {}).", + success.height, + success.result_hash.to_hex() + ); + } + SubmitVerdict::UpstreamBusy => { + println!( + "[submit] pool is busy at height {}: pausing submits for it for {}s.", + success.height, + POOL_BUSY_COOLDOWN.as_secs() + ); + } + SubmitVerdict::DuplicateSubmission => { + println!( + "[submit] height {} was already credited for this nonce (hash {}).", + success.height, + success.result_hash.to_hex() + ); + } + _ => { + println!( + "\n\n████████████████ [MINING SUBMIT FAILED] block height {} was NOT confirmed accepted\n██ after {} attempts (hash {}). Check the node/connection.", + success.height, + MAX_SUBMIT_ATTEMPTS, + success.result_hash.to_hex() + ); + } + } + println!("▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔"); + verdict +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +enum TemplateSubmitState { + /// Nothing outstanding: the next winner for this template goes to the node. + NotSubmitted, + /// A winner for this template is with the submit thread right now. + InFlight, + /// The node gave a definitive verdict: nothing more can be won here. + Settled, +} + +struct TemplateGateEntry { + state: TemplateSubmitState, + /// Winners dropped without any HTTP call, because one was already in flight + /// for this template or the node had already settled it. + suppressed: u64, + /// The upstream answered `kind:"share"` for this template, so it CREDITS every + /// winner instead of holding one block. Redundant-winner suppression stops for + /// it: each further share is separately payable work, and dropping one is + /// money the miner earned and never gets. + share_paying: bool, + /// Earliest instant a further winner for this template may be submitted. Set + /// only by a `busy` answer, so an overloaded pool is not hammered by a rig + /// finding hundreds of nonces a second. + cooldown_until: Option, + outcome: &'static str, + /// Template install counter as of the moment this template settled. If the + /// upstream serves the very same job again afterwards it has plainly not + /// consumed it, and the gate re-opens (see `maintain`). + settled_at_install: u64, + /// The single operator line for this template has already been printed. + reported: bool, +} + +impl TemplateGateEntry { + fn new() -> TemplateGateEntry { + TemplateGateEntry { + state: TemplateSubmitState::NotSubmitted, + suppressed: 0, + share_paying: false, + cooldown_until: None, + outcome: SubmitVerdict::NoNodeVerdict.outcome_text(), + settled_at_install: 0, + reported: false, + } + } +} + +/// Per-template submit gate, keyed by (height, parent hash) - the same identity a +/// `BlockMiningResult` already carries. +/// +/// At a low difficulty a fast GPU finds MANY nonces meeting one template's target, +/// but a height holds exactly ONE block: `mint::api::miner_success` drops the +/// pending entry for that height the moment it accepts a block, so every later +/// winner for the same template can only come back as "pending block height N not +/// found". Submitting them anyway costs one blocking HTTP round trip each, which +/// is what filled the submit queue, then the result channel, then blocked the +/// workers and stopped the GPU. +/// +/// So the FIRST winner for a template is always submitted (that is the payout, +/// and it must never regress), while later winners for the SAME template are +/// dropped only once the node has taken it or proved it gone. +/// +/// None of that holds against a POOL, which credits a PPLNS share for EVERY +/// winner instead of holding one block, so the gate stops suppressing as soon as +/// the upstream proves it pays that way. +#[derive(Default)] +struct SubmitGate { + templates: HashMap<(u64, Vec), TemplateGateEntry>, + /// The upstream has credited a share at least once, so it is a pool and it + /// pays per winner. Latched for the whole session and inherited by every new + /// template: a fresh template starting out "solo" again would suppress the + /// burst of shares found before its first verdict comes back, and those are + /// earned income. The rpcaddr cannot change while the miner runs, and a + /// definitive verdict still settles any template, so an upstream that later + /// behaves like a fullnode is still gated from its first answer onwards. + upstream_credits_shares: bool, +} + +type SubmitGateHandle = Arc>; + +fn template_key(res: &BlockMiningResult) -> (u64, Vec) { + (res.height, res.prevhash.clone()) +} + +impl SubmitGate { + /// True when this winner must go to the upstream. False means it is a + /// redundant winner for a template already in flight or already settled, so + /// dropping it costs nothing; it is counted for the one summary line. + fn admit(&mut self, res: &BlockMiningResult) -> bool { + self.admit_at(res, Instant::now()) + } + + /// `admit` with the clock passed in, so the busy back-off is testable. + fn admit_at(&mut self, res: &BlockMiningResult, now: Instant) -> bool { + let pays_per_winner = self.upstream_credits_shares; + let entry = self + .templates + .entry(template_key(res)) + .or_insert_with(TemplateGateEntry::new); + if entry.state == TemplateSubmitState::Settled { + entry.suppressed = entry.suppressed.saturating_add(1); + return false; + } + if entry.cooldown_until.is_some_and(|until| now < until) { + // The upstream said it is overloaded moments ago. Submitting again + // right now would only be refused again and add to the load. + entry.suppressed = entry.suppressed.saturating_add(1); + return false; + } + if entry.share_paying || pays_per_winner { + // This upstream PAYS per winner, so there is no such thing as a + // redundant one: a height holds a single block, but a pool credits + // every share against the same template. Suppressing here is the + // miner refusing its own income. + entry.state = TemplateSubmitState::InFlight; + return true; + } + if entry.state == TemplateSubmitState::NotSubmitted { + entry.state = TemplateSubmitState::InFlight; + return true; + } + entry.suppressed = entry.suppressed.saturating_add(1); + false + } + + /// Record what the upstream said about the winner that was in flight. + fn record_verdict( + &mut self, + res: &BlockMiningResult, + verdict: SubmitVerdict, + install_seq: u64, + ) { + self.record_verdict_at(res, verdict, install_seq, Instant::now()); + } + + /// `record_verdict` with the clock passed in, so the busy back-off is testable. + fn record_verdict_at( + &mut self, + res: &BlockMiningResult, + verdict: SubmitVerdict, + install_seq: u64, + now: Instant, + ) { + if verdict.credits_share() { + // Latched, and outside the entry lookup so it survives even if this + // template has already been pruned: an upstream that credited one + // share pays for the rest of them too, and the answer to the NEXT + // winner cannot arrive before that winner has to be admitted. + self.upstream_credits_shares = true; + } + let Some(entry) = self.templates.get_mut(&template_key(res)) else { + // Already pruned: the template is far behind the tip, so there is + // nothing left to gate. + return; + }; + if verdict.credits_share() { + entry.share_paying = true; + } + // Any answer that is not "busy" clears the back-off, so a pool that has + // recovered is used again at once. + entry.cooldown_until = verdict.retry_cooldown().map(|wait| now + wait); + if verdict.ends_template() { + entry.state = TemplateSubmitState::Settled; + entry.outcome = verdict.outcome_text(); + entry.settled_at_install = install_seq; + } else if entry.state != TemplateSubmitState::Settled { + // Nothing proves this template is dead, so re-open it: a network + // hiccup, a credited share, a busy or duplicate answer, or a rejection + // about one submission, must never permanently suppress a height. + // A template that IS settled keeps that outcome: settling is + // definitive, and on a share-paying template a second submission can + // still be in flight when the first one settles it. + entry.state = TemplateSubmitState::NotSubmitted; + entry.outcome = verdict.outcome_text(); + } + } + + /// Re-open a settled template the upstream is STILL serving, print the ONE + /// line per template the operator gets, then forget templates too old to + /// submit anywhere. A line is emitted only once the miner has moved off that + /// template, so the suppressed count in it is the final one. + fn maintain(&mut self, live: Option<&(u64, Vec)>, live_height: u64, install_seq: u64) { + for (key, entry) in self.templates.iter_mut() { + let still_live = + live.is_some_and(|(hei, prevhash)| *hei == key.0 && *prevhash == key.1); + if still_live { + // A fullnode that took our block moves on, so it never serves that + // job again. An upstream that DOES serve it again has not consumed + // it (a pool grading work at one height, or a block reorged back + // out), and our own bookkeeping must not be what silences a live + // height. Re-opening costs at most one submit per template refresh. + if entry.state == TemplateSubmitState::Settled + && install_seq > entry.settled_at_install + { + entry.state = TemplateSubmitState::NotSubmitted; + } + continue; + } + if entry.reported { + continue; + } + // Nothing was dropped for this template, so there is nothing to + // report: the submit path already printed its own outcome. + if entry.suppressed > 0 { + if entry.state == TemplateSubmitState::Settled { + println!( + "\n[Mining] height {} {}, suppressed {} redundant winners.", + key.0, entry.outcome, entry.suppressed + ); + } else { + // Never settled, so these were not provably dead: they were + // held back while a submission was in flight or while the + // upstream asked for a pause. Say so instead of calling them + // redundant. + println!( + "\n[Mining] height {} {}, held back {} further winners.", + key.0, entry.outcome, entry.suppressed + ); + } + } + entry.reported = true; + } + self.prune(live_height); + } + + /// Bound the memory: keep only the last `SUBMIT_GATE_HEIGHT_WINDOW` heights, + /// and never more than `SUBMIT_GATE_MAX_TEMPLATES` entries in total. + fn prune(&mut self, live_height: u64) { + if live_height > 0 { + let floor = live_height.saturating_sub(SUBMIT_GATE_HEIGHT_WINDOW); + self.templates.retain(|(hei, _), _| *hei >= floor); + } + while self.templates.len() > SUBMIT_GATE_MAX_TEMPLATES { + let Some(oldest) = self.templates.keys().min().cloned() else { + break; + }; + self.templates.remove(&oldest); + } + } +} + +/// A poisoned gate must never stop submissions: recover the state and carry on. +fn lock_gate(gate: &SubmitGateHandle) -> MutexGuard<'_, SubmitGate> { + gate.lock().unwrap_or_else(|e| e.into_inner()) +} + +fn admit_for_submit(gate: &SubmitGateHandle, win: &BlockMiningResult) -> bool { + lock_gate(gate).admit(win) +} + +fn record_submit_verdict(gate: &SubmitGateHandle, win: &BlockMiningResult, verdict: SubmitVerdict) { + let install_seq = MINING_TEMPLATE_INSTALL_SEQ.load(Relaxed); + lock_gate(gate).record_verdict(win, verdict, install_seq); +} + +/// Re-open, retire and prune templates. Runs on the result thread every drain +/// tick, so the gate cannot grow while mining continues. +fn maintain_submit_gate(gate: &SubmitGateHandle) { + let live = live_template_identity(); + let live_height = MINING_BLOCK_HEIGHT.load(Relaxed); + let install_seq = MINING_TEMPLATE_INSTALL_SEQ.load(Relaxed); + lock_gate(gate).maintain(live.as_ref(), live_height, install_seq); +} + +/// Queue a winning result for the submit thread. +/// +/// With the gate above at most one winner per template is ever outstanding, so +/// the queue holds a handful of entries and can no longer fill in the redundant +/// winner case that used to jam the whole pipeline. The inline submit is kept +/// purely as a last resort (the submit thread is gone, i.e. shutdown), because a +/// dropped winner is lost money. +fn queue_block_mining_success( + cnf: &PoWorkConf, + submit_tx: &mpsc::SyncSender>, + gate: &SubmitGateHandle, + win: &Arc, +) { + match submit_tx.try_send(win.clone()) { + Ok(()) => {} + Err(mpsc::TrySendError::Full(win)) => { + eprintln!( + "[Mining] Submit queue full, submitting height {} inline.", + win.height + ); + submit_inline_without_verdict(cnf, gate, &win); + } + Err(mpsc::TrySendError::Disconnected(win)) => { + submit_inline_without_verdict(cnf, gate, &win); + } + } +} + +/// Last-resort submit on the calling thread. It uses the plain submitter, which +/// reports nothing back, so the template is re-opened instead of being left in +/// flight: no height may stay suppressed on the strength of a submit whose +/// outcome we never learned. +fn submit_inline_without_verdict( + cnf: &PoWorkConf, + gate: &SubmitGateHandle, + win: &Arc, +) { + super::push_block_mining_success(cnf, win); + record_submit_verdict(gate, win, SubmitVerdict::NoNodeVerdict); +} + pub(crate) fn start_block_mining_workers( cnf: &PoWorkConf, stop_flag: Option>, ) -> bool { - let (res_tx, res_rx) = mpsc::channel(); - let miner_backends = build_miner_backends(cnf); + let (res_tx, res_rx) = mpsc::sync_channel(RESULT_CHANNEL_CAPACITY); + let miner_backends = build_miner_backends(cnf, &stop_flag); if miner_backends.is_empty() { return false; } @@ -209,11 +1003,40 @@ pub(crate) fn start_block_mining_workers( ); } + // Submitting a winner is a blocking HTTP round-trip (with retries), so it runs + // on a dedicated thread: doing it inline would stall the 123ms result drain and + // back the result channel up whenever a pool serves an easy share target. + let (submit_tx, submit_rx) = + mpsc::sync_channel::>(SUBMIT_QUEUE_CAPACITY); + // One gate shared by the result thread (which admits winners) and the submit + // thread (which reports the node verdict back). + let submit_gate: SubmitGateHandle = Arc::new(Mutex::new(SubmitGate::default())); + let cnf_submit = cnf.clone(); + let submit_thread_guard = cnf.runtime.track_mining_thread(); + let gate_submit = submit_gate.clone(); + spawn(move || { + let _submit_thread_guard = submit_thread_guard; + while let Ok(win) = submit_rx.recv() { + // A panic inside the submit must not leave the template stuck "in + // flight", which would suppress every later winner for it, so the + // verdict starts as "no node verdict" (re-open) and is recorded + // outside the firewall. + let mut verdict = SubmitVerdict::NoNodeVerdict; + guard_mining_iteration("block submit thread", || { + verdict = submit_block_mining_success(&cnf_submit, &win); + }); + record_submit_verdict(&gate_submit, &win, verdict); + } + }); + let cnf1 = cnf.clone(); let worker_qty = miner_backends.len(); let stop_flag_res = stop_flag.clone(); let result_thread_guard = cnf.runtime.track_mining_thread(); + let gate_results = submit_gate; spawn(move || { + // submit_tx lives in this thread: when the drain loop returns on shutdown it + // drops, which is what tells the submit thread to finish and exit. let _result_thread_guard = result_thread_guard; let rate_clock = Instant::now(); let mut rate_tracker = HashrateTracker::default(); @@ -221,17 +1044,34 @@ pub(crate) fn start_block_mining_workers( let mut rstx = res_rx; loop { if super::should_stop(&stop_flag_res) { + // Final drain: do not abandon target-meeting winners still in the + // channel just because shutdown was requested. This is the only + // other body on this thread that can submit, so it gets the same + // panic firewall as the drain loop below. + guard_mining_iteration("block shutdown drain", || { + drain_winners_for_shutdown(&mut rstx, &submit_tx, &gate_results); + }); return; } let now_ms = rate_clock.elapsed().as_millis() as u64; - deal_block_mining_results( - &cnf1, - &mut most_hash, - &mut rstx, - worker_qty, - &mut rate_tracker, - now_ms, - ); + // A panic here must never end the thread: this is the only path that + // submits winning blocks, so losing it means silent total payout loss. + guard_mining_iteration("block result thread", || { + deal_block_mining_results( + &cnf1, + &mut most_hash, + &mut rstx, + worker_qty, + &mut rate_tracker, + now_ms, + &submit_tx, + &gate_results, + ); + // Retire settled templates (one operator line each) and keep the + // gate bounded. This runs every tick, so it also happens while no + // result at all is coming in. + maintain_submit_gate(&gate_results); + }); std::thread::sleep(Duration::from_millis(123)); } }); @@ -247,13 +1087,15 @@ pub(crate) fn start_block_mining_workers( if super::should_stop(&stop_flag_miner) { return; } - run_block_mining_item( - &cnf2, - thrid, - rstx.clone(), - backend.clone(), - &stop_flag_miner, - ); + guard_mining_iteration("block mining worker", || { + run_block_mining_item( + &cnf2, + thrid, + rstx.clone(), + backend.clone(), + &stop_flag_miner, + ); + }); std::thread::sleep(Duration::from_millis(9)); } }); @@ -275,11 +1117,29 @@ pub(crate) fn set_pending_block_stuff(height: u64, res: JV) -> Result<(), String let target_array: [u8; HASH_WIDTH] = target_bytes .try_into() .map_err(|v: Vec| format!("invalid target_hash length: {}", v.len()))?; + // Defense in depth: an upstream (pool or node) that hands us an absurd target + // would install an unmineable template and drive the display math into + // pathological ranges. Reject it so the pull loop simply retries. + let leading_zero_bytes = target_array.iter().take_while(|byte| **byte == 0).count(); + if leading_zero_bytes >= MAX_TARGET_LEADING_ZERO_BYTES { + return Err(format!( + "implausible target_hash with {leading_zero_bytes} leading zero bytes" + )); + } let target_hash = Hash::from(target_array); let intro_bytes = decode("block_intro")?; let block_intro = BlockIntro::build(&intro_bytes).map_err(|e| format!("invalid block_intro: {e}"))?; + // JSON height drives x16rs repeat count; consensus uses the height embedded in + // the intro. Reject mismatches so we never mine with the wrong repeat or + // submit under a height the header does not claim. + let intro_height = block_intro.height().uint(); + if height != intro_height { + return Err(format!( + "pending height {height} does not match block_intro height {intro_height}" + )); + } let coinbase_bytes = decode("coinbase_body")?; let coinbase_tx = TransactionCoinbase::build(&coinbase_bytes) .map_err(|e| format!("invalid coinbase_body: {e}"))?; @@ -300,20 +1160,122 @@ pub(crate) fn set_pending_block_stuff(height: u64, res: JV) -> Result<(), String } let new_stuff = BlockMiningStuff { height, + epoch: 0, target_hash, block_intro, coinbase_tx, mkrl_list, }; + install_block_mining_stuff(new_stuff) +} + +/// Install a freshly fetched template. +/// +/// The fullnode re-serializes `block_intro` on EVERY `/query/miner/pending` +/// request: it increments the served coinbase nonce, recomputes the merkle root +/// and re-serializes the intro, and the merkle root lives INSIDE the intro. So +/// the raw bytes differ on every poll even when the tip, the parent and the +/// transaction set are all unchanged. The miner replaces that merkle root itself +/// from its own coinbase nonce before hashing, so it is not part of the job at +/// all. +/// +/// A job is therefore identified by (height, parent hash). Only a change of that +/// pair means workers must give up what they are doing, so only that bumps the +/// epoch. The refreshed bytes are installed either way. Bumping on every poll +/// used to void one finished batch per worker per poll, winners included. +fn install_block_mining_stuff(mut new_stuff: BlockMiningStuff) -> Result<(), String> { + let height = new_stuff.height; let mut guard = MINING_BLOCK_STUFF .write() .map_err(|e| format!("mining state lock poisoned: {e}"))?; + let job_changed = + guard.height != height || guard.block_intro.prevhash() != new_stuff.block_intro.prevhash(); + // Keep the epoch inside the template so workers see the same generation the + // bytes belong to. The atomic mirrors it for the cheap loop-exit check. + new_stuff.epoch = if job_changed { + MINING_BLOCK_EPOCH.fetch_add(1, Relaxed).saturating_add(1) + } else { + guard.epoch + }; *guard = new_stuff.into(); MINING_BLOCK_HEIGHT.store(height, Relaxed); + // Every install counts here, re-serializations included: the submit gate needs + // to know the upstream handed us this job again, not whether it changed. + MINING_TEMPLATE_INSTALL_SEQ.fetch_add(1, Relaxed); Ok(()) } -fn build_miner_backends(cnf: &PoWorkConf) -> Vec { +pub(crate) fn current_mining_epoch() -> u64 { + MINING_BLOCK_EPOCH.load(Relaxed) +} + +/// Record whether the upstream is serving stale (no longer refreshed) work. +/// Returns true when the state actually changed, so the caller can log the +/// transition once instead of on every poll. +pub(crate) fn set_upstream_stale(stale: bool) -> bool { + UPSTREAM_STALE.swap(stale, Relaxed) != stale +} + +/// True while the installed template is known to be dead work. +pub(crate) fn upstream_is_stale() -> bool { + UPSTREAM_STALE.load(Relaxed) +} + +/// Seconds to wait BEFORE each GPU init attempt after the first. A driver that is +/// still loading after a boot, a resume, or a TDR reset routinely needs tens of +/// seconds to become enumerable, and treating the very first probe as final turns +/// a self-clearing condition into a permanent outage on an unattended rig. +const GPU_INIT_RETRY_DELAYS_SECS: [u64; 4] = [5, 15, 30, 60]; + +/// How many GPU init attempts to make. Retrying only helps when the requested +/// backend is actually compiled in; a missing feature is deterministic, so waiting +/// two minutes to reach the same answer would only stall startup. +fn gpu_init_attempts(gpu_requested: bool, gpu_backend_compiled_in: bool) -> usize { + if gpu_requested && gpu_backend_compiled_in { + GPU_INIT_RETRY_DELAYS_SECS.len() + 1 + } else { + 1 + } +} + +/// CPU-assist threads belong to ANY GPU rig, not only an OpenCL one. This decision +/// used to live inside the OpenCL arm, so a CUDA rig had no non-GPU backend at +/// all: once the CUDA session-disable latch quarantined the card, the whole rig +/// was left with a single bounded recovery thread and earned nothing. +fn cpu_assist_thread_count(cnf: &PoWorkConf, gpu_backends: usize) -> usize { + if !cnf.cpu_assist || cnf.supervene == 0 || gpu_backends == 0 { + return 0; + } + cnf.efficiency.spawn_supervene(cnf.supervene) as usize +} + +/// Sleep, but wake as soon as a supervisor asks us to stop. Returns true when the +/// wait ended because of a stop request, so startup never holds shutdown. +fn sleep_unless_stopped( + stop_flag: &Option>, + total: Duration, +) -> bool { + let deadline = Instant::now() + total; + loop { + if super::should_stop(stop_flag) { + return true; + } + let now = Instant::now(); + if now >= deadline { + return false; + } + std::thread::sleep( + deadline + .saturating_duration_since(now) + .min(Duration::from_millis(200)), + ); + } +} + +fn build_gpu_backends(cnf: &PoWorkConf) -> Vec { + // Both arms below are cfg-gated, so a build with neither GPU feature enabled + // never mutates this. + #[allow(unused_mut)] let mut backends = Vec::new(); if cnf.usecuda { @@ -393,26 +1355,54 @@ fn build_miner_backends(cnf: &PoWorkConf) -> Vec { "\n[Warn] use_opencl=true but app built without `ocl` feature, fallback to CPU miner." ); } + } - if cnf.cpu_assist && cnf.supervene > 0 && !backends.is_empty() { - let thrnum = cnf.efficiency.spawn_supervene(cnf.supervene) as usize; - println!( - "\n[Start] Create #{} Ryzen CPU assist threads (hybrid GPU+CPU, active={}).", - thrnum, - cnf.runtime.active_cpu_assist.load(Relaxed) - ); - for i in 0..thrnum { - backends.push(MinerBackend::Cpu { - assist_idx: Some(i as u32), - }); - } + backends +} + +fn build_miner_backends( + cnf: &PoWorkConf, + stop_flag: &Option>, +) -> Vec { + let gpu_requested = cnf.usecuda || cnf.useopencl; + let compiled_in = + (cnf.usecuda && cfg!(feature = "cuda")) || (cnf.useopencl && cfg!(feature = "ocl")); + let attempts = gpu_init_attempts(gpu_requested, compiled_in); + let mut backends = Vec::new(); + for attempt in 1..=attempts { + backends = build_gpu_backends(cnf); + if !backends.is_empty() || attempt == attempts { + break; + } + let wait = GPU_INIT_RETRY_DELAYS_SECS[attempt - 1]; + println!( + "\n[Start] GPU not ready yet (attempt {attempt}/{attempts}). Waiting {wait}s and trying again - this is normal shortly after a boot, a resume, or a driver reset." + ); + if sleep_unless_stopped(stop_flag, Duration::from_secs(wait)) { + return Vec::new(); + } + } + + // Any GPU rig gets CPU-assist threads, so a quarantined card still leaves real + // work running instead of taking the whole rig to zero. + let assist_threads = cpu_assist_thread_count(cnf, backends.len()); + if assist_threads > 0 { + println!( + "\n[Start] Create #{} Ryzen CPU assist threads (hybrid GPU+CPU, active={}).", + assist_threads, + cnf.runtime.active_cpu_assist.load(Relaxed) + ); + for i in 0..assist_threads { + backends.push(MinerBackend::Cpu { + assist_idx: Some(i as u32), + }); } } if backends.is_empty() { - if cnf.useopencl { + if gpu_requested { eprintln!( - "[Fatal] OpenCL was requested but no usable GPU backend initialized; refusing silent CPU fallback." + "[Fatal] a GPU miner was requested but no usable GPU backend initialized after {attempts} attempt(s); refusing silent CPU fallback (you would pay for GPU power while mining slowly on the CPU). If the GPU works in other software, wait a minute and press Start again; otherwise check the driver/CUDA runtime, or set the backend to CPU." ); return backends; } @@ -441,7 +1431,18 @@ fn backend_nonce_space(_cnf: &PoWorkConf, backend: &MinerBackend) -> u32 { } #[cfg(feature = "cuda")] MinerBackend::Cuda(res) => { - let wg = res.workgroups.min(_cnf.workgroups); + // Match run_batch: the planned window must reflect the same effective + // work-groups (OOM/error backoff) and thermal cap the batch will use, + // otherwise the nonce accounting overstates what the GPU covered. + let thermal = _cnf + .runtime + .thermal_workgroups_cap() + .unwrap_or(u32::MAX); + let wg = res + .effective_wg() + .min(_cnf.workgroups) + .min(thermal) + .max(1); wg.saturating_mul(x16rs_cuda::DEFAULT_LOCAL_SIZE) .saturating_mul(res.unit_size) .max(1) @@ -461,7 +1462,7 @@ fn next_nonce_space(current: u32, use_secs: f64, is_gpu_backend: bool) -> u32 { fn run_block_mining_item( _cnf: &PoWorkConf, thrid: usize, - result_ch_tx: mpsc::Sender>, + result_ch_tx: mpsc::SyncSender>, backend: MinerBackend, stop_flag: &Option>, ) { @@ -472,8 +1473,26 @@ fn run_block_mining_item( std::thread::sleep(Duration::from_millis(2000)); return; } + // The upstream told us it can no longer refresh this template. Hashing it + // cannot win a block or earn a share, so idle until fresh work arrives + // instead of paying for power that buys nothing. + if upstream_is_stale() { + std::thread::sleep(Duration::from_millis(1000)); + return; + } - let mining_hei = MINING_BLOCK_HEIGHT.load(Relaxed); + // Height, epoch and the bytes to mine all come from ONE snapshot. Reading them + // separately let a template install land in between, which labelled the batch + // with a generation it was not mined under. + let stuff = match MINING_BLOCK_STUFF.read() { + Ok(stuff) => stuff.clone(), + Err(e) => { + eprintln!("[Mining] Block state lock failed: {e}"); + return; + } + }; + let mining_hei = stuff.height; + let mining_epoch = stuff.epoch; if mining_hei == 0 { std::thread::sleep(Duration::from_millis(111)); return; @@ -506,14 +1525,11 @@ fn run_block_mining_item( MinerBackend::Cuda(_) => true, _ => false, }; - let stuff = match MINING_BLOCK_STUFF.read() { - Ok(stuff) => stuff.clone(), - Err(e) => { - eprintln!("[Mining] Block state lock failed: {e}"); - return; - } - }; - let height = stuff.height; + let height = mining_hei; + // Fixed for the life of this template: solo mining resolves to None here and + // never touches the GPU share path at all. + let share_target = pool_share_target(_cnf, &stuff); + let prevhash = stuff.block_intro.prevhash().to_vec(); let mut coinbase_tx = stuff.coinbase_tx.clone(); coinbase_tx.set_nonce(coinbase_nonce); let mut block_intro = stuff.block_intro.clone(); @@ -521,8 +1537,15 @@ fn run_block_mining_item( coinbase_tx.hash(), &stuff.mkrl_list, )); + // The header bytes are invariant across batches at this height (the kernel + // varies only the nonce field), so serialize ONCE and clone per batch instead + // of re-serializing every iteration. + let block_intro_bin = block_intro.serialize(); loop { - if super::should_stop(stop_flag) || _cnf.runtime.thermal_pause_active() { + if super::should_stop(stop_flag) + || _cnf.runtime.thermal_pause_active() + || upstream_is_stale() + { return; } if nonce_start >= nonce_limit { @@ -536,17 +1559,17 @@ fn run_block_mining_item( let remain = nonce_limit.saturating_sub(nonce_start); let current_nonce_space = nonce_space.min(remain).max(1); let ctn = Instant::now(); - let block_intro_bin = block_intro.serialize(); let batch_ctx = BatchCtx { height, - block_intro: block_intro_bin, + block_intro: block_intro_bin.clone(), nonce_start, nonce_space: current_nonce_space, configured_wg: _cnf.workgroups, localsize: _cnf.localsize, unitsize: _cnf.unitsize, thermal_wg_cap: _cnf.runtime.thermal_workgroups_cap(), + share_target, }; let cpu_mine = &do_group_block_mining; let batch = match &backend { @@ -575,10 +1598,17 @@ fn run_block_mining_item( let gpu_ns = batch.gpu_nonce_space; let cpu_ns = batch.cpu_nonce_space; + // A FINISHED batch is never thrown away, whatever the template did while it + // ran. Those hashes were really computed: the sample keeps the hashrate + // display honest, and any nonce that met the target is a payout the node + // still accepts at the height it was mined for. Stopping the loop is the + // job-switch response (see the re-check at the end of this loop); dropping + // the result is not. let use_secs = ctn.elapsed().as_secs_f64(); let mlres = BlockMiningResult { worker_id: thrid, height, + prevhash: prevhash.clone(), nonce_start, nonce_space: current_nonce_space, gpu_nonce_space: gpu_ns, @@ -590,24 +1620,100 @@ fn run_block_mining_item( use_secs, is_gpu: is_gpu_backend, }; - if result_ch_tx.send(mlres.into()).is_err() { + if !send_mining_result(&result_ch_tx, mlres.into()) { return; } + // Against a pool every nonce under the share target is separately + // payable, so each one gets its own submission. Reporting only the batch + // best made GPU credit track batch cadence instead of hashrate. + // + // These carry NO nonce accounting and no elapsed time: the head result + // above already counted the whole batch once, and `record_sample` + // returns early on a zero duration, so the hashrate display cannot be + // inflated by counting one batch once per share it found. + for share in &batch.shares { + let share_res = BlockMiningResult { + worker_id: thrid, + height, + prevhash: prevhash.clone(), + nonce_start, + nonce_space: 0, + gpu_nonce_space: 0, + cpu_nonce_space: 0, + head_nonce: share.nonce, + coinbase_nonce: coinbase_nonce.to_vec(), + result_hash: share.hash.to_vec(), + target_hash: stuff.target_hash.to_vec(), + use_secs: 0.0, + is_gpu: is_gpu_backend, + }; + if !send_mining_result(&result_ch_tx, share_res.into()) { + return; + } + } + report_share_undersampling(batch.share_overflow); + nonce_space = next_nonce_space(current_nonce_space, use_secs, is_gpu_backend); - let Some(nst) = nonce_start.checked_add(current_nonce_space) else { + // Advance by the nonces actually mined, not the planned window. After a GPU + // error/OOM the bounded recovery only covers a prefix of the window, so + // skipping the whole planned span would leave large unmined nonce holes. + // Advancing by the covered count keeps coverage contiguous (the next batch + // re-tries the remainder). For a normal batch the covered count equals the + // planned window, so this is a no-op there. + let mined = gpu_ns.saturating_add(cpu_ns); + let advance = if mined > 0 { + mined.min(current_nonce_space) + } else { + current_nonce_space + }; + let Some(nst) = nonce_start.checked_add(advance) else { return; }; nonce_start = nst; + // Any height change (including a reorg to a lower tip) or a real job change + // means this template is done, so stop starting NEW batches and let the next + // outer loop iteration pick the fresh one up. The batch that just finished + // was already sent: a job switch ends the loop, it never voids mined work. let check_hei = MINING_BLOCK_HEIGHT.load(Relaxed); - if check_hei > mining_hei { + if check_hei != mining_hei || current_mining_epoch() != mining_epoch { return; } } } +/// On shutdown, still submit any target-meeting results already in the channel. +/// +/// Hand-off only: `queue_block_mining_success` falls back to a blocking inline +/// submit (up to five attempts at the 30 s RPC timeout) when the queue is full, +/// and holding shutdown on network I/O would look like a hung miner. The submit +/// thread drains whatever is queued before it exits, so a successful `try_send` +/// is still a real submission. The per-template gate applies here too, so a +/// backlog of redundant winners cannot turn shutdown into hundreds of doomed +/// round trips. +fn drain_winners_for_shutdown( + result_ch_rx: &mut mpsc::Receiver>, + submit_tx: &mpsc::SyncSender>, + gate: &SubmitGateHandle, +) { + let live_template = live_template_identity(); + while let Ok(res) = result_ch_rx.try_recv() { + if result_meets_target(&res) + && !result_is_orphaned(&res, live_template.as_ref()) + && admit_for_submit(gate, &res) + { + if let Err(e) = submit_tx.try_send(res.clone()) { + eprintln!( + "[Mining] Shutdown could not queue a winning result at height {}: {e}", + res.height + ); + } + } + } +} + pub(crate) fn do_group_block_mining( height: u64, mut block_intro: Vec, @@ -635,31 +1741,51 @@ fn deal_block_mining_results( worker_qty: usize, rate_tracker: &mut HashrateTracker, now_ms: u64, + submit_tx: &mpsc::SyncSender>, + gate: &SubmitGateHandle, ) { let vene = worker_qty.max(1) as u32; + let drain_cap = (vene as usize * 4).max(RESULT_DRAIN_MIN_PER_TICK); let mut deal_hei = 0u64; let mut most = Arc::new(BlockMiningResult::new()); let mut total_nonce_space = 0u64; let mut gpu_nonce_space = 0u64; let mut cpu_nonce_space = 0u64; let mut recv_count = 0; + // Every result that independently meets its own target is real money: a winning + // block when solo, a creditable PPLNS share when pooled. Draining keeps only the + // single strongest hash for stats, so collect ALL of them separately, otherwise + // every winner but one in this drain window would be silently lost. + let mut winners: Vec> = Vec::new(); + let live_template = live_template_identity(); while let Ok(res) = result_ch_rx.try_recv() { deal_hei = res.height; - total_nonce_space += res.nonce_space as u64; + // Prefer actual mined counts when recovery mined only a prefix of the plan. if res.gpu_nonce_space > 0 || res.cpu_nonce_space > 0 { + total_nonce_space += res.gpu_nonce_space as u64 + res.cpu_nonce_space as u64; gpu_nonce_space += res.gpu_nonce_space as u64; cpu_nonce_space += res.cpu_nonce_space as u64; } else if res.is_gpu { + total_nonce_space += res.nonce_space as u64; gpu_nonce_space += res.nonce_space as u64; } else { + total_nonce_space += res.nonce_space as u64; cpu_nonce_space += res.nonce_space as u64; } rate_tracker.record_result(&res, now_ms); if hash_more_power(&res.result_hash, &most.result_hash) { most = res.clone(); } + // Submit everything that met its own target. The height is NOT a filter: + // the node looks the template up by the submitted height and keeps several + // recent ones, so a solution found just before the tip advanced is still + // money. Only a same-height reorg (a different parent at this very height) + // is genuinely unreconstructable, and only that is dropped. + if result_meets_target(&res) && !result_is_orphaned(&res, live_template.as_ref()) { + record_winner(&mut winners, &res); + } recv_count += 1; - if recv_count >= vene as usize * 4 { + if recv_count >= drain_cap { break; } } @@ -695,7 +1821,7 @@ fn deal_block_mining_results( if should_pause_for_profit(&cnf.efficiency, hac1day, &cnf.gpu_profile, active_cpu) { cnf.runtime.paused_unprofitable.store(true, Relaxed); println!( - "\n[efficiency] Mining paused — estimated cost exceeds HAC revenue. Set pause_if_unprofitable=false or lower power draw." + "\n[efficiency] Mining paused: estimated cost exceeds HAC revenue. Set pause_if_unprofitable=false or lower power draw." ); } else { cnf.runtime.paused_unprofitable.store(false, Relaxed); @@ -736,8 +1862,24 @@ fn deal_block_mining_results( "", &cnf.efficiency.stats_file, ); - if cnf.debug == 1 || hash_more_power(&most.result_hash, &most.target_hash) { - super::push_block_mining_success(cnf, &most); + if !winners.is_empty() { + // Submit every distinct winner the node could still take. The gate drops + // only the ones it can prove are dead: a second winner for a template the + // node has already settled cannot be accepted at any price, because a + // height holds exactly one block. Everything else still goes out. + for w in &winners { + if !admit_for_submit(gate, w) { + continue; + } + queue_block_mining_success(cnf, submit_tx, gate, w); + } + } else if cnf.debug == 1 { + // Debug mode exercises the submit path even without a genuine winner. It + // goes through the same gate, so it submits once per template instead of + // once per drain tick. + if admit_for_submit(gate, &most) { + queue_block_mining_success(cnf, submit_tx, gate, &most); + } } may_print_turn_to_nex_block_mining(deal_hei, Some(most_hash)); } @@ -767,6 +1909,224 @@ pub(crate) fn may_print_turn_to_nex_block_mining(curr_hei: u64, most_hash: Optio mod tests { use super::*; + /// The installed template lives in process-global statics, so every test that + /// installs one has to run alone. + fn mining_state_guard() -> std::sync::MutexGuard<'static, ()> { + static LOCK: std::sync::Mutex<()> = std::sync::Mutex::new(()); + LOCK.lock().unwrap_or_else(|e| e.into_inner()) + } + + /// One `/query/miner/pending` body. `mrklroot` is the field the fullnode + /// rewrites on every single request, and the one the miner replaces itself. + fn pending_template_json(height: u64, prevhash: u8, mrklroot: u8) -> JV { + let mut intro = BlockIntro::default(); + intro.head.height = BlockHeight::from(height); + intro.head.prevhash = Hash::from([prevhash; HASH_WIDTH]); + intro.head.mrklroot = Hash::from([mrklroot; HASH_WIDTH]); + serde_json::json!({ + "height": height, + "target_hash": hex::encode([0x0fu8; HASH_WIDTH]), + "block_intro": hex::encode(intro.serialize()), + "coinbase_body": hex::encode(TransactionCoinbase::default().serialize()), + "mkrl_modify_list": Vec::::new(), + }) + } + + fn installed_mrklroot() -> Vec { + MINING_BLOCK_STUFF + .read() + .unwrap() + .block_intro + .mrklroot() + .to_vec() + } + + #[test] + fn a_reserialized_template_is_installed_without_voiding_in_flight_work() { + let _guard = mining_state_guard(); + // The fullnode increments the served coinbase nonce and recomputes the + // merkle root on EVERY pending request, and the merkle root sits inside the + // serialized intro. Calling that a new job bumped the epoch on every poll, + // and every bump destroyed one finished batch per worker, winners included. + set_pending_block_stuff(41, pending_template_json(41, 0x11, 0xa1)).unwrap(); + let installed_epoch = current_mining_epoch(); + + set_pending_block_stuff(41, pending_template_json(41, 0x11, 0xa2)).unwrap(); + assert_eq!( + current_mining_epoch(), + installed_epoch, + "a merely re-serialized template must not count as a new job" + ); + assert_eq!( + installed_mrklroot(), + vec![0xa2u8; HASH_WIDTH], + "the refreshed template must still be installed" + ); + + // A same-height reorg replaces the parent: that IS a new job. + set_pending_block_stuff(41, pending_template_json(41, 0x22, 0xa3)).unwrap(); + assert_eq!(current_mining_epoch(), installed_epoch + 1); + + // So is the tip advancing. + set_pending_block_stuff(42, pending_template_json(42, 0x33, 0xa4)).unwrap(); + assert_eq!(current_mining_epoch(), installed_epoch + 2); + assert_eq!(current_mining_height(), 42); + } + + #[test] + fn a_cuda_rig_also_gets_cpu_assist_threads() { + // The assist block used to live inside the OpenCL arm, so a CUDA rig had no + // non-GPU backend at all: when the session-disable latch quarantined the + // card, the whole rig produced nothing. + let mut cnf = PoWorkConf::test_defaults("127.0.0.1:1".to_string(), 2, 16); + cnf.cpu_assist = true; + cnf.usecuda = true; + cnf.useopencl = false; + assert!(cpu_assist_thread_count(&cnf, 1) > 0); + cnf.usecuda = false; + cnf.useopencl = true; + assert!(cpu_assist_thread_count(&cnf, 1) > 0); + + // No GPU backend means the CPU-only path below owns the thread count, and + // an operator who turned the assist off still gets none. + assert_eq!(cpu_assist_thread_count(&cnf, 0), 0); + cnf.cpu_assist = false; + assert_eq!(cpu_assist_thread_count(&cnf, 1), 0); + cnf.cpu_assist = true; + cnf.supervene = 0; + assert_eq!(cpu_assist_thread_count(&cnf, 1), 0); + } + + #[test] + fn gpu_init_is_retried_before_it_is_called_fatal() { + // A driver still loading after a boot or a TDR reset is self-clearing, so + // one failed probe must not stop an unattended rig for good. + assert_eq!(gpu_init_attempts(true, true), 5); + // Nothing to wait for when the backend is not compiled in, or not wanted. + assert_eq!(gpu_init_attempts(true, false), 1); + assert_eq!(gpu_init_attempts(false, true), 1); + // The waits stay bounded so a genuinely dead GPU is still reported. + assert!(GPU_INIT_RETRY_DELAYS_SECS.iter().sum::() <= 120); + assert!(GPU_INIT_RETRY_DELAYS_SECS.windows(2).all(|w| w[0] < w[1])); + } + + #[test] + fn a_startup_retry_wait_never_holds_shutdown() { + let stop = Some(Arc::new(std::sync::atomic::AtomicBool::new(true))); + let started = Instant::now(); + assert!(sleep_unless_stopped(&stop, Duration::from_secs(60))); + assert!(started.elapsed() < Duration::from_secs(1)); + assert!(!sleep_unless_stopped(&None, Duration::from_millis(1))); + } + + #[test] + fn only_a_same_height_reorg_makes_a_solution_dead_work() { + let mut res = BlockMiningResult::default(); + res.height = 100; + res.prevhash = vec![0x11u8; HASH_WIDTH]; + res.target_hash = vec![0x0f; HASH_WIDTH]; + res.result_hash = vec![0x01; HASH_WIDTH]; + assert!(result_meets_target(&res)); + + // Tip advanced past us: the node looks the template up by the SUBMITTED + // height and keeps several recent heights, so this is still a payout. + assert!(!result_is_orphaned( + &res, + Some(&(101u64, vec![0x99u8; HASH_WIDTH])) + )); + // Same job: obviously live. + assert!(!result_is_orphaned( + &res, + Some(&(100u64, vec![0x11u8; HASH_WIDTH])) + )); + // Same height, different parent: the node evicted our template, so the + // solution cannot be reconstructed there any more. + assert!(result_is_orphaned( + &res, + Some(&(100u64, vec![0x22u8; HASH_WIDTH])) + )); + // Unknown live template must never cost a payout. + assert!(!result_is_orphaned(&res, None)); + } + + #[test] + fn a_winner_mined_just_before_the_tip_advanced_is_still_submitted() { + let _guard = mining_state_guard(); + // Live template is height 501 on parent 0x22; the winner was mined at 500. + set_pending_block_stuff(501, pending_template_json(501, 0x22, 0xb1)).unwrap(); + + let cnf = PoWorkConf::test_defaults("127.0.0.1:1".to_string(), 1, 16); + let (res_tx, mut res_rx) = mpsc::sync_channel::>(4); + let (submit_tx, submit_rx) = mpsc::sync_channel::>(4); + + let mut win = BlockMiningResult::default(); + win.height = 500; + win.prevhash = vec![0x11u8; HASH_WIDTH]; + win.nonce_space = 1; + win.use_secs = 0.5; + win.head_nonce = 77; + win.coinbase_nonce = vec![0x05; HASH_WIDTH]; + win.target_hash = vec![0x0f; HASH_WIDTH]; + win.result_hash = vec![0x01; HASH_WIDTH]; + res_tx.send(Arc::new(win)).unwrap(); + + let mut most_hash = vec![255u8; HASH_WIDTH]; + let mut tracker = HashrateTracker::default(); + deal_block_mining_results( + &cnf, + &mut most_hash, + &mut res_rx, + 1, + &mut tracker, + 1, + &submit_tx, + &test_gate(), + ); + + let submitted = submit_rx + .try_recv() + .expect("a solution found a moment before the tip advanced is still accepted by the node at its own height and must be submitted"); + assert_eq!(submitted.height, 500); + assert_eq!(submitted.head_nonce, 77); + } + + #[test] + fn a_same_height_reorg_winner_is_not_submitted() { + let _guard = mining_state_guard(); + // The node repacked height 600 on a different parent, so our solution can + // no longer be rebuilt from its cached template. + set_pending_block_stuff(600, pending_template_json(600, 0x22, 0xc1)).unwrap(); + + let cnf = PoWorkConf::test_defaults("127.0.0.1:1".to_string(), 1, 16); + let (res_tx, mut res_rx) = mpsc::sync_channel::>(4); + let (submit_tx, submit_rx) = mpsc::sync_channel::>(4); + + let mut orphaned = BlockMiningResult::default(); + orphaned.height = 600; + orphaned.prevhash = vec![0x11u8; HASH_WIDTH]; + orphaned.nonce_space = 1; + orphaned.use_secs = 0.5; + orphaned.head_nonce = 5; + orphaned.coinbase_nonce = vec![0x07; HASH_WIDTH]; + orphaned.target_hash = vec![0x0f; HASH_WIDTH]; + orphaned.result_hash = vec![0x01; HASH_WIDTH]; + res_tx.send(Arc::new(orphaned)).unwrap(); + + let mut most_hash = vec![255u8; HASH_WIDTH]; + let mut tracker = HashrateTracker::default(); + deal_block_mining_results( + &cnf, + &mut most_hash, + &mut res_rx, + 1, + &mut tracker, + 1, + &submit_tx, + &test_gate(), + ); + assert!(submit_rx.try_recv().is_err()); + } + #[test] fn malformed_pending_block_is_rejected_without_panicking() { assert!(set_pending_block_stuff(1, serde_json::json!({})).is_err()); @@ -774,6 +2134,125 @@ mod tests { assert!(set_pending_block_stuff(1, invalid_hex).is_err()); } + fn test_gate() -> SubmitGateHandle { + Arc::new(Mutex::new(SubmitGate::default())) + } + + /// A winner for one template identity (height, parent hash). + fn winner_for_template(height: u64, prevhash: u8, head_nonce: u32) -> Arc { + let mut res = BlockMiningResult::default(); + res.height = height; + res.prevhash = vec![prevhash; HASH_WIDTH]; + res.head_nonce = head_nonce; + res.nonce_space = 1; + res.use_secs = 0.5; + res.coinbase_nonce = vec![0x05; HASH_WIDTH]; + res.target_hash = vec![0x0f; HASH_WIDTH]; + res.result_hash = vec![0x01; HASH_WIDTH]; + Arc::new(res) + } + + fn winner_for_test( + height: u64, + coinbase_nonce: u8, + head_nonce: u32, + hash: u8, + ) -> Arc { + let mut res = BlockMiningResult::default(); + res.height = height; + res.head_nonce = head_nonce; + res.coinbase_nonce = vec![coinbase_nonce; HASH_WIDTH]; + res.result_hash = vec![hash; HASH_WIDTH]; + res.target_hash = vec![0x0f; HASH_WIDTH]; + Arc::new(res) + } + + #[test] + fn every_distinct_share_at_one_height_is_kept_for_submission() { + // A pool credits each (height, coinbase_nonce, block_nonce) separately, so + // keeping only the strongest hash per height silently loses earned shares. + let mut winners = Vec::new(); + let strong = winner_for_test(7, 1, 100, 0x01); + let weak = winner_for_test(7, 2, 200, 0x0e); + let same_worker_next_batch = winner_for_test(7, 1, 300, 0x05); + record_winner(&mut winners, &strong); + record_winner(&mut winners, &weak); + record_winner(&mut winners, &same_worker_next_batch); + assert_eq!(winners.len(), 3); + + // Only an exact repeat of the same triple is a true duplicate. + record_winner(&mut winners, &winner_for_test(7, 2, 200, 0x0e)); + assert_eq!(winners.len(), 3); + } + + #[test] + fn a_hash_landing_exactly_on_target_still_counts_as_a_winner() { + let mut res = BlockMiningResult::default(); + res.target_hash = vec![0x0f; HASH_WIDTH]; + res.result_hash = res.target_hash.clone(); + assert!(result_meets_target(&res)); + res.result_hash = vec![0x10; HASH_WIDTH]; + assert!(!result_meets_target(&res)); + res.target_hash = Vec::new(); + assert!(!result_meets_target(&res)); + } + + #[test] + fn a_panicking_iteration_never_ends_the_mining_thread() { + let previous_hook = std::panic::take_hook(); + std::panic::set_hook(Box::new(|_| {})); + let mut iterations = 0u32; + for round in 0..3 { + guard_mining_iteration("test loop", || { + if round == 1 { + panic!("simulated result thread panic"); + } + }); + iterations += 1; + } + std::panic::set_hook(previous_hook); + assert_eq!(iterations, 3); + } + + #[test] + fn an_implausible_upstream_target_is_rejected() { + let mut degenerate = [0u8; HASH_WIDTH]; + degenerate[31] = 1; + let stuff = serde_json::json!({"target_hash": hex::encode(degenerate)}); + assert!(set_pending_block_stuff(1, stuff).is_err()); + let all_zero = serde_json::json!({"target_hash": hex::encode([0u8; HASH_WIDTH])}); + assert!(set_pending_block_stuff(1, all_zero).is_err()); + } + + #[test] + fn a_full_result_queue_drops_statistics_but_never_a_winner() { + let (tx, rx) = mpsc::sync_channel::>(1); + let mut losing = BlockMiningResult::new(); + losing.target_hash = vec![0x0f; HASH_WIDTH]; + assert!(send_mining_result(&tx, Arc::new(losing.clone()))); + // Queue is full now: a statistics-only batch is dropped, not blocked. + assert!(send_mining_result(&tx, Arc::new(losing))); + assert_eq!(rx.try_recv().map(|r| r.height), Ok(0)); + + let winner = winner_for_test(9, 1, 5, 0x01); + assert!(send_mining_result(&tx, winner)); + assert_eq!(rx.try_recv().map(|r| r.height), Ok(9)); + drop(rx); + assert!(!send_mining_result(&tx, winner_for_test(9, 1, 6, 0x01))); + } + + #[test] + fn stale_upstream_work_is_flagged_and_reported_once_per_transition() { + assert!(!upstream_is_stale()); + assert!(set_upstream_stale(true)); + assert!(upstream_is_stale()); + // Repeating the same state is not a transition, so the operator gets one + // message per outage instead of one per poll. + assert!(!set_upstream_stale(true)); + assert!(set_upstream_stale(false)); + assert!(!upstream_is_stale()); + } + #[test] fn gpu_nonce_window_never_expands_into_a_cpu_tail() { let gpu_window = 64 * 256 * 64; @@ -814,4 +2293,650 @@ mod tests { assert!(tracker.totals(WORKER_RATE_STALE_MS).total_hps() > 0.0); assert_eq!(tracker.totals(WORKER_RATE_STALE_MS + 1).total_hps(), 0.0); } + + #[test] + fn only_the_first_winner_of_a_template_is_ever_submitted() { + let _guard = mining_state_guard(); + // Reproduced on real hardware: at a low difficulty one GPU finds hundreds + // of valid nonces per template, but a height holds exactly ONE block. Every + // extra submit cost a blocking HTTP round trip, which filled the submit + // queue, then the result channel, then blocked the workers and stopped the + // GPU. Only the first one may leave; the rest are provably dead. + set_pending_block_stuff(700, pending_template_json(700, 0x11, 0xd1)).unwrap(); + let cnf = PoWorkConf::test_defaults("127.0.0.1:1".to_string(), 1, 16); + let (res_tx, mut res_rx) = mpsc::sync_channel::>(16); + let (submit_tx, submit_rx) = mpsc::sync_channel::>(16); + for head_nonce in 0..9u32 { + res_tx + .send(winner_for_template(700, 0x11, head_nonce)) + .unwrap(); + } + + let gate = test_gate(); + let mut most_hash = vec![255u8; HASH_WIDTH]; + let mut tracker = HashrateTracker::default(); + deal_block_mining_results( + &cnf, + &mut most_hash, + &mut res_rx, + 8, + &mut tracker, + 1, + &submit_tx, + &gate, + ); + + assert_eq!( + submit_rx.try_recv().map(|w| w.head_nonce), + Ok(0), + "the first winner for a template is the payout and must always be submitted" + ); + assert!( + submit_rx.try_recv().is_err(), + "a redundant winner for the same template can only come back as 'pending block height N not found'" + ); + assert_eq!( + lock_gate(&gate).templates[&(700u64, vec![0x11u8; HASH_WIDTH])].suppressed, + 8 + ); + } + + #[test] + fn a_same_height_reorg_template_still_gets_its_own_submit() { + let mut gate = SubmitGate::default(); + assert!(gate.admit(&winner_for_template(701, 0x11, 1))); + assert!(!gate.admit(&winner_for_template(701, 0x11, 2))); + // A different parent at the same height is a DIFFERENT template: the node + // holds a different pending block for it, so this is a real chance at the + // height and must never be gated by the first template. + assert!(gate.admit(&winner_for_template(701, 0x22, 3))); + assert!(!gate.admit(&winner_for_template(701, 0x22, 4))); + } + + #[test] + fn a_winner_at_a_new_height_is_never_suppressed() { + let mut gate = SubmitGate::default(); + assert!(gate.admit(&winner_for_template(702, 0x11, 1))); + gate.record_verdict( + &winner_for_template(702, 0x11, 1), + SubmitVerdict::Accepted, + 1, + ); + assert!(!gate.admit(&winner_for_template(702, 0x11, 2))); + assert!(gate.admit(&winner_for_template(703, 0x11, 1))); + } + + #[test] + fn a_template_the_upstream_keeps_serving_is_re_opened() { + // A fullnode that took our block never serves that job again. An upstream + // that hands us the SAME job after settling it has not consumed it (a pool + // grading work at one height, or a block reorged back out), and our own + // bookkeeping must not be what silences a live height. + let mut gate = SubmitGate::default(); + let live = (704u64, vec![0x11u8; HASH_WIDTH]); + let win = winner_for_template(704, 0x11, 1); + assert!(gate.admit(&win)); + gate.record_verdict(&win, SubmitVerdict::Accepted, 7); + assert!(!gate.admit(&winner_for_template(704, 0x11, 2))); + + // Same job, nothing re-served: it stays settled. + gate.maintain(Some(&live), 704, 7); + assert!(!gate.admit(&winner_for_template(704, 0x11, 3))); + + // Served again: one more winner gets its chance, and only one. + gate.maintain(Some(&live), 704, 8); + assert!(gate.admit(&winner_for_template(704, 0x11, 4))); + assert!(!gate.admit(&winner_for_template(704, 0x11, 5))); + } + + #[test] + fn a_transport_failure_re_opens_the_template() { + let mut gate = SubmitGate::default(); + let win = winner_for_template(800, 0x11, 1); + assert!(gate.admit(&win)); + assert!(!gate.admit(&winner_for_template(800, 0x11, 2))); + + // No answer from the node proves nothing about the template. + gate.record_verdict(&win, SubmitVerdict::NoNodeVerdict, 1); + assert!( + gate.admit(&winner_for_template(800, 0x11, 3)), + "a network hiccup must never suppress a height for good" + ); + // Neither does a rejection that is about one submission only. + gate.record_verdict(&win, SubmitVerdict::SubmissionRejected, 1); + assert!(gate.admit(&winner_for_template(800, 0x11, 4))); + + // A definitive verdict does settle it, and keeps it settled. + gate.record_verdict(&win, SubmitVerdict::Accepted, 1); + assert!(!gate.admit(&winner_for_template(800, 0x11, 5))); + gate.record_verdict(&win, SubmitVerdict::NoNodeVerdict, 1); + assert!(!gate.admit(&winner_for_template(800, 0x11, 6))); + } + + #[test] + fn the_submit_gate_stays_bounded_as_the_height_advances() { + let mut gate = SubmitGate::default(); + for height in 1..=500u64 { + let win = winner_for_template(height, 0x11, 1); + gate.admit(&win); + gate.record_verdict(&win, SubmitVerdict::Accepted, height); + gate.maintain(Some(&(height, vec![0x11u8; HASH_WIDTH])), height, height); + assert!(gate.templates.len() as u64 <= SUBMIT_GATE_HEIGHT_WINDOW + 1); + } + + // A reorg storm at one standing height cannot grow it either. + let mut gate = SubmitGate::default(); + for parent in 0..(SUBMIT_GATE_MAX_TEMPLATES as u32 * 3) { + let mut res = BlockMiningResult::default(); + res.height = 900; + res.prevhash = parent.to_be_bytes().to_vec(); + gate.admit(&res); + gate.maintain(Some(&(900, Vec::new())), 900, parent as u64); + } + assert!(gate.templates.len() <= SUBMIT_GATE_MAX_TEMPLATES); + } + + #[test] + fn a_settled_template_is_reported_once_with_its_final_count() { + let mut gate = SubmitGate::default(); + let key = (900u64, vec![0x11u8; HASH_WIDTH]); + let win = winner_for_template(900, 0x11, 1); + assert!(gate.admit(&win)); + for head_nonce in 2..12u32 { + assert!(!gate.admit(&winner_for_template(900, 0x11, head_nonce))); + } + gate.record_verdict(&win, SubmitVerdict::Accepted, 3); + + // Still the live template, so the count is not final yet: no line. + gate.maintain(Some(&key), 900, 3); + assert!(!gate.templates[&key].reported); + + // The miner moved on: exactly one line, carrying the whole count. + gate.maintain(Some(&(901, vec![0x33u8; HASH_WIDTH])), 901, 4); + assert!(gate.templates[&key].reported); + assert_eq!(gate.templates[&key].suppressed, 10); + gate.maintain(Some(&(901, vec![0x33u8; HASH_WIDTH])), 901, 5); + assert!(gate.templates[&key].reported); + } + + #[test] + fn only_a_node_verdict_about_the_template_settles_it() { + let verdict = |body: &str| classify_submit_response(body).map(|(v, _)| v); + assert_eq!( + verdict("{\"ret\":0,\"height\":3,\"mining\":\"success\"}"), + Some(SubmitVerdict::Accepted) + ); + assert_eq!( + verdict("{\"ret\":1,\"err\":\"pending block height 3 not found\"}"), + Some(SubmitVerdict::TemplateGone) + ); + assert_eq!( + verdict("{\"ret\":1,\"err\":\"pending block not ready\"}"), + Some(SubmitVerdict::TemplateGone) + ); + assert_eq!( + verdict( + "{\"ret\":1,\"err\":\"difficulty check failed: expected at least 0f but got ff\"}" + ), + Some(SubmitVerdict::TemplateGone) + ); + // The pool relay wraps an unrecognized upstream body as `msg`. + assert_eq!( + verdict("{\"ret\":1,\"msg\":\"pending block height 9 not found\"}"), + Some(SubmitVerdict::TemplateGone) + ); + // About this submission only: the template still deserves another try. + assert_eq!( + verdict("{\"ret\":1,\"err\":\"coinbase nonce format invalid\"}"), + Some(SubmitVerdict::SubmissionRejected) + ); + // No `ret` at all is a front-end hiccup, not a node decision. + assert!(verdict("502 Bad Gateway").is_none()); + assert!(verdict("{\"msg\":\"gateway timeout\"}").is_none()); + + assert!(SubmitVerdict::Accepted.ends_template()); + assert!(SubmitVerdict::TemplateGone.ends_template()); + assert!(!SubmitVerdict::SubmissionRejected.ends_template()); + assert!(!SubmitVerdict::NoNodeVerdict.ends_template()); + } + + #[test] + fn both_upstream_dialects_are_classified() { + // Measured against a real pool: 526 submits in 180 seconds, 519 of them + // answered {"kind":"stale","ret":1} with NO error text at all. Reading only + // the fullnode's error strings turned every one of those into "a rejection + // about this one submission", which re-opened a template the pool had + // already declared dead and paid for another round trip with the next + // winner, and the next, and the next. + let verdict = |body: &str| classify_submit_response(body).map(|(v, _)| v); + + // ---- pool dialect, exactly as its miner-API arm wraps it ---- + assert_eq!( + verdict("{\"ret\":0,\"kind\":\"block\"}"), + Some(SubmitVerdict::Accepted) + ); + assert_eq!( + verdict("{\"ret\":0,\"kind\":\"share\"}"), + Some(SubmitVerdict::ShareCredited) + ); + assert_eq!( + verdict("{\"ret\":1,\"kind\":\"stale\"}"), + Some(SubmitVerdict::TemplateGone) + ); + assert_eq!( + verdict("{\"ret\":1,\"kind\":\"busy\"}"), + Some(SubmitVerdict::UpstreamBusy) + ); + assert_eq!( + verdict("{\"ret\":1,\"kind\":\"duplicate\"}"), + Some(SubmitVerdict::DuplicateSubmission) + ); + assert_eq!( + verdict("{\"ret\":1,\"kind\":\"invalid\"}"), + Some(SubmitVerdict::SubmissionRejected) + ); + + // ---- the pool's own object form, which keeps the extra fields ---- + assert_eq!( + verdict("{\"ret\":0,\"kind\":\"block\",\"solved_height\":9}"), + Some(SubmitVerdict::Accepted) + ); + assert_eq!( + verdict("{\"ret\":0,\"kind\":\"share\",\"accepted\":1234}"), + Some(SubmitVerdict::ShareCredited) + ); + assert_eq!( + verdict("{\"ret\":1,\"kind\":\"stale\",\"height\":9}"), + Some(SubmitVerdict::TemplateGone) + ); + assert_eq!( + verdict("{\"ret\":1,\"kind\":\"busy\",\"err\":\"too many shares this height\"}"), + Some(SubmitVerdict::UpstreamBusy) + ); + assert_eq!( + verdict("{\"ret\":1,\"kind\":\"invalid\",\"err\":\"above share target\"}"), + Some(SubmitVerdict::SubmissionRejected) + ); + + // ---- fullnode dialect, unchanged ---- + assert_eq!( + verdict("{\"ret\":0,\"height\":3,\"mining\":\"success\"}"), + Some(SubmitVerdict::Accepted) + ); + assert_eq!( + verdict("{\"ret\":1,\"err\":\"pending block height 3 not found\"}"), + Some(SubmitVerdict::TemplateGone) + ); + assert_eq!( + verdict("{\"ret\":1,\"err\":\"pending block not ready\"}"), + Some(SubmitVerdict::TemplateGone) + ); + assert_eq!( + verdict("{\"ret\":1,\"err\":\"difficulty check failed\"}"), + Some(SubmitVerdict::TemplateGone) + ); + assert_eq!( + verdict("{\"ret\":1,\"err\":\"coinbase nonce format invalid\"}"), + Some(SubmitVerdict::SubmissionRejected) + ); + // An unmarked success keeps the meaning it always had here. + assert_eq!(verdict("{\"ret\":0}"), Some(SubmitVerdict::Accepted)); + // A kind nobody here knows falls back to the same rules, so a future pool + // word can never be read as a settle it did not mean. + assert_eq!( + verdict("{\"ret\":1,\"kind\":\"something-new\"}"), + Some(SubmitVerdict::SubmissionRejected) + ); + // No `ret` at all stays a transport-level non-verdict. + assert!(verdict("502 Bad Gateway").is_none()); + assert!(verdict("{\"kind\":\"stale\"}").is_none()); + + // The reason text still names what happened, for the submit log line. + let (_, reason) = classify_submit_response("{\"ret\":1,\"kind\":\"stale\"}").unwrap(); + assert!(reason.contains("stale"), "the log line must name the kind"); + + // ---- what each verdict does to the template ---- + assert!(SubmitVerdict::Accepted.ends_template()); + assert!(SubmitVerdict::TemplateGone.ends_template()); + assert!(!SubmitVerdict::ShareCredited.ends_template()); + assert!(!SubmitVerdict::UpstreamBusy.ends_template()); + assert!(!SubmitVerdict::DuplicateSubmission.ends_template()); + assert!(!SubmitVerdict::SubmissionRejected.ends_template()); + assert!(!SubmitVerdict::NoNodeVerdict.ends_template()); + assert!(SubmitVerdict::ShareCredited.credits_share()); + assert!(!SubmitVerdict::Accepted.credits_share()); + assert_eq!( + SubmitVerdict::UpstreamBusy.retry_cooldown(), + Some(POOL_BUSY_COOLDOWN) + ); + assert_eq!(SubmitVerdict::ShareCredited.retry_cooldown(), None); + assert_eq!(SubmitVerdict::SubmissionRejected.retry_cooldown(), None); + + // A template the pool called stale must not be reported as a rejected + // submission: the operator line is the only thing they see. + let stale_line = SubmitVerdict::TemplateGone.outcome_text(); + assert!(stale_line.contains("stale")); + assert!(!stale_line.contains("rejection")); + } + + #[test] + fn every_pool_share_for_one_template_is_submitted_and_paid() { + // THE money guarantee. On mainnet the pool's share target is 2^24 easier + // than the network target, so "share" is the ordinary answer and "block" + // is rare. Each share is credited work, so treating the first one as a + // settle would drop every later share for that template on the floor: + // systematic underpayment of the miner, invisible in the logs. + let mut gate = SubmitGate::default(); + let key = (950u64, vec![0x11u8; HASH_WIDTH]); + let (share, _) = classify_submit_response("{\"ret\":0,\"kind\":\"share\"}").unwrap(); + assert_eq!(share, SubmitVerdict::ShareCredited); + + // The pool answers every single submission with a credited share. + let mut submitted = 0u32; + for head_nonce in 0..200u32 { + let win = winner_for_template(950, 0x11, head_nonce); + if gate.admit(&win) { + submitted += 1; + gate.record_verdict(&win, share, 1); + } + } + assert_eq!( + submitted, 200, + "every share a pool credits is money: not one may be dropped locally" + ); + assert_eq!( + gate.templates[&key].suppressed, 0, + "a credited share is never a redundant winner" + ); + assert_ne!( + gate.templates[&key].state, + TemplateSubmitState::Settled, + "a share credit must never settle the template" + ); + + // The real drain admits a whole batch before any verdict comes back. Once + // the upstream has proved it pays per share, none of that batch may be + // dropped either. + let mut batch = 0u32; + for head_nonce in 200..260u32 { + if gate.admit(&winner_for_template(950, 0x11, head_nonce)) { + batch += 1; + } + } + assert_eq!( + batch, 60, + "a burst of shares must not be gated by the first" + ); + assert_eq!(gate.templates[&key].suppressed, 0); + + // The NEXT template inherits it. Otherwise every job change would start + // out "solo" again and eat the burst of shares found before its first + // verdict came back, which is earned income, once per template. + let mut fresh = 0u32; + for head_nonce in 0..40u32 { + if gate.admit(&winner_for_template(951, 0x22, head_nonce)) { + fresh += 1; + } + } + assert_eq!( + fresh, 40, + "a pool that pays per share keeps paying at the next height" + ); + assert_eq!( + gate.templates[&(951u64, vec![0x22u8; HASH_WIDTH])].suppressed, + 0 + ); + } + + #[test] + fn a_share_paying_template_submits_every_winner_of_a_drain() { + let _guard = mining_state_guard(); + // Same guarantee, through the real drain path: with a fullnode only the + // first winner of a template leaves, with a share-paying pool all of them do. + set_pending_block_stuff(710, pending_template_json(710, 0x11, 0xd1)).unwrap(); + let cnf = PoWorkConf::test_defaults("127.0.0.1:1".to_string(), 1, 16); + let (res_tx, mut res_rx) = mpsc::sync_channel::>(16); + let (submit_tx, submit_rx) = mpsc::sync_channel::>(16); + + // One earlier share told us this upstream credits them. + let gate = test_gate(); + let primer = winner_for_template(710, 0x11, 99); + assert!(lock_gate(&gate).admit(&primer)); + lock_gate(&gate).record_verdict(&primer, SubmitVerdict::ShareCredited, 1); + + for head_nonce in 0..9u32 { + res_tx + .send(winner_for_template(710, 0x11, head_nonce)) + .unwrap(); + } + let mut most_hash = vec![255u8; HASH_WIDTH]; + let mut tracker = HashrateTracker::default(); + deal_block_mining_results( + &cnf, + &mut most_hash, + &mut res_rx, + 8, + &mut tracker, + 1, + &submit_tx, + &gate, + ); + + let mut queued = Vec::new(); + while let Ok(win) = submit_rx.try_recv() { + queued.push(win.head_nonce); + } + assert_eq!( + queued, + (0..9u32).collect::>(), + "every winner of a share-paying template is separately payable work" + ); + assert_eq!( + lock_gate(&gate).templates[&(710u64, vec![0x11u8; HASH_WIDTH])].suppressed, + 0 + ); + } + + #[test] + fn an_overflowing_share_list_is_counted_and_reported_not_silently_dropped() { + // (b) of the brief. The list is bounded on purpose; what must never + // happen is losing shares QUIETLY. The kernel counts every hit including + // the ones that did not fit, and the host adds them up so the operator + // can see the miner is being paid for less than it mined. + let before = SHARE_HITS_DROPPED.load(Relaxed); + report_share_undersampling(0); + assert_eq!( + SHARE_HITS_DROPPED.load(Relaxed), + before, + "a batch that fitted has nothing to report" + ); + report_share_undersampling(7_976); + assert_eq!(SHARE_HITS_DROPPED.load(Relaxed), before + 7_976); + // The LINE is throttled, the COUNT is not: a pool serving a target far + // too easy for the card would otherwise either spam the log or hide the + // real cost, and the running total is the honest number. + report_share_undersampling(24); + assert_eq!(SHARE_HITS_DROPPED.load(Relaxed), before + 8_000); + assert!(SHARE_OVERFLOW_LOG_INTERVAL >= Duration::from_secs(1)); + } + + #[test] + fn solo_mining_never_asks_the_gpu_for_a_share_list() { + let _guard = mining_state_guard(); + // (d) of the brief, at the one place that decides it. With no pool there + // is no share target, so the OpenCL kernel is launched with + // share_capacity=0, skips the appending block and the host moves not one + // extra byte over the bus: a solo rig runs exactly what it ran before. + set_pending_block_stuff(820, pending_template_json(820, 0x11, 0xe1)).unwrap(); + let stuff = MINING_BLOCK_STUFF.read().unwrap().clone(); + + let mut cnf = PoWorkConf::test_defaults("127.0.0.1:1".to_string(), 1, 16); + assert!(cnf.worker_param().is_empty(), "test defaults must be solo"); + assert_eq!(pool_share_target(&cnf, &stuff), None); + + // Pooled: the served target_hash IS the share target (HBIT puts it in + // that field), so there is no second threshold that could disagree. + cnf.pool_worker = "1AVRuFXNFi3rdMrPH4hdqSgFrEBnWisWaS".to_string(); + assert_eq!( + pool_share_target(&cnf, &stuff), + Some([0x0fu8; HASH_WIDTH]), + "a pooled miner filters on exactly the target it was served" + ); + } + + #[test] + fn a_gpu_batch_of_pool_shares_is_not_metered_out_by_the_drain_tick() { + let _guard = mining_state_guard(); + // The drain used to take four results per worker per tick, so a single + // GPU worker handed the pool four shares every 123 ms however many it + // found: the whole fix would have been throttled away one layer down. + set_pending_block_stuff(830, pending_template_json(830, 0x11, 0xe2)).unwrap(); + let cnf = PoWorkConf::test_defaults("127.0.0.1:1".to_string(), 1, 16); + let (res_tx, mut res_rx) = + mpsc::sync_channel::>(RESULT_CHANNEL_CAPACITY); + let (submit_tx, submit_rx) = + mpsc::sync_channel::>(SUBMIT_QUEUE_CAPACITY); + + // One earlier share proves this upstream credits them. + let gate = test_gate(); + let primer = winner_for_template(830, 0x11, u32::MAX); + assert!(lock_gate(&gate).admit(&primer)); + lock_gate(&gate).record_verdict(&primer, SubmitVerdict::ShareCredited, 1); + + let shares = RESULT_DRAIN_MIN_PER_TICK as u32; + for head_nonce in 0..shares { + res_tx + .send(winner_for_template(830, 0x11, head_nonce)) + .unwrap(); + } + let mut most_hash = vec![255u8; HASH_WIDTH]; + let mut tracker = HashrateTracker::default(); + // ONE worker, so the old bound would have been four. + deal_block_mining_results( + &cnf, + &mut most_hash, + &mut res_rx, + 1, + &mut tracker, + 1, + &submit_tx, + &gate, + ); + + let mut queued = Vec::new(); + while let Ok(win) = submit_rx.try_recv() { + queued.push(win.head_nonce); + } + assert_eq!( + queued, + (0..shares).collect::>(), + "one tick must carry a whole batch of shares, not four of them" + ); + // And the floor is exactly the submit queue, so a bigger drain can never + // provoke the blocking inline submit on the result thread. + assert_eq!(RESULT_DRAIN_MIN_PER_TICK, SUBMIT_QUEUE_CAPACITY); + } + + #[test] + fn a_pool_stale_reply_settles_the_template() { + // The 519 wasted round trips: "stale" carries no error text, so it used to + // read as a rejection of one submission and re-opened the template. + let mut gate = SubmitGate::default(); + let key = (960u64, vec![0x11u8; HASH_WIDTH]); + let win = winner_for_template(960, 0x11, 1); + let (stale, _) = classify_submit_response("{\"ret\":1,\"kind\":\"stale\"}").unwrap(); + assert!(gate.admit(&win)); + gate.record_verdict(&win, stale, 1); + assert!( + !gate.admit(&winner_for_template(960, 0x11, 2)), + "a template the pool called stale can never take another winner" + ); + assert_eq!(gate.templates[&key].state, TemplateSubmitState::Settled); + assert_eq!( + gate.templates[&key].outcome, + SubmitVerdict::TemplateGone.outcome_text() + ); + + // A duplicate is about ONE submission (the same height, coinbase nonce and + // block nonce), so a DIFFERENT nonce is still wanted: it must not settle. + let mut gate = SubmitGate::default(); + let win = winner_for_template(961, 0x11, 1); + let (dup, _) = classify_submit_response("{\"ret\":1,\"kind\":\"duplicate\"}").unwrap(); + assert!(gate.admit(&win)); + gate.record_verdict(&win, dup, 1); + assert!(gate.admit(&winner_for_template(961, 0x11, 2))); + assert_ne!( + gate.templates[&(961u64, vec![0x11u8; HASH_WIDTH])].state, + TemplateSubmitState::Settled + ); + } + + #[test] + fn a_busy_pool_is_backed_off_without_killing_its_template() { + let mut gate = SubmitGate::default(); + let key = (970u64, vec![0x11u8; HASH_WIDTH]); + let win = winner_for_template(970, 0x11, 1); + let (busy, _) = classify_submit_response("{\"ret\":1,\"kind\":\"busy\"}").unwrap(); + let t0 = Instant::now(); + assert!(gate.admit_at(&win, t0)); + gate.record_verdict_at(&win, busy, 1, t0); + assert_ne!( + gate.templates[&key].state, + TemplateSubmitState::Settled, + "an overloaded pool has not consumed the template" + ); + assert!( + !gate.admit_at( + &winner_for_template(970, 0x11, 2), + t0 + Duration::from_millis(123) + ), + "a pool that just said it is overloaded must not be hammered" + ); + assert!( + gate.admit_at(&winner_for_template(970, 0x11, 3), t0 + POOL_BUSY_COOLDOWN), + "the pause is a back-off, not a settle: the template is still alive" + ); + + // Any other answer clears the pause at once, so a recovered pool is used + // again immediately. + gate.record_verdict_at( + &win, + SubmitVerdict::ShareCredited, + 1, + t0 + POOL_BUSY_COOLDOWN, + ); + assert!(gate.admit_at(&winner_for_template(970, 0x11, 4), t0 + POOL_BUSY_COOLDOWN)); + } + + #[test] + fn the_fullnode_gate_behaviour_is_unchanged() { + // Solo mining must behave exactly as it did: a height holds one block, so + // an accepted one settles the template, and a transport non-verdict never + // silences a height. + let (accepted, _) = + classify_submit_response("{\"ret\":0,\"height\":3,\"mining\":\"success\"}").unwrap(); + assert_eq!(accepted, SubmitVerdict::Accepted); + let mut gate = SubmitGate::default(); + let win = winner_for_template(980, 0x11, 1); + assert!(gate.admit(&win)); + gate.record_verdict(&win, accepted, 1); + assert!( + !gate.admit(&winner_for_template(980, 0x11, 2)), + "an accepted block still settles the height" + ); + assert_eq!( + gate.templates[&(980u64, vec![0x11u8; HASH_WIDTH])].suppressed, + 1 + ); + + assert!(classify_submit_response("502 Bad Gateway").is_none()); + let mut gate = SubmitGate::default(); + let win = winner_for_template(981, 0x11, 1); + assert!(gate.admit(&win)); + gate.record_verdict(&win, SubmitVerdict::NoNodeVerdict, 1); + assert!( + gate.admit(&winner_for_template(981, 0x11, 2)), + "a network hiccup must never permanently silence a height" + ); + } } diff --git a/app/src/cuda_pow.rs b/app/src/cuda_pow.rs index 21b0cb0e..2a9d54fd 100644 --- a/app/src/cuda_pow.rs +++ b/app/src/cuda_pow.rs @@ -1,9 +1,128 @@ -use x16rs_cuda::{CudaMiner, CudaResult}; +use x16rs_cuda::{CudaBatchOutput, CudaMiner, CudaResult}; pub struct CudaMiningResources { pub miner: CudaMiner, pub workgroups: u32, pub unit_size: u32, + /// Effective work-groups after OOM/error backoff. Starts at `workgroups` and + /// is halved on a batch error, ramped back up on success, mirroring the + /// OpenCL GpuOomState. Be honest about what this buys: a smaller grid only + /// mitigates launch-timeout / TDR pressure. A block-size resource fault + /// (cudaErrorInvalidConfiguration / LaunchOutOfResources) and a real + /// out-of-memory failure are NOT fixed by fewer work groups, and a sticky + /// context fault is grid independent, so those keep failing until the + /// quarantine parks the card for a backoff window (see `quarantine` below). + pub eff_wg: std::sync::atomic::AtomicU32, + /// Never back off below this many work-groups. + pub floor_wg: u32, + /// Time-based quarantine with exponential backoff and automatic re-probe, + /// shared with the OpenCL backend so both cards are treated identically. It + /// also owns the consecutive-failure count, which only a clean batch resets. + pub quarantine: crate::gpu_oom::GpuQuarantine, +} + +impl CudaMiningResources { + /// Current effective work-groups, clamped to [floor, configured]. + pub fn effective_wg(&self) -> u32 { + let max = self.workgroups.max(1); + let floor = self.floor_wg.max(1).min(max); + self.eff_wg + .load(std::sync::atomic::Ordering::Relaxed) + .clamp(floor, max) + } + + /// Halve the effective work-groups toward the floor after a batch error/OOM. + pub fn record_error(&self) -> u32 { + let floor = self.floor_wg.max(1); + let cur = self.eff_wg.load(std::sync::atomic::Ordering::Relaxed); + let next = (cur / 2).max(floor); + self.eff_wg + .store(next, std::sync::atomic::Ordering::Relaxed); + next + } + + /// Ramp the effective work-groups back up toward the configured maximum after + /// a clean batch, so throughput recovers once memory pressure clears. + pub fn record_success(&self) { + if self.quarantine.record_success() { + println!( + "[CUDA] GPU RECOVERED: the re-probe succeeded, quarantine cleared and GPU mining has resumed." + ); + } + let cur = self.eff_wg.load(std::sync::atomic::Ordering::Relaxed); + if cur < self.workgroups { + let next = cur.saturating_add((cur / 4).max(1)).min(self.workgroups); + self.eff_wg + .store(next, std::sync::atomic::Ordering::Relaxed); + } + } + + /// Count one failed batch and report whether it parked the card. + pub fn note_batch_failure(&self) -> crate::gpu_oom::GpuFailureReport { + self.quarantine.record_failure(std::time::Instant::now()) + } + + /// Print the one-time operator alert when a failure armed the quarantine. + pub fn announce_quarantine(&self, report: &crate::gpu_oom::GpuFailureReport, detail: &str) { + let Some(entry) = report.quarantined else { + return; + }; + // Drop straight to the floor so the automatic re-probe runs on the + // smallest, most likely to succeed grid. + self.eff_wg + .store(self.floor_wg.max(1), std::sync::atomic::Ordering::Relaxed); + eprintln!( + "[CUDA] ALERT GPU quarantined after {} consecutive failed batches ({} this session): no GPU work for {}, then the card is re-probed automatically. Mining continues on capped CPU recovery. {} Check the driver, cooling, power and the PCIe riser.", + report.consecutive_failures, + report.total_failures, + crate::gpu_oom::format_backoff(entry.retry_in), + detail + ); + } + + /// Gate one batch. True while the card is quarantined and must not be given + /// work. When the backoff expires this lets exactly one re-probe batch through + /// at floor work-groups, so the card recovers without a restart. + pub fn quarantine_blocks_batch(&self) -> bool { + match self.quarantine.gate(std::time::Instant::now()) { + crate::gpu_oom::GpuGate::Run => false, + crate::gpu_oom::GpuGate::Skip { + level, + retry_in, + total_failures, + notify, + } => { + if notify { + eprintln!( + "[CUDA] GPU QUARANTINED (level {level}, {total_failures} failed batches): no GPU work for another {}, mining continues on capped CPU recovery. The card is re-probed automatically, no restart needed.", + crate::gpu_oom::format_backoff(retry_in) + ); + } + true + } + crate::gpu_oom::GpuGate::Reprobe { + level, + total_failures, + } => { + let wg = self.floor_wg.max(1); + self.eff_wg.store(wg, std::sync::atomic::Ordering::Relaxed); + println!( + "[CUDA] GPU quarantine (level {level}, {total_failures} failed batches) expired: re-probing the device at work_groups={wg}." + ); + false + } + } + } + + /// Quarantine state for the panel / stats (None while mining normally). + pub fn quarantine_status(&self) -> Option { + self.quarantine.status(std::time::Instant::now()) + } + + /// One-line GPU health for a non-technical operator. + pub fn quarantine_note(&self) -> String { + self.quarantine.describe(std::time::Instant::now()) + } } pub fn initialize_cuda( @@ -13,7 +132,7 @@ pub fn initialize_cuda( ) -> Vec> { if !CudaMiner::is_available() { eprintln!( - "[CUDA] x16rs-cuda built without kernels — rebuild with: cargo build -p poworker --features cuda" + "[CUDA] x16rs-cuda built without kernels; rebuild with: cargo build -p poworker --features cuda" ); return Vec::new(); } @@ -42,6 +161,9 @@ pub fn initialize_cuda( miner, workgroups, unit_size, + eff_wg: std::sync::atomic::AtomicU32::new(workgroups), + floor_wg: (workgroups / 16).max(1), + quarantine: crate::gpu_oom::GpuQuarantine::new(), })] } Err(e) => { @@ -51,6 +173,10 @@ pub fn initialize_cuda( } } +/// One CUDA block batch, best result only. +/// +/// Kept exactly as it was for the callers that want one hash per batch and nothing +/// else. Pool mining goes through [`do_group_block_mining_cuda_shares`]. pub fn do_group_block_mining_cuda( cuda: &CudaMiningResources, height: u64, @@ -58,6 +184,29 @@ pub fn do_group_block_mining_cuda( nonce_start: u32, workgroups: u32, ) -> CudaResult<(u32, [u8; 32])> { - cuda.miner - .mine_block_batch(height, &block_intro, nonce_start, workgroups) -} \ No newline at end of file + do_group_block_mining_cuda_shares(cuda, height, block_intro, nonce_start, workgroups, None) + .map(|out| out.best) +} + +/// One CUDA block batch, plus every nonce that beat `share_target`. +/// +/// `share_target` is `None` for solo mining, and that is the whole guarantee that +/// solo behaviour is unchanged: the kernel is launched with share_capacity=0, skips +/// the appending block, and the host neither uploads a target nor reads a share +/// buffer back. +pub fn do_group_block_mining_cuda_shares( + cuda: &CudaMiningResources, + height: u64, + block_intro: Vec, + nonce_start: u32, + workgroups: u32, + share_target: Option<&[u8; 32]>, +) -> CudaResult { + cuda.miner.mine_block_batch_shares( + height, + &block_intro, + nonce_start, + workgroups, + share_target, + ) +} diff --git a/app/src/diabider.rs b/app/src/diabider.rs index 7f3ababc..2ca420b0 100644 --- a/app/src/diabider.rs +++ b/app/src/diabider.rs @@ -176,7 +176,12 @@ fn check_bidding_step( // raise fee let mut my_tx = my_bid_txp.tx_clone(); my_tx.set_fee(new_bid_fee.clone()); - let _ = my_tx.fill_sign(&engcnf.dmer_bid_account); + // A failed sign must NOT be submitted: an unsigned / mis-signed raised-bid tx + // is rejected by the node, wasting the bid window and the fee bump. + if let Err(e) = my_tx.fill_sign(&engcnf.dmer_bid_account) { + printerr!("raise-bid tx sign error: {}", e); + retry!(3); + } let txp = TxPkg::create(my_tx); // submit tx diff --git a/app/src/diaworker.rs b/app/src/diaworker.rs index 7cf78bd5..97e22cc9 100644 --- a/app/src/diaworker.rs +++ b/app/src/diaworker.rs @@ -1,5 +1,5 @@ use std::sync::Arc; -use std::sync::atomic::{AtomicU32, Ordering::*}; +use std::sync::atomic::{AtomicBool, AtomicU32, AtomicU64, Ordering::*}; use std::sync::{RwLock, mpsc}; use std::thread::*; @@ -9,6 +9,9 @@ use reqwest::blocking::Client as HttpClient; use serde_json::Value as JV; use crate::efficiency::*; +// Same panic firewall the block miner uses: one result thread owns every +// submission, so a panic there would silently end all payouts. +use crate::mining_guard::guard_mining_iteration; use basis::difficulty::*; use field::*; @@ -16,7 +19,7 @@ use mint::action::*; use mint::genesis::*; use sys::*; -use crate::hash_util::diamond_more_power; +use crate::hash_util::diamond_better; #[cfg(feature = "ocl")] use crate::gpu_oom::GpuBatchError; @@ -61,7 +64,21 @@ impl DiaWorkConf { let active = efficiency.initial_active_supervene(configured_supervene); let runtime = MiningRuntimeState::new(0, active); // HACD is officially CPU/full-node mining. Legacy GPU keys are ignored - // so a stale or hand-edited config cannot activate the OpenCL path. + // so a stale or hand-edited config cannot activate the OpenCL path. Warn + // loudly if the config still carries GPU keys, so it is clear they do + // nothing here (rather than silently forcing CPU). + let wants_gpu = ["useopencl", "usecuda"].iter().any(|k| { + matches!( + ini_must(sec, k, "").trim().to_lowercase().as_str(), + "true" | "1" | "yes" + ) + }) || ini_must_u64(sec, "workgroups", 0) > 0; + if wants_gpu { + println!( + "[diamond] NOTE: HACD (diamond) mining is CPU / full-node only; the GPU keys \ + in this config (useopencl / usecuda / workgroups) are ignored." + ); + } DiaWorkConf { rpcaddr: ini_must(sec, "connect", "127.0.0.1:8081"), api_token: ini_must(sec, "api_token", "").trim().to_string(), @@ -118,10 +135,27 @@ mod config_tests { /*************************************/ const HASH_WIDTH: usize = 32; +// Length of a diamond hash string (x16rs DMD_M): 10 leading '0' chars followed by +// the 6-char diamond name. This is a fixed mainnet consensus constant. +pub(crate) const DIAMOND_HASH_LEN: usize = 16; const MINING_INTERVAL: f64 = 3.0; // 3 secs +/// Bounded result channel: an unbounded queue grows without limit whenever the +/// drain thread stalls. Under backpressure a statistics-only batch may be +/// dropped, but a batch carrying a mined diamond is real money and always waits. +const RESULT_CHANNEL_CAPACITY: usize = 1024; +/// Mined diamonds queued for the dedicated submit thread. A diamond is rare, so +/// this only has to absorb a burst while one submission is in flight. +const SUBMIT_QUEUE_CAPACITY: usize = 64; // current mining diamond number static MINING_DIAMOND_NUM: AtomicU32 = AtomicU32::new(0); +/// Bumped once per installed diamond job, exactly like MINING_BLOCK_EPOCH in the +/// block miner. A chain reorg can move the diamond tip DOWN (next_num < mining_num) +/// or replace born.hash while the number stays the same, and the number alone +/// cannot express either case, so a worker that only watched the number kept +/// grinding a dead (number, prev_hash) pair. Every worker snapshots this too and +/// restarts on any change. +static MINING_DIAMOND_EPOCH: AtomicU64 = AtomicU64::new(0); use std::sync::LazyLock; static HTTP_CLIENT: LazyLock = LazyLock::new(|| { @@ -130,6 +164,51 @@ static HTTP_CLIENT: LazyLock = LazyLock::new(|| { }); static MINING_DIAMOND_STUFF: LazyLock> = LazyLock::new(|| RwLock::default()); +fn current_diamond_epoch() -> u64 { + MINING_DIAMOND_EPOCH.load(Acquire) +} + +/// Publish a diamond job (number + prev_hash) for the workers. Returns true only +/// when something actually changed, so an unchanged re-poll never disturbs a +/// running worker. The epoch is bumped on every real change, which is what makes +/// a reorg to a LOWER number, or to a different born.hash at the SAME number, +/// visible to a worker that already snapshotted its job. +fn install_diamond_job(next_num: u32, new_hash: Hash) -> bool { + let mut stuff = MINING_DIAMOND_STUFF + .write() + .unwrap_or_else(|e| e.into_inner()); + if MINING_DIAMOND_NUM.load(Acquire) == next_num && *stuff == new_hash { + return false; // nothing changed + } + *stuff = new_hash; + // Release: publish the STUFF write above before the number and the epoch, so + // a reader that sees either (with an Acquire load) also sees the matching + // prev_hash. + MINING_DIAMOND_NUM.store(next_num, Release); + MINING_DIAMOND_EPOCH.fetch_add(1, Release); + true +} + +/// Snapshot the live job as one consistent `(number, epoch, prev_hash)` triple. +/// install_diamond_job holds the STUFF write lock across all three updates, so +/// reading them under the read lock rules out a worker pairing the number of one +/// job with the prev_hash of another. +fn snapshot_diamond_job() -> (u32, u64, Hash) { + let stuff = MINING_DIAMOND_STUFF + .read() + .unwrap_or_else(|e| e.into_inner()); + ( + MINING_DIAMOND_NUM.load(Acquire), + current_diamond_epoch(), + stuff.clone(), + ) +} + +/// True while the job a worker snapshotted is still the live one. +fn diamond_job_is_current(snapshot_number: u32, snapshot_epoch: u64) -> bool { + snapshot_number == MINING_DIAMOND_NUM.load(Acquire) && snapshot_epoch == current_diamond_epoch() +} + #[allow(dead_code)] #[derive(Debug, Clone, Default)] pub(crate) struct DiamondMiningResult { @@ -138,17 +217,102 @@ pub(crate) struct DiamondMiningResult { nonce_space: u64, u64_nonce: u64, msg_nonce: Vec, - dia_str: [u8; 10], + dia_str: [u8; DIAMOND_HASH_LEN], is_success: Option, use_secs: f64, is_gpu: bool, gpu_batch_ok: bool, } +fn should_stop(stop_flag: &Option>) -> bool { + stop_flag.as_ref().map(|f| f.load(Relaxed)).unwrap_or(false) +} + +/// Spawn a diamond mining thread that is registered with the runtime, exactly +/// like the block miner registers its own threads, so a shutdown supervisor can +/// wait for every HACD thread with `MiningRuntimeState::active_mining_threads`. +/// The guard is taken BEFORE the thread starts, so the count can never read zero +/// while a thread is still on its way up, and it is released by Drop, so a thread +/// that returns (or unwinds) always acknowledges instead of hanging the wait. +fn spawn_tracked_diamond_thread(runtime: &Arc, body: F) +where + F: FnOnce() + Send + 'static, +{ + let thread_guard = runtime.track_mining_thread(); + spawn(move || { + let _thread_guard = thread_guard; + body(); + }); +} + +/// Hand a batch result to the drain thread. Returns false only when the drain +/// side is gone, so the worker knows to exit. Under backpressure a +/// statistics-only result is dropped with a log, but a result carrying a mined +/// diamond is a payout and waits for space instead. +fn send_diamond_result( + result_ch_tx: &mpsc::SyncSender, + res: DiamondMiningResult, +) -> bool { + match result_ch_tx.try_send(res) { + Ok(()) => true, + Err(mpsc::TrySendError::Full(res)) => { + if res.is_success.is_some() { + return result_ch_tx.send(res).is_ok(); + } + eprintln!( + "[Mining] Diamond result queue full, dropped a statistics-only batch at number {}.", + res.number + ); + true + } + Err(mpsc::TrySendError::Disconnected(_)) => false, + } +} + +/// Disposition of a finished GPU batch: `(send the result, report a device error)`. +/// +/// Device health and payout are two different questions. A driver-level failure +/// (a queue.finish() error, an OOM, a corrupt read-back) says nothing about a +/// DiamondMint the host has ALREADY verified on the CPU - that mint is a pure +/// function of prev_hash/nonce/address/custom_message - so it is always forwarded +/// and only the health signal is downgraded. Letting one flag gate both is how a +/// driver hiccup silently threw away hours to days of mined value. +#[cfg_attr(not(feature = "ocl"), allow(dead_code))] +fn diamond_batch_disposition(gpu_batch_ok: bool, carries_payout: bool) -> (bool, bool) { + (gpu_batch_ok || carries_payout, !gpu_batch_ok) +} + +/// Queue a mined diamond for the submit thread. If the queue is full or the +/// thread is gone we submit inline rather than drop it: a dropped diamond is +/// lost money. +fn queue_diamond_mining_success( + cnf: &DiaWorkConf, + submit_tx: &mpsc::SyncSender, + success: DiamondMint, +) { + match submit_tx.try_send(success) { + Ok(()) => {} + Err(mpsc::TrySendError::Full(success)) => { + eprintln!( + "[Mining] Diamond submit queue full, submitting number {} inline.", + *success.d.number + ); + push_diamond_mining_success(cnf, success); + } + Err(mpsc::TrySendError::Disconnected(success)) => { + push_diamond_mining_success(cnf, success); + } + } +} + /* * Diamond worker */ pub fn diaworker() { + diaworker_with_stop(None) +} + +pub fn diaworker_with_stop(stop_flag: Option>) { let cnfp = "./diaworker.config.ini".to_string(); let inicnf = sys::load_config(cnfp.clone()); let mut cnf = DiaWorkConf::new(&inicnf); @@ -161,7 +325,7 @@ pub fn diaworker() { // cnf.supervene = 1; // test end - let (res_tx, res_rx) = mpsc::channel(); + let (res_tx, res_rx) = mpsc::sync_channel(RESULT_CHANNEL_CAPACITY); // init load_init(&mut cnf); @@ -217,13 +381,37 @@ pub fn diaworker() { #[cfg(not(feature = "ocl"))] let vene: u32 = cnf.supervene; + // Submitting a mined diamond is a blocking HTTP round-trip (with retries), so + // it runs on a dedicated thread: doing it inline would stall the 77ms result + // drain and back the result channel up. + let (submit_tx, submit_rx) = mpsc::sync_channel::(SUBMIT_QUEUE_CAPACITY); + let cnf_submit = cnf.clone(); + spawn_tracked_diamond_thread(&cnf.runtime, move || { + while let Ok(success) = submit_rx.recv() { + guard_mining_iteration("diamond submit thread", || { + push_diamond_mining_success(&cnf_submit, success); + }); + } + }); + // deal results let cnf1 = cnf.clone(); - spawn(move || { - let mut most_dia_str = [b'W'; 10]; + let stop_flag_res = stop_flag.clone(); + spawn_tracked_diamond_thread(&cnf.runtime, move || { + // submit_tx lives in this thread: when the drain loop returns on shutdown + // it drops, which is what tells the submit thread to finish and exit. + let mut most_dia_str = [b'W'; DIAMOND_HASH_LEN]; let mut rstx = res_rx; loop { - deal_diamond_mining_results(&cnf1, &mut most_dia_str, &mut rstx, vene); + if should_stop(&stop_flag_res) { + return; + } + // A panic here must never end the thread: this is the only path that + // submits mined diamonds, so losing it means silent total payout loss + // while the miner still looks like it is running. + guard_mining_iteration("diamond result thread", || { + deal_diamond_mining_results(&cnf1, &mut most_dia_str, &mut rstx, vene, &submit_tx); + }); delay_continue_ms!(77); } }); @@ -252,10 +440,22 @@ pub fn diaworker() { let gpu = OpenclGpuHandle::new(resource, gpu_snapshot, scan.clone()); gpu.configure_oom_floor(vram, cnf.localsize, cnf.unitsize, cnf.workgroups, &arch); let cnf2 = cnf.clone(); - let rstx: mpsc::Sender = res_tx.clone(); - spawn(move || { + let rstx: mpsc::SyncSender = res_tx.clone(); + let stop_flag_worker = stop_flag.clone(); + spawn_tracked_diamond_thread(&cnf.runtime, move || { loop { - run_diamond_worker_thread_opencl(&cnf2, thrid, rstx.clone(), gpu.clone()); + if should_stop(&stop_flag_worker) { + return; + } + guard_mining_iteration("diamond GPU mining worker", || { + run_diamond_worker_thread_opencl( + &cnf2, + thrid, + rstx.clone(), + gpu.clone(), + &stop_flag_worker, + ); + }); delay_continue_ms!(9); } }); @@ -271,9 +471,15 @@ pub fn diaworker() { for thrid in 0..thrnum { let cnf2 = cnf.clone(); let rstx = res_tx.clone(); - spawn(move || { + let stop_flag_worker = stop_flag.clone(); + spawn_tracked_diamond_thread(&cnf.runtime, move || { loop { - run_diamond_worker_thread(&cnf2, thrid, rstx.clone()); + if should_stop(&stop_flag_worker) { + return; + } + guard_mining_iteration("diamond mining worker", || { + run_diamond_worker_thread(&cnf2, thrid, rstx.clone(), &stop_flag_worker); + }); delay_continue_ms!(9); } }); @@ -291,9 +497,20 @@ pub fn diaworker() { for thrid in 0..thrnum { let cnf2 = cnf.clone(); let rstx = res_tx.clone(); - spawn(move || { + let stop_flag_worker = stop_flag.clone(); + spawn_tracked_diamond_thread(&cnf.runtime, move || { loop { - run_diamond_worker_thread(&cnf2, thrid, rstx.clone()); + if should_stop(&stop_flag_worker) { + return; + } + guard_mining_iteration("diamond CPU assist worker", || { + run_diamond_worker_thread( + &cnf2, + thrid, + rstx.clone(), + &stop_flag_worker, + ); + }); delay_continue_ms!(9); } }); @@ -306,9 +523,15 @@ pub fn diaworker() { for thrid in 0..thrnum { let cnf2 = cnf.clone(); let rstx = res_tx.clone(); - spawn(move || { + let stop_flag_worker = stop_flag.clone(); + spawn_tracked_diamond_thread(&cnf.runtime, move || { loop { - run_diamond_worker_thread(&cnf2, thrid, rstx.clone()); + if should_stop(&stop_flag_worker) { + return; + } + guard_mining_iteration("diamond mining worker", || { + run_diamond_worker_thread(&cnf2, thrid, rstx.clone(), &stop_flag_worker); + }); delay_continue_ms!(9); } }); @@ -317,6 +540,9 @@ pub fn diaworker() { // pull loop loop { + if should_stop(&stop_flag) { + return; + } if !is_within_idle_schedule(cnf.efficiency.idle_start_hour, cnf.efficiency.idle_end_hour) { delay_continue!(5); } @@ -324,20 +550,25 @@ pub fn diaworker() { delay_continue!(3); } // HACD is CPU-only; GPU temperature polling does not apply here. - pull_and_push_diamond(&cnf); + // This is the ONLY code that refreshes MINING_DIAMOND_NUM / + // MINING_DIAMOND_STUFF, so a panic here would unwind diaworker_with_stop + // and leave every worker grinding the job it last snapshotted, forever and + // invisibly. Same firewall as every other loop in this file. + guard_mining_iteration("diamond pull loop", || pull_and_push_diamond(&cnf)); delay_continue!(MINING_INTERVAL as u64); } } fn deal_diamond_mining_results( cnf: &DiaWorkConf, - most_dia_str: &mut [u8; 10], + most_dia_str: &mut [u8; DIAMOND_HASH_LEN], result_ch_rx: &mut mpsc::Receiver, vene: u32, + submit_tx: &mpsc::SyncSender, ) { let mut deal_number = 0u32; let mut most = DiamondMiningResult::default(); - most.dia_str = [b'w'; 10]; + most.dia_str = [b'w'; DIAMOND_HASH_LEN]; let mut total_nonce_space = 0u64; let mut gpu_nonce_space = 0u64; let mut cpu_nonce_space = 0u64; @@ -352,12 +583,12 @@ fn deal_diamond_mining_results( cpu_nonce_space += res.nonce_space as u64; } total_use_secs += res.use_secs; - if diamond_more_power(&res.dia_str, &most.dia_str) { + if diamond_better(&res.dia_str, &most.dia_str) { most = res.clone(); } // upload success if let Some(success) = &res.is_success { - push_diamond_mining_success(cnf, success.clone()); + queue_diamond_mining_success(cnf, submit_tx, success.clone()); } recv_count += 1; if recv_count >= vene as usize * 4 { @@ -368,15 +599,20 @@ fn deal_diamond_mining_results( return; } // total most - if diamond_more_power(&most.dia_str, most_dia_str) { + if diamond_better(&most.dia_str, most_dia_str) { *most_dia_str = most.dia_str.clone(); } // print hashrate let diastr = String::from_utf8_lossy(&most.dia_str).into_owned(); let most_diastr = String::from_utf8_lossy(most_dia_str).into_owned(); - let avg_use_secs = total_use_secs / recv_count as f64; - let nonce_rates = if avg_use_secs.is_finite() && avg_use_secs > 0.0 { - total_nonce_space as f64 / avg_use_secs + // Aggregate hashrate = total nonces / wall-clock. The workers run in parallel, + // so wall-clock ~= total_use_secs / (parallel workers). Dividing by recv_count + // (batches drained) instead would multiply the rate by the number of batches + // each worker sent per drain, wildly overcounting for sequential batches. + let parallelism = (vene.max(1)) as f64; + let wall_secs = total_use_secs / parallelism; + let nonce_rates = if wall_secs.is_finite() && wall_secs > 0.0 { + total_nonce_space as f64 / wall_secs } else { 0.0 }; @@ -386,7 +622,7 @@ fn deal_diamond_mining_results( if should_pause_for_diamond_profit(&cnf.efficiency, &cnf.gpu_profile, active_cpu) { cnf.runtime.paused_unprofitable.store(true, Relaxed); println!( - "\n[efficiency] HACD mining paused — daily power cost exceeds configured revenue target (hac_price)." + "\n[efficiency] HACD mining paused: daily power cost exceeds configured revenue target (hac_price)." ); } else { cnf.runtime.paused_unprofitable.store(false, Relaxed); @@ -394,7 +630,7 @@ fn deal_diamond_mining_results( let paused = cnf.runtime.paused_unprofitable.load(Relaxed); // HACD is strictly CPU-only, so there is no GPU power draw. Using the shared // estimate_gpu_watts("") here would print a phantom ~280 W (Unknown vendor) - // for a CPU miner — report CPU-only wattage instead. + // for a CPU miner, so report CPU-only wattage instead. let gpu_w = 0.0; let watts = gpu_w + active_cpu as f64 * cnf.efficiency.cpu_watts_per_thread; let hashrate_show = rates_to_show(nonce_rates); @@ -432,13 +668,16 @@ fn deal_diamond_mining_results( may_print_turn_to_nex_diamond_mining(deal_number, Some(most_dia_str)); } -fn may_print_turn_to_nex_diamond_mining(curr_number: u32, most_dia_str: Option<&mut [u8; 10]>) { - let mining_number = MINING_DIAMOND_NUM.load(Relaxed); +fn may_print_turn_to_nex_diamond_mining( + curr_number: u32, + most_dia_str: Option<&mut [u8; DIAMOND_HASH_LEN]>, +) { + let mining_number = MINING_DIAMOND_NUM.load(Acquire); if mining_number <= curr_number { return; // not turn } if let Some(most_dia_str) = most_dia_str { - *most_dia_str = [b'W'; 10]; // reset + *most_dia_str = [b'W'; DIAMOND_HASH_LEN]; // reset } println!( @@ -452,12 +691,13 @@ fn may_print_turn_to_nex_diamond_mining(curr_number: u32, most_dia_str: Option<& fn run_diamond_worker_thread( cnf: &DiaWorkConf, _thrid: usize, - result_ch_tx: mpsc::Sender, + result_ch_tx: mpsc::SyncSender, + stop_flag: &Option>, ) { if mining_is_gated(&cnf.runtime, &cnf.efficiency) { delay_return_ms!(2000); } - let cmdn = MINING_DIAMOND_NUM.load(Relaxed); + let (cmdn, current_mining_epoch, current_mining_block_hash) = snapshot_diamond_job(); if cmdn == 0 { delay_return_ms!(99); // not yet } @@ -473,12 +713,6 @@ fn run_diamond_worker_thread( let mut nonce_space: u64 = 15000; let current_mining_number: u32 = cmdn; - let current_mining_block_hash: Hash = { - MINING_DIAMOND_STUFF - .read() - .unwrap_or_else(|e| e.into_inner()) - .clone() - }; // start mining let mut custom_nonce = [0u8; HASH_WIDTH]; @@ -494,6 +728,14 @@ fn run_diamond_worker_thread( let mut nonce_start = 0; loop { + // This inner loop only ends when the diamond number turns over, which can + // be many minutes away, so it has to observe the stop flag itself for a + // shutdown supervisor to see this thread acknowledge in good time. With no + // stop flag (the standalone diaworker binary) this is always false, so the + // mining behavior is unchanged. + if should_stop(stop_flag) { + return; + } let ctn = Instant::now(); // println!("- nonce_start: {}", nonce_start); let mut result = do_diamond_group_mining( @@ -509,7 +751,7 @@ fn run_diamond_worker_thread( result.use_secs = use_secs; result.is_gpu = false; result.gpu_batch_ok = true; - if result_ch_tx.send(result).is_err() { + if !send_diamond_result(&result_ch_tx, result) { return; } let Some(ns) = nonce_start.checked_add(nonce_space) else { @@ -521,9 +763,11 @@ fn run_diamond_worker_thread( } nonce_space = nonce_space.max(1); - // check next - if current_mining_number < MINING_DIAMOND_NUM.load(Relaxed) { - return; // turn to next number + // check next: a strict-advance test would miss a reorg that lowered the + // tip or replaced born.hash at the same number, and every hash after that + // would be unacceptable by construction. + if !diamond_job_is_current(current_mining_number, current_mining_epoch) { + return; // turn to the new job } } } @@ -532,25 +776,20 @@ fn run_diamond_worker_thread( fn run_diamond_worker_thread_opencl( cnf: &DiaWorkConf, _thrid: usize, - result_ch_tx: mpsc::Sender, + result_ch_tx: mpsc::SyncSender, gpu: std::sync::Arc, + stop_flag: &Option>, ) { if mining_is_gated(&cnf.runtime, &cnf.efficiency) { delay_return_ms!(2000); } - let cmdn = MINING_DIAMOND_NUM.load(Relaxed); + let (cmdn, current_mining_epoch, current_mining_block_hash) = snapshot_diamond_job(); if cmdn == 0 { delay_return_ms!(99); // not yet } let rwd_addr = cnf.rewardaddr.clone(); let current_mining_number: u32 = cmdn; - let current_mining_block_hash: Hash = { - MINING_DIAMOND_STUFF - .read() - .unwrap_or_else(|e| e.into_inner()) - .clone() - }; let mut custom_nonce = [0u8; HASH_WIDTH]; if let Err(e) = getrandom::fill(&mut custom_nonce) { @@ -561,6 +800,18 @@ fn run_diamond_worker_thread_opencl( let mut nonce_start = 0; loop { + // Same reason as the CPU worker: acknowledge shutdown without waiting for + // the diamond number to turn over. + if should_stop(stop_flag) { + return; + } + // Gate each batch on the device health check, the same way the block + // backend does. While the card is held back, on_batch_error is a no-op, so + // launching batches anyway would run with every OOM back-off and + // context-rebuild path silently switched off. + if gpu.gpu_is_disabled() { + delay_return_ms!(1000); + } let wg_cap = gpu.workgroups(cnf.workgroups, cnf.runtime.thermal_workgroups_cap()); let gpu_nonce_space = (wg_cap as u64) .saturating_mul(cnf.localsize as u64) @@ -585,7 +836,13 @@ fn run_diamond_worker_thread_opencl( result.nonce_space = gpu_nonce_space; let use_secs = Instant::now().duration_since(ctn).as_millis() as f64 / 1000.0; result.use_secs = use_secs; - if !result.gpu_batch_ok { + let (send_result, report_batch_error) = + diamond_batch_disposition(result.gpu_batch_ok, result.is_success.is_some()); + if report_batch_error { + // Payout first, health second: see diamond_batch_disposition. + if send_result { + let _ = send_diamond_result(&result_ch_tx, result); + } gpu.on_batch_error( GpuBatchError::Other("diamond OpenCL batch failed".into()), cnf.efficiency.oom_fallback, @@ -595,7 +852,7 @@ fn run_diamond_worker_thread_opencl( delay_return_ms!(50); } gpu.on_batch_success(cnf.workgroups, &cnf.runtime); - if result_ch_tx.send(result).is_err() { + if !send_diamond_result(&result_ch_tx, result) { return; } @@ -604,7 +861,8 @@ fn run_diamond_worker_thread_opencl( }; nonce_start = ns; - if current_mining_number < MINING_DIAMOND_NUM.load(Relaxed) { + // Same reorg-aware test as the CPU worker. + if !diamond_job_is_current(current_mining_number, current_mining_epoch) { return; } } @@ -632,7 +890,7 @@ fn do_diamond_group_mining( nonce_space, u64_nonce: 0, msg_nonce: custom_nonce.to_vec(), - dia_str: [b'W'; 10], + dia_str: [b'W'; DIAMOND_HASH_LEN], is_success: None, use_secs: 0.0, is_gpu: false, @@ -640,11 +898,11 @@ fn do_diamond_group_mining( }; let mut most_firhx = [0u8; HASH_WIDTH]; let mut most_resxh = [0u8; HASH_WIDTH]; - let mut most_diastr = [b'W'; 10]; + let mut most_diastr = [b'W'; DIAMOND_HASH_LEN]; let mut most_noncebytes = [0u8; 8]; // start mining - for nonce in nonce_start..nonce_start + nonce_space { + for nonce in nonce_start..nonce_start.saturating_add(nonce_space) { // std::thread::sleep(std::time::Duration::from_micros(333)); // test let nonce_bytes = nonce.to_be_bytes(); let (firhx, resxh, diastr) = @@ -664,7 +922,7 @@ fn do_diamond_group_mining( most_noncebytes = nonce_bytes; break; } - if diamond_more_power(&diastr, &most.dia_str) { + if diamond_better(&diastr, &most.dia_str) { most.u64_nonce = nonce; most.dia_str = diastr.clone(); most_firhx = firhx; @@ -689,15 +947,41 @@ fn do_diamond_group_mining( most } +/// The diamond pre-image: prev_hash(32) || nonce big-endian(8) || address(21) || +/// custom_message(0 or 32). Byte-identical to what `x16rs::mine_diamond` builds +/// on the CPU, and its LENGTH is what selects the SHA3 padding branch inside +/// sha3_256_hash_diamond on the GPU (61 without a custom message, 93 with), so +/// both sides build it here and nowhere else. +#[cfg_attr(not(feature = "ocl"), allow(dead_code))] +pub(crate) fn diamond_pre_image( + prevblockhash: &Hash, + nonce_bytes: &[u8; 8], + rwdaddr: &Address, + custom_message: &[u8], +) -> Vec { + let prevhash: &[u8; HASH_WIDTH] = prevblockhash; + let address: &[u8; 21] = rwdaddr; + [ + prevhash.as_slice(), + nonce_bytes.as_slice(), + address.as_slice(), + custom_message, + ] + .concat() +} + pub(crate) fn check_diamer_success( number: u32, firhx: [u8; HASH_WIDTH], resxh: [u8; HASH_WIDTH], - diastr: [u8; 10], + diastr: [u8; DIAMOND_HASH_LEN], ) -> Option<[u8; 6]> { - if let None = x16rs::check_diamond_hash_result(&diastr) { + // The 6-char name is derived by x16rs from positions DMD_L..DMD_M of the diamond + // string; take its result directly instead of hand-slicing so this stays correct + // no matter the leading-zero prefix length (mainnet DMD_L=10, DMD_M=16). + let Some(name) = x16rs::check_diamond_hash_result(&diastr) else { return None; - } + }; if !x16rs::check_diamond_difficulty(number, &firhx, &resxh) { return None; } @@ -710,10 +994,6 @@ pub(crate) fn check_diamer_success( number ); flush!("\n▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔\n"); - // The 6-char name is the tail after the leading-zero prefix: positions DMD_L..DMD_M of the - // diamond string. That prefix is 4 on this reduced-difficulty testnet build (upstream: 10). - let mut name = [0u8; 6]; - name.copy_from_slice(&diastr[4..]); Some(name) } @@ -764,7 +1044,7 @@ fn load_init(cnf: &mut DiaWorkConf) { } fn pull_and_push_diamond(cnf: &DiaWorkConf) { - let mining_num = MINING_DIAMOND_NUM.load(Relaxed); + let mining_num = MINING_DIAMOND_NUM.load(Acquire); let urlapi_latest = format!("http://{}/query/latest", &cnf.rpcaddr); // get next number @@ -786,16 +1066,21 @@ fn pull_and_push_diamond(cnf: &DiaWorkConf) { // println!("mining next num: {} {}", &mining_num, &next_num); if next_num == 1 { // println!("get latest: next_num == 1"); - *MINING_DIAMOND_STUFF - .write() - .unwrap_or_else(|e| e.into_inner()) = genesis_block_hash(); - MINING_DIAMOND_NUM.store(next_num, Relaxed); + install_diamond_job(next_num, genesis_block_hash()); return; // first mining } - if next_num <= mining_num { - return; // no change + // Advance, or roll back after a chain reorg that lowered the diamond tip. + // Same number: still re-fetch prev_hash in case born.hash was replaced. + if next_num == mining_num { + // Refresh prev_hash for the same number when the node reorged the tip. + // Cheap GET; skip heavy work only when hash is unchanged. + } else if next_num < mining_num { + println!( + "[HACD] diamond tip reorg: number {} -> {}, refreshing job", + mining_num, next_num + ); } - // query next! + // query prev diamond (or re-query when number did not advance) let urlapi_diamond = format!( "http://{}/query/diamond?number={}", &cnf.rpcaddr, @@ -830,10 +1115,12 @@ fn pull_and_push_diamond(cnf: &DiaWorkConf) { let Ok(hash_bytes) = hx.try_into() else { delay_return!(30); }; - *MINING_DIAMOND_STUFF - .write() - .unwrap_or_else(|e| e.into_inner()) = Hash::from(hash_bytes); - MINING_DIAMOND_NUM.store(next_num, Relaxed); + let new_hash = Hash::from(hash_bytes); + // Same number and same prev_hash: nothing to do. Anything else is a new job + // and bumps the epoch, which is what tells the running workers to restart. + if !install_diamond_job(next_num, new_hash) { + return; + } // print first req msg if mining_num == 0 { may_print_turn_to_nex_diamond_mining(mining_num, None); @@ -843,41 +1130,310 @@ fn pull_and_push_diamond(cnf: &DiaWorkConf) { fn push_diamond_mining_success(cnf: &DiaWorkConf, success: DiamondMint) { let urlapi_success = format!("http://{}/submit/diamondminer/success", &cnf.rpcaddr); let actionbody = success.serialize(); - // println!("\n\ncurl {}?hexbody=true -X POST -d '{}'", &urlapi_success, &actionbody.to_hex()); - let body = - match crate::rpc_http::post_text(&HTTP_CLIENT, &urlapi_success, &cnf.api_token, actionbody) - { - Ok(t) => t, + // Submitting the mined diamond is the whole payoff. Match the block submit + // path: retry transport failures AND unrecognized HTTP-200 bodies (proxy + // HTML / truncated JSON); stop only on a clear node accept or reject. + const MAX_SUBMIT_ATTEMPTS: u32 = 5; + let mut last = String::new(); + for attempt in 1..=MAX_SUBMIT_ATTEMPTS { + match crate::rpc_http::post_text( + &HTTP_CLIENT, + &urlapi_success, + &cnf.api_token, + actionbody.clone(), + ) { + Ok(body) => { + // Snippet only: a proxy error page can be up to MAX_RPC_BODY_BYTES + // (2 MiB) and this string is printed on the failure path. + last = body.chars().take(200).collect(); + let Ok(res) = serde_json::from_str::(&body) else { + let snippet: String = body.chars().take(120).collect(); + println!( + "[HACD submit] attempt {attempt}/{MAX_SUBMIT_ATTEMPTS} unrecognized response, retrying: {snippet}" + ); + if attempt < MAX_SUBMIT_ATTEMPTS { + std::thread::sleep(std::time::Duration::from_millis( + 500u64 * attempt as u64, + )); + } + continue; + }; + let jstr = |k: &str| res[k].as_str().unwrap_or(""); + let tx_err = jstr("err"); + if !tx_err.is_empty() { + println!( + "ㄨㄨㄨㄨ Failed submit tx diamond mint to mainnet\n ERROR: {}\n", + tx_err + ); + return; + } + let tx_hash = jstr("tx_hash"); + if tx_hash.len() == 64 { + println!( + "Success submit tx diamond mint {} ({}) to mainnet, \n get tx hash: {}\n", + success.d.diamond.to_readable(), + *success.d.number, + tx_hash + ); + return; + } + // JSON but no usable tx_hash — treat as transient front-end noise. + println!( + "[HACD submit] attempt {attempt}/{MAX_SUBMIT_ATTEMPTS} missing tx_hash, retrying" + ); + if attempt < MAX_SUBMIT_ATTEMPTS { + std::thread::sleep(std::time::Duration::from_millis(500u64 * attempt as u64)); + } + } Err(e) => { - println!("Error: cannot submit diamond success to {urlapi_success}: {e}"); - return; + last = format!("transport error: {e}"); + println!( + "Error: attempt {attempt}/{MAX_SUBMIT_ATTEMPTS} cannot submit diamond success to {urlapi_success}: {e}" + ); + if attempt < MAX_SUBMIT_ATTEMPTS { + std::thread::sleep(std::time::Duration::from_millis(500u64 * attempt as u64)); + } } - }; - let Ok(res) = serde_json::from_str::(&body) else { - println!("Error: invalid JSON from {urlapi_success}"); - return; - }; - let jstr = |k: &str| res[k].as_str().unwrap_or(""); - let tx_err = jstr("err"); - if tx_err.len() > 0 { - println!( - "ㄨㄨㄨㄨ Failed submit tx diamond mint to mainnet\n ERROR: {}\n", - tx_err - ); - return; - } - let tx_hash = jstr("tx_hash"); - if tx_hash.len() != 64 { - return; // err + } } println!( - "Success submit tx diamond mint {} ({}) to mainnet, \n get tx hash: {}\n", - success.d.diamond.to_readable(), - *success.d.number, - tx_hash + "ㄨㄨㄨㄨ Failed submit tx diamond mint after {MAX_SUBMIT_ATTEMPTS} attempts ({last}). Check the node/connection." ); } +#[cfg(test)] +mod result_channel_tests { + use super::*; + + fn mined_diamond() -> DiamondMint { + DiamondMint::with(DiamondName::from(*b"ABCDEF"), DiamondNumber::from(1u32)) + } + + fn statistics_result(number: u32) -> DiamondMiningResult { + let mut res = DiamondMiningResult::default(); + res.number = number; + res + } + + fn success_result(number: u32) -> DiamondMiningResult { + let mut res = statistics_result(number); + res.is_success = Some(mined_diamond()); + res + } + + #[test] + fn a_full_diamond_queue_drops_statistics_but_never_a_mined_diamond() { + let (tx, rx) = mpsc::sync_channel::(1); + assert!(send_diamond_result(&tx, statistics_result(5))); + // Queue is full now: a statistics-only batch is dropped, not blocked. + assert!(send_diamond_result(&tx, statistics_result(6))); + assert_eq!(rx.try_recv().map(|r| r.number), Ok(5)); + + assert!(send_diamond_result(&tx, success_result(9))); + assert_eq!(rx.try_recv().map(|r| r.number), Ok(9)); + drop(rx); + assert!(!send_diamond_result(&tx, success_result(9))); + } + + #[test] + fn a_mined_diamond_waits_for_queue_space_instead_of_being_dropped() { + let (tx, rx) = mpsc::sync_channel::(1); + assert!(send_diamond_result(&tx, statistics_result(1))); + + // The queue is full and this result is a payout, so the worker must block + // until the drain makes room rather than throw the diamond away. + let sender = spawn(move || send_diamond_result(&tx, success_result(42))); + assert_eq!(rx.recv().map(|r| r.number), Ok(1)); + assert!(sender.join().unwrap()); + let delivered = rx.recv().unwrap(); + assert_eq!(delivered.number, 42); + assert!(delivered.is_success.is_some()); + } + + #[test] + fn a_panicking_drain_iteration_never_ends_the_diamond_result_thread() { + let previous_hook = std::panic::take_hook(); + std::panic::set_hook(Box::new(|_| {})); + let (tx, rx) = mpsc::sync_channel::(4); + let mut drained = 0u32; + for round in 0..3u32 { + assert!(send_diamond_result(&tx, statistics_result(round))); + guard_mining_iteration("diamond result test loop", || { + while rx.try_recv().is_ok() { + drained += 1; + } + if round == 1 { + panic!("simulated diamond result thread panic"); + } + }); + } + std::panic::set_hook(previous_hook); + assert_eq!(drained, 3); + } + + #[test] + fn a_tracked_diamond_thread_acknowledges_shutdown_when_it_exits() { + // The shutdown supervisor waits on active_mining_threads(), so a HACD + // thread must be counted before it starts and must release the count when + // it returns. + let runtime = MiningRuntimeState::new(0, 1); + let (started_tx, started_rx) = mpsc::channel::<()>(); + let (release_tx, release_rx) = mpsc::channel::<()>(); + spawn_tracked_diamond_thread(&runtime, move || { + let _ = started_tx.send(()); + let _ = release_rx.recv(); + }); + assert_eq!(started_rx.recv(), Ok(())); + assert_eq!(runtime.active_mining_threads(), 1); + + drop(release_tx); + let deadline = Instant::now() + Duration::from_secs(5); + while runtime.active_mining_threads() > 0 && Instant::now() < deadline { + sleep(Duration::from_millis(5)); + } + assert_eq!(runtime.active_mining_threads(), 0); + } + + #[test] + fn a_panicking_pull_iteration_never_ends_the_diamond_pull_loop() { + // The pull loop is the ONLY code that refreshes the mining job. Without a + // firewall a panic there unwinds diaworker_with_stop and every worker + // keeps grinding whatever job it last snapshotted, forever and invisibly. + let previous_hook = std::panic::take_hook(); + std::panic::set_hook(Box::new(|_| {})); + let mut refreshes = 0u32; + for round in 0..3u32 { + guard_mining_iteration("diamond pull loop", || { + if round == 1 { + panic!("simulated diamond pull loop panic"); + } + refreshes += 1; + }); + } + std::panic::set_hook(previous_hook); + assert_eq!(refreshes, 2); + } + + #[test] + fn a_failed_gpu_batch_still_forwards_a_verified_diamond() { + // Healthy batch: forwarded, no error reported. + assert_eq!(diamond_batch_disposition(true, false), (true, false)); + assert_eq!(diamond_batch_disposition(true, true), (true, false)); + // Failed batch with nothing mined: just a health signal. + assert_eq!(diamond_batch_disposition(false, false), (false, true)); + // The defect: a driver error on a batch that already produced a + // CPU-verified DiamondMint used to drop the mint on the floor. The mint + // must still be sent AND the device still reported unhealthy. + assert_eq!(diamond_batch_disposition(false, true), (true, true)); + } + + #[test] + fn a_stop_flag_is_observed_by_the_diamond_loops() { + assert!(!should_stop(&None)); + let flag = Arc::new(AtomicBool::new(false)); + let stop = Some(flag.clone()); + assert!(!should_stop(&stop)); + flag.store(true, Relaxed); + assert!(should_stop(&stop)); + } +} + +#[cfg(test)] +mod diamond_job_reorg_tests { + use super::*; + + fn hash_of(byte: u8) -> Hash { + Hash::from([byte; HASH_WIDTH]) + } + + fn snapshot() -> (u32, u64) { + // Exactly what a worker thread takes at entry. + let (number, epoch, _prev_hash) = snapshot_diamond_job(); + (number, epoch) + } + + fn live_prev_hash() -> Hash { + snapshot_diamond_job().2 + } + + #[test] + fn the_diamond_pre_image_matches_the_cpu_and_hits_only_the_two_sha3_branches() { + // sha3_256_hash_diamond picks its SHA3 padding purely from the pre-image + // LENGTH: 61 without a custom message, 93 with one. Any third length would + // be hashed with the wrong pad, so pin both the lengths and the fact that + // this builder reproduces x16rs::mine_diamond byte for byte. (A full GPU + // vector still needs real hardware; this is the CPU half of that net.) + let prev = Hash::from([7u8; HASH_WIDTH]); + let addr = Address::from([3u8; 21]); + let custom = Hash::from([9u8; HASH_WIDTH]); + let nonce_bytes = 0x0123_4567_89ab_cdefu64.to_be_bytes(); + + for number in [ + 1u32, + DIAMOND_ABOVE_NUMBER_OF_CREATE_BY_CUSTOM_MESSAGE, + DIAMOND_ABOVE_NUMBER_OF_CREATE_BY_CUSTOM_MESSAGE + 1, + 60000, + ] { + let with_custom = number > DIAMOND_ABOVE_NUMBER_OF_CREATE_BY_CUSTOM_MESSAGE; + let custom_nonce: &[u8] = if with_custom { custom.as_bytes() } else { &[] }; + let pre_image = diamond_pre_image(&prev, &nonce_bytes, &addr, custom_nonce); + assert_eq!(pre_image.len(), if with_custom { 93 } else { 61 }); + + let (ssshash, reshash, diastr) = + x16rs::mine_diamond(number, &prev, &nonce_bytes, &addr, custom_nonce); + // Same SHA3 input as consensus. + assert_eq!(x16rs::calculate_hash(pre_image), ssshash); + // Same medium hash and same name from that SHA3. + let repeat = x16rs::mine_diamond_hash_repeat(number); + assert_eq!(x16rs::x16rs_hash(repeat, &ssshash), reshash); + assert_eq!(x16rs::diamond_hash(&reshash), diastr); + } + } + + #[test] + fn a_diamond_tip_reorg_retires_the_snapshotted_job_immediately() { + // These statics are process-wide, so this is the only test that installs + // jobs; it walks the whole reorg sequence in one function. + assert!(install_diamond_job(100, hash_of(1))); + let (snap_num, snap_epoch) = snapshot(); + assert_eq!(snap_num, 100); + assert!(diamond_job_is_current(snap_num, snap_epoch)); + + // An unchanged re-poll (every MINING_INTERVAL) must not disturb a worker. + assert!(!install_diamond_job(100, hash_of(1))); + assert!(diamond_job_is_current(snap_num, snap_epoch)); + + // Case B: the node reorged and born.hash for the SAME number changed. + // Every hash the worker produces from here on carries a prev_hash from the + // dead branch and is rejected by the submit endpoint. + assert!(install_diamond_job(100, hash_of(2))); + assert_eq!(live_prev_hash(), hash_of(2)); + // The old exit test was `snapshot < live`, which cannot see this at all. + assert!(!(snap_num < MINING_DIAMOND_NUM.load(Acquire))); + assert!(!diamond_job_is_current(snap_num, snap_epoch)); + + // Case A: the reorg rolled the diamond tip BACK to a lower number. + let (snap_num, snap_epoch) = snapshot(); + assert!(install_diamond_job(98, hash_of(3))); + let live_num = MINING_DIAMOND_NUM.load(Acquire); + assert!(live_num < snap_num, "the tip must have rolled back"); + // Again the old strict-advance test would have kept the worker grinding + // number 100 through the entire 98 -> 99 -> 100 recovery. + assert!(!(snap_num < live_num)); + assert!(!diamond_job_is_current(snap_num, snap_epoch)); + + // Climbing back up is a new job every step, including the step that lands + // on the number the worker had originally snapshotted. + let (snap_num, snap_epoch) = snapshot(); + assert!(install_diamond_job(99, hash_of(4))); + assert!(!diamond_job_is_current(snap_num, snap_epoch)); + let (snap_num, snap_epoch) = snapshot(); + assert!(install_diamond_job(100, hash_of(5))); + assert!(!diamond_job_is_current(snap_num, snap_epoch)); + assert!(diamond_job_is_current(100, current_diamond_epoch())); + } +} + fn run_diamond_mining_benchmark(cnf: &DiaWorkConf, config_path: &str) { #[cfg(not(feature = "ocl"))] { @@ -892,7 +1448,7 @@ fn run_diamond_mining_benchmark(cnf: &DiaWorkConf, config_path: &str) { return; } println!( - "[benchmark] HACD: GPU tuning uses same profiles as HAC — run poworker benchmark or share ini." + "[benchmark] HACD: GPU tuning uses same profiles as HAC; run poworker benchmark or share ini." ); let scan = crate::opencl_diag::scan_opencl(); let init_unitsize = cnf.unitsize.max(128); diff --git a/app/src/efficiency.rs b/app/src/efficiency.rs index 51459533..56a8da2f 100644 --- a/app/src/efficiency.rs +++ b/app/src/efficiency.rs @@ -76,7 +76,9 @@ impl EfficiencyConf { let sec = sys::ini_section(ini, "efficiency"); let mode_raw = ini_must(sec, "mode", "profit"); let supervene_max = ini_must_u64(sec, "supervene_max", 0) as u32; - let supervene_min = ini_must_u64(sec, "supervene_min", 2) as u32; + // Default floor of 1, not 2: an explicit `supervene=1` must mean one + // thread, not be silently bumped to two by the minimum. + let supervene_min = ini_must_u64(sec, "supervene_min", 1) as u32; EfficiencyConf { mode: EfficiencyMode::from_str(&mode_raw), power_cost_kwh: ini_must_f64(sec, "power_cost_kwh", 0.15), @@ -84,7 +86,7 @@ impl EfficiencyConf { cpu_watts_per_thread: ini_must_f64(sec, "cpu_watts_per_thread", 8.0), hac_price: ini_must_f64(sec, "hac_price", 0.0), dynamic_supervene: ini_must_bool(sec, "dynamic_supervene", true), - supervene_min: supervene_min.max(0), + supervene_min, supervene_max, oom_fallback: ini_must_bool(sec, "oom_fallback", true), max_temp_c: ini_must_u64(sec, "max_temp_c", 0) as u32, @@ -109,10 +111,15 @@ impl EfficiencyConf { return 0; } let mut sv = configured.max(1); - if self.supervene_max > 0 { - sv = sv.min(self.supervene_max); - } - sv.max(self.supervene_min) + let hi = if self.supervene_max > 0 { + self.supervene_max + } else { + u32::MAX + }; + sv = sv.min(hi); + // Apply the floor, but never let a misconfigured min exceed the max: a + // `supervene_min > supervene_max` must not spawn more threads than the cap. + sv.max(self.supervene_min.min(hi)) } pub fn spawn_supervene(&self, configured: u32) -> u32 { diff --git a/app/src/gpu_oom.rs b/app/src/gpu_oom.rs index a7741258..0e14b081 100644 --- a/app/src/gpu_oom.rs +++ b/app/src/gpu_oom.rs @@ -1,6 +1,9 @@ -//! Per-GPU OpenCL work_groups OOM recovery (halving + optional ramp-back). +//! Per-GPU OpenCL work_groups OOM recovery (halving + optional ramp-back) and the +//! time-based GPU quarantine shared by the OpenCL and CUDA backends. -use std::sync::atomic::{AtomicBool, AtomicU32, Ordering::Relaxed}; +use std::sync::Mutex; +use std::sync::atomic::{AtomicBool, AtomicU32, AtomicU64, Ordering::Relaxed}; +use std::time::{Duration, Instant}; use crate::gpu_arch::ArchLimits; @@ -8,6 +11,11 @@ use crate::gpu_arch::ArchLimits; pub const OOM_FLOOR_WG: u32 = 512; /// Successful GPU batches before restoring base work_groups after OOM reduction. pub const OOM_RECOVERY_BATCHES: u32 = 16; +/// Clean batches before an experimental arch steps work_groups back up one level. +/// Much slower than [`OOM_RECOVERY_BATCHES`] because these arches went OOM at the +/// base size once already, but not "never", which cost half the card for the +/// whole session after a single transient CL_OUT_OF_RESOURCES. +pub const OOM_SLOW_RAMP_BATCHES: u32 = OOM_RECOVERY_BATCHES * 4; /// Per-device work_groups state — lives on [`crate::opencl_gpu::OpenclGpuHandle`]. pub struct GpuOomState { @@ -76,7 +84,7 @@ impl GpuOomState { let next = (cur / 2).max(floor); if next < cur { eprintln!( - "[efficiency] OpenCL error — reducing work_groups {} -> {} (floor={})", + "[efficiency] OpenCL error - reducing work_groups {} -> {} (floor={})", cur, next, floor ); self.effective_workgroups.store(next, Relaxed); @@ -90,15 +98,26 @@ impl GpuOomState { self.oom_floor_wg.max(1) } + /// Adopt the work_groups a rebuilt context actually allocated. + /// + /// This also re-arms the ramp-back whenever the new size is below base. Without + /// that, any path that drops the device to the floor WITHOUT going through + /// `record_error` (a context rebuild, the quarantine re-probe) would leave + /// `oom_reduced` false, `record_success` would return early forever, and the + /// card would stay at the floor for the rest of the session. pub fn sync_effective(&mut self, wg: u32) { let clamped = wg.max(1); self.effective_workgroups.store(clamped, Relaxed); + self.oom_reduced + .store(clamped < self.base_workgroups.max(1), Relaxed); + self.success_batches_since_oom.store(0, Relaxed); } + /// One clean batch. Restores work_groups after an OOM reduction: in one step + /// on standard arches, one level at a time on the experimental ones. The + /// experimental arches used to have NO ramp at all, so a single transient + /// CL_OUT_OF_RESOURCES halved throughput for the rest of the process. pub fn record_success(&self) { - if !self.oom_ramp_to_base { - return; - } let cur = self.effective_workgroups.load(Relaxed); let base = self.base_workgroups.max(1); if cur >= base { @@ -110,15 +129,283 @@ impl GpuOomState { return; } let n = self.success_batches_since_oom.fetch_add(1, Relaxed) + 1; - if n >= OOM_RECOVERY_BATCHES { - self.effective_workgroups.store(base, Relaxed); - self.oom_reduced.store(false, Relaxed); - self.success_batches_since_oom.store(0, Relaxed); - println!("[efficiency] GPU stable — restored work_groups to {}", base); + if self.oom_ramp_to_base { + if n >= OOM_RECOVERY_BATCHES { + self.effective_workgroups.store(base, Relaxed); + self.oom_reduced.store(false, Relaxed); + self.success_batches_since_oom.store(0, Relaxed); + println!("[efficiency] GPU stable - restored work_groups to {}", base); + } + return; + } + // Experimental arch (RDNA4): step up one level after a long clean run and + // back off again on the next error, so 32 -> 48 -> 64 rather than 32 for + // the whole session. The arch floor still applies on the way down. + if n < OOM_SLOW_RAMP_BATCHES { + return; + } + self.success_batches_since_oom.store(0, Relaxed); + let next = cur.saturating_add((cur / 2).max(1)).min(base); + if next > cur { + self.effective_workgroups.store(next, Relaxed); + if next >= base { + self.oom_reduced.store(false, Relaxed); + } + println!( + "[efficiency] GPU stable for {} batches - raising work_groups {} -> {}", + n, cur, next + ); } } } +/// Consecutive failed GPU batches before a device may be quarantined. +pub const GPU_QUARANTINE_MIN_FAILURES: u32 = 20; + +/// The run of failures must ALSO have lasted this long before the card is parked. +/// A failed batch costs only the bounded CPU recovery, so 20 failures on their own +/// are reached in well under a minute, which is inside a Windows TDR storm, a +/// driver update or a brief thermal excursion. Requiring both conditions keeps a +/// healthy card mining through a fast burst. +pub const GPU_QUARANTINE_MIN_ELAPSED: Duration = Duration::from_secs(120); + +/// First quarantine interval. Doubles on every failed re-probe. +pub const GPU_QUARANTINE_BASE_BACKOFF: Duration = Duration::from_secs(60); + +/// Longest quarantine interval. The card is re-probed at this cadence forever, so +/// a dead card costs one failed batch every half hour while a card that recovers +/// (driver reinstalled, case cooled down) resumes mining unattended. +pub const GPU_QUARANTINE_MAX_BACKOFF: Duration = Duration::from_secs(30 * 60); + +/// How often the "still quarantined" reminder is printed for the operator. +pub const GPU_QUARANTINE_NOTICE_INTERVAL: Duration = Duration::from_secs(60); + +/// Quarantine interval for a 1-based level, doubling up to the cap. +pub fn gpu_quarantine_backoff(level: u32) -> Duration { + let shift = level.saturating_sub(1).min(31); + let secs = GPU_QUARANTINE_BASE_BACKOFF + .as_secs() + .saturating_mul(1u64 << shift); + Duration::from_secs(secs).min(GPU_QUARANTINE_MAX_BACKOFF) +} + +/// Compact "90s" / "8m" / "2m 30s" for operator-facing messages. +pub fn format_backoff(d: Duration) -> String { + let secs = d.as_secs(); + if secs < 60 { + format!("{}s", secs) + } else if secs % 60 == 0 { + format!("{}m", secs / 60) + } else { + format!("{}m {}s", secs / 60, secs % 60) + } +} + +/// Whether a batch may run right now on a device that may be quarantined. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum GpuGate { + /// Normal operation. + Run, + /// The quarantine interval just expired: this batch re-probes the card. + Reprobe { level: u32, total_failures: u64 }, + /// Still quarantined: skip the device and mine the bounded CPU recovery. + Skip { + level: u32, + retry_in: Duration, + total_failures: u64, + /// True at most once per notice interval, so a non-technical operator can + /// see WHY the hashrate dropped without the log being flooded. + notify: bool, + }, +} + +/// Set on the failure that armed (or re-armed) the quarantine. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct GpuQuarantineEntry { + pub level: u32, + pub retry_in: Duration, +} + +/// Outcome of reporting one failed GPU batch. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct GpuFailureReport { + pub consecutive_failures: u32, + pub total_failures: u64, + /// `Some` only on the call that armed the quarantine, so the loud operator + /// alert is printed exactly once per level. + pub quarantined: Option, +} + +/// Read-only quarantine state for the panel / stats. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct GpuQuarantineStatus { + pub level: u32, + pub retry_in: Duration, + pub total_failures: u64, +} + +#[derive(Clone, Copy, Debug, Default)] +struct QuarantineInner { + consecutive_failures: u32, + level: u32, + first_failure_at: Option, + quarantine_until: Option, + reprobing: bool, + last_notice_at: Option, +} + +/// Time-based GPU quarantine shared by the OpenCL and CUDA backends. +/// +/// This replaces the write-once "disable the GPU for the session" latch, which had +/// no clearing path: a card that failed 20 batches during a driver reset stayed +/// off until a human restarted the process, which on an unattended miner is a +/// total GPU-income outage for hours. Here the card is parked for a growing +/// interval and then re-probed, forever. A genuinely dead card ends up quarantined +/// almost all of the time and costs nothing; a card recovering from a 30-second +/// TDR comes back on its own while the operator sleeps. +pub struct GpuQuarantine { + inner: Mutex, + total_failures: AtomicU64, +} + +impl Default for GpuQuarantine { + fn default() -> Self { + Self::new() + } +} + +impl GpuQuarantine { + pub fn new() -> Self { + Self { + inner: Mutex::new(QuarantineInner::default()), + total_failures: AtomicU64::new(0), + } + } + + fn lock(&self) -> std::sync::MutexGuard<'_, QuarantineInner> { + self.inner.lock().unwrap_or_else(|e| e.into_inner()) + } + + /// Gate one batch at `now`. + /// + /// Once the interval expires this opens the re-probe window and reports + /// [`GpuGate::Reprobe`] to exactly one caller. With several GPU threads a few + /// more may slip in beside the probe, which is bounded (one failed batch each, + /// then the quarantine re-arms at the next level) and cannot deadlock, unlike + /// a single-probe token that a crashing worker could leak. + pub fn gate(&self, now: Instant) -> GpuGate { + let mut st = self.lock(); + let total_failures = self.total_failures.load(Relaxed); + match st.quarantine_until { + None => GpuGate::Run, + Some(until) if now < until => { + let notify = match st.last_notice_at { + Some(last) => { + now.saturating_duration_since(last) >= GPU_QUARANTINE_NOTICE_INTERVAL + } + None => true, + }; + if notify { + st.last_notice_at = Some(now); + } + GpuGate::Skip { + level: st.level, + retry_in: until.saturating_duration_since(now), + total_failures, + notify, + } + } + Some(_) => { + st.quarantine_until = None; + st.reprobing = true; + st.last_notice_at = Some(now); + GpuGate::Reprobe { + level: st.level, + total_failures, + } + } + } + } + + /// Report one failed batch at `now`. + pub fn record_failure(&self, now: Instant) -> GpuFailureReport { + let total_failures = self.total_failures.fetch_add(1, Relaxed).saturating_add(1); + let mut st = self.lock(); + let consecutive_failures = st.consecutive_failures.saturating_add(1); + st.consecutive_failures = consecutive_failures; + let first = *st.first_failure_at.get_or_insert(now); + let persisted = now.saturating_duration_since(first); + // A failed re-probe re-arms straight away at the next level. Otherwise the + // run has to be long in COUNT and in TIME, so a burst cannot kill a card. + let trip = st.reprobing + || (consecutive_failures >= GPU_QUARANTINE_MIN_FAILURES + && persisted >= GPU_QUARANTINE_MIN_ELAPSED); + if !trip { + return GpuFailureReport { + consecutive_failures, + total_failures, + quarantined: None, + }; + } + st.reprobing = false; + st.level = st.level.saturating_add(1); + let retry_in = gpu_quarantine_backoff(st.level); + st.quarantine_until = Some(now.checked_add(retry_in).unwrap_or(now)); + st.last_notice_at = Some(now); + GpuFailureReport { + consecutive_failures, + total_failures, + quarantined: Some(GpuQuarantineEntry { + level: st.level, + retry_in, + }), + } + } + + /// Report one clean batch. Returns true when this cleared an armed quarantine + /// or an open re-probe window, i.e. the card just came back. + pub fn record_success(&self) -> bool { + let mut st = self.lock(); + let recovered = st.reprobing || st.level > 0 || st.quarantine_until.is_some(); + *st = QuarantineInner::default(); + recovered + } + + /// Current quarantine state for the panel / stats (None while mining). + pub fn status(&self, now: Instant) -> Option { + let st = self.lock(); + let until = st.quarantine_until?; + Some(GpuQuarantineStatus { + level: st.level, + retry_in: until.saturating_duration_since(now), + total_failures: self.total_failures.load(Relaxed), + }) + } + + /// One line a non-technical operator can read straight off the dashboard. + pub fn describe(&self, now: Instant) -> String { + match self.status(now) { + None => "ok".to_string(), + Some(status) => format!( + "quarantined (level {}), retry in {}, {} failed batches", + status.level, + format_backoff(status.retry_in), + status.total_failures + ), + } + } + + /// Consecutive failed batches since the last clean one. + pub fn consecutive_failures(&self) -> u32 { + self.lock().consecutive_failures + } + + /// Failed batches over the whole session. + pub fn total_failures(&self) -> u64 { + self.total_failures.load(Relaxed) + } +} + #[derive(Clone, Debug, PartialEq, Eq)] pub enum GpuBatchError { OutOfResources, @@ -227,4 +514,222 @@ mod tests { "gfx1201 must not ramp back to OOM-prone base" ); } + + #[test] + fn gfx1201_slow_ramp_recovers_from_one_transient_oom() { + // One transient CL_OUT_OF_RESOURCES used to cost half the card for the + // whole session, because record_success returned immediately for arches + // with oom_ramp_to_base = false. + let mut st = GpuOomState::new(64); + st.configure_floor(16 * 1024 * 1024 * 1024, 256, 64, 64, "gfx1201"); + st.record_error(64, true); + assert_eq!(st.effective_workgroups.load(Relaxed), 32); + + for _ in 0..OOM_SLOW_RAMP_BATCHES { + st.record_success(); + } + assert_eq!( + st.effective_workgroups.load(Relaxed), + 48, + "a long clean run must step gfx1201 back up one level" + ); + for _ in 0..OOM_SLOW_RAMP_BATCHES { + st.record_success(); + } + assert_eq!( + st.effective_workgroups.load(Relaxed), + 64, + "a second clean run must reach the configured base again" + ); + // And the arch floor still applies on the way back down. + st.record_error(64, true); + st.record_error(64, true); + assert_eq!(st.effective_workgroups.load(Relaxed), 32); + } + + #[test] + fn a_floor_forced_by_a_rebuild_or_re_probe_can_still_ramp_back() { + // The quarantine re-probe and the context rebuild both push the device to + // the floor through sync_effective, not record_error. If that left the + // ramp disarmed, a recovered card would mine at the floor forever. + let mut st = GpuOomState::new(2048); + st.configure_floor(16 * 1024 * 1024 * 1024, 256, 96, 2048, "gfx1100"); + st.sync_effective(st.floor_wg()); + assert_eq!(st.effective_workgroups.load(Relaxed), 512); + for _ in 0..OOM_RECOVERY_BATCHES { + st.record_success(); + } + assert_eq!( + st.effective_workgroups.load(Relaxed), + 2048, + "a card forced to the floor must recover its configured work_groups" + ); + } + + #[test] + fn slow_ramp_backs_off_again_on_the_next_error() { + let mut st = GpuOomState::new(64); + st.configure_floor(16 * 1024 * 1024 * 1024, 256, 64, 64, "gfx1201"); + st.record_error(64, true); + for _ in 0..OOM_SLOW_RAMP_BATCHES { + st.record_success(); + } + assert_eq!(st.effective_workgroups.load(Relaxed), 48); + st.record_error(64, true); + assert_eq!(st.effective_workgroups.load(Relaxed), 32); + } + + /// Fail one batch every 10s until the device is parked; returns when that + /// happened and for how long. + fn quarantine_after_a_long_failing_run( + q: &GpuQuarantine, + start: Instant, + ) -> (Instant, Duration) { + for i in 0..GPU_QUARANTINE_MIN_FAILURES { + let now = start + Duration::from_secs(10) * i; + if let Some(entry) = q.record_failure(now).quarantined { + return (now, entry.retry_in); + } + } + panic!("a long, slow run of failures must quarantine the device"); + } + + #[test] + fn a_fast_burst_of_failures_does_not_quarantine_a_healthy_card() { + // 20 failed batches take well under a minute (a failing batch fails at + // enqueue and the only cost is the bounded CPU recovery), which is inside + // one Windows TDR storm or a driver update. That must not park the card. + let q = GpuQuarantine::new(); + let start = Instant::now(); + for i in 0..GPU_QUARANTINE_MIN_FAILURES * 3 { + let now = start + Duration::from_millis(500 * i as u64); + let report = q.record_failure(now); + assert!( + report.quarantined.is_none(), + "failure {} inside a 30s burst must not quarantine the GPU", + report.consecutive_failures + ); + } + assert_eq!(q.gate(start + Duration::from_secs(31)), GpuGate::Run); + } + + #[test] + fn a_persistent_failing_run_quarantines_and_re_probes() { + let q = GpuQuarantine::new(); + let start = Instant::now(); + let (armed_at, retry_in) = quarantine_after_a_long_failing_run(&q, start); + assert_eq!(retry_in, GPU_QUARANTINE_BASE_BACKOFF); + + match q.gate(armed_at) { + GpuGate::Skip { level, notify, .. } => { + assert_eq!(level, 1); + // The loud ALERT was printed by the failure that armed it, so the + // gate stays quiet until the reminder interval is up. + assert!(!notify); + } + other => panic!("expected the device to be skipped, got {:?}", other), + } + assert!(matches!( + q.gate(armed_at + Duration::from_secs(1)), + GpuGate::Skip { notify: false, .. } + )); + assert!(q.status(armed_at).is_some()); + assert!(q.describe(armed_at).starts_with("quarantined (level 1)")); + + // Timer expired: exactly one caller is handed the re-probe. + let expired = armed_at + GPU_QUARANTINE_BASE_BACKOFF + Duration::from_secs(1); + assert!(matches!(q.gate(expired), GpuGate::Reprobe { level: 1, .. })); + assert_eq!(q.gate(expired), GpuGate::Run); + + // The probe failed, so the card is parked again at level 2 (2m). While a + // window longer than the notice interval runs, a non-technical operator + // gets a periodic reminder of WHY the hashrate dropped, not silence. + assert!(q.record_failure(expired).quarantined.is_some()); + assert!(matches!( + q.gate(expired + GPU_QUARANTINE_NOTICE_INTERVAL), + GpuGate::Skip { + level: 2, + notify: true, + .. + } + )); + } + + #[test] + fn a_failed_re_probe_doubles_the_backoff_up_to_the_cap() { + let q = GpuQuarantine::new(); + let start = Instant::now(); + let (armed_at, _) = quarantine_after_a_long_failing_run(&q, start); + + let mut now = armed_at; + let mut level = 1u32; + let mut seen = vec![GPU_QUARANTINE_BASE_BACKOFF]; + for _ in 0..8 { + now += gpu_quarantine_backoff(level) + Duration::from_secs(1); + assert!(matches!(q.gate(now), GpuGate::Reprobe { .. })); + // One failed probe must re-arm immediately, without waiting for + // another 20 failures over two minutes. + let entry = q + .record_failure(now) + .quarantined + .expect("a failed re-probe must re-arm the quarantine"); + level = entry.level; + seen.push(entry.retry_in); + } + assert_eq!( + &seen[..5], + &[ + Duration::from_secs(60), + Duration::from_secs(120), + Duration::from_secs(240), + Duration::from_secs(480), + Duration::from_secs(960), + ] + ); + assert_eq!(*seen.last().unwrap(), GPU_QUARANTINE_MAX_BACKOFF); + // Never permanently disabled: the card keeps being re-probed at the cap. + now += GPU_QUARANTINE_MAX_BACKOFF + Duration::from_secs(1); + assert!(matches!(q.gate(now), GpuGate::Reprobe { .. })); + } + + #[test] + fn a_successful_re_probe_clears_the_quarantine_completely() { + let q = GpuQuarantine::new(); + let start = Instant::now(); + let (armed_at, _) = quarantine_after_a_long_failing_run(&q, start); + let expired = armed_at + GPU_QUARANTINE_BASE_BACKOFF + Duration::from_secs(1); + assert!(matches!(q.gate(expired), GpuGate::Reprobe { .. })); + + assert!( + q.record_success(), + "recovery must be announced exactly once" + ); + assert!(!q.record_success()); + assert_eq!(q.gate(expired), GpuGate::Run); + assert_eq!(q.consecutive_failures(), 0); + assert!(q.status(expired).is_none()); + assert_eq!(q.describe(expired), "ok"); + + // Back to the full forgiving trigger: one later failure must not re-park. + assert!( + q.record_failure(expired + Duration::from_secs(1)) + .quarantined + .is_none() + ); + } + + #[test] + fn total_failures_survive_recovery_for_the_operator_alert() { + let q = GpuQuarantine::new(); + let start = Instant::now(); + quarantine_after_a_long_failing_run(&q, start); + let before = q.total_failures(); + assert!(before >= GPU_QUARANTINE_MIN_FAILURES as u64); + q.record_success(); + assert_eq!( + q.total_failures(), + before, + "session failure total must keep counting so a flapping card is visible" + ); + } } diff --git a/app/src/hash_util.rs b/app/src/hash_util.rs index 69fcb31e..a9b02f4a 100644 --- a/app/src/hash_util.rs +++ b/app/src/hash_util.rs @@ -25,7 +25,11 @@ pub fn hash_left_zero_pad3(dst: &[u8]) -> Vec { break; } } - dst[0..idx + 3].to_vec() + // Clamp the end: a degenerate hash whose first non-zero byte sits in the last + // two bytes (or an input shorter than 3 bytes) would otherwise slice past the + // end and panic inside the sole result/submit thread. + let end = (idx + 3).min(dst.len()); + dst[..end].to_vec() } pub fn diamond_more_power(dst: &[u8], src: &[u8]) -> bool { @@ -42,3 +46,87 @@ pub fn diamond_more_power(dst: &[u8], src: &[u8]) -> bool { } false } + +/// Mainnet consensus name shape: exactly DMD_L leading '0' chars followed by +/// DMD_M-DMD_L non-'0' chars. Mirrors `x16rs::check_diamond_hash_result` (and +/// `diamond_is_valid_name` in x16rs_diamond.cl) without allocating. +pub fn diamond_name_is_valid(dia: &[u8]) -> bool { + const DMD_L: usize = 10; + const DMD_M: usize = 16; + if dia.len() != DMD_M { + return false; + } + dia[..DMD_L].iter().all(|c| *c == b'0') && dia[DMD_L..].iter().all(|c| *c != b'0') +} + +/// True when candidate `dst` should replace the current best `src`. +/// +/// `diamond_more_power` on its own is pure more-leading-zeros-wins, so it ranks +/// an 11+ zero overshoot (which can never be minted) above a real diamond. The +/// GPU kernel already ranks valid-first; this is the same rule on the host so +/// kernel, host and the console "best so far" cannot disagree. +pub fn diamond_better(dst: &[u8], src: &[u8]) -> bool { + match (diamond_name_is_valid(dst), diamond_name_is_valid(src)) { + (true, false) => true, + (false, true) => false, + _ => diamond_more_power(dst, src), + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn zero_pad3_never_slices_past_the_end() { + // An upstream-supplied target with 30+ leading zero bytes used to panic + // here and kill the result/submit thread. + let mut degenerate = [0u8; 32]; + degenerate[31] = 1; + assert_eq!(hash_left_zero_pad3(°enerate).len(), 32); + let mut near_end = [0u8; 32]; + near_end[30] = 1; + assert_eq!(hash_left_zero_pad3(&near_end).len(), 32); + assert_eq!(hash_left_zero_pad3(&[0u8; 32]).len(), 3); + assert!(hash_left_zero_pad3(&[]).is_empty()); + assert_eq!(hash_left_zero_pad3(&[0u8, 1u8]).len(), 2); + } + + #[test] + fn zero_pad3_keeps_three_bytes_after_the_first_non_zero() { + let mut normal = [0u8; 32]; + normal[2] = 9; + assert_eq!(hash_left_zero_pad3(&normal), vec![0u8, 0u8, 9u8, 0u8, 0u8]); + } + + const MINTABLE: &[u8; 16] = b"0000000000WTYUIA"; + const OVERSHOOT: &[u8; 16] = b"00000000000TYUIA"; + const WEAK: &[u8; 16] = b"000000000WTYUIAH"; + + #[test] + fn a_valid_diamond_name_is_exactly_ten_zeros_then_six_non_zeros() { + assert!(diamond_name_is_valid(MINTABLE)); + // 11 leading zeros: "more power" but the node rejects it. + assert!(!diamond_name_is_valid(OVERSHOOT)); + // 9 leading zeros: below the bar. + assert!(!diamond_name_is_valid(WEAK)); + assert!(!diamond_name_is_valid(b"0000000000WTYUI")); + assert!(!diamond_name_is_valid(&[0u8; 16])); + } + + #[test] + fn the_ranking_prefers_a_mintable_diamond_over_a_stronger_overshoot() { + // This is the defect: the raw leading-zero rule ranks the unmintable + // 11-zero hash above the real diamond, so the GPU work group and the + // console "best" both report a hash the miner would never submit. + assert!(diamond_more_power(OVERSHOOT, MINTABLE)); + assert!(!diamond_better(OVERSHOOT, MINTABLE)); + assert!(diamond_better(MINTABLE, OVERSHOOT)); + // Same rule as the kernel: valid beats invalid, otherwise fall back to + // leading zeros. + assert!(diamond_better(MINTABLE, WEAK)); + assert!(!diamond_better(WEAK, MINTABLE)); + assert!(diamond_better(OVERSHOOT, WEAK)); + assert!(!diamond_better(MINTABLE, MINTABLE)); + } +} diff --git a/app/src/lib.rs b/app/src/lib.rs index aa771f3e..03e7da04 100644 --- a/app/src/lib.rs +++ b/app/src/lib.rs @@ -5,6 +5,7 @@ pub mod gpu_arch; pub mod gpu_oom; pub mod hash_util; pub mod mining_batch; +pub mod mining_guard; pub mod mining_runtime; pub mod mining_stats; pub mod panel_tuning; diff --git a/app/src/mining_batch.rs b/app/src/mining_batch.rs index 5a97ee5f..e0b62f03 100644 --- a/app/src/mining_batch.rs +++ b/app/src/mining_batch.rs @@ -7,16 +7,45 @@ use crate::hash_util::hash_more_power; #[cfg(feature = "ocl")] use crate::gpu_oom::GpuBatchError; -#[cfg(feature = "ocl")] +#[cfg(any(feature = "ocl", feature = "cuda"))] use crate::mining_runtime::MiningRuntimeState; #[cfg(feature = "ocl")] use crate::opencl_gpu::OpenclGpuHandle; #[cfg(feature = "ocl")] -use crate::opencl_gpu::block::do_group_block_mining_opencl; +use crate::opencl_gpu::block::do_group_block_mining_opencl_shares; -#[cfg(any(feature = "ocl", test))] +#[cfg(any(feature = "ocl", feature = "cuda", test))] const GPU_ERROR_CPU_RECOVERY_NONCES: u32 = 100_000; +/// Block intro serialization is a fixed 89-byte layout, shared by the OpenCL +/// kernel (opencl_gpu/block.rs) and CUDA (`x16rs_cuda::STUFF_BYTES`). The nonce +/// lives at bytes 79..83, so a shorter or longer intro would be hashed over a +/// different message length than the kernel used and the resulting mismatch would +/// be charged to the card as a "GPU integrity error". +pub const BLOCK_INTRO_BYTES: usize = 89; + +/// What a failed CUDA batch should do besides falling back to the CPU. +#[cfg(any(feature = "cuda", test))] +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct CudaFailureAction { + /// Halve the effective work groups. Skipped for a sticky context fault: that + /// fault is grid independent and x16rs-cuda rebuilds the context itself, so + /// shrinking the launch would only cost throughput once the card recovers. + pub halve_workgroups: bool, +} + +/// Decide what to do about a failed CUDA batch from the fault class. +/// +/// Whether the card is parked is NOT decided here: both backends now defer to the +/// shared [`crate::gpu_oom::GpuQuarantine`], which needs the failures to persist in +/// count AND in time, and which re-probes instead of latching the card off. +#[cfg(any(feature = "cuda", test))] +pub fn cuda_failure_action(sticky: bool) -> CudaFailureAction { + CudaFailureAction { + halve_workgroups: !sticky, + } +} + /// Inputs for one block-mining batch across CPU / CUDA / OpenCL backends. pub struct BatchCtx { pub height: u64, @@ -27,6 +56,51 @@ pub struct BatchCtx { pub localsize: u32, pub unitsize: u32, pub thermal_wg_cap: Option, + /// Threshold for the pool share list, or `None` for solo mining. + /// + /// A pool serves its SHARE target as the template `target_hash`, so this is + /// that same value and there is no second threshold to carry. `None` is what + /// keeps solo mining byte identical on BOTH GPU backends: the kernel is + /// launched with share_capacity=0, skips the appending block entirely, and the + /// host neither writes nor reads a share buffer. + pub share_target: Option<[u8; 32]>, +} + +/// One nonce whose hash beat the share target, already re-hashed on the CPU. +/// +/// Against a pool each of these is a separately payable PPLNS share, which is +/// why a batch reports all of them instead of only its strongest. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct MinedShare { + pub nonce: u32, + pub hash: [u8; 32], +} + +/// What a GPU backend produced for one batch, before the CPU tail is merged in. +pub struct GpuBatchOutcome { + /// The single strongest nonce/hash. Solo mining and the block-found path + /// read only this, and it is produced exactly as it always was. + pub best: (u32, [u8; 32]), + pub gpu_nonce_space: u32, + /// Verified extra shares. Always empty for solo mining; both GPU backends + /// fill it when the miner is pooled. + pub shares: Vec, + /// Hits the kernel counted but could not store. Non-zero means the fixed + /// capacity was exceeded and the miner is being paid for less than it mined. + pub share_overflow: u64, +} + +impl GpuBatchOutcome { + /// A best-only outcome, i.e. what every backend produced before the pool + /// share list existed. + pub fn best_only(best: (u32, [u8; 32]), gpu_nonce_space: u32) -> GpuBatchOutcome { + GpuBatchOutcome { + best, + gpu_nonce_space, + shares: Vec::new(), + share_overflow: 0, + } + } } /// Result of one batch including GPU/CPU nonce accounting for stats. @@ -35,6 +109,34 @@ pub struct BatchResult { pub result_hash: [u8; 32], pub gpu_nonce_space: u32, pub cpu_nonce_space: u32, + /// Payable nonces BESIDES `head_nonce`, which is submitted on its own. Empty + /// for every solo batch and every CPU batch. + pub shares: Vec, + /// Hits the GPU counted but could not store. + pub share_overflow: u64, +} + +impl BatchResult { + /// Every submission this batch owes the upstream: the head result plus each + /// further share. One entry per payable nonce, which is the whole point - + /// reporting one per batch tied pool credit to batch cadence, not hashrate. + pub fn submission_nonces(&self) -> Vec { + let mut nonces = Vec::with_capacity(self.shares.len() + 1); + nonces.push(self.head_nonce); + nonces.extend(self.shares.iter().map(|share| share.nonce)); + nonces + } +} + +/// Split a kernel share readback into what was stored and what was lost. +/// +/// `hits` is the kernel's atomic counter, which counts EVERY nonce under the +/// share target including those that did not fit. Returning the overflow instead +/// of silently truncating is what lets the miner tell its operator it is +/// undersampling rather than quietly losing income. +pub fn split_share_readback(hits: u64, capacity: usize) -> (usize, u64) { + let capacity = capacity as u64; + (hits.min(capacity) as usize, hits.saturating_sub(capacity)) } pub struct GpuBatchPlan { @@ -87,12 +189,12 @@ pub fn merge_cpu_tail( } } -/// Compare two 32-byte hashes; returns true if `candidate` beats `current`. -pub fn hash_beats(candidate: &[u8; 32], current: &[u8; 32]) -> bool { - hash_more_power(candidate, current) -} -/// Verify the GPU's best nonce/hash pair before it can reach submission. -pub fn verify_gpu_best_result( +/// Verify one GPU nonce/hash pair before it can reach submission. +/// +/// Used for the batch's best result AND for every entry of the pool share list: +/// a card returning garbage has to be caught here, not forwarded to the pool as +/// a hundred bad shares that get the miner throttled or banned. +pub fn verify_gpu_nonce_result( height: u64, block_intro: &[u8], nonce_start: u32, @@ -108,8 +210,11 @@ pub fn verify_gpu_best_result( best.0, nonce_start, gpu_nonce_space )); } - if block_intro.len() < 83 { - return Err("block intro is too short for nonce verification".to_string()); + if block_intro.len() != BLOCK_INTRO_BYTES { + return Err(format!( + "block intro must be {BLOCK_INTRO_BYTES} bytes for nonce verification, got {}", + block_intro.len() + )); } let mut verify_intro = block_intro.to_vec(); @@ -126,21 +231,57 @@ pub fn verify_gpu_best_result( Ok(()) } +/// Verify every entry of a GPU share list and reject the batch if any is wrong. +/// +/// Two things are checked per entry: that the CPU reproduces the hash the card +/// reported for that nonce, and that the hash really does beat the share target +/// the kernel was told to filter on. A card that gets either wrong is faulty, and +/// the existing policy for a faulty card is to fail the whole batch rather than +/// pick out the entries that happen to look right. +pub fn verify_gpu_shares( + height: u64, + block_intro: &[u8], + nonce_start: u32, + gpu_nonce_space: u32, + share_target: &[u8; 32], + raw: &[(u32, [u8; 32])], +) -> Result, String> { + let mut shares = Vec::with_capacity(raw.len()); + for entry in raw { + verify_gpu_nonce_result(height, block_intro, nonce_start, gpu_nonce_space, entry)?; + // Equal-inclusive, the same test the node and the pool apply: a hash + // landing exactly on target is payable. + if hash_more_power(share_target, &entry.1) { + return Err(format!( + "GPU listed nonce {} as a share but its hash {} is above the share target {}", + entry.0, + hex::encode(entry.1), + hex::encode(share_target) + )); + } + shares.push(MinedShare { + nonce: entry.0, + hash: entry.1, + }); + } + Ok(shares) +} + /// Finish a GPU batch: merge CPU tail nonces into the best hash. pub fn finish_gpu_batch( height: u64, block_intro: Vec, nonce_start: u32, nonce_space: u32, - gpu_best: (u32, [u8; 32]), - gpu_nonce_space: u32, + gpu: GpuBatchOutcome, cpu_mine: impl Fn(u64, Vec, u32, u32) -> (u32, [u8; 32]), ) -> BatchResult { + let gpu_nonce_space = gpu.gpu_nonce_space; let tail_space = nonce_space.saturating_sub(gpu_nonce_space); let (head_nonce, result_hash) = if tail_space > 0 { let tail_start = nonce_start.saturating_add(gpu_nonce_space); merge_cpu_tail( - gpu_best, + gpu.best, height, block_intro, tail_start, @@ -148,13 +289,19 @@ pub fn finish_gpu_batch( cpu_mine, ) } else { - gpu_best + gpu.best }; + // The head result is submitted on its own, so keeping it in the list too + // would buy a second HTTP round trip and a `duplicate` answer for it. + let mut shares = gpu.shares; + shares.retain(|share| share.nonce != head_nonce); BatchResult { head_nonce, result_hash, gpu_nonce_space, cpu_nonce_space: tail_space, + shares, + share_overflow: gpu.share_overflow, } } @@ -172,11 +319,13 @@ pub fn cpu_batch_fallback( result_hash, gpu_nonce_space: 0, cpu_nonce_space: nonce_space, + shares: Vec::new(), + share_overflow: 0, } } /// Recover a bounded prefix after an OpenCL failure and skip the rest of the failed window. -#[cfg(any(feature = "ocl", test))] +#[cfg(any(feature = "ocl", feature = "cuda", test))] fn cpu_gpu_error_recovery( height: u64, block_intro: Vec, @@ -242,6 +391,18 @@ impl BlockMinerBackend for OpenclBlockBackend { ctx: &BatchCtx, cpu_mine: &dyn Fn(u64, Vec, u32, u32) -> (u32, [u8; 32]), ) -> BatchResult { + // Quarantined, not disabled: the card is parked for a growing interval and + // re-probed automatically, so a driver reset or a thermal excursion no + // longer costs the whole session's GPU income. + if self.gpu.quarantine_blocks_batch() { + return cpu_gpu_error_recovery( + ctx.height, + ctx.block_intro.clone(), + ctx.nonce_start, + ctx.nonce_space, + cpu_mine, + ); + } let wg_cap = self.gpu.workgroups(ctx.configured_wg, ctx.thermal_wg_cap); let Some(plan) = plan_gpu_batch(ctx.nonce_space, wg_cap, ctx.localsize, ctx.unitsize) else { @@ -256,7 +417,7 @@ impl BlockMinerBackend for OpenclBlockBackend { let gpu_result = { let opencl = self.gpu.lock_resources(); - do_group_block_mining_opencl( + do_group_block_mining_opencl_shares( &opencl, ctx.height, ctx.block_intro.clone(), @@ -264,6 +425,7 @@ impl BlockMinerBackend for OpenclBlockBackend { plan.workgroups_eff, ctx.localsize, ctx.unitsize, + ctx.share_target.as_ref(), ) }; @@ -280,39 +442,63 @@ impl BlockMinerBackend for OpenclBlockBackend { cpu_mine, ) } - Ok(best) => { - if let Err(message) = verify_gpu_best_result( + Ok(output) => { + // The best result is verified exactly as before, and every share + // in the list goes through the same check. Whichever fails, the + // batch is charged to the card and none of it is submitted. + let verified = verify_gpu_nonce_result( ctx.height, &ctx.block_intro, ctx.nonce_start, plan.gpu_nonce_space, - &best, - ) { - let integrity_error = - GpuBatchError::Other(format!("GPU integrity error: {message}")); - eprintln!("[OpenCL] {}", integrity_error.display()); - self.gpu.on_batch_error( - integrity_error, - false, - ctx.configured_wg, - &self.runtime, - ); - return cpu_gpu_error_recovery( + &output.best, + ) + .and_then(|()| match ctx.share_target.as_ref() { + Some(target) => verify_gpu_shares( ctx.height, - ctx.block_intro.clone(), + &ctx.block_intro, ctx.nonce_start, - ctx.nonce_space, - cpu_mine, - ); - } + plan.gpu_nonce_space, + target, + &output.shares, + ), + None => Ok(Vec::new()), + }); + let shares = match verified { + Ok(shares) => shares, + Err(message) => { + let integrity_error = + GpuBatchError::Other(format!("GPU integrity error: {message}")); + eprintln!("[OpenCL] {}", integrity_error.display()); + self.gpu.on_batch_error( + integrity_error, + false, + ctx.configured_wg, + &self.runtime, + ); + return cpu_gpu_error_recovery( + ctx.height, + ctx.block_intro.clone(), + ctx.nonce_start, + ctx.nonce_space, + cpu_mine, + ); + } + }; + let (_, share_overflow) = + split_share_readback(output.share_hits, crate::opencl_gpu::SHARE_LIST_CAPACITY); self.gpu.on_batch_success(ctx.configured_wg, &self.runtime); finish_gpu_batch( ctx.height, ctx.block_intro.clone(), ctx.nonce_start, ctx.nonce_space, - best, - plan.gpu_nonce_space, + GpuBatchOutcome { + best: output.best, + gpu_nonce_space: plan.gpu_nonce_space, + shares, + share_overflow, + }, cpu_mine, ) } @@ -322,7 +508,7 @@ impl BlockMinerBackend for OpenclBlockBackend { #[cfg(feature = "cuda")] pub struct CudaBlockBackend { - pub cuda: Arc, + pub cuda: Arc, pub configured_wg: u32, pub runtime: Arc, } @@ -338,7 +524,27 @@ impl BlockMinerBackend for CudaBlockBackend { ctx: &BatchCtx, cpu_mine: &dyn Fn(u64, Vec, u32, u32) -> (u32, [u8; 32]), ) -> BatchResult { - let wg_cap = self.cuda.workgroups.min(self.configured_wg); + // Honor the OOM/error backoff AND the thermal governor's graded cap, so a + // hot or memory-pressured NVIDIA card steps down smoothly instead of + // running full tilt until the hard thermal pause. + if self.cuda.quarantine_blocks_batch() { + // The card is parked for a growing backoff (see the alert below) and + // will be re-probed automatically. Keep the bounded CPU recovery so + // the miner still makes progress without hammering a sick device. + return cpu_gpu_error_recovery( + ctx.height, + ctx.block_intro.clone(), + ctx.nonce_start, + ctx.nonce_space, + cpu_mine, + ); + } + let wg_cap = self + .cuda + .effective_wg() + .min(self.configured_wg) + .min(ctx.thermal_wg_cap.unwrap_or(u32::MAX)) + .max(1); let localsize = x16rs_cuda::DEFAULT_LOCAL_SIZE; let Some(plan) = plan_gpu_batch(ctx.nonce_space, wg_cap, localsize, self.cuda.unit_size) else { @@ -351,17 +557,30 @@ impl BlockMinerBackend for CudaBlockBackend { ); }; - match crate::do_group_block_mining_cuda( + match crate::poworker::do_group_block_mining_cuda_shares( &self.cuda, ctx.height, ctx.block_intro.clone(), ctx.nonce_start, plan.workgroups_eff, + ctx.share_target.as_ref(), ) { Err(e) => { eprintln!("[CUDA] batch failed: {e}"); self.runtime.record_gpu_error_event(); - cpu_batch_fallback( + let report = self.cuda.note_batch_failure(); + let action = cuda_failure_action(e.is_sticky()); + if action.halve_workgroups { + // Back off the effective work-groups so the next batch tries a + // smaller, likely-runnable size instead of failing forever. + let reduced = self.cuda.record_error(); + eprintln!("[CUDA] reducing work_groups to {reduced} after error"); + } + self.cuda + .announce_quarantine(&report, &format!("Last error: {e}")); + // Cap CPU recovery like the OpenCL path so a CUDA error cannot + // make one CPU thread grind the whole GPU-sized nonce window. + cpu_gpu_error_recovery( ctx.height, ctx.block_intro.clone(), ctx.nonce_start, @@ -369,15 +588,77 @@ impl BlockMinerBackend for CudaBlockBackend { cpu_mine, ) } - Ok(best) => finish_gpu_batch( - ctx.height, - ctx.block_intro.clone(), - ctx.nonce_start, - ctx.nonce_space, - best, - plan.gpu_nonce_space, - cpu_mine, - ), + Ok(output) => { + // Re-verify the GPU's best hash on the CPU, like OpenCL, so a + // faulty card cannot make us submit a wrong solution. Every entry + // of the share list goes through the SAME check (verify_gpu_shares + // calls verify_gpu_nonce_result per entry and additionally proves + // the hash really beats the target the kernel filtered on), so a + // card returning garbage is caught here and not forwarded to the + // pool as a hundred bad shares that get the miner throttled. + let verified = verify_gpu_nonce_result( + ctx.height, + &ctx.block_intro, + ctx.nonce_start, + plan.gpu_nonce_space, + &output.best, + ) + .and_then(|()| match ctx.share_target.as_ref() { + Some(target) => verify_gpu_shares( + ctx.height, + &ctx.block_intro, + ctx.nonce_start, + plan.gpu_nonce_space, + target, + &output.shares, + ), + None => Ok(Vec::new()), + }); + let shares = match verified { + Ok(shares) => shares, + Err(message) => { + eprintln!("[CUDA] GPU integrity error: {message}"); + self.runtime.record_gpu_error_event(); + // A card returning hashes the CPU cannot reproduce is as + // dead as one that fails to launch, so it shares the same + // budget. + let report = self.cuda.note_batch_failure(); + self.cuda.record_error(); + self.cuda.announce_quarantine( + &report, + "The card is returning hashes the CPU cannot reproduce; check for an overclock, bad memory or a failing driver.", + ); + return cpu_gpu_error_recovery( + ctx.height, + ctx.block_intro.clone(), + ctx.nonce_start, + ctx.nonce_space, + cpu_mine, + ); + } + }; + // Clean batch: ramp effective work-groups back toward the max. + self.cuda.record_success(); + // The kernel's counter is the TOTAL number of hits, so anything it + // could not store is lost income and has to be reported, not + // silently truncated. Same split as the OpenCL arm, against the + // kernel's own capacity. + let (_, share_overflow) = + split_share_readback(output.share_hits, x16rs_cuda::SHARE_LIST_CAPACITY); + finish_gpu_batch( + ctx.height, + ctx.block_intro.clone(), + ctx.nonce_start, + ctx.nonce_space, + GpuBatchOutcome { + best: output.best, + gpu_nonce_space: plan.gpu_nonce_space, + shares, + share_overflow, + }, + cpu_mine, + ) + } } } } @@ -396,12 +677,12 @@ mod tests { let result_hash = x16rs::block_hash(height, &verified_intro); let valid = (result_nonce, result_hash); - verify_gpu_best_result(height, &block_intro, nonce_start, 256, &valid).unwrap(); + verify_gpu_nonce_result(height, &block_intro, nonce_start, 256, &valid).unwrap(); let mut bad_hash = result_hash; bad_hash[0] ^= 1; assert!( - verify_gpu_best_result( + verify_gpu_nonce_result( height, &block_intro, nonce_start, @@ -411,7 +692,7 @@ mod tests { .is_err() ); assert!( - verify_gpu_best_result( + verify_gpu_nonce_result( height, &block_intro, nonce_start, @@ -422,6 +703,87 @@ mod tests { ); } + #[test] + fn a_sticky_cuda_fault_does_not_halve_the_grid() { + // A sticky context fault is grid independent and x16rs-cuda rebuilds the + // context itself, so halving the work groups would only cost throughput. + assert!(!cuda_failure_action(true).halve_workgroups); + assert!(cuda_failure_action(false).halve_workgroups); + } + + #[test] + fn a_run_of_failed_cuda_batches_quarantines_instead_of_killing_the_card() { + use crate::gpu_oom::{ + GPU_QUARANTINE_BASE_BACKOFF, GPU_QUARANTINE_MIN_ELAPSED, GPU_QUARANTINE_MIN_FAILURES, + GpuGate, GpuQuarantine, + }; + use std::time::{Duration, Instant}; + + // CUDA used to disable the card on a bare count, with no at-floor and no + // elapsed-time precondition, and with no way back short of a restart. + // Both backends now share this policy, so a fast burst is survivable and a + // real fault is only ever a timed quarantine. + let quarantine = GpuQuarantine::new(); + let start = Instant::now(); + for i in 0..GPU_QUARANTINE_MIN_FAILURES * 2 { + let now = start + Duration::from_millis(300 * i as u64); + assert!( + quarantine.record_failure(now).quarantined.is_none(), + "a burst of {} failures in under a minute must not park a CUDA card", + i + 1 + ); + } + + let quarantine = GpuQuarantine::new(); + let mut armed = None; + for i in 0..GPU_QUARANTINE_MIN_FAILURES { + let now = start + Duration::from_secs(10) * i; + if let Some(entry) = quarantine.record_failure(now).quarantined { + armed = Some((now, entry)); + break; + } + } + let (armed_at, entry) = armed.expect("a persistent failing run must park the card"); + assert_eq!(entry.retry_in, GPU_QUARANTINE_BASE_BACKOFF); + assert!(armed_at.saturating_duration_since(start) >= GPU_QUARANTINE_MIN_ELAPSED); + assert!(matches!( + quarantine.gate(armed_at), + GpuGate::Skip { level: 1, .. } + )); + // And it always comes back: the card is re-probed, never latched off. + assert!(matches!( + quarantine.gate(armed_at + GPU_QUARANTINE_BASE_BACKOFF + Duration::from_secs(1)), + GpuGate::Reprobe { .. } + )); + assert!(quarantine.record_success()); + } + + #[test] + fn nonce_verification_rejects_a_short_intro_instead_of_blaming_the_gpu() { + // An 83..88 byte intro used to pass the guard, get bytes 79..83 rewritten + // and then be hashed over a different message length than the kernel used. + // The guaranteed mismatch was reported as a GPU integrity error and + // charged against the card's failure budget for a host-side bug. + let height = 1u64; + let short_intro = vec![0u8; 85]; + let err = verify_gpu_nonce_result(height, &short_intro, 0, 256, &(7, [0u8; 32])) + .expect_err("an 85-byte intro must be rejected as a host-side length bug"); + assert!( + err.contains("89 bytes"), + "the error must name the 89-byte invariant, got: {err}" + ); + assert!( + verify_gpu_nonce_result( + height, + &vec![0u8; BLOCK_INTRO_BYTES + 1], + 0, + 256, + &(7, [0u8; 32]) + ) + .is_err() + ); + } + #[test] fn gpu_error_recovery_is_bounded_and_accounts_only_mined_nonces() { use std::cell::Cell; @@ -443,4 +805,232 @@ mod tests { assert_eq!(result.cpu_nonce_space, GPU_ERROR_CPU_RECOVERY_NONCES); assert_eq!(result.head_nonce, 7); } + + /// Hash of `nonce` under a fixed 89-byte intro, i.e. what an honest card + /// returns and what the CPU check recomputes. + fn honest_hit(height: u64, block_intro: &[u8], nonce: u32) -> (u32, [u8; 32]) { + let mut intro = block_intro.to_vec(); + intro[79..83].copy_from_slice(&nonce.to_be_bytes()); + (nonce, x16rs::block_hash(height, &intro)) + } + + #[test] + fn a_full_share_buffer_reports_what_it_could_not_store() { + // The capacity is fixed, so what matters is that the kernel's counter is + // the TOTAL and the overflow is surfaced. Losing shares silently is + // losing money silently, which is the whole defect being fixed. + assert_eq!(split_share_readback(0, 1024), (0, 0)); + assert_eq!(split_share_readback(7, 1024), (7, 0)); + assert_eq!(split_share_readback(1024, 1024), (1024, 0)); + assert_eq!(split_share_readback(1025, 1024), (1024, 1)); + assert_eq!(split_share_readback(9_000, 1024), (1024, 7_976)); + // A degenerate capacity must not make the overflow look like zero. + assert_eq!(split_share_readback(5, 0), (0, 5)); + + // And the overflow travels with the batch instead of being dropped on + // the floor between the kernel and the host. + let batch = finish_gpu_batch( + 1, + vec![0u8; BLOCK_INTRO_BYTES], + 0, + 256, + GpuBatchOutcome { + best: (3, [1u8; 32]), + gpu_nonce_space: 256, + shares: Vec::new(), + share_overflow: 7_976, + }, + |_, _, nonce_start, _| (nonce_start, [0u8; 32]), + ); + assert_eq!(batch.share_overflow, 7_976); + } + + #[test] + fn a_batch_of_hits_yields_one_submission_each_not_one_for_the_batch() { + // The measured defect: 34 billion hashes produced 77 submissions because + // the card reported one result per batch. Every hit under the share + // target has to become its own submission or pool credit tracks batch + // cadence instead of hashrate. + let height = 7u64; + let block_intro = vec![0u8; BLOCK_INTRO_BYTES]; + let easiest_target = [0xffu8; 32]; + let raw: Vec<(u32, [u8; 32])> = (0..64u32) + .map(|nonce| honest_hit(height, &block_intro, nonce)) + .collect(); + + let shares = + verify_gpu_shares(height, &block_intro, 0, 256, &easiest_target, &raw).unwrap(); + assert_eq!(shares.len(), 64); + + // The head result is whichever nonce the reduction picked; it is + // submitted on its own, so the list must not repeat it. + let best = raw[9]; + let batch = finish_gpu_batch( + height, + block_intro.clone(), + 0, + 256, + GpuBatchOutcome { + best, + gpu_nonce_space: 256, + shares, + share_overflow: 0, + }, + |_, _, nonce_start, _| (nonce_start, [0xffu8; 32]), + ); + let submissions = batch.submission_nonces(); + assert_eq!( + submissions.len(), + 64, + "64 payable nonces must produce 64 submissions, not one" + ); + assert_eq!(submissions[0], best.0); + let mut sorted = submissions.clone(); + sorted.sort_unstable(); + sorted.dedup(); + assert_eq!(sorted.len(), 64, "no nonce may be submitted twice"); + } + + #[test] + fn a_share_the_card_cannot_prove_fails_the_batch_instead_of_reaching_the_pool() { + let height = 7u64; + let block_intro = vec![0u8; BLOCK_INTRO_BYTES]; + let easiest_target = [0xffu8; 32]; + let good = honest_hit(height, &block_intro, 11); + + // A hash the CPU does not reproduce. + let mut wrong_hash = good; + wrong_hash.0 = 12; + assert!( + verify_gpu_shares(height, &block_intro, 0, 256, &easiest_target, &[good, wrong_hash]) + .is_err() + ); + // A nonce outside the batch window. + assert!( + verify_gpu_shares( + height, + &block_intro, + 0, + 8, + &easiest_target, + &[honest_hit(height, &block_intro, 900)] + ) + .is_err() + ); + // An honest hash the card listed even though it is ABOVE the target it + // was told to filter on: the compare is broken, so nothing is forwarded. + let strict_target = [0u8; 32]; + assert!( + verify_gpu_shares(height, &block_intro, 0, 256, &strict_target, &[good]).is_err() + ); + } + + #[test] + fn solo_mining_returns_the_same_single_result_it_always_did() { + // (d) of the brief: with no pool there is no share target, the OpenCL + // kernel is launched with share_capacity=0 and skips the list entirely, + // so a batch carries exactly one result, the same nonce and the same + // hash as before the list existed. + let height = 7u64; + let block_intro = vec![0u8; BLOCK_INTRO_BYTES]; + let best = honest_hit(height, &block_intro, 42); + + let solo_ctx = BatchCtx { + height, + block_intro: block_intro.clone(), + nonce_start: 0, + nonce_space: 256, + configured_wg: 1, + localsize: 256, + unitsize: 1, + thermal_wg_cap: None, + share_target: None, + }; + assert!( + solo_ctx.share_target.is_none(), + "a solo template must never carry a share target" + ); + + let batch = finish_gpu_batch( + height, + block_intro, + solo_ctx.nonce_start, + solo_ctx.nonce_space, + GpuBatchOutcome::best_only(best, 256), + |_, _, nonce_start, _| (nonce_start, [0xffu8; 32]), + ); + assert_eq!(batch.head_nonce, best.0); + assert_eq!(batch.result_hash, best.1); + assert!(batch.shares.is_empty()); + assert_eq!(batch.share_overflow, 0); + assert_eq!(batch.submission_nonces(), vec![best.0]); + assert_eq!(batch.gpu_nonce_space, 256); + assert_eq!(batch.cpu_nonce_space, 0); + } + + #[test] + #[cfg(feature = "cuda")] + fn the_cuda_arm_splits_the_readback_against_the_cuda_kernels_own_capacity() { + // The CUDA arm must not carry the OpenCL capacity or a hardcoded one: the + // number that decides how much of the list is live is the one the CUDA + // kernel was launched with, and the rest is reported as lost income. + let cap = x16rs_cuda::SHARE_LIST_CAPACITY; + assert_eq!(split_share_readback(0, cap), (0, 0)); + assert_eq!(split_share_readback(cap as u64, cap), (cap, 0)); + assert_eq!(split_share_readback(cap as u64 + 5, cap), (cap, 5)); + // And the kernel's own clamp agrees with the host's split, so the host can + // never ask for more entries than the device actually wrote. + assert_eq!(x16rs_cuda::stored_share_count(cap as u64 + 5), cap); + assert_eq!( + split_share_readback(cap as u64 + 5, cap).0, + x16rs_cuda::stored_share_count(cap as u64 + 5) + ); + } + + #[test] + #[cfg(feature = "cuda")] + fn a_solo_cuda_batch_asks_the_kernel_for_no_share_list_at_all() { + // (f) of the brief for the CUDA path: a solo template carries no share + // target, share_capacity_for turns that into 0, and 0 is what makes the + // kernel skip the appending block and the host skip the target upload, the + // counter clear and the readback. + let solo = BatchCtx { + height: 7, + block_intro: vec![0u8; BLOCK_INTRO_BYTES], + nonce_start: 0, + nonce_space: 256, + configured_wg: 1, + localsize: x16rs_cuda::DEFAULT_LOCAL_SIZE, + unitsize: 1, + thermal_wg_cap: None, + share_target: None, + }; + assert_eq!(x16rs_cuda::share_capacity_for(solo.share_target.as_ref()), 0); + + let pooled_target = [0x0fu8; 32]; + assert_eq!( + x16rs_cuda::share_capacity_for(Some(&pooled_target)), + x16rs_cuda::SHARE_LIST_CAPACITY as u32 + ); + } + + #[test] + #[cfg(feature = "cuda")] + fn the_cuda_reduction_still_needs_its_full_256_thread_block() { + // The kernel's shared local_nonces[256] and its power-of-two tree reduction + // make this structural: a smaller block silently corrupts the reduction and + // reports a wrong best nonce. The share list is an output path beside that + // reduction and must never be a reason to launch a different block size. + assert_eq!(x16rs_cuda::DEFAULT_LOCAL_SIZE, 256); + } + + #[test] + fn a_cpu_batch_never_carries_shares() { + let batch = cpu_batch_fallback(1, vec![0u8; BLOCK_INTRO_BYTES], 5, 16, |_, _, ns, _| { + (ns, [0u8; 32]) + }); + assert!(batch.shares.is_empty()); + assert_eq!(batch.share_overflow, 0); + assert_eq!(batch.submission_nonces(), vec![5]); + } } diff --git a/app/src/mining_guard.rs b/app/src/mining_guard.rs new file mode 100644 index 00000000..4d0df781 --- /dev/null +++ b/app/src/mining_guard.rs @@ -0,0 +1,58 @@ +//! Panic firewall shared by the block (HAC) and diamond (HACD) mining threads. +//! +//! Both miners have exactly one thread that drains results and submits winners. +//! A panic on that thread would end every payout while the process still looked +//! perfectly healthy, so each loop iteration runs inside `catch_unwind` here. + +pub(crate) fn panic_reason(payload: &(dyn std::any::Any + Send)) -> &str { + if let Some(text) = payload.downcast_ref::<&'static str>() { + text + } else if let Some(text) = payload.downcast_ref::() { + text.as_str() + } else { + "unknown panic payload" + } +} + +/// Panic firewall for the long-lived mining threads. The result thread is the +/// only path that submits winning work, so a panic that ended it would stop +/// every payout while the process still looked perfectly healthy. Contain the +/// panic, log it, and let the loop keep running. +pub(crate) fn guard_mining_iteration(label: &str, body: impl FnOnce()) { + if let Err(payload) = std::panic::catch_unwind(std::panic::AssertUnwindSafe(body)) { + eprintln!( + "[Mining] {} panicked and was contained: {}", + label, + panic_reason(&*payload) + ); + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn a_panicking_iteration_is_contained_and_reported() { + let previous_hook = std::panic::take_hook(); + std::panic::set_hook(Box::new(|_| {})); + let mut iterations = 0u32; + for round in 0..3 { + guard_mining_iteration("test loop", || { + if round == 1 { + panic!("simulated result thread panic"); + } + }); + iterations += 1; + } + std::panic::set_hook(previous_hook); + assert_eq!(iterations, 3); + } + + #[test] + fn a_panic_payload_of_any_shape_yields_a_reason() { + assert_eq!(panic_reason(&"static text"), "static text"); + assert_eq!(panic_reason(&"owned text".to_string()), "owned text"); + assert_eq!(panic_reason(&7u32), "unknown panic payload"); + } +} diff --git a/app/src/mining_runtime.rs b/app/src/mining_runtime.rs index f18d903f..1af63fdb 100644 --- a/app/src/mining_runtime.rs +++ b/app/src/mining_runtime.rs @@ -1,4 +1,4 @@ -//! Process-wide mining runtime: CPU assist, thermal cap, profit pause — not per-GPU OOM. +//! Process-wide mining runtime: CPU assist, thermal cap, profit pause - not per-GPU OOM. use std::sync::Arc; use std::sync::atomic::{ @@ -102,9 +102,21 @@ impl MiningRuntimeState { } fn store_workgroup_stat(target: &AtomicU32, value: u32) { - if value > 0 { - target.store(value, Relaxed); + if value == 0 { + return; } + // These are the MINIMUM (worst-case) work_groups reached, aggregated across + // every GPU. A plain store would overwrite with whichever device reported + // last, hiding a more constrained device; keep the smallest non-zero value + // seen instead. The live/current value is tracked separately in + // `reported_effective_wg`. + let _ = target.fetch_update(Relaxed, Relaxed, |cur| { + if cur == 0 || value < cur { + Some(value) + } else { + None + } + }); } /// Report per-GPU OOM work_groups; aggregates latest across devices for panel stats. @@ -165,7 +177,7 @@ impl MiningRuntimeState { self.thermal_cap_wg.store(wg, Relaxed); if !self.throttled.swap(true, Relaxed) || prev != wg { println!( - "[efficiency] Thermal {}C >= {}C — cap work_groups to {} (configured {})", + "[efficiency] Thermal {}C >= {}C - cap work_groups to {} (configured {})", temp_c, max_temp_c, wg, @@ -178,7 +190,7 @@ impl MiningRuntimeState { self.thermal_cap_wg.store(0, Relaxed); self.throttled.store(false, Relaxed); println!( - "[efficiency] Thermal OK ({}C) — removed work_groups thermal cap", + "[efficiency] Thermal OK ({}C) - removed work_groups thermal cap", temp_c ); } @@ -338,6 +350,11 @@ fn hottest_sensor_reading(sensors: &[crate::efficiency::GpuTempSensorBackend]) - hottest } +/// How many times to retry spawning the thermal monitor thread before giving up +/// and fail-closing. A working sensor is already detected by this point, so a +/// spawn failure is a transient OS resource problem worth retrying. +const THERMAL_SPAWN_ATTEMPTS: u32 = 4; + /// Start a vendor-specific cached GPU sensor monitor before mining workers run. pub fn start_thermal_monitor( runtime: &Arc, @@ -413,55 +430,80 @@ pub fn start_thermal_monitor( initial_hottest.unwrap_or(eff.max_temp_c as f32 + 5.0), ); - let monitor_runtime = runtime.clone(); let max_temp_c = eff.max_temp_c; let throttle_wg = eff.throttle_workgroups; - let monitor_guard = runtime.track_mining_thread(); - let spawn_result = std::thread::Builder::new() - .name("hac-thermal-monitor".to_string()) - .spawn(move || { - let _monitor_guard = monitor_guard; - let mut consecutive_misses = 0u32; - loop { - if wait_for_thermal_poll(&stop_flag) { - return; - } - match hottest_sensor_reading(&sensors) { - Some(temp) => { - if consecutive_misses > 0 { - println!( - "[Thermal] Sensor recovered after {} missed sample(s)", - consecutive_misses + // Share the sensors across retry attempts without moving them into a spawn + // that might fail (which would consume them). + let sensors = std::sync::Arc::new(sensors); + + let mut spawned = false; + for attempt in 1..=THERMAL_SPAWN_ATTEMPTS { + let monitor_runtime = runtime.clone(); + let guard_runtime = runtime.clone(); + let sensors = sensors.clone(); + let stop_flag = stop_flag.clone(); + let spawn_result = std::thread::Builder::new() + .name("hac-thermal-monitor".to_string()) + .spawn(move || { + let _monitor_guard = guard_runtime.track_mining_thread(); + let mut consecutive_misses = 0u32; + loop { + if wait_for_thermal_poll(&stop_flag) { + return; + } + match hottest_sensor_reading(&sensors) { + Some(temp) => { + if consecutive_misses > 0 { + println!( + "[Thermal] Sensor recovered after {} missed sample(s)", + consecutive_misses + ); + } + consecutive_misses = 0; + monitor_runtime.observe_thermal_temperature( + max_temp_c, + throttle_wg, + configured_wg, + temp, ); } - consecutive_misses = 0; - monitor_runtime.observe_thermal_temperature( - max_temp_c, - throttle_wg, - configured_wg, - temp, - ); - } - None => { - consecutive_misses = consecutive_misses.saturating_add(1); - if consecutive_misses == 1 { - eprintln!( - "[Thermal] Sensor read failed; preserving the current safety state" + None => { + consecutive_misses = consecutive_misses.saturating_add(1); + if consecutive_misses == 1 { + eprintln!( + "[Thermal] Sensor read failed; preserving the current safety state" + ); + } + monitor_runtime.observe_thermal_sensor_miss( + consecutive_misses, + throttle_wg, + configured_wg, ); } - monitor_runtime.observe_thermal_sensor_miss( - consecutive_misses, - throttle_wg, - configured_wg, - ); } } + }); + match spawn_result { + Ok(_) => { + spawned = true; + break; } - }); - if let Err(error) = spawn_result { + Err(error) => { + eprintln!( + "[Thermal] monitor spawn attempt {attempt}/{THERMAL_SPAWN_ATTEMPTS} failed: {error}" + ); + if attempt < THERMAL_SPAWN_ATTEMPTS { + std::thread::sleep(std::time::Duration::from_millis(250u64 * attempt as u64)); + } + } + } + } + // Only fail-close (pause all mining) after exhausting retries - a single + // transient spawn failure must not permanently halt a working miner. + if !spawned { runtime.fail_closed_thermal( configured_wg, - &format!("cannot start thermal monitor thread: {error}"), + "cannot start thermal monitor thread after all retries", ); } } diff --git a/app/src/mining_stats.rs b/app/src/mining_stats.rs index 3e5ebcd0..6006ceb0 100644 --- a/app/src/mining_stats.rs +++ b/app/src/mining_stats.rs @@ -229,8 +229,23 @@ pub fn write_mining_stats(path: &str, stats: &MiningStatsSnapshot) { if path.is_empty() { return; } - if let Ok(json) = serde_json::to_vec_pretty(stats) { - let _ = atomic_write_private(Path::new(path), &json); + // This runs on every stats update, so surface failures WITHOUT spamming: log + // the first error of a failing streak and stay quiet until it recovers. A + // silently unwritten stats file is why a panel would show a frozen miner. + use std::sync::atomic::{AtomicBool, Ordering::Relaxed}; + static WARNED: AtomicBool = AtomicBool::new(false); + let result = serde_json::to_vec_pretty(stats) + .map_err(|e| format!("serialize: {e}")) + .and_then(|json| { + atomic_write_private(Path::new(path), &json).map_err(|e| format!("write {path}: {e}")) + }); + match result { + Ok(()) => WARNED.store(false, Relaxed), + Err(e) => { + if !WARNED.swap(true, Relaxed) { + eprintln!("[stats] cannot update stats file ({e}); suppressing until it recovers"); + } + } } } diff --git a/app/src/opencl_dia.rs b/app/src/opencl_dia.rs index 19e581f6..6cf996a7 100644 --- a/app/src/opencl_dia.rs +++ b/app/src/opencl_dia.rs @@ -5,13 +5,14 @@ use mint::action::DIAMOND_ABOVE_NUMBER_OF_CREATE_BY_CUSTOM_MESSAGE; use mint::action::DiamondMint; use x16rs::calculate_hash; use x16rs::diamond_hash; +use x16rs::x16rs_hash; -use crate::hash_util::diamond_more_power; +use crate::hash_util::diamond_better; use crate::opencl_gpu::{ OpenCLResources, enqueue_diamond_kernel, read_diamond_gpu_results, write_stuff_to_gpu, }; -use super::{DiamondMiningResult, HASH_WIDTH, check_diamer_success}; +use super::{DIAMOND_HASH_LEN, DiamondMiningResult, check_diamer_success, diamond_pre_image}; pub(crate) fn do_diamond_group_mining_opencl( opencl: &OpenCLResources, @@ -26,8 +27,6 @@ pub(crate) fn do_diamond_group_mining_opencl( unit_size: u32, ) -> DiamondMiningResult { let empthbytes = [0u8; 0]; - let prevhash: &[u8; HASH_WIDTH] = prevblockhash; - let address: &[u8; 21] = rwdaddr; let custom_nonce: &[u8] = match number > DIAMOND_ABOVE_NUMBER_OF_CREATE_BY_CUSTOM_MESSAGE { true => custom_message.as_bytes(), false => &empthbytes, @@ -38,20 +37,30 @@ pub(crate) fn do_diamond_group_mining_opencl( nonce_space, u64_nonce: 0, msg_nonce: custom_nonce.to_vec(), - dia_str: [b'W'; 10], + dia_str: [b'W'; DIAMOND_HASH_LEN], is_success: None, use_secs: 0.0, is_gpu: true, gpu_batch_ok: false, }; let repeat = x16rs::mine_diamond_hash_repeat(number) as u32; - let stuff = [ - prevhash.to_vec(), - [0u8; 8].to_vec(), - address.to_vec(), - custom_nonce.as_ref().to_vec(), - ] - .concat(); + // The kernel overwrites the 8 nonce bytes per work item; upload with a zero + // nonce. Same builder the CPU recompute below uses, so the two cannot drift. + let stuff = diamond_pre_image(prevblockhash, &[0u8; 8], rwdaddr, custom_nonce); + let stuff_len = stuff.len() as u32; + // sha3_256_hash_diamond has exactly two padding layouts: 61 bytes without a + // custom message and 93 bytes with one. Any other length would be hashed with + // the wrong pad and silently produce a hash that is not the consensus one, so + // refuse the batch instead of mining garbage. + debug_assert!(stuff_len == 61 || stuff_len == 93); + if stuff_len != 61 && stuff_len != 93 { + eprintln!( + "[OpenCL] diamond pre-image length {} is neither 61 nor 93; skipping batch", + stuff_len + ); + most.gpu_batch_ok = false; + return most; + } let write_event = match write_stuff_to_gpu(opencl, &stuff, None) { Ok(ev) => ev, @@ -69,6 +78,7 @@ pub(crate) fn do_diamond_group_mining_opencl( unit_size, num_work_groups, local_work_size, + stuff_len, Some(&write_event), ) { Ok(ev) => ev, @@ -86,46 +96,80 @@ pub(crate) fn do_diamond_group_mining_opencl( return most; } + let mut found_success = false; + // How many of the returned work-group bests actually reproduce on the CPU. A + // batch where NOTHING reproduces is a broken device, not a healthy empty + // batch, and must not be reported to the health tracker as a success. + let mut verified_count = 0usize; for i in 0..num_work_groups as usize { - let hash_bytes = &hashes[i * 32..(i * 32) + 32].try_into().unwrap(); - let dia_str = diamond_hash(hash_bytes); - let nonce_bytes = nonces[i].to_be_bytes(); - let stuff = [ - prevblockhash.as_slice(), - nonce_bytes.as_slice(), - address.as_slice(), - custom_message.as_ref(), - ] - .concat(); + // Real bounds guards. The previous `hashes[i * 32..]` + try_into pair read + // like one, but an out-of-range index panics inside the slice expression + // before try_into is ever reached, so its Err arm could never fire. + let Some(chunk) = hashes.get(i * 32..(i * 32) + 32) else { + continue; + }; + let Ok(hash_bytes) = <[u8; 32]>::try_from(chunk) else { + continue; + }; + let Some(&nonce) = nonces.get(i) else { + continue; + }; + // Reject nonces outside the batch window (corrupt GPU read). + if nonce.wrapping_sub(nonce_start) >= nonce_space { + continue; + } + let nonce_bytes = nonce.to_be_bytes(); + // Re-hash with the SAME custom-message bytes the GPU kernel was fed: + // `custom_nonce` is gated by number (empty at or below the custom-message + // threshold, matching node consensus). + let stuff = diamond_pre_image(prevblockhash, &nonce_bytes, rwdaddr, custom_nonce); let ssshash: [u8; 32] = calculate_hash(stuff); + // Full medium-hash recompute (parity with block verify_gpu_best_result). + let expected_medium = x16rs_hash(repeat as i32, &ssshash); + if expected_medium != hash_bytes { + continue; + } + verified_count += 1; + let dia_str = diamond_hash(&hash_bytes); - if let Some(dia_name) = check_diamer_success(number, ssshash, *hash_bytes, dia_str) { + if let Some(dia_name) = check_diamer_success(number, ssshash, hash_bytes, dia_str) { let name = DiamondName::from(dia_name); let number = DiamondNumber::from(number); let mut diamint = DiamondMint::with(name, number); diamint.d.prev_hash = prevblockhash.clone(); - diamint.d.nonce = Fixed8::from(nonces[i].to_be_bytes()); + diamint.d.nonce = Fixed8::from(nonce_bytes); diamint.d.address = rwdaddr.clone(); diamint.d.custom_message = custom_message.clone(); most.dia_str = dia_str; - most.u64_nonce = nonces[i]; + most.u64_nonce = nonce; most.is_success = Some(diamint); most.gpu_batch_ok = true; - return most; - } else if diamond_more_power(&dia_str, &most.dia_str) { + found_success = true; + break; + } else if diamond_better(&dia_str, &most.dia_str) { most.dia_str = dia_str; - most.u64_nonce = nonces[i]; + most.u64_nonce = nonce; } } + // Always finish the queue when required — including the early success path — + // so AMD RDNA/duplicate ICD does not leave work outstanding. if opencl.needs_queue_finish { if let Err(e) = opencl.queue.finish() { eprintln!("[OpenCL] diamond queue finish: {}", e); + // Report the driver failure, but NEVER discard `most.is_success`: that + // DiamondMint was already verified on the CPU above (x16rs_hash + // recompute, then check_diamond_hash_result + check_diamond_difficulty + // via check_diamer_success). Its validity is a pure function of + // prev_hash/nonce/address/custom_message and does not depend on the + // command queue, so dropping it here would be pure loss of mined value. most.gpu_batch_ok = false; return most; } } - most.gpu_batch_ok = true; + if !found_success { + most.gpu_batch_ok = verified_count > 0 || num_work_groups == 0; + } most } diff --git a/app/src/opencl_diag.rs b/app/src/opencl_diag.rs index 586eb79f..2d97cfde 100644 --- a/app/src/opencl_diag.rs +++ b/app/src/opencl_diag.rs @@ -110,14 +110,48 @@ pub fn discrete_device_indices(platforms: &[OpenClPlatformEntry], platform_id: u .unwrap_or_default() } +/// Enumerate OpenCL platforms without dying when there are none. +/// +/// `Platform::list()` PANICS. On a machine that has the ICD loader but no vendor +/// driver, which is every fresh Linux install, every container and every WSL +/// session, it aborts with +/// `Platform::list: Error retrieving platform list: ApiWrapper(GetPlatformIdsPlatformListUnavailable(10))` +/// and a path into the ocl crate. +/// +/// That machine is exactly the one whose owner was told, by the release README, +/// to run `./list_opencl` to find out whether their driver is installed. So the +/// one tool whose job is to diagnose a missing OpenCL driver was the tool that +/// crashed when the driver was missing, and it crashed without saying why. +/// +/// No platforms is not an error here, it is an answer: the callers report it. +#[cfg(feature = "ocl")] +pub fn platform_list() -> Vec { + match ocl::core::get_platform_ids() { + Ok(ids) => ids.into_iter().map(ocl::Platform::new).collect(), + Err(_) => Vec::new(), + } +} + #[cfg(feature = "ocl")] pub fn scan_opencl() -> OpenClScan { - use ocl::{Device, Platform}; + use ocl::Device; let mut platforms = Vec::new(); let mut warnings = Vec::new(); - for (pi, platform) in Platform::list().iter().enumerate() { + let found = platform_list(); + if found.is_empty() { + warnings.push( + "No OpenCL platform was found. The loader is present but no vendor driver is \ + registered, so nothing here can use a GPU. On Linux install the OpenCL runtime \ + for your card (Mesa/ROCm for AMD, the NVIDIA driver for NVIDIA) and check that \ + /etc/OpenCL/vendors contains an .icd file; on Windows reinstall the GPU driver. \ + HACD diamond mining is CPU-only and is unaffected." + .to_string(), + ); + } + + for (pi, platform) in found.iter().enumerate() { let name = platform.name().unwrap_or_else(|_| "?".into()); let vendor = platform.vendor().unwrap_or_else(|_| "?".into()); let version = platform.version().unwrap_or_else(|_| "?".into()); diff --git a/app/src/opencl_gpu/block.rs b/app/src/opencl_gpu/block.rs index c3e7ad91..73bc42be 100644 --- a/app/src/opencl_gpu/block.rs +++ b/app/src/opencl_gpu/block.rs @@ -1,9 +1,32 @@ use crate::gpu_oom::GpuBatchError; use crate::hash_util::hash_more_power; use crate::opencl_gpu::{ - OpenCLResources, enqueue_mining_kernel, read_block_gpu_results, write_stuff_to_gpu, + OpenCLResources, SHARE_LIST_CAPACITY, enqueue_mining_kernel, read_block_gpu_results, + read_share_gpu_results, write_share_inputs_to_gpu, write_stuff_to_gpu, }; +/// Block intro serialization is a fixed 89-byte layout (matches CUDA STUFF_BYTES). +const BLOCK_INTRO_BYTES: usize = 89; + +/// Everything one OpenCL block batch produced. +pub struct OpenclBatchOutput { + /// The batch's single strongest nonce/hash, from the work-group tree + /// reduction. Unchanged, and still the only thing solo mining reads. + pub best: (u32, [u8; 32]), + /// Every nonce whose hash beat the share target, up to the buffer capacity. + /// Always empty when the caller passed no share target. + pub shares: Vec<(u32, [u8; 32])>, + /// How many hits the kernel counted in TOTAL, including the ones that did + /// not fit in `shares`. Greater than `shares.len()` means the batch is + /// undersampling and the miner is earning less than it mined. + pub share_hits: u64, +} + +/// One block batch, best result only. +/// +/// Kept exactly as it was for the benchmark and auto-tune callers, which want +/// one hash per batch and nothing else. Pool mining goes through +/// [`do_group_block_mining_opencl_shares`]. pub fn do_group_block_mining_opencl( opencl: &OpenCLResources, height: u64, @@ -13,6 +36,42 @@ pub fn do_group_block_mining_opencl( local_work_size: u32, unit_size: u32, ) -> std::result::Result<(u32, [u8; 32]), GpuBatchError> { + do_group_block_mining_opencl_shares( + opencl, + height, + block_intro, + nonce_start, + num_work_groups, + local_work_size, + unit_size, + None, + ) + .map(|out| out.best) +} + +/// One block batch, plus the list of every nonce that beat `share_target`. +/// +/// `share_target` is `None` for solo mining. That is the whole guarantee that +/// solo behaviour is untouched: the kernel is launched with share_capacity=0, +/// which skips the appending block, and the host writes and reads not one extra +/// byte over the bus. +#[allow(clippy::too_many_arguments)] +pub fn do_group_block_mining_opencl_shares( + opencl: &OpenCLResources, + height: u64, + block_intro: Vec, + nonce_start: u32, + num_work_groups: u32, + local_work_size: u32, + unit_size: u32, + share_target: Option<&[u8; 32]>, +) -> std::result::Result { + if block_intro.len() != BLOCK_INTRO_BYTES { + return Err(GpuBatchError::from_message(&format!( + "block intro must be {BLOCK_INTRO_BYTES} bytes, got {}", + block_intro.len() + ))); + } let mut most_nonce = 0u32; let mut most_hash = [255u8; 32]; let repeat = x16rs::block_hash_repeat(height) as u32; @@ -20,6 +79,15 @@ pub fn do_group_block_mining_opencl( let write_event = write_stuff_to_gpu(opencl, &block_intro, None) .map_err(|e| GpuBatchError::from_message(&e))?; + let (share_capacity, ready_event) = match share_target { + Some(target) => { + let event = write_share_inputs_to_gpu(opencl, target, Some(&write_event)) + .map_err(|e| GpuBatchError::from_message(&e))?; + (SHARE_LIST_CAPACITY as u32, event) + } + None => (0u32, write_event), + }; + let kernel_event = enqueue_mining_kernel( opencl, nonce_start, @@ -27,7 +95,8 @@ pub fn do_group_block_mining_opencl( unit_size, num_work_groups, local_work_size, - Some(&write_event), + share_capacity, + Some(&ready_event), )?; let mut hashes = vec![0u8; opencl.buffer_best_hashes.len()]; @@ -43,6 +112,22 @@ pub fn do_group_block_mining_opencl( } } + let mut shares = Vec::new(); + let mut share_hits = 0u64; + if share_capacity > 0 { + let mut share_nonces: Vec = Vec::new(); + let mut share_hashes: Vec = Vec::new(); + share_hits = + read_share_gpu_results(opencl, &kernel_event, &mut share_nonces, &mut share_hashes) + .map_err(|e| GpuBatchError::from_message(&e))?; + shares.reserve(share_nonces.len()); + for (i, nonce) in share_nonces.iter().enumerate() { + let mut hash = [0u8; 32]; + hash.copy_from_slice(&share_hashes[i * 32..(i * 32) + 32]); + shares.push((*nonce, hash)); + } + } + if opencl.needs_queue_finish { opencl .queue @@ -50,5 +135,168 @@ pub fn do_group_block_mining_opencl( .map_err(|e| GpuBatchError::from_message(&format!("queue finish: {}", e)))?; } - Ok((most_nonce, most_hash)) + Ok(OpenclBatchOutput { + best: (most_nonce, most_hash), + shares, + share_hits, + }) +} + +#[cfg(test)] +mod gpu_tests { + use super::*; + use crate::opencl_gpu::initialize_opencl; + + /// Mainnet repeat count is 16 from height 750000 on, so the kernel under + /// test here runs the same algorithm mix a real block does. + const REPEAT16_HEIGHT: u64 = 800_000; + const WORK_GROUPS: u32 = 2; + const LOCAL_SIZE: u32 = 256; + const UNIT_SIZE: u32 = 4; + const BATCH_NONCES: u32 = WORK_GROUPS * LOCAL_SIZE * UNIT_SIZE; + const NONCE_START: u32 = 4_000; + + fn opencl_dir() -> String { + std::env::var("HACASH_OPENCL_DIR").unwrap_or_else(|_| { + std::path::Path::new(env!("CARGO_MANIFEST_DIR")) + .parent() + .expect("app crate must have a workspace parent") + .join("x16rs") + .join("opencl") + .to_string_lossy() + .into_owned() + }) + } + + fn cpu_hash(intro: &[u8], nonce: u32) -> [u8; 32] { + let mut stuff = intro.to_vec(); + stuff[79..83].copy_from_slice(&nonce.to_be_bytes()); + x16rs::block_hash(REPEAT16_HEIGHT, &stuff) + } + + #[test] + #[ignore = "requires a physical OpenCL GPU; set HACASH_RUN_OPENCL_INTEGRATION=1"] + fn the_share_list_matches_the_cpu_and_leaves_the_best_result_untouched() { + assert_eq!( + std::env::var_os("HACASH_RUN_OPENCL_INTEGRATION").as_deref(), + Some(std::ffi::OsStr::new("1")), + "set HACASH_RUN_OPENCL_INTEGRATION=1 before running this ignored GPU test" + ); + let platform: u32 = std::env::var("HACASH_OPENCL_PLATFORM_ID") + .ok() + .and_then(|v| v.parse().ok()) + .unwrap_or(0); + let devices = std::env::var("HACASH_OPENCL_DEVICE_IDS").unwrap_or_else(|_| "0".to_string()); + let mut resources = initialize_opencl( + false, + &opencl_dir(), + &platform, + &devices, + &WORK_GROUPS, + &LOCAL_SIZE, + &UNIT_SIZE, + None, + false, + ); + assert!(!resources.is_empty(), "no usable OpenCL device"); + let opencl = resources.remove(0); + + // The kernel's sha3_256_hash folds the 89-byte message's padding into a + // constant word that hard-codes byte 88 as 0x00 (a real block intro ends + // in the two zero bytes of its transaction count), so a test intro has + // to honour that or the card and the CPU hash different messages. + let mut intro: Vec = (0..BLOCK_INTRO_BYTES as u8).collect(); + intro[88] = 0; + + // What the CPU says about this exact window, which is the authority. + let mut cpu: Vec<(u32, [u8; 32])> = (NONCE_START..NONCE_START + BATCH_NONCES) + .map(|nonce| (nonce, cpu_hash(&intro, nonce))) + .collect(); + let cpu_best = *cpu + .iter() + .min_by(|a, b| a.1.cmp(&b.1)) + .expect("non empty window"); + + // 1. SOLO: no share target, so the kernel skips the whole added block. + // The single best result has to be the CPU's, byte for byte. + let solo = do_group_block_mining_opencl_shares( + &opencl, + REPEAT16_HEIGHT, + intro.clone(), + NONCE_START, + WORK_GROUPS, + LOCAL_SIZE, + UNIT_SIZE, + None, + ) + .expect("solo batch"); + assert_eq!( + solo.best.1, + cpu_hash(&intro, solo.best.0), + "the hash the card reports for its own nonce must be the CPU's" + ); + assert_eq!(solo.best, cpu_best, "solo best must match the CPU exactly"); + assert!(solo.shares.is_empty(), "solo must never build a share list"); + assert_eq!(solo.share_hits, 0); + + // 2. POOL, easiest possible target: every nonce is payable, so the + // counter sees the whole window and the list reports the overflow. + let pool = do_group_block_mining_opencl_shares( + &opencl, + REPEAT16_HEIGHT, + intro.clone(), + NONCE_START, + WORK_GROUPS, + LOCAL_SIZE, + UNIT_SIZE, + Some(&[0xffu8; 32]), + ) + .expect("pool batch"); + assert_eq!( + pool.best, cpu_best, + "adding the share list must not perturb the reduction" + ); + assert_eq!( + pool.share_hits, BATCH_NONCES as u64, + "the counter must see every hit, not only the stored ones" + ); + assert_eq!(pool.shares.len(), SHARE_LIST_CAPACITY.min(BATCH_NONCES as usize)); + let mut seen: Vec = pool.shares.iter().map(|(nonce, _)| *nonce).collect(); + seen.sort_unstable(); + seen.dedup(); + assert_eq!(seen.len(), pool.shares.len(), "no nonce may be listed twice"); + for (nonce, hash) in &pool.shares { + assert!( + (NONCE_START..NONCE_START + BATCH_NONCES).contains(nonce), + "share nonce {nonce} is outside the batch window" + ); + assert_eq!(*hash, cpu_hash(&intro, *nonce), "share hash must match the CPU"); + } + + // 3. POOL, a target only three nonces beat: exactly those three, and + // nothing else, may come back. + cpu.sort_by(|a, b| a.1.cmp(&b.1)); + let strict_target = cpu[2].1; + let expected: Vec = { + let mut want: Vec = cpu[..3].iter().map(|(nonce, _)| *nonce).collect(); + want.sort_unstable(); + want + }; + let strict = do_group_block_mining_opencl_shares( + &opencl, + REPEAT16_HEIGHT, + intro.clone(), + NONCE_START, + WORK_GROUPS, + LOCAL_SIZE, + UNIT_SIZE, + Some(&strict_target), + ) + .expect("strict batch"); + assert_eq!(strict.best, cpu_best); + assert_eq!(strict.share_hits, 3); + let mut got: Vec = strict.shares.iter().map(|(nonce, _)| *nonce).collect(); + got.sort_unstable(); + assert_eq!(got, expected, "the kernel must list exactly the payable nonces"); + } } diff --git a/app/src/opencl_gpu/handle.rs b/app/src/opencl_gpu/handle.rs index 1b376867..25dac0f9 100644 --- a/app/src/opencl_gpu/handle.rs +++ b/app/src/opencl_gpu/handle.rs @@ -4,7 +4,9 @@ use std::sync::Mutex; use std::sync::atomic::AtomicU32; use crate::gpu_arch::ArchLimits; -use crate::gpu_oom::{GpuBatchError, GpuOomState}; +use crate::gpu_oom::{ + GpuBatchError, GpuGate, GpuOomState, GpuQuarantine, GpuQuarantineStatus, format_backoff, +}; use crate::mining_runtime::MiningRuntimeState; use crate::opencl_diag::OpenClScan; @@ -45,7 +47,16 @@ pub struct OpenclGpuHandle { inner: Mutex, snapshot: OpenclGpuSnapshot, oom: Mutex, + /// Failures since the last clean batch OR the last successful context rebuild. + /// Drives the rebuild cadence only. The quarantine keeps its own count, which a + /// rebuild must NOT reset, otherwise a card that rebuilds every few errors is + /// retried forever with no alert. consecutive_errors: AtomicU32, + /// Time-based quarantine with exponential backoff and automatic re-probe. It + /// replaces the old write-once session latch, which had no clearing path: a + /// card that failed 20 batches during a driver reset stayed off until someone + /// restarted the process. Identical policy to the CUDA backend. + quarantine: GpuQuarantine, cached_scan: Mutex>, } @@ -61,10 +72,97 @@ impl OpenclGpuHandle { snapshot, oom: Mutex::new(GpuOomState::new(base_wg)), consecutive_errors: AtomicU32::new(0), + quarantine: GpuQuarantine::new(), cached_scan: Mutex::new(Some(scan)), }) } + /// Gate one batch. True while this device is quarantined and must not be given + /// work; the caller mines the bounded CPU recovery window instead. When the + /// backoff expires this rebuilds the context at floor work_groups and lets one + /// re-probe batch through, so the card recovers with nobody watching. + pub fn quarantine_blocks_batch(&self) -> bool { + match self.quarantine.gate(std::time::Instant::now()) { + GpuGate::Run => false, + GpuGate::Skip { + level, + retry_in, + total_failures, + notify, + } => { + if notify { + eprintln!( + "[OpenCL] GPU QUARANTINED (level {level}, {total_failures} failed batches): no GPU work for another {}, mining continues on capped CPU recovery. The card is re-probed automatically, no restart needed.", + format_backoff(retry_in) + ); + } + true + } + GpuGate::Reprobe { + level, + total_failures, + } => { + let wg = self.prepare_reprobe(); + println!( + "[OpenCL] GPU quarantine (level {level}, {total_failures} failed batches) expired: re-probing the device at work_groups={wg}." + ); + false + } + } + } + + /// Compatibility name for [`Self::quarantine_blocks_batch`], kept for the + /// diamond worker's call site. There is no longer a permanent "disabled" + /// state: this is true only while the device is inside a backoff window. + pub fn gpu_is_disabled(&self) -> bool { + self.quarantine_blocks_batch() + } + + /// Quarantine state for the panel / stats (None while mining normally). + pub fn quarantine_status(&self) -> Option { + self.quarantine.status(std::time::Instant::now()) + } + + /// One-line GPU health for a non-technical operator, e.g. + /// `quarantined (level 3), retry in 8m, 61 failed batches`. + pub fn quarantine_note(&self) -> String { + self.quarantine.describe(std::time::Instant::now()) + } + + /// Drop to floor work_groups and rebuild the context before a re-probe, so the + /// probe runs on the smallest, most likely to succeed configuration after the + /// driver has had the whole backoff window to recover. + fn prepare_reprobe(&self) -> u32 { + let mut res = self.lock_resources(); + let floor = { + let mut oom = self.oom.lock().unwrap_or_else(|e| e.into_inner()); + let floor = oom.floor_wg(); + oom.sync_effective(floor); + floor + }; + soft_recover_opencl(&mut res); + let scan = self + .cached_scan + .lock() + .unwrap_or_else(|e| e.into_inner()) + .clone(); + if let Some(scan) = scan { + match rebuild_opencl_gpu(&self.snapshot, floor, &scan) { + Ok(new_res) => { + let synced_wg = new_res.workgroups; + *res = new_res; + drop(res); + if let Ok(mut oom) = self.oom.lock() { + oom.sync_effective(synced_wg); + } + return synced_wg; + } + Err(e) => eprintln!("[OpenCL] re-probe context rebuild failed: {}", e), + } + } + floor + } + pub fn configure_oom_floor( &self, vram_bytes: u64, @@ -115,14 +213,35 @@ impl OpenclGpuHandle { ) { runtime.record_gpu_error_event(); use std::sync::atomic::Ordering::Relaxed; + let report = self.quarantine.record_failure(std::time::Instant::now()); let mut res = self.lock_resources(); let res_wg = res.workgroups; let arch_limits = ArchLimits::for_slug(&res.arch_slug); let experimental = arch_limits.is_experimental(); - let oom = self.oom.lock().unwrap_or_else(|e| e.into_inner()); + let mut oom = self.oom.lock().unwrap_or_else(|e| e.into_inner()); let cur_eff = oom.effective_wg(); let at_floor = cur_eff <= oom.floor_wg(); let n = self.consecutive_errors.fetch_add(1, Relaxed) + 1; + // Same policy as CUDA, and deliberately NOT conditioned on being at the + // work_groups floor: an arch whose floor is never reached used to be + // retried forever with no alert at all. Park the device for a growing + // interval instead of latching it off for the process. + if let Some(entry) = report.quarantined { + let floor = oom.floor_wg(); + oom.sync_effective(floor); + drop(oom); + soft_recover_opencl(&mut res); + drop(res); + runtime.report_gpu_workgroups(floor, runtime.thermal_workgroups_cap(), configured_wg); + eprintln!( + "[OpenCL] ALERT GPU quarantined after {} consecutive failed batches ({} this session): no GPU work for {}, then the card is re-probed automatically. Mining continues on capped CPU recovery. Last error: {}. Check the driver, cooling, power and the PCIe riser.", + report.consecutive_failures, + report.total_failures, + format_backoff(entry.retry_in), + err.display() + ); + return; + } let retry_only = experimental && err.is_out_of_resources() && oom_fallback && !at_floor && n < 3; let next_wg = if retry_only { @@ -172,16 +291,68 @@ impl OpenclGpuHandle { pub fn on_batch_success(&self, configured_wg: u32, runtime: &MiningRuntimeState) { use std::sync::atomic::Ordering::Relaxed; self.consecutive_errors.store(0, Relaxed); + if self.quarantine.record_success() { + println!( + "[OpenCL] GPU RECOVERED: the re-probe succeeded, quarantine cleared and GPU mining has resumed." + ); + } self.oom .lock() .unwrap_or_else(|e| e.into_inner()) .record_success(); + self.grow_context_to_effective(configured_wg); runtime.report_gpu_workgroups( self.effective_wg(), runtime.thermal_workgroups_cap(), configured_wg, ); } + + /// After an OOM ramp-back the OOM state can allow more work_groups than the + /// current context has buffers for, and `enqueue_mining_kernel` rejects + /// anything above `res.workgroups`. Rebuild once at the larger size, otherwise + /// the ramp is purely cosmetic and the card stays at reduced throughput for + /// the rest of the session. + fn grow_context_to_effective(&self, configured_wg: u32) { + let want = self.effective_wg().min(configured_wg.max(1)); + let mut res = self.lock_resources(); + if want <= res.workgroups { + return; + } + let scan = self + .cached_scan + .lock() + .unwrap_or_else(|e| e.into_inner()) + .clone(); + let rebuilt = match scan { + Some(scan) => rebuild_opencl_gpu(&self.snapshot, want, &scan), + None => Err("no cached OpenCL scan".to_string()), + }; + match rebuilt { + Ok(new_res) => { + let synced_wg = new_res.workgroups; + *res = new_res; + drop(res); + if let Ok(mut oom) = self.oom.lock() { + oom.sync_effective(synced_wg); + } + println!("[OpenCL] Restored GPU context at work_groups={}", synced_wg); + } + Err(e) => { + // Clamp the OOM state back to what the live context can run, so a + // failed ramp-up is not retried on every following batch. + let capped = res.workgroups; + drop(res); + if let Ok(mut oom) = self.oom.lock() { + oom.sync_effective(capped); + } + eprintln!( + "[OpenCL] work_groups ramp-up rebuild failed, staying at {}: {}", + capped, e + ); + } + } + } } fn rebuild_opencl_gpu( diff --git a/app/src/opencl_gpu/init.rs b/app/src/opencl_gpu/init.rs index 5a7642e0..2971ea71 100644 --- a/app/src/opencl_gpu/init.rs +++ b/app/src/opencl_gpu/init.rs @@ -8,7 +8,7 @@ use crate::efficiency::clamp_workgroups_for_vram_with_floor; use crate::gpu_arch::{self, ArchLimits, GpuVendor}; use crate::opencl_diag::OpenClScan; use ocl::enums::DeviceInfo; -use ocl::{Context, Device, Platform, Program}; +use ocl::{Context, Device, Program}; use super::compile::{ OPENCL_CACHE_PREFIX, compile_program_from_source, newest_opencl_source_mtime, @@ -148,13 +148,23 @@ pub fn initialize_opencl( } let amd_icd_count = crate::opencl_diag::count_amd_platforms(&scan.platforms); - let platforms = Platform::list(); + // Not Platform::list(): that panics when no vendor driver is registered, + // which would take the miner down instead of letting it fall back to CPU. + let platforms = crate::opencl_diag::platform_list(); let Some(platform) = platforms.get(resolved_platform as usize).cloned() else { - eprintln!( - "[OpenCL] Platform {} is unavailable ({} platform(s) detected)", - resolved_platform, - platforms.len() - ); + if platforms.is_empty() { + eprintln!( + "[OpenCL] No OpenCL platform is available on this machine. Install the GPU \ + driver's OpenCL runtime, then run ./list_opencl to confirm it. Mining \ + continues on the CPU." + ); + } else { + eprintln!( + "[OpenCL] Platform {} is unavailable ({} platform(s) detected)", + resolved_platform, + platforms.len() + ); + } return Vec::new(); }; @@ -232,7 +242,9 @@ pub fn initialize_opencl( wg = clamped; } } - let num_work_items = wg * localsize; + // Saturate rather than overflow: a large configured work_groups * localsize + // could wrap a u32, producing a tiny/zero global size and silent misconfig. + let num_work_items = wg.saturating_mul(*localsize); let global_work_size = num_work_items; println!("-----------------------------------------"); diff --git a/app/src/opencl_gpu/mod.rs b/app/src/opencl_gpu/mod.rs index 2f09fe58..92db6122 100644 --- a/app/src/opencl_gpu/mod.rs +++ b/app/src/opencl_gpu/mod.rs @@ -9,6 +9,7 @@ mod resources; pub use handle::{OpenclGpuHandle, OpenclGpuSnapshot, opencl_snapshot_from_resource}; pub use init::initialize_opencl; pub use resources::{ - OpenCLResources, enqueue_diamond_kernel, enqueue_mining_kernel, read_block_gpu_results, - read_diamond_gpu_results, write_stuff_to_gpu, + OpenCLResources, SHARE_LIST_CAPACITY, enqueue_diamond_kernel, enqueue_mining_kernel, + read_block_gpu_results, read_diamond_gpu_results, read_share_gpu_results, + write_share_inputs_to_gpu, write_stuff_to_gpu, }; diff --git a/app/src/opencl_gpu/resources.rs b/app/src/opencl_gpu/resources.rs index aaf4b5c1..075fc1d4 100644 --- a/app/src/opencl_gpu/resources.rs +++ b/app/src/opencl_gpu/resources.rs @@ -11,6 +11,36 @@ use ocl::{Buffer, Context, Device, Event, Kernel, Program, Queue}; pub(crate) const HASH_WIDTH: usize = 32; pub(crate) const STUFF_BUFFER_CAP: usize = 512; +/// How many pool shares one GPU batch can hand back. +/// +/// Arithmetic, not a guess, from a measured gfx1201: +/// +/// A batch covers `work_groups * local_size * unit_size` nonces. Auto-tune on +/// that card picked 48 * 256 * 96 = 1,179,648 nonces at 98.6 MH/s, so about +/// 12 ms per batch; the `amd_profit` profile's ceiling of 1536 work groups is +/// 37.7 M nonces, about 0.38 s per batch. The expected hits per batch is +/// therefore +/// +/// ```text +/// lambda = batch seconds * shares per second +/// ``` +/// +/// and the second factor is what a pool CHOOSES: it sets `share_bits` so one +/// worker submits on the order of one share a second (HBIT ships 20, its +/// 4096-share PPLNS window and its per-height cap all assume that order). At one +/// share per second that is lambda = 0.012 on the tuned batch and lambda = 0.4 on +/// the largest one, so 1024 carries three to five orders of magnitude of +/// headroom. The mean would only REACH 1024 if a pool served this single card +/// about 2700 shares a second, which is past the pool's own per-height cap and +/// past what one submit thread doing an HTTP round trip per share could post +/// anyway. Overflow here is not a tail event, it is a misconfigured pool, and the +/// host says so out loud. +/// +/// The cost is 1024 * (4 + 32) bytes = 36 KiB of device memory per GPU, next to +/// the 1.2 GiB `buffer_global_hashes` already takes at that ceiling, and at most +/// 1024 CPU re-hashes per batch for the integrity check every entry must pass. +pub const SHARE_LIST_CAPACITY: usize = 1024; + pub(crate) fn pinned_host_write_flags() -> MemFlags { MemFlags::new() .alloc_host_ptr() @@ -87,6 +117,14 @@ pub struct OpenCLResources { buffer_global_hashes: Buffer, buffer_global_order: Buffer, pub buffer_best_hashes: Buffer, + /// Share target the kernel appends against (32 bytes). Only written when the + /// miner is pooled; a solo batch never touches it. + buffer_share_target: Buffer, + /// Single atomic counter: how many nonces beat the share target in this + /// batch, INCLUDING the ones that did not fit in the list below. + buffer_share_found: Buffer, + buffer_share_nonces: Buffer, + buffer_share_hashes: Buffer, /// Reused input buffer — avoids per-kernel GPU allocation. buffer_stuff: Buffer, /// Cached OpenCL kernel — rebuilt only when `unit_size` changes. @@ -137,8 +175,22 @@ fn build_block_kernel( .arg(&res.buffer_global_order) .arg(&res.buffer_best_hashes) .arg(&res.buffer_best_nonces) + .arg(&res.buffer_share_target) + .arg(&res.buffer_share_found) + .arg(&res.buffer_share_nonces) + .arg(&res.buffer_share_hashes) + // share_capacity, set per batch. 0 = solo: the kernel skips the list. + .arg(0u32) .build() - .map_err(|e| format!("kernel build: {}", e)) + // The pool share list added five arguments to x16rs_main, so an + // opencl_dir left over from an older bundle fails here with an argument + // error. Say that outright: a bare driver message reads like a broken + // card and would send the operator hunting the wrong fault. + .map_err(|e| { + format!( + "kernel build: {e}. If this mentions an invalid argument, the opencl_dir points at an x16rs_main.cl older than this miner; point it at the x16rs/opencl folder shipped with this build." + ) + }) } fn build_diamond_kernel( @@ -157,6 +209,7 @@ fn build_diamond_kernel( .arg(&res.buffer_global_order) .arg(&res.buffer_best_hashes) .arg(&res.buffer_best_nonces_diamond) + .arg(0u32) // stuff_len: 61 or 93 .build() .map_err(|e| format!("kernel build: {}", e)) } @@ -246,6 +299,89 @@ pub fn read_block_gpu_results( Ok(()) } +/// Load the share target and clear the hit counter before a POOL batch. +/// +/// The two writes are chained onto `wait` and onto each other, so the returned +/// event alone is enough for the kernel to depend on even on the out-of-order +/// queue. Never called on the solo path: `share_capacity` is 0 there and the +/// kernel does not read either buffer. +pub fn write_share_inputs_to_gpu( + res: &OpenCLResources, + share_target: &[u8; HASH_WIDTH], + wait: Option<&Event>, +) -> std::result::Result { + let mut target_event = Event::empty(); + let mut cmd = res + .buffer_share_target + .write(&share_target[..]) + .enew(&mut target_event); + if let Some(dep) = wait { + cmd = cmd.ewait(dep); + } + cmd.enq() + .map_err(|e| format!("share target write: {}", e))?; + + let mut counter_event = Event::empty(); + res.buffer_share_found + .write(&[0u32][..]) + .ewait(&target_event) + .enew(&mut counter_event) + .enq() + .map_err(|e| format!("share counter write: {}", e))?; + Ok(counter_event) +} + +/// Read back how many nonces beat the share target, then the entries that fit. +/// +/// The counter is read first because it decides how much of the list is live: +/// pulling the whole 36 KiB every batch would be pure PCIe traffic for a batch +/// that found nothing. `found` is the TOTAL, so a value above the capacity is +/// exactly the undersampling signal the host has to report. +pub fn read_share_gpu_results( + res: &OpenCLResources, + wait: &Event, + nonces: &mut Vec, + hashes: &mut Vec, +) -> std::result::Result { + let mut found = [0u32; 1]; + let mut found_event = Event::empty(); + res.buffer_share_found + .read(&mut found[..]) + .ewait(wait) + .enew(&mut found_event) + .enq() + .map_err(|e| format!("read share count enqueue: {}", e))?; + wait_event(&found_event, "share count read")?; + + let total = found[0] as u64; + let stored = (total.min(SHARE_LIST_CAPACITY as u64)) as usize; + nonces.clear(); + hashes.clear(); + if stored == 0 { + return Ok(total); + } + nonces.resize(stored, 0u32); + hashes.resize(stored * HASH_WIDTH, 0u8); + + let mut nonce_event = Event::empty(); + let mut hash_event = Event::empty(); + res.buffer_share_nonces + .read(&mut nonces[..]) + .ewait(wait) + .enew(&mut nonce_event) + .enq() + .map_err(|e| format!("read share nonces enqueue: {}", e))?; + res.buffer_share_hashes + .read(&mut hashes[..]) + .ewait(wait) + .enew(&mut hash_event) + .enq() + .map_err(|e| format!("read share hashes enqueue: {}", e))?; + wait_event(&nonce_event, "share nonce read")?; + wait_event(&hash_event, "share hash read")?; + Ok(total) +} + pub fn read_diamond_gpu_results( res: &OpenCLResources, wait: &Event, @@ -272,6 +408,9 @@ pub fn read_diamond_gpu_results( } /// Block mining kernel (u32 nonce). +/// +/// `share_capacity` is 0 for solo mining, which makes the kernel skip the pool +/// share list entirely and do exactly what it did before the list existed. pub fn enqueue_mining_kernel( res: &OpenCLResources, nonce_start: u32, @@ -279,8 +418,15 @@ pub fn enqueue_mining_kernel( unit_size: u32, num_work_groups: u32, local_work_size: u32, + share_capacity: u32, wait: Option<&Event>, ) -> std::result::Result { + if share_capacity as usize > SHARE_LIST_CAPACITY { + return Err(GpuBatchError::Other(format!( + "share_capacity {} exceeds allocated share list {}", + share_capacity, SHARE_LIST_CAPACITY + ))); + } run_cached_kernel( res, unit_size, @@ -297,12 +443,16 @@ pub fn enqueue_mining_kernel( kernel .set_arg(3, unit_size) .map_err(|e| format!("set_arg unit_size: {}", e))?; + kernel + .set_arg(12, share_capacity) + .map_err(|e| format!("set_arg share_capacity: {}", e))?; Ok(()) }, ) } /// Diamond mining kernel (u64 nonce). +/// `stuff_len` is the prehash byte length (61 without custom message, 93 with). pub fn enqueue_diamond_kernel( res: &OpenCLResources, nonce_start: u64, @@ -310,6 +460,7 @@ pub fn enqueue_diamond_kernel( unit_size: u32, num_work_groups: u32, local_work_size: u32, + stuff_len: u32, wait: Option<&Event>, ) -> std::result::Result { run_cached_kernel( @@ -328,6 +479,9 @@ pub fn enqueue_diamond_kernel( kernel .set_arg(3, unit_size) .map_err(|e| format!("set_arg unit_size: {}", e))?; + kernel + .set_arg(8, stuff_len) + .map_err(|e| format!("set_arg stuff_len: {}", e))?; Ok(()) }, ) @@ -450,6 +604,32 @@ pub(crate) fn build_opencl_resources( .len(STUFF_BUFFER_CAP) .build() .map_err(|e| format!("buffer_stuff: {}", e))?; + let buffer_share_target = Buffer::::builder() + .queue(queue.clone()) + .flags(pinned_host_write_flags()) + .len(HASH_WIDTH) + .build() + .map_err(|e| format!("buffer_share_target: {}", e))?; + // Read/write both ways: the host clears the counter before each pool batch + // and reads it after, so the host-read-only readback flags do not fit here. + let buffer_share_found = Buffer::::builder() + .queue(queue.clone()) + .flags(ocl::core::MEM_READ_WRITE) + .len(1) + .build() + .map_err(|e| format!("buffer_share_found: {}", e))?; + let buffer_share_nonces = Buffer::::builder() + .queue(queue.clone()) + .flags(readback_flags) + .len(SHARE_LIST_CAPACITY) + .build() + .map_err(|e| format!("buffer_share_nonces: {}", e))?; + let buffer_share_hashes = Buffer::::builder() + .queue(queue.clone()) + .flags(readback_flags) + .len(HASH_WIDTH * SHARE_LIST_CAPACITY) + .build() + .map_err(|e| format!("buffer_share_hashes: {}", e))?; if out_of_order { println!("[OpenCL] Pinned host buffers enabled for stuff + readback"); } @@ -470,6 +650,10 @@ pub(crate) fn build_opencl_resources( buffer_global_hashes, buffer_global_order, buffer_best_hashes, + buffer_share_target, + buffer_share_found, + buffer_share_nonces, + buffer_share_hashes, buffer_stuff, kernel_slot: Mutex::new(KernelSlot { kernel: None, diff --git a/app/src/poworker.rs b/app/src/poworker.rs index cb7a29d7..f3961ef9 100644 --- a/app/src/poworker.rs +++ b/app/src/poworker.rs @@ -33,6 +33,10 @@ pub struct PoWorkConf { pub rpcaddr: String, /// Optional fullnode API token (`X-Api-Token`) when server requires auth. pub api_token: String, + /// Optional payout address announced to a POOL as `&worker=
` so it + /// can credit this miner's shares. Empty (default) = solo mining: the URLs + /// stay byte-identical to what a plain fullnode expects. + pub pool_worker: String, pub supervene: u32, // cpu core (configured) pub noncemax: u32, pub noticewait: u64, // new block notice wait @@ -54,7 +58,35 @@ pub struct PoWorkConf { pub runtime: Arc, } +/// Percent-encode a query-string component, leaving only the RFC 3986 unreserved +/// characters. HAC addresses are already safe, but this guards against a +/// misconfigured worker id breaking or injecting into the request URL. +fn percent_encode_component(s: &str) -> String { + let mut out = String::with_capacity(s.len()); + for b in s.bytes() { + match b { + b'A'..=b'Z' | b'a'..=b'z' | b'0'..=b'9' | b'-' | b'_' | b'.' | b'~' => { + out.push(b as char) + } + _ => out.push_str(&format!("%{b:02X}")), + } + } + out +} + impl PoWorkConf { + /// `&worker=` suffix appended to pool requests so the pool + /// can credit shares to us. Empty string when solo mining. The value is + /// percent-encoded so a misconfigured worker id cannot break or inject into + /// the query string. + pub fn worker_param(&self) -> String { + if self.pool_worker.is_empty() { + String::new() + } else { + format!("&worker={}", percent_encode_component(&self.pool_worker)) + } + } + pub fn new(ini: &IniObj) -> PoWorkConf { let sec = &ini_section(ini, "default"); // default = root let sec_gpu = &ini_section(ini, "gpu"); @@ -66,6 +98,7 @@ impl PoWorkConf { let cnf = PoWorkConf { rpcaddr: ini_must(sec, "connect", "127.0.0.1:8081"), api_token: ini_must(sec, "api_token", "").trim().to_string(), + pool_worker: ini_must(sec, "pool_worker", "").trim().to_string(), supervene: configured_supervene, noncemax: ini_must_u64(sec, "nonce_max", u32::MAX as u64) as u32, noticewait: ini_must_u64(sec, "notice_wait", 45), @@ -162,7 +195,7 @@ pub fn poworker_with_stop(cnf: PoWorkConf, stop_flag: Option>) { if cnf.runtime.paused_unprofitable.load(Relaxed) { delay_continue_ms!(3000); } - pull_pending_block_stuff(&cnf); + pull_pending_block_stuff(&cnf, &stop_flag); delay_continue_ms!(25); } } @@ -173,14 +206,88 @@ fn should_stop(stop_flag: &Option>) -> bool { /////////////////////////////// -fn pull_pending_block_stuff(cnf: &PoWorkConf) { +/// Per-request long-poll wait for the miner notice. When a supervisor can ask us +/// to stop, cap it so the loop re-checks the stop flag every few seconds instead +/// of blocking for the whole configured notice window. +const NOTICE_POLL_WAIT_WITH_STOP: u64 = 3; + +/// How long to wait before polling again while the upstream says it is serving +/// work it can no longer refresh. Long enough not to hammer a struggling pool or +/// node, short enough to pick mining back up as soon as the outage clears. +const STALE_UPSTREAM_BACKOFF_SECS: u64 = 10; + +/// Minimum period of one pending + notice cycle. A cooperating upstream already +/// blocks inside the notice long-poll for the configured wait, so this floor never +/// slows a healthy miner down. It exists only for the pathological upstream: one +/// that answers the notice immediately (a saturated pool bridge, or a server with +/// no long-poll at all). The cycle costs TWO requests, so without a floor a single +/// miner would issue tens of requests per second forever and make the overload it +/// is suffering from worse. +const NOTICE_CYCLE_MIN: Duration = Duration::from_millis(200); + +/// Anti-spin sleep still owed by a notice cycle. Zero when the notice reported +/// that the tip reached the height we are mining on top of: genuinely new work is +/// money and is never delayed. +fn notice_cycle_backoff(spent: Duration, new_work_ready: bool) -> Duration { + if new_work_ready { + return Duration::ZERO; + } + NOTICE_CYCLE_MIN.saturating_sub(spent) +} + +/// Detect the "upstream is serving work it can no longer refresh" answer that the +/// pool bridge returns on `/query/miner/pending` and `/query/miner/notice` during +/// an upstream node outage. Both bodies stay backward compatible in shape (a +/// plain `err` string), so anything else, including an unknown or malformed error +/// body, is treated as an ordinary transient error and must not crash the miner. +fn upstream_stale_reason(res: &JV) -> Option<&str> { + let err = res["err"].as_str()?; + if err.is_empty() { + return None; + } + if err.to_ascii_lowercase().contains("stale") { + Some(err) + } else { + None + } +} + +/// Report an upstream stale-work outage once per transition and pause the mining +/// threads: the installed template can no longer win a block or earn a share, so +/// continuing to hash it only burns power. +fn enter_upstream_stale(reason: &str, source: &str) { + if block_mining_runtime::set_upstream_stale(true) { + println!( + "\n[Mining] PAUSED: {source} reports stale work ({reason}). The template can no longer win anything, so hashing is idle until fresh work arrives. Check the pool or node this miner connects to." + ); + } +} + +/// Clear the stale-work pause once real work is available again. +fn leave_upstream_stale() { + if block_mining_runtime::set_upstream_stale(false) { + println!("\n[Mining] Fresh work received, mining resumes."); + } +} + +fn notice_poll_wait(notice_wait: u64, can_stop: bool) -> u64 { + if can_stop { + notice_wait.min(NOTICE_POLL_WAIT_WITH_STOP) + } else { + notice_wait + } +} + +fn pull_pending_block_stuff(cnf: &PoWorkConf, stop_flag: &Option>) { + let cycle_start = Instant::now(); let curr_hei = block_mining_runtime::current_mining_height(); // query pending let urlapi_pending = format!( - "http://{}/query/miner/pending?stuff=true&t={}", + "http://{}/query/miner/pending?stuff=true&t={}{}", &cnf.rpcaddr, - sys::curtimes() + sys::curtimes(), + cnf.worker_param() ); let jsdata = match crate::rpc_http::get_text(&HTTP_CLIENT, &urlapi_pending, &cnf.api_token, None) { @@ -204,25 +311,49 @@ fn pull_pending_block_stuff(cnf: &PoWorkConf) { let jstr = |k| res[k].as_str().unwrap_or(""); let jnum = |k| res[k].as_u64().unwrap_or(0); let JV::String(ref _blkhd) = res["block_intro"] else { + if let Some(reason) = upstream_stale_reason(&res) { + enter_upstream_stale(reason, "pending work"); + delay_return!(STALE_UPSTREAM_BACKOFF_SECS); + } println!("Error: get block stuff error: {}", jstr("err")); delay_return!(15); }; let pending_height = jnum("height"); - // set pending block stuff - if pending_height > curr_hei { - if let Err(e) = block_mining_runtime::set_pending_block_stuff(pending_height, res) { - println!("Error: invalid block data from {urlapi_pending}: {e}"); - delay_return!(10); - } - if curr_hei == 0 { - block_mining_runtime::may_print_turn_to_nex_block_mining(curr_hei, None); - } + // Install the refreshed template unconditionally. Comparing the raw block_intro + // hex to decide would call EVERY poll a template change: the fullnode + // increments the served coinbase nonce and recomputes the merkle root on every + // /query/miner/pending request, and the merkle root lives inside the serialized + // intro. Whether this is really a NEW job - and so whether workers must give up + // what they are mining - is decided inside set_pending_block_stuff from the + // job's identity (height + parent hash), which that re-serialization leaves + // untouched. A same-height reorg still changes the parent and is still detected. + if let Err(e) = block_mining_runtime::set_pending_block_stuff(pending_height, res) { + println!("Error: invalid block data from {urlapi_pending}: {e}"); + delay_return!(10); + } + // Real work was served and installed, so any stale-work pause lifts. + leave_upstream_stale(); + if curr_hei == 0 { + block_mining_runtime::may_print_turn_to_nex_block_mining(curr_hei, None); } // with notice let mut rpid = vec![0].repeat(16); + // The stop flag is only observable BETWEEN notice requests, so cap the + // per-request wait when a supervisor can ask us to stop (see + // notice_poll_wait); otherwise an in-flight long-poll would hold shutdown + // for the whole notice window. + let poll_wait = notice_poll_wait(cnf.noticewait, stop_flag.is_some()); + let poll_timeout = if stop_flag.is_some() { + Duration::from_secs(poll_wait.saturating_add(5)) + } else { + Duration::from_secs(300) + }; loop { + if should_stop(stop_flag) { + return; + } if let Err(e) = getrandom::fill(&mut rpid) { println!("Error: cannot generate request id: {e}"); delay_return!(1); @@ -230,7 +361,7 @@ fn pull_pending_block_stuff(cnf: &PoWorkConf) { let urlapi_notice = format!( "http://{}/query/miner/notice?wait={}&height={}&rqid={}", &cnf.rpcaddr, - &cnf.noticewait, + &poll_wait, pending_height, &hex::encode(&rpid) ); @@ -239,7 +370,7 @@ fn pull_pending_block_stuff(cnf: &PoWorkConf) { &HTTP_CLIENT, &urlapi_notice, &cnf.api_token, - Some(Duration::from_secs(300)), + Some(poll_timeout), ) { Ok(t) => t, Err(e) => { @@ -254,31 +385,47 @@ fn pull_pending_block_stuff(cnf: &PoWorkConf) { println!("Error: invalid miner notice JSON at {urlapi_notice}"); delay_return!(1); }; + // The notice body carries the LAST known height alongside the error, so it + // must be checked before the height comparison below: otherwise a stale + // notice at our own height would look like new work and spin the miner + // between notice and pending while the workers grind dead work. + if let Some(reason) = upstream_stale_reason(&res2) { + enter_upstream_stale(reason, "block notice"); + delay_return!(STALE_UPSTREAM_BACKOFF_SECS); + } let jnum = |k| res2[k].as_u64().unwrap_or(0); let res_hei = jnum("height"); - // println!("\n++++++++ {} {} {}\n", &jsdata, res_hei, current_height); - if res_hei >= pending_height { - // next block discover - break; + // Fullnode notice reports chain tip height, while we long-poll with the + // pending (tip+1) height. When tip advances to pending_height, break for + // the next job. On timeout (or any other reply) also leave the wait so the + // outer loop re-fetches /query/miner/pending: that is how same-height tip + // reorgs (intro change without tip height advance) are detected. A second + // tight spin on the same tip would never re-read block_intro. + // + // Nothing new means the cycle must not be allowed to run free: it costs two + // requests, and an upstream that answers the notice instantly would + // otherwise be hammered at tens of requests per second (see NOTICE_CYCLE_MIN). + let new_work_ready = pending_height > 0 && res_hei >= pending_height; + let backoff = notice_cycle_backoff(cycle_start.elapsed(), new_work_ready); + if !backoff.is_zero() { + std::thread::sleep(backoff); } - // No new block yet. A compliant server long-polls (so this rarely loops), - // but a fast-returning/incompatible server or notice_wait=0 would otherwise - // spin this at 100% CPU — cap the spin with a small delay before re-polling. - std::thread::sleep(Duration::from_millis(200)); + break; } } fn push_block_mining_success(cnf: &PoWorkConf, success: &block_mining_runtime::BlockMiningResult) { let urlapi_success = format!( - "http://{}/submit/miner/success?height={}&block_nonce={}&coinbase_nonce={}&t={}", + "http://{}/submit/miner/success?height={}&block_nonce={}&coinbase_nonce={}&t={}{}", &cnf.rpcaddr, success.height, success.head_nonce, success.coinbase_nonce.to_hex(), - sys::curtimes() + sys::curtimes(), + cnf.worker_param() ); // Submitting the winning block is the entire payoff of solo mining, and the - // result was already drained from the channel — so a single transient network + // result was already drained from the channel, so a single transient network // error must not silently lose it. Retry transport failures with backoff, and // only claim SUCCESS once the node confirms acceptance (ret == 0). const MAX_SUBMIT_ATTEMPTS: u32 = 5; @@ -289,19 +436,37 @@ fn push_block_mining_success(cnf: &PoWorkConf, success: &block_mining_runtime::B Ok(body) => { let parsed = serde_json::from_str::(&body).ok(); let ret = parsed.as_ref().and_then(|j| j["ret"].as_i64()); - last = body; - if ret == Some(0) { - accepted = true; - } else if ret.is_some() { - // Deterministic node rejection (stale height, invalid, etc.): - // retrying will not help, so stop and report it honestly. - let err = parsed - .as_ref() - .and_then(|j| j["err"].as_str()) - .unwrap_or(""); - println!("[submit] node rejected height {}: {}", success.height, err); + last = body.clone(); + match ret { + Some(0) => { + accepted = true; + break; + } + Some(_) => { + // Deterministic node rejection (stale height, invalid, + // etc.): retrying will not help, so stop and report it. + let err = parsed + .as_ref() + .and_then(|j| j["err"].as_str()) + .unwrap_or(""); + println!("[submit] node rejected height {}: {}", success.height, err); + break; + } + None => { + // HTTP 200 but no parseable `ret` (proxy/load-balancer error + // page, truncated or non-JSON body). This is NOT a node + // decision, so treat it as transient and retry: a winning + // block is not discarded on a front-end hiccup. + let snippet: String = body.chars().take(120).collect(); + println!( + "[submit] attempt {}/{} unrecognized response, retrying: {}", + attempt, MAX_SUBMIT_ATTEMPTS, snippet + ); + if attempt < MAX_SUBMIT_ATTEMPTS { + std::thread::sleep(Duration::from_millis(500u64 * attempt as u64)); + } + } } - break; } Err(e) => { last = format!("transport error: {e}"); @@ -723,7 +888,7 @@ fn run_block_mining_benchmark(cnf: &PoWorkConf, config_path: &str) { let Some(base) = pick_benchmark_result(&bench_results, cnf.efficiency.mode) else { println!( - "[benchmark] No successful tuning points — config unchanged (check OpenCL driver)." + "[benchmark] No successful tuning points; config unchanged (check OpenCL driver)." ); continue; }; @@ -1068,6 +1233,78 @@ mod tests { ); } + #[test] + fn an_upstream_stale_answer_is_recognized_on_both_miner_endpoints() { + // Exact bodies the pool bridge returns during an upstream node outage. + let pending = serde_json::json!({"err": "upstream stale; work is not being refreshed"}); + assert_eq!( + upstream_stale_reason(&pending), + Some("upstream stale; work is not being refreshed") + ); + let notice = serde_json::json!({ + "err": "upstream stale; work is not being refreshed", + "height": 1234 + }); + assert!(upstream_stale_reason(¬ice).is_some()); + } + + #[test] + fn an_unknown_or_missing_error_body_is_not_treated_as_stale() { + // Anything else must stay an ordinary transient error, and no shape may + // panic the miner. + assert_eq!(upstream_stale_reason(&serde_json::json!({})), None); + assert_eq!(upstream_stale_reason(&serde_json::json!({"err": ""})), None); + assert_eq!( + upstream_stale_reason(&serde_json::json!({"err": "no job yet; wait for upstream fullnode"})), + None + ); + assert_eq!(upstream_stale_reason(&serde_json::json!({"err": 7})), None); + assert_eq!( + upstream_stale_reason(&serde_json::json!({"err": {"nested": true}})), + None + ); + assert_eq!(upstream_stale_reason(&serde_json::json!(null)), None); + assert_eq!( + upstream_stale_reason(&serde_json::json!({"height": 9, "block_intro": "00"})), + None + ); + } + + #[test] + fn the_notice_cycle_has_an_anti_spin_floor_that_never_delays_new_work() { + // An upstream that answers the notice immediately must not be hammered: + // the cycle costs a pending request plus a notice request, so hold it to a + // minimum period instead of running it as fast as the network allows. + assert_eq!( + notice_cycle_backoff(Duration::from_millis(5), false), + NOTICE_CYCLE_MIN - Duration::from_millis(5) + ); + assert_eq!( + notice_cycle_backoff(Duration::ZERO, false), + NOTICE_CYCLE_MIN + ); + // A cooperating upstream already spent the whole notice window, so the + // floor costs it nothing. + assert_eq!( + notice_cycle_backoff(Duration::from_secs(45), false), + Duration::ZERO + ); + // New work is a payout waiting to be mined and is never delayed. + assert_eq!( + notice_cycle_backoff(Duration::from_millis(5), true), + Duration::ZERO + ); + } + + #[test] + fn notice_long_poll_is_capped_when_a_stop_flag_can_arrive() { + assert_eq!(notice_poll_wait(45, true), NOTICE_POLL_WAIT_WITH_STOP); + assert_eq!(notice_poll_wait(300, true), NOTICE_POLL_WAIT_WITH_STOP); + assert_eq!(notice_poll_wait(1, true), 1); + assert_eq!(notice_poll_wait(45, false), 45); + assert_eq!(notice_poll_wait(300, false), 300); + } + #[test] fn cpu_group_mining_result_matches_manual_scan() { let height = 1u64; diff --git a/app/src/rpc_http.rs b/app/src/rpc_http.rs index 586774a2..02dc4dae 100644 --- a/app/src/rpc_http.rs +++ b/app/src/rpc_http.rs @@ -77,8 +77,22 @@ fn read_limited(reader: R, declared_length: Option) -> Result Result { + let status = resp.status(); let declared_length = resp.content_length(); - read_limited(resp, declared_length) + let body = read_limited(resp, declared_length)?; + // reqwest returns Ok for ANY HTTP status. A 5xx (or 408/429) is a transient + // server/proxy failure, NOT an application reply — surface it as an error so + // callers retry (e.g. a winning block submit) instead of mistaking a 502 + // error page for a response and dropping the block. Deterministic 4xx replies + // are passed through so the caller can read the node's error body. + if status.is_server_error() + || status == reqwest::StatusCode::REQUEST_TIMEOUT + || status == reqwest::StatusCode::TOO_MANY_REQUESTS + { + let snippet: String = body.chars().take(200).collect(); + return Err(format!("upstream HTTP {status}: {snippet}")); + } + Ok(body) } #[cfg(test)] diff --git a/app/tests/poworker_template_churn_sim.rs b/app/tests/poworker_template_churn_sim.rs new file mode 100644 index 00000000..c7b34705 --- /dev/null +++ b/app/tests/poworker_template_churn_sim.rs @@ -0,0 +1,114 @@ +//! The fullnode re-serializes `block_intro` on EVERY `/query/miner/pending` +//! request: it increments the served coinbase nonce, recomputes the merkle root +//! from it and re-serializes the intro, and the merkle root sits inside the intro. +//! The miner replaces that merkle root itself before hashing, so a re-serialized +//! template is the SAME job. +//! +//! When the miner treated it as a new job it reinstalled on every poll, bumped the +//! template generation, and every worker then threw away the batch it had just +//! finished - the winning nonce included. This test mines against a sim that churns +//! the template the way the real node does and asserts that mining output keeps +//! flowing ACROSS repeated polls, not just before the first one. + +use std::sync::atomic::{AtomicBool, Ordering}; +use std::sync::{Arc, Mutex, OnceLock}; +use std::thread; +use std::time::{Duration, Instant}; + +use app::poworker::{PoWorkConf, poworker_with_stop}; +use testkit::sim::miner_api::{MinerApiSim, MinerPendingStuff}; + +/// The mining runtime keeps its installed template in process-global state, so the +/// sim-driven tests in this binary must not overlap. +fn test_guard() -> std::sync::MutexGuard<'static, ()> { + static LOCK: OnceLock> = OnceLock::new(); + LOCK.get_or_init(|| Mutex::new(())) + .lock() + .unwrap_or_else(|e| e.into_inner()) +} + +fn fetch_block_intro(rpcaddr: &str) -> String { + let url = format!("http://{rpcaddr}/query/miner/pending?stuff=true"); + let body = reqwest::blocking::get(&url) + .expect("query the simulated miner api") + .text() + .expect("read the pending body"); + let parsed: serde_json::Value = + serde_json::from_str(&body).expect("parse the pending body as json"); + parsed["block_intro"] + .as_str() + .expect("pending answer carries a block_intro") + .to_string() +} + +#[test] +fn a_miner_polling_a_template_mutating_node_still_submits_its_winners() { + let _guard = test_guard(); + + let sim = MinerApiSim::start(MinerPendingStuff::easy_for_test(1)); + + // Guard the guard: if the sim ever stops churning the template, this whole + // test silently stops proving anything, which is exactly how the defect + // survived a green suite in the first place. + let first = fetch_block_intro(sim.rpcaddr()); + let second = fetch_block_intro(sim.rpcaddr()); + assert_ne!( + first, second, + "the sim must re-serialize block_intro per request like the real fullnode" + ); + + let stop = Arc::new(AtomicBool::new(false)); + // Batches deliberately longer than the miner's pending poll period, so several + // template refreshes land INSIDE every batch. That is the production shape: a + // CPU batch self-tunes to ~3 s while the node is polled far more often. + let cnf = PoWorkConf::test_defaults(sim.rpcaddr().to_string(), 1, 20_000); + + let stop2 = stop.clone(); + let worker = thread::spawn(move || { + poworker_with_stop(cnf, Some(stop2)); + }); + + // Mining has to work at all before the churn question is even meaningful. + if !sim.wait_for_submit(1, Duration::from_secs(45)) { + stop.store(true, Ordering::Relaxed); + thread::sleep(Duration::from_millis(80)); + panic!( + "the miner submitted nothing at all against a node that refreshes its template on every poll" + ); + } + + // From here on, measure output ACROSS template refreshes: wait until the miner + // has re-fetched pending several more times and count the winners it produced + // in that window. Reinstalling on every poll used to void the in-flight batch + // of every worker, so this window produced nothing at all. + const REQUIRED_POLLS: usize = 5; + const REQUIRED_SUBMITS: usize = 3; + let polls_before = sim.pending_count(); + let submits_before = sim.submit_count(); + let deadline = Instant::now() + Duration::from_secs(45); + while Instant::now() < deadline { + if sim.pending_count() >= polls_before + REQUIRED_POLLS + && sim.submit_count() >= submits_before + REQUIRED_SUBMITS + { + break; + } + thread::sleep(Duration::from_millis(20)); + } + let polls = sim.pending_count().saturating_sub(polls_before); + let submits = sim.submit_count().saturating_sub(submits_before); + stop.store(true, Ordering::Relaxed); + thread::sleep(Duration::from_millis(80)); + + assert!( + polls >= REQUIRED_POLLS, + "expected the miner to re-fetch pending at least {REQUIRED_POLLS} more times, saw {polls}" + ); + assert!( + submits >= REQUIRED_SUBMITS, + "mining output collapsed across template refreshes: only {submits} winners submitted over {polls} pending polls" + ); + assert_eq!(sim.last_submit().get("height"), Some(&"1".to_string())); + + drop(sim); + let _ = worker.join(); +} diff --git a/basis/src/config/engine.rs b/basis/src/config/engine.rs index bf12ffa1..5ab6d04d 100644 --- a/basis/src/config/engine.rs +++ b/basis/src/config/engine.rs @@ -3,6 +3,36 @@ +/// Strict variant of `ini_must_u64` for the keys where silently falling back to the default +/// picks the wrong chain or the wrong limit. `ini_must_u64` cannot tell "key absent" from +/// "key present but unparseable" and returns the default for both, so a typo like +/// `chain_id = 5abc` would resolve to 0, which is mainnet, and the node would run mainnet +/// rules while the operator believes it is on a side chain. An absent or empty value still +/// takes the default; anything present but not a number is a hard startup failure. +fn engine_ini_u64_strict(sec: &HashMap>, sec_name: &str, key: &str, dv: u64) -> u64 { + let Some(raw) = sec.get(key).and_then(|v| v.as_deref()) else { + return dv; + }; + // the section map can also be built by hand, so drop an inline comment here too: + // `chain_id = 0 ; mainnet` must not be treated as a typo + let val = raw + .split(|c| c == ';' || c == '#') + .next() + .unwrap_or("") + .trim(); + if val.is_empty() { + return dv; + } + match val.parse::() { + Ok(n) => n, + Err(_) => panic!( + "[Config Error] [{}] {} is present but is not a valid number: {:?}", + sec_name, key, raw + ), + } +} + + #[derive(Clone)] pub struct EngineConf { pub max_block_txs: usize, @@ -124,8 +154,8 @@ impl EngineConf { contract_cache_size: 0.0, }; // setup lowest_fee - if ini_must(sec_server, "lowest_fee", "").len() > 0 { - let lfepr = ini_must_amount(sec_server, "lowest_fee").compress(2, AmtCpr::Grow) + if ini_must(sec_server, "lowest_fee", "").trim().len() > 0 { + let lfepr = ini_must_amount_required(sec_server, "server", "lowest_fee").compress(2, AmtCpr::Grow) .unwrap().to_238_u64().unwrap() / 166; // =6024, simple hac trs size cnf.lowest_fee_purity = lfepr; println!("[Config] node accepted lowest fee purity {}.", lfepr); @@ -135,22 +165,33 @@ impl EngineConf { cnf.fast_sync = ini_must_bool(sec, "fast_sync", false); let sec_mint = &ini_section(ini, "mint"); - cnf.chain_id = ini_must_u64(sec_mint, "chain_id", 0) as u32; - cnf.sync_maxh = ini_must_u64(sec_mint, "height_max", 0); - cnf.dev_count_switch = ini_must_u64(sec_mint, "dev_count_switch", 0) as usize; + // chain_id selects the consensus rule set, so it is parsed strictly and its range is + // checked: `as u32` alone would fold 4294967296 back to 0, which is mainnet. + let chain_id = engine_ini_u64_strict(sec_mint, "mint", "chain_id", 0); + if chain_id > u32::MAX as u64 { + panic!("[Config Error] [mint] chain_id {} is out of range, it must be between 0 and {}.", + chain_id, u32::MAX) + } + cnf.chain_id = chain_id as u32; + cnf.sync_maxh = engine_ini_u64_strict(sec_mint, "mint", "height_max", 0); + cnf.dev_count_switch = engine_ini_u64_strict(sec_mint, "mint", "dev_count_switch", 0) as usize; cnf.show_miner_name = ini_must_bool(sec_mint, "show_miner_name", false); let sec_vm = &ini_section(ini, "vm"); cnf.vm_log_enable = ini_must_bool(sec_vm, "log_enable", false); cnf.vm_log_can_delete = ini_must_bool(sec_vm, "log_can_delete", false); - cnf.vm_log_open_height = ini_must_u64(sec_vm, "log_open_height", 0) as u64; + cnf.vm_log_open_height = engine_ini_u64_strict(sec_vm, "vm", "log_open_height", 0); cnf.vm_log_delete_auth_hash = ini_must(sec_vm, "log_delete_auth_hash", ""); // HAC miner let sec_miner = &ini_section(ini, "miner"); cnf.miner_enable = ini_must_bool(sec_miner, "enable", false); if cnf.miner_enable { - cnf.miner_reward_address = ini_must_address(sec_miner, "reward"); + cnf.miner_reward_address = ini_must_address_required(sec_miner, "miner", "reward"); + if !cnf.miner_reward_address.is_privakey() { + panic!("miner reward address {} must be PRIVAKEY type but got version {}", + cnf.miner_reward_address.to_readable(), cnf.miner_reward_address.version()) + } let msg = ini_must_maxlen(sec_miner, "message", "", 16); let msgapp = vec![' ' as u8].repeat(16-msg.len()); let msg: [u8; 16] = vec![msg.as_bytes().to_vec(), msgapp].concat().try_into().unwrap(); @@ -161,23 +202,27 @@ impl EngineConf { let sec_dmer = &ini_section(ini, "diamondminer"); cnf.dmer_enable = ini_must_bool(sec_dmer, "enable", false); if cnf.dmer_enable { - cnf.dmer_reward_address = ini_must_address(sec_dmer, "reward"); + cnf.dmer_reward_address = ini_must_address_required(sec_dmer, "diamondminer", "reward"); if !cnf.dmer_reward_address.is_privakey() { panic!("diamond miner reward address {} must be PRIVAKEY type but got version {}", cnf.dmer_reward_address.to_readable(), cnf.dmer_reward_address.version()) } - cnf.dmer_bid_account = ini_must_account(sec_dmer, "bid_password"); - cnf.dmer_bid_min = ini_must_amount(sec_dmer, "bid_min").compress(2, AmtCpr::Grow).unwrap(); - cnf.dmer_bid_max = ini_must_amount(sec_dmer, "bid_max").compress(2, AmtCpr::Grow).unwrap(); - cnf.dmer_bid_step = ini_must_amount(sec_dmer, "bid_step").compress(2, AmtCpr::Grow).unwrap(); + cnf.dmer_bid_account = ini_must_account_required(sec_dmer, "bid_password"); + cnf.dmer_bid_min = ini_must_amount_required(sec_dmer, "diamondminer", "bid_min").compress(2, AmtCpr::Grow).unwrap(); + cnf.dmer_bid_max = ini_must_amount_required(sec_dmer, "diamondminer", "bid_max").compress(2, AmtCpr::Grow).unwrap(); + cnf.dmer_bid_step = ini_must_amount_required(sec_dmer, "diamondminer", "bid_step").compress(2, AmtCpr::Grow).unwrap(); } // tx pool + // An unset `maxs` must stay empty: the caller overlays this list onto its own + // defaults, so the old fallback of 100 per unparsed entry silently shrank a mining + // node's pool from 2000 to 100 slots whenever the key was simply absent. let sec_txpool = &ini_section(ini, "txpool"); - cnf.txpool_maxs = ini_must(sec_txpool, "maxs", "").replace(" ", "").split(",").map(|a|{ + let txpool_maxs = ini_must(sec_txpool, "maxs", "").replace(" ", ""); + cnf.txpool_maxs = txpool_maxs.split(",").filter(|a| !a.is_empty()).map(|a|{ match a.parse::() { Ok(n) => n, - _ => 100, + _ => panic!("[Config Error] [txpool] maxs entry {:?} is not a valid number.", a), } }).collect(); @@ -219,4 +264,187 @@ mod tests { Address::from_readable(&reward).unwrap() ); } + + #[test] + fn negation_word_keeps_the_miners_off() { + for word in ["no", "off", "disabled"] { + let mut ini = IniObj::new(); + ini.insert( + "miner".to_owned(), + HashMap::from([("enable".to_owned(), Some(word.to_owned()))]), + ); + ini.insert( + "diamondminer".to_owned(), + HashMap::from([("enable".to_owned(), Some(word.to_owned()))]), + ); + let cnf = EngineConf::new(&ini); + assert_eq!(cnf.miner_enable, false, "enable = {} must stay off", word); + assert_eq!(cnf.dmer_enable, false, "enable = {} must stay off", word); + } + } + + #[test] + fn diamond_miner_refuses_to_start_without_a_bid_password() { + let mut ini = IniObj::new(); + ini.insert( + "diamondminer".to_owned(), + HashMap::from([ + ("enable".to_owned(), Some("true".to_owned())), + ("reward".to_owned(), Some("1MzNY1oA3kfgYi75zquj3SRUPYztzXHzK9".to_owned())), + ]), + ); + let res = std::panic::catch_unwind(|| EngineConf::new(&ini)); + assert!(res.is_err(), "an unset bid_password must not build a spending account"); + } + + #[test] + fn miner_reward_address_must_be_privakey() { + let reward = Address::from([Address::SCRIPTMH; Address::SIZE]).to_readable(); + let mut ini = IniObj::new(); + ini.insert( + "miner".to_owned(), + HashMap::from([ + ("enable".to_owned(), Some("true".to_owned())), + ("reward".to_owned(), Some(reward)), + ]), + ); + let res = std::panic::catch_unwind(|| EngineConf::new(&ini)); + assert!(res.is_err(), "a non PRIVAKEY miner reward must be rejected at config load"); + } + + fn ini_of(section: &str, pairs: &[(&str, Option<&str>)]) -> IniObj { + let mut ini = IniObj::new(); + ini.insert( + section.to_owned(), + pairs + .iter() + .map(|(k, v)| (k.to_string(), v.map(|s| s.to_string()))) + .collect(), + ); + ini + } + + // the example address the missing reward key used to fall back to + const OLD_DEFAULT_REWARD: &str = "1AVRuFXNFi3rdMrPH4hdqSgFrEBnWisWaS"; + + #[test] + fn hac_miner_refuses_to_start_without_a_reward_address() { + for reward in [None, Some(""), Some(" ")] { + let mut pairs: Vec<(&str, Option<&str>)> = vec![("enable", Some("true"))]; + if let Some(val) = reward { + pairs.push(("reward", Some(val))); + } + let ini = ini_of("miner", &pairs); + let res = std::panic::catch_unwind(|| EngineConf::new(&ini)); + assert!( + res.is_err(), + "reward {:?} must stop startup, never mine the coinbase to {}", + reward, OLD_DEFAULT_REWARD + ); + } + } + + #[test] + fn diamond_miner_refuses_to_start_without_a_reward_address() { + let ini = ini_of("diamondminer", &[ + ("enable", Some("true")), + ("bid_password", Some("a-real-wallet-password")), + ("bid_min", Some("1")), + ("bid_max", Some("2")), + ("bid_step", Some("1:244")), + ]); + let res = std::panic::catch_unwind(|| EngineConf::new(&ini)); + assert!( + res.is_err(), + "a missing diamond reward must stop startup, never pay {}", + OLD_DEFAULT_REWARD + ); + } + + #[test] + fn diamond_miner_refuses_to_start_without_the_bid_amounts() { + let full: [(&str, Option<&str>); 6] = [ + ("enable", Some("true")), + ("reward", Some("1MzNY1oA3kfgYi75zquj3SRUPYztzXHzK9")), + ("bid_password", Some("a-real-wallet-password")), + ("bid_min", Some("1")), + ("bid_max", Some("2")), + ("bid_step", Some("1:244")), + ]; + // the complete config builds + let cnf = EngineConf::new(&ini_of("diamondminer", &full)); + assert_eq!(cnf.dmer_enable, true); + // dropping any single bid amount must stop startup instead of bidding a placeholder + for missing in ["bid_min", "bid_max", "bid_step"] { + let pairs: Vec<(&str, Option<&str>)> = + full.iter().filter(|(k, _)| *k != missing).cloned().collect(); + let ini = ini_of("diamondminer", &pairs); + let res = std::panic::catch_unwind(|| EngineConf::new(&ini)); + assert!(res.is_err(), "a missing {} must stop startup", missing); + } + } + + #[test] + fn unparseable_chain_id_must_not_silently_become_mainnet() { + let ini = ini_of("mint", &[("chain_id", Some("5abc"))]); + let res = std::panic::catch_unwind(|| EngineConf::new(&ini)); + assert!(res.is_err(), "a typo in chain_id must not run mainnet rules by accident"); + } + + #[test] + fn chain_id_outside_the_u32_range_is_rejected() { + // 4294967296 folds back to 0 = mainnet under a bare `as u32` + let ini = ini_of("mint", &[("chain_id", Some("4294967296"))]); + let res = std::panic::catch_unwind(|| EngineConf::new(&ini)); + assert!(res.is_err(), "an out of range chain_id must not wrap around to mainnet"); + } + + #[test] + fn absent_or_empty_chain_id_keeps_the_mainnet_default() { + assert!(EngineConf::new(&IniObj::new()).is_mainnet()); + assert!(EngineConf::new(&ini_of("mint", &[("chain_id", Some(" "))])).is_mainnet()); + assert!(EngineConf::new(&ini_of("mint", &[("chain_id", None)])).is_mainnet()); + } + + #[test] + fn chain_id_tolerates_surrounding_space_and_inline_comment() { + let cnf = EngineConf::new(&ini_of("mint", &[("chain_id", Some(" 7 ; sidechain"))])); + assert_eq!(cnf.chain_id, 7); + assert_eq!(cnf.is_mainnet(), false); + } + + #[test] + fn unparseable_mint_and_vm_numbers_are_rejected() { + let cases: [(&str, &str); 3] = [ + ("mint", "height_max"), + ("mint", "dev_count_switch"), + ("vm", "log_open_height"), + ]; + for (section, key) in cases { + let ini = ini_of(section, &[(key, Some("12x"))]); + let res = std::panic::catch_unwind(|| EngineConf::new(&ini)); + assert!(res.is_err(), "[{}] {} = 12x must not fall back to the default", section, key); + } + } + + #[test] + fn txpool_maxs_stays_empty_when_unset_so_callers_keep_their_own_limits() { + assert_eq!(EngineConf::new(&IniObj::new()).txpool_maxs, Vec::::new()); + assert_eq!( + EngineConf::new(&ini_of("txpool", &[("maxs", Some(" "))])).txpool_maxs, + Vec::::new() + ); + } + + #[test] + fn txpool_maxs_reads_a_list_and_rejects_garbage() { + let cnf = EngineConf::new(&ini_of("txpool", &[("maxs", Some("2000, 100"))])); + assert_eq!(cnf.txpool_maxs, vec![2000usize, 100usize]); + // a trailing separator is tolerated + let cnf = EngineConf::new(&ini_of("txpool", &[("maxs", Some("2000,100,"))])); + assert_eq!(cnf.txpool_maxs, vec![2000usize, 100usize]); + let ini = ini_of("txpool", &[("maxs", Some("2000,lots"))]); + let res = std::panic::catch_unwind(|| EngineConf::new(&ini)); + assert!(res.is_err(), "a malformed txpool limit must be reported, not defaulted"); + } } diff --git a/deploy/Dockerfile b/deploy/Dockerfile new file mode 100644 index 00000000..a4ef4438 --- /dev/null +++ b/deploy/Dockerfile @@ -0,0 +1,58 @@ +# HBIT: the Hacash full node and the payout pool, built once and run as two +# services on the same box. +# +# No GPU here on purpose. This image is for a VPS or a mini PC whose job is to +# hold the chain and pay miners; the mining happens on other people's hardware. +# Leaving OpenCL and CUDA out keeps the image small and removes a whole class of +# driver problem from a machine that must simply stay up. + +# ---------------------------------------------------------------- build stage +FROM rust:1-bookworm AS build + +# Everything the default (non-GPU) feature set needs to link. +RUN apt-get update \ + && apt-get install -y --no-install-recommends pkg-config libssl-dev clang \ + && rm -rf /var/lib/apt/lists/* + +WORKDIR /src +COPY . . + +# Cap the compiler parallelism, because the default is one job per core and each +# rustc on this tree can hold well over a gigabyte. +# +# On the machines this image is FOR, a VPS or a mini PC with 2 to 8 GB, an +# unbounded build is not slow, it is fatal: the kernel kills rustc and the build +# fails with a message about a signal that says nothing about memory. It bites on +# large machines too. Building this on a 32 core host killed the Docker engine +# outright, mid-compile, leaving only an RPC EOF behind. +# +# 4 keeps the peak near 6 to 8 GB. Raise it when the builder has the memory for +# it, roughly 2 GB per job: +# docker build --build-arg CARGO_JOBS=16 -f deploy/Dockerfile -t hbit:local . +ARG CARGO_JOBS=4 +ENV CARGO_BUILD_JOBS=${CARGO_JOBS} + +# --locked so the image is built from the committed dependency set and a +# release cannot silently pick up a different one. +RUN cargo build --locked --release --bin fullnode \ + && cargo build --locked --release -p hbit-pool --bin hbit-pool-server --bin hbit-pool-payout + +# -------------------------------------------------------------- runtime stage +FROM debian:bookworm-slim AS runtime + +RUN apt-get update \ + && apt-get install -y --no-install-recommends ca-certificates curl \ + && rm -rf /var/lib/apt/lists/* + +# A dedicated unprivileged user. The wallet and the chain belong to it, not to +# root, so a flaw in either program is not a flaw with root's reach. +RUN useradd --system --create-home --home-dir /var/lib/hbit --shell /usr/sbin/nologin hbit + +COPY --from=build /src/target/release/fullnode /usr/local/bin/fullnode +COPY --from=build /src/target/release/hbit-pool-server /usr/local/bin/hbit-pool-server +COPY --from=build /src/target/release/hbit-pool-payout /usr/local/bin/hbit-pool-payout +COPY --from=build /src/docs/POOL-OPERATOR.md /usr/share/doc/hbit/POOL-OPERATOR.md +COPY --from=build /src/docs/POOL-README.md /usr/share/doc/hbit/POOL-README.md + +USER hbit +WORKDIR /var/lib/hbit diff --git a/deploy/README.md b/deploy/README.md new file mode 100644 index 00000000..abfbd1bf --- /dev/null +++ b/deploy/README.md @@ -0,0 +1,217 @@ +# Running HBIT on a VPS or a mini PC + +This is the deployment path for the pool operator. The pool is not something +miners download; it is a service you run beside your own full node, and miners +point at it. + +Two ways are provided and they do the same job. Pick one. + +| | Docker Compose | systemd | +|---|---|---| +| Setup | one file, one command | install binaries and units | +| Node RPC exposure | unreachable by construction | depends on your firewall | +| Upgrade | rebuild the image, recreate | replace the binary, restart | +| Needs | Docker on the host | nothing extra | + +**Docker is the safer default**, for one specific reason covered below. + +--- + +## The thing to get right, whichever you pick + +The node's RPC is the **miner API**. Anything that can reach it can ask for +block templates and submit blocks. It must never be reachable from the internet. +The pool's port must be. + +With Compose that is structural: the node service publishes no port at all, so +its RPC exists only on the private network the pool shares with it. There is no +rule to forget. + +With systemd it is your firewall's job, and a firewall nobody verified is a +firewall nobody has. The verification step below is not optional. + +--- + +## Docker Compose + +### First run + +```bash +git clone https://github.com/Moskyera/fullnodedev.git +cd fullnodedev +``` + +Set the node's reward address. The node **refuses to start** until you do, on +purpose: an earlier version of this project shipped a config with somebody +else's address in it, and anyone who ran it mined into a stranger's wallet. + +```bash +$EDITOR deploy/node/hacash.config.ini # fill in reward = your own address +``` + +Create the wallet passphrase. It is one half of the wallet; the key file the +pool creates is the other, and neither is worth anything alone. + +```bash +mkdir -p deploy/secrets +install -m 0400 /dev/null deploy/secrets/wallet-passphrase +printf '%s' 'a passphrase you have written down somewhere else' \ + > deploy/secrets/wallet-passphrase +``` + +`deploy/secrets/` is gitignored so it cannot be committed. Write the passphrase +on paper too, somewhere that is not this machine. + +Set the chain and the port you want in `deploy/docker-compose.yml` if the +defaults (mainnet, 9777) are not what you want, then: + +```bash +docker compose -f deploy/docker-compose.yml up -d --build +docker compose -f deploy/docker-compose.yml logs -f pool +``` + +The first start syncs the chain, which takes a while and publishes nothing until +it is done. The pool waits for the node to answer before it serves any work. + +### Back up before you tell anyone to mine on it + +The `pool-data` volume holds the wallet, the PPLNS share window and the +pending-payout ledger. Losing the node volume costs a resync. Losing this one +costs every coin the pool holds and the record of who is owed what. + +```bash +docker compose -f deploy/docker-compose.yml stop pool +docker run --rm -v hbit_pool-data:/d -v "$PWD":/out debian:bookworm-slim \ + tar czf /out/hbit-pool-backup-$(date +%F).tgz -C /d . +docker compose -f deploy/docker-compose.yml start pool +``` + +Copy that archive **and** the passphrase somewhere that is not this machine. +Then prove the backup works before you rely on it: restore it into an empty +volume and run `hbit-pool-payout` with no `--commit`, which pays nothing and +prints the wallet address. If that address matches your pool's, the backup is +real. + +### Paying out by hand + +The server settles on a timer. To settle manually you must stop it first: both +programs take an exclusive lock on the wallet, because two things settling one +wallet is how a pool pays the same window twice. + +```bash +docker compose -f deploy/docker-compose.yml stop pool +docker compose -f deploy/docker-compose.yml run --rm --entrypoint hbit-pool-payout pool \ + http://node:8080 pool-wallet.key mainnet # dry run, pays nothing +# read the split, then repeat with --commit +docker compose -f deploy/docker-compose.yml start pool +``` + +### Upgrading + +```bash +git pull +docker compose -f deploy/docker-compose.yml up -d --build +``` + +Never rename or replace anything in the `pool-data` volume during an upgrade. +The wallet file, its `.state.json` and its `.settle.lock` are matched by name; a +renamed wallet means the pool starts empty while the money sits in a file +nothing reads any more. + +--- + +## systemd + +For a host without Docker. Units are in `deploy/systemd/`. + +```bash +sudo useradd --system --home-dir /var/lib/hbit --shell /usr/sbin/nologin hbit +sudo mkdir -p /opt/hbit /var/lib/hbit/node /var/lib/hbit/pool /etc/hbit +sudo chown -R hbit:hbit /var/lib/hbit + +cargo build --locked --release --bin fullnode +cargo build --locked --release -p hbit-pool --bin hbit-pool-server --bin hbit-pool-payout +sudo install -m 0755 target/release/fullnode /opt/hbit/ +sudo install -m 0755 target/release/hbit-pool-server /opt/hbit/ +sudo install -m 0755 target/release/hbit-pool-payout /opt/hbit/ +sudo install -m 0755 deploy/hbit-wait-for-node.sh /opt/hbit/ + +sudo install -m 0400 -o hbit -g hbit /dev/null /etc/hbit/wallet-passphrase +sudo -u hbit tee /etc/hbit/wallet-passphrase >/dev/null <<< 'your passphrase' + +sudo cp deploy/systemd/*.service /etc/systemd/system/ +sudo systemctl daemon-reload +sudo systemctl enable --now hacash-node hbit-pool +journalctl -u hbit-pool -f +``` + +The passphrase is in a `0400` file rather than an `Environment=` line because +unit files are commonly world readable and `systemctl show hbit-pool` prints +every environment value. + +### The firewall, which is the part that matters + +Open the pool port. Keep the node's RPC closed. + +```bash +sudo ufw allow 9777/tcp # miners +sudo ufw allow 3337/tcp # chain p2p +sudo ufw deny 8080/tcp # node RPC: never from outside +sudo ufw enable +``` + +Then **verify from another machine**, because the rule you believe you wrote is +not evidence: + +```bash +nmap -Pn -p 8080,9777,3337 YOUR.VPS.IP +``` + +9777 and 3337 open, 8080 filtered or closed. If 8080 answers, stop the pool and +fix it before anyone mines on this. + +Also make sure the node's own config binds its RPC to `127.0.0.1` when it is not +in a container. `deploy/node/hacash.config.ini` binds `0.0.0.0`, which is right +inside Docker and wrong on a bare host. + +--- + +## Operating it + +Healthy log, roughly every settle interval: + +``` +[settle] holding back N unit(s) of block income that is not yet buried 16 deep +[settle] submitted payout tx paying N miner(s) U units; the node holds it +[reorg] our block N orphaned (chain holds ) +``` + +Orphans are normal: it means the pool noticed one of its blocks losing a race +and did not pay out on income that no longer exists. + +A payout that stays pending across several cycles gets a warning naming the +cause. The pool mines coinbase-only blocks unless the node has transactions to +pack, so a payout confirms when a block includes it. + +From another machine, the two endpoints a miner cares about: + +```bash +curl -s http://YOUR.VPS.IP:9777/terms +curl -s "http://YOUR.VPS.IP:9777/earnings?worker=" +``` + +`/terms` reports the pool's real scheme, fee, minimum and maturity, read out of +the same constants settlement uses, so what it advertises cannot drift from what +it does. + +--- + +## Telling miners about it + +Miners point the panel or `poworker` at `YOUR.VPS.IP:9777` and set +`pool_worker` to their own HAC address, which is also their payout address. The +pool refuses a share from an address it could not pay, so nobody mines for an id +that would be dropped at settlement. + +To have HBIT appear in the panel's pool list with your address filled in, ship a +`pools.json` as described in `docs/POOL-OPERATOR.md`. diff --git a/deploy/docker-compose.yml b/deploy/docker-compose.yml new file mode 100644 index 00000000..d2c8b0ed --- /dev/null +++ b/deploy/docker-compose.yml @@ -0,0 +1,122 @@ +# HBIT on a VPS or a mini PC: one full node, one payout pool, one command. +# +# THE POINT OF RUNNING IT THIS WAY is what is NOT published. The node's RPC is +# the miner API: anything that can reach it can ask for block templates and +# submit blocks. On a bare host that port is one forgotten firewall rule away +# from the internet. Here it is only ever reachable on the private compose +# network, because nothing maps it to the host. The pool port is the single +# published port, and it is the only one that has any business being open. +# +# Start: docker compose -f deploy/docker-compose.yml up -d +# Watch: docker compose -f deploy/docker-compose.yml logs -f pool +# Stop: docker compose -f deploy/docker-compose.yml down +# +# Before the first start, read deploy/README.md. It covers the passphrase, the +# backup you must take, and the payout procedure. Running a pool means holding +# other people's money. + +name: hbit + +services: + node: + build: + context: .. + dockerfile: deploy/Dockerfile + image: hbit:local + command: ["fullnode"] + working_dir: /var/lib/hbit/node + restart: unless-stopped + volumes: + # The chain. Losing it costs a resync, not money. + - node-data:/var/lib/hbit/node + # hacash.config.ini is resolved next to the executable's working dir. + # Read only: the node must never be able to rewrite its own consensus + # settings, and a config change should be a deliberate act on the host. + - ./node/hacash.config.ini:/var/lib/hbit/node/hacash.config.ini:ro + # NO ports: mapping. The RPC stays on the compose network where only the + # pool can reach it. Publishing 8080 here would hand the miner API to the + # internet, which is the single worst mistake available on this box. + expose: + - "8080" + healthcheck: + # Answering /query/latest is the honest test: the process being alive says + # nothing about whether it can serve the pool. + test: ["CMD", "curl", "-fsS", "http://127.0.0.1:8080/query/latest"] + interval: 15s + timeout: 5s + retries: 20 + start_period: 120s + + pool: + image: hbit:local + depends_on: + node: + # Not just "started". The pool refuses to serve work it cannot verify + # against the node, so starting it before the node answers just makes it + # exit and retry. + condition: service_healthy + working_dir: /var/lib/hbit/pool + restart: unless-stopped + # [settle_secs] + # Listening on 0.0.0.0 INSIDE the container is correct and not an exposure: + # only the published port below reaches the outside world. + # + # share_bits is how many powers of two EASIER a share is than a block, so + # what a share costs is (block work - share_bits). Measured on the live chain + # on 2026-07-27, a block costs 2^42, which makes 20 the right value here for + # two reasons. + # + # Margin. The pool refuses to start when a share would cost less than 2^16, + # because below that PPLNS credit tracks how fast a worker completes an HTTP + # round trip rather than how much it hashes. 24 leaves a share at 2^18: only + # two bits of headroom, so a network difficulty drop of 4x would take this + # pool offline at its next restart. 20 leaves 2^22, which is six bits. + # + # Payout memory. PPLNS pays on the last 4096 shares, so easier shares make + # that window cover LESS time. At 2^18 a single card of ordinary speed + # produces about 26 shares a second, and with ten miners attached the whole + # window turns over in roughly fifteen seconds: a miner that drops off for + # half a minute loses everything it was owed. At 2^22 the same pool keeps + # about four minutes of history, which is the difference between a payout + # scheme and a lottery on connection stability. + # + # Raise it only if miners report too few shares to be paid smoothly, and + # never past (block work - 16), which the pool enforces and explains. + command: + - "hbit-pool-server" + - "http://node:8080" + - "pool-wallet.key" + - "0.0.0.0:9777" + - "20" + - "mainnet" + - "300" + environment: + # A path, never the passphrase itself. An Environment value shows up in + # docker inspect and in any process listing inside the container; a file + # mounted read only does not. + HBIT_WALLET_PASSWORD_FILE: /run/secrets/hbit_wallet_passphrase + secrets: + - hbit_wallet_passphrase + volumes: + # The wallet, the PPLNS share window and the pending-payout ledger all + # live here. THIS is the volume that holds money and the record of who is + # owed what. Back it up. A lost chain is an afternoon; a lost wallet is + # everything the pool holds and everything it owes. + - pool-data:/var/lib/hbit/pool + ports: + # host:container. This is the only port reachable from outside, and it is + # the one miners connect to. + - "9777:9777" + +secrets: + hbit_wallet_passphrase: + # Create it before the first start, and keep a copy somewhere that is not + # this machine: + # install -m 0400 /dev/null deploy/secrets/wallet-passphrase + # printf '%s' 'your passphrase' > deploy/secrets/wallet-passphrase + # Without it the pool refuses to start rather than writing a plaintext key. + file: ./secrets/wallet-passphrase + +volumes: + node-data: + pool-data: diff --git a/deploy/hbit-wait-for-node.sh b/deploy/hbit-wait-for-node.sh new file mode 100644 index 00000000..4d06883c --- /dev/null +++ b/deploy/hbit-wait-for-node.sh @@ -0,0 +1,72 @@ +#!/bin/sh +# Block until a Hacash fullnode answers, then exit 0. +# +# Why this exists: systemd's After= only orders the START of two units, it does +# not wait for the first one to be ready. hbit-pool-server refuses to start when +# the node is not answering (correctly: it will not serve work against a node +# that is not there), so on a reboot, and on the first boot of a box whose chain +# is still syncing, the pool would fail and be restarted until the node caught +# up. This turns that into one wait with one message. +# +# It never prompts and never reads a secret. Everything it needs is an argument. +# +# usage: hbit-wait-for-node.sh [seconds_to_wait] +# e.g. http://127.0.0.1:8080 - the node's API, which is the +# [server] listen port in its hacash.config.ini. +# [seconds_to_wait] give up after this many seconds and exit non-zero, so +# systemd restarts the unit and the wait starts again. +# Default 3600. 0 means wait forever. +set -eu + +NODE=${1:-} +LIMIT=${2:-3600} + +if [ -z "$NODE" ]; then + echo "hbit-wait-for-node.sh: no node URL given." >&2 + echo "What to do: pass the node's API base URL, e.g." >&2 + echo " hbit-wait-for-node.sh http://127.0.0.1:8080 3600" >&2 + exit 2 +fi + +case "$LIMIT" in +*[!0-9]* | "") + echo "hbit-wait-for-node.sh: [seconds_to_wait] must be a whole number (got '$LIMIT')." >&2 + exit 2 + ;; +esac + +# Strip a trailing slash so the URL below never becomes a double slash. +NODE=$(printf '%s' "$NODE" | sed 's:/*$::') + +if ! command -v curl >/dev/null 2>&1; then + echo "hbit-wait-for-node.sh: curl is not installed, so this cannot tell whether" >&2 + echo "the node is up, and the pool must not be started blind." >&2 + echo "What to do: apt-get install -y curl (or dnf install -y curl), then" >&2 + echo " systemctl restart hbit-pool" >&2 + exit 2 +fi + +WAITED=0 +STEP=5 +while :; do + # -f makes a non-2xx answer a failure, --max-time keeps one hung request from + # becoming a hung service. The body is discarded except for the readiness + # check: /query/latest answers {"height":N,"diamond":N} when the node is up. + if curl -fsS --max-time 5 "$NODE/query/latest" 2>/dev/null | grep -q '"height"'; then + echo "node at $NODE is answering after ${WAITED}s; starting the pool" + exit 0 + fi + if [ "$LIMIT" -ne 0 ] && [ "$WAITED" -ge "$LIMIT" ]; then + echo "no Hacash fullnode answered at $NODE after ${WAITED}s." >&2 + echo "The pool was NOT started, so nothing was mined and nothing was paid." >&2 + echo "What to do: systemctl status hacash-node, and journalctl -u hacash-node -n 50" >&2 + exit 1 + fi + # One line a minute, not one every 5 seconds: this can legitimately run for + # hours on a first sync, and a journal full of dots helps nobody. + if [ $((WAITED % 60)) -eq 0 ]; then + echo "waiting for the Hacash fullnode at $NODE (${WAITED}s so far)" + fi + sleep "$STEP" + WAITED=$((WAITED + STEP)) +done diff --git a/deploy/node/hacash.config.ini b/deploy/node/hacash.config.ini new file mode 100644 index 00000000..ba171db6 --- /dev/null +++ b/deploy/node/hacash.config.ini @@ -0,0 +1,51 @@ +; Full node for the HBIT pool, running in a container. +; +; Mining is OFF here. This node's job is to hold the chain and answer the pool's +; template and submit requests. The mining happens on other people's hardware. +; +; The RPC below binds 0.0.0.0, which is correct INSIDE a container and would be +; dangerous on a bare host. The compose file publishes no port for this service, +; so the RPC is reachable only from the pool container on the private network. +; If you ever run this node outside Docker, change bind to 127.0.0.1 and put the +; pool on the same machine, or you are handing the miner API to the internet. + +[node] +listen = 3337 +boots = 54.193.49.59:3337, 182.92.163.225:3337, 54.219.80.127:3337 +not_find_nodes = false +fast_sync = false +; NOT true. Measured 2026-07-27: a chain synced with fast_sync = true +; stops dead at a block whose state it never wrote, with +; [Block Sync Warning] insert N failed: diamond status HTAKES not found +; and nothing retries, so the node sits there forever looking healthy. +; Turning the flag off afterwards does not repair it: the state was never +; written. A clean sync with it OFF reached the tip in seven minutes with +; no errors, which is the only reason this is not still true. + +; No [mint] section. chain_id defaults to 0, which is mainnet. + +[server] +enable = true +listen = 8080 +bind = 0.0.0.0 + +[miner] +; ============================================================================ +; YOU MUST FILL IN `reward` BELOW BEFORE THE FIRST START. Until you do, the node +; REFUSES to start and the pool cannot serve work. That refusal is deliberate: +; an earlier version of this project shipped a config with somebody else's +; address already in it, and anyone who ran it mined into a stranger's wallet. +; There is no default and there will never be one. +; ============================================================================ +; +; `enable = true` is what turns on /query/miner/pending and +; /submit/miner/success, which is how the pool gets block templates and submits +; the blocks its miners find. It does NOT make this node mine by itself. +; +; The address below is almost vestigial in a pool setup: the pool pays miners +; from its OWN wallet, so this one only ever receives anything if you point a +; miner straight at this node instead of at the pool. Use an address you +; control anyway, because "almost never" is not "never". +enable = true +reward = +message = hbit-pool-node diff --git a/deploy/systemd/hacash-node.service b/deploy/systemd/hacash-node.service new file mode 100644 index 00000000..99e5c957 --- /dev/null +++ b/deploy/systemd/hacash-node.service @@ -0,0 +1,79 @@ +# Hacash fullnode, as the always-on half of an HBIT pool box. +# +# Install to /etc/systemd/system/hacash-node.service (deploy/install.sh does it). +# Every path here is absolute on purpose: a service manager gives a process no +# terminal, no login shell and no working directory of its own. +# +# Anything you change here needs `systemctl daemon-reload` afterwards. +# Every directive is checked in docs/POOL-OPERATOR.md, section "The unit files, +# line by line". Do not paste a hardening line you cannot explain to yourself. + +[Unit] +Description=Hacash fullnode (the chain the HBIT pool mines on) +Documentation=file:/opt/hbit/docs/POOL-OPERATOR.md +# network-online.target means "an address is configured", which is what the P2P +# code needs before it dials boot nodes. +Wants=network-online.target +After=network-online.target +# No start rate limiting. The alternative is a node that gives up after five +# quick failures and then stays down silently, which on a money box is worse +# than one that retries every 15 seconds and prints the reason every time. +StartLimitIntervalSec=0 + +[Service] +Type=simple +User=hbit +Group=hbit +WorkingDirectory=/var/lib/hbit/node +# The config path is the ONLY argument this binary accepts, and it is worth +# passing: with no argument the node looks for hacash.config.ini NEXT TO THE +# BINARY (not in the working directory), and /opt is read-only for this service. +ExecStart=/opt/hbit/bin/hacash /etc/hbit/hacash.config.ini +Restart=always +RestartSec=15s +# The node installs its clean-shutdown handler on SIGINT (Ctrl-C) only. Sending +# the systemd default SIGTERM would kill it where it stands, mid database write. +# With this line `systemctl stop` runs the same path Ctrl-C does, and the journal +# ends with "[Exit] Hacash fullnode closed." +KillSignal=SIGINT +# Closing the chain database can take a while on a large chain. After this, +# systemd escalates to SIGKILL. +TimeoutStopSec=180 +SyslogIdentifier=hacash-node +# Files this service creates are owner-only. Nothing here is secret today, but a +# chain database is not something other local accounts need to read. +UMask=0077 + +# ---- hardening: see POOL-OPERATOR.md for one line on each ---- +NoNewPrivileges=yes +PrivateTmp=yes +ProtectSystem=strict +ProtectHome=yes +# The ONLY directory this service may write. The pool wallet lives in a sibling +# directory, so the node cannot touch it even though both run as the same user. +ReadWritePaths=/var/lib/hbit/node +# The node has no business reading the pool's passphrase. This makes the path +# not merely unwritable but absent inside this service's mount namespace. +InaccessiblePaths=/etc/hbit/wallet-passphrase +PrivateDevices=yes +ProtectKernelTunables=yes +ProtectKernelModules=yes +ProtectControlGroups=yes +RestrictSUIDSGID=yes +RestrictRealtime=yes +# AF_NETLINK is in the list on purpose: glibc's name resolution uses it to pick a +# source address, so leaving it out breaks DNS in ways that look like a network +# outage. +RestrictAddressFamilies=AF_UNIX AF_INET AF_INET6 AF_NETLINK +LockPersonality=yes +RemoveIPC=yes +# The ports here are all above 1024, so no capability is needed to bind them. +CapabilityBoundingSet= +# Optional, and NOT enabled by default because a wrong filter kills the node with +# SIGSYS and no explanation. If you want it, add it, restart, and watch a full +# sync before you trust it: +# SystemCallFilter=@system-service +# SystemCallErrorNumber=EPERM + +[Install] +WantedBy=multi-user.target diff --git a/deploy/systemd/hbit-pool.service b/deploy/systemd/hbit-pool.service new file mode 100644 index 00000000..02707ee2 --- /dev/null +++ b/deploy/systemd/hbit-pool.service @@ -0,0 +1,87 @@ +# HBIT mining pool: serves work to other people's miners and pays them. +# +# Install to /etc/systemd/system/hbit-pool.service (deploy/install.sh does it). +# +# THIS SERVICE HOLDS A WALLET THAT OWES REAL MONEY TO REAL PEOPLE. Read the two +# rules before you edit it: +# 1. The passphrase is NOT in this file. It is in a 0400 file named by +# HBIT_WALLET_PASSWORD_FILE below. Unit files are commonly world readable +# and `systemctl show hbit-pool` prints every Environment= value. +# 2. The wallet path in ExecStart is the pool's identity. Change it and the +# pool starts empty, with a new wallet, while the money and the record of +# who is owed what sit in files nothing reads any more. +# +# Anything you change here needs `systemctl daemon-reload` afterwards. + +[Unit] +Description=HBIT mining pool (serves work, counts shares, pays miners) +Documentation=file:/opt/hbit/docs/POOL-OPERATOR.md +# Wants, not Requires. Requires would stop the pool whenever the node unit fails, +# and a unit stopped that way does NOT come back when its dependency returns. The +# pool is meant to outlive a node hiccup, and ExecStartPre below is what actually +# waits for the node to be ready. +Wants=network-online.target hacash-node.service +After=network-online.target hacash-node.service +# No start rate limiting: a pool that gives up stays down until a human notices, +# and nobody is logged in. Retrying every 30 seconds prints the refusal in the +# journal each time and heals itself the moment the cause is fixed. +StartLimitIntervalSec=0 + +[Service] +Type=simple +User=hbit +Group=hbit +WorkingDirectory=/var/lib/hbit/pool +# A PATH, not a secret. This is safe in a unit file and safe in `systemctl show`. +# It also means a pool started before the passphrase file exists REFUSES to start +# rather than quietly creating a plaintext wallet. +Environment=HBIT_WALLET_PASSWORD_FILE=/etc/hbit/wallet-passphrase +# Ordering is not readiness. After= only means the node was started first, not +# that it answers yet, and hbit-pool-server exits rather than serving work +# against a node that is not there. This waits for the node's API, up to an hour. +ExecStartPre=/opt/hbit/bin/hbit-wait-for-node.sh http://127.0.0.1:8080 3600 +# node | wallet file | listen | share_bits | chain +# 0.0.0.0:9777 is the public face of the pool and the ONLY port that belongs on +# the internet. See the firewall section of POOL-OPERATOR.md. +ExecStart=/opt/hbit/bin/hbit-pool-server http://127.0.0.1:8080 /var/lib/hbit/pool/pool-wallet.key 0.0.0.0:9777 24 mainnet +Restart=always +RestartSec=30s +# Must be larger than the ExecStartPre bound above, or systemd would kill the +# wait before it gives up on its own. +TimeoutStartSec=3900 +TimeoutStopSec=30 +SyslogIdentifier=hbit-pool +# The wallet key file is forced to 0600 by the pool itself. This covers the +# accounting file next to it, which records what every miner is owed. +UMask=0077 + +# ---- hardening: see POOL-OPERATOR.md for one line on each ---- +NoNewPrivileges=yes +PrivateTmp=yes +ProtectSystem=strict +ProtectHome=yes +# The ONLY directory this service may write: the wallet, the accounting file and +# the settle lock. Not the chain database, which is the node's. +ReadWritePaths=/var/lib/hbit/pool +PrivateDevices=yes +ProtectKernelTunables=yes +ProtectKernelModules=yes +ProtectControlGroups=yes +RestrictSUIDSGID=yes +RestrictRealtime=yes +# AF_NETLINK is in the list on purpose: glibc's name resolution uses it to pick a +# source address, so leaving it out breaks DNS in ways that look like a network +# outage. +RestrictAddressFamilies=AF_UNIX AF_INET AF_INET6 AF_NETLINK +LockPersonality=yes +RemoveIPC=yes +# Port 9777 is above 1024, so no capability is needed to bind it. +CapabilityBoundingSet= +# Optional, and NOT enabled by default because a wrong filter kills the pool with +# SIGSYS and no explanation. If you want it, add it, restart, and watch a full +# settlement cycle before you trust it: +# SystemCallFilter=@system-service +# SystemCallErrorNumber=EPERM + +[Install] +WantedBy=multi-user.target diff --git a/docs/COMMUNITY-POOL-DESIGN.md b/docs/COMMUNITY-POOL-DESIGN.md new file mode 100644 index 00000000..61fdc91c --- /dev/null +++ b/docs/COMMUNITY-POOL-DESIGN.md @@ -0,0 +1,217 @@ +# Hacash Community Pool — Design + +A trust-minimized mining pool for newcomers, designed within what the Hacash +mainnet actually allows today (verified against the node source, 2026-07). +Goal: small GPUs (RTX 3050, RX 9060 XT) get **smooth, frequent, fair** payouts — +without the operator taking meaningful custody of anyone's funds. + +This document is the plan. It is intentionally phased so we ship value early and +add trust-minimization on top, rather than building everything before anything +works. + +--- + +## 1. Philosophy + +- **Newcomer-first.** A user picks the pool in the panel, pastes their HAC + address, presses Start. No CLI, no config files. +- **Honest.** We never call something a "payout pool" until it pays fairly, and + we always disclose the custody model in plain language in the UI. +- **Trust-minimized, not custodial-forever.** Custody is bounded, guarded, and + ultimately escapable by miners (see §6). We never hold more than a short + settlement window's worth, and never behind a single key. +- **No consensus fork.** Everything here runs on the node **as-is**. No changes + to Hacash consensus; the pool is off-node software plus normal transactions. + +--- + +## 2. The constraints we must design within (verified facts) + +These are the "physics". Every design choice below follows from them. + +| Fact | Consequence | Source | +|------|-------------|--------| +| Coinbase is **single-output**: one PRIVAKEY address, `reward == block_reward(height)` | A block reward cannot be split on-chain among many miners | `mint/src/check/coinbase.rs:12-18,114-142` | +| Consensus does **not** bind the coinbase to node config — `/submit/block` accepts any valid block with any PRIVAKEY coinbase | The pool can assemble blocks off-node and choose the coinbase address | `mint/src/api/submit_block.rs`; `mint/src/check/coinbase.rs:114-142` | +| Stock template API (`/query/miner/pending`) hardwires coinbase to `[miner] reward`; worker submits only 2 nonces | To choose coinbase we must build templates ourselves, not ask the node | `mint/src/check/block_build.rs:27-31`; `mint/src/api/miner_success.rs` | +| **No "share" concept** anywhere — PoW is validated only against the full network target | Share accounting is 100% off-node (pool ↔ worker) | `mint/src/api/miner_success.rs:50`; `mint/src/check/block_accept.rs:28` | +| A normal transfer can be **any fractional amount**, and one tx carries up to **200 actions** (`TX_ACTIONS_MAX=200`) | The pool can pay ~200 miners fractional amounts in one cheap tx | `basis/src/component/action.rs` (TX_ACTIONS_MAX); `protocol/src/action/hacash.rs:4,55` (HacToTrs 1, HacFromToTrs 14) | +| "Istanbul" (mainnet activation at height **765432**, already live) unlocked: **type3 multisig** (≤200 signers), **VM contracts** (40/41/44), **P2SH scriptmh** (P2SHScriptProve 46) + VM native hashes + `ViewCheckSign` + `HeightScope`/`BalanceFloor` guards | We have a real toolkit to harden custody | `protocol/src/upgrade.rs:8,33-42,88-141`; `vm/src/action/p2sh.rs:86`; `vm/src/native/hash.rs` | +| Payment channels ship **cooperative open/close only**; trustless unilateral exit is modeled but **unregistered** | True off-chain streaming needs a node change — out of scope for v1/v2 | `mint/src/action/mod.rs:30-36`; `field/src/component/channel.rs` | +| Economics: block reward **8 HAC**, block time **5 min**, network **~26 GH/s** | A 10-GPU pool finds ~1 block / 3 h; a 3050 alone would wait ~11 days per block | `mint/src/genesis/reward.rs`; `mint/src/config.rs:9`; explorer.hacash.org | + +**The core insight:** the "minimum payout = one whole block" barrier applies only +to the **coinbase**. If the pool receives whole blocks and **redistributes via +normal transfers**, it can pay each miner their exact fractional share, often and +cheaply. That is the whole design. The cost is custody between settlements, which +§6 bounds and hardens. + +--- + +## 3. Architecture + +``` + miners (poworker, modified) pool operator (e.g. home M6 + fullnode) + ┌───────────────────────┐ work + shares ┌──────────────────────────────────┐ + │ GPU/CPU, own wallet │ ───────────────▶ │ hac-pool daemon │ + │ mines pool templates │ ◀─────────────── │ • share validator (share target)│ + │ submits SHARES │ share target │ • share accounting (PPLNS) │ + └───────────────────────┘ │ • block assembler (own coinbase)│ + │ • settlement engine (batched) │ + full-solution ─────────────────────────────▶│ • treasury (multisig / P2SH-HTLC)│ + └───────────────┬──────────────────┘ + │ /submit/block, batched transfers + ▼ + Hacash fullnode (unchanged) +``` + +Components (all off-node): + +1. **Share validator.** Advertises a pool-chosen **share target** below the + network target; validates each submitted share by re-running + `x16rs::block_hash` over the reconstructed 89-byte block header and comparing + to the share target (`hash_bigger_than`). Reusable primitives already exist. +2. **Share accounting.** PPLNS-style: keep a sliding window of the last *N* + shares; each miner's credit = their share count in the window. Transparent + (published log). +3. **Block assembler.** Re-implements the ~85 lines of `impl_packing_next_block` + off-node: pulls `prevhash/height/difficulty/timestamp` from `/query/block/intro` + + `/query/latest`, calls `create_coinbase_tx(height, msg, POOL_ADDRESS)` + (already address-parameterized), builds `BlockV1`, computes the merkle root, + exposes the header nonce slot. Coinbase pays the **pool treasury** (see §6). +4. **Settlement engine.** On a cadence (e.g. every *S* blocks), computes each + miner's owed HAC from the PPLNS window and pays up to 200 miners in one + batched transaction (`HacToTrs`), directly to their own wallets. +5. **Treasury.** Where block rewards land before settlement. Hardened per §6. + +--- + +## 4. Share accounting model (PPLNS) + +- Pool sets `share_target = network_target × D` where `D` (e.g. 1/1024) makes + shares common enough for smooth accounting but not spammy. +- Each valid share credits the submitting miner 1 unit in a rolling window of the + last *N* shares (window ≈ a few multiples of "shares per found block"). +- When the pool finds a **full-network solution** (a share that also beats the + real target), it submits the block via `/submit/block`; the 8 HAC lands in the + treasury. +- A miner's entitlement over any period = (their shares in window) / (total shares + in window) × (HAC the pool earned in that period). This is **PPLNS**: fair, + hop-resistant, and standard. +- Everything is published (share log + per-miner running balance + settlement + tx hashes) so anyone can verify payouts match work. + +--- + +## 5. Settlement model + +- **Batched fractional transfers.** Every *S* blocks (tunable; e.g. 12 blocks ≈ + 1 hour), pay every miner whose accrued balance ≥ `min_payout` (dust floor, + e.g. 0.01 HAC) in **one** `HacToTrs` tx carrying up to 200 outputs. +- **Sub-block granularity achieved.** A 3050 (~8 MH/s) accrues ~its fair fraction + of every 8 HAC the pool earns, and gets it every settlement window — hours, not + ~11 days. This is the entire point. +- **Fees.** Hacash fees are negligible today; one batched tx per window per ~200 + miners is cheap. Fee is paid by the treasury (a tiny pool fee %, disclosed). +- **Carry-over.** Balances below `min_payout` roll to the next window. + +--- + +## 6. Trust / custody — bounded, guarded, escapable + +This is a **custodial** design between settlements (the pool holds rewards before +redistributing) — there is no fully-trustless smooth option on Hacash today. We +minimize the trust in three stacked layers, shipped in order: + +- **Layer A — Bound it (Phase 1).** Settle frequently (small *S*). The treasury + never holds more than ~*S* blocks' worth. Publish everything. Trust = "operator + won't run off with < one hour of pooled reward, in public." +- **Layer B — Guard it (Phase 2).** Make the treasury a **type3 multisig** + (`ReqSignList`, ≤200 signers). Funds move only with M-of-N community signatures, + so no single operator key can move pooled funds. + Evidence: `protocol/src/transaction/type3.rs`, `protocol/src/action/reqsign.rs`. +- **Layer C — Make it escapable (Phase 3).** Escrow each miner's accrued balance + in a **P2SH lockbox** (`P2SHScriptProve` 46) that the miner can **self-claim** + (hashlock / their key) with a **height-locked refund** — an HTLC-shaped escrow. + If the operator disappears, miners claim their owed balance themselves. This is + the closest thing to trustless the chain offers without a node change. + Evidence: `vm/src/action/p2sh.rs:86`; `vm/src/native/hash.rs`; `HeightScope` + `protocol/src/action/chain.rs:33`; `ViewCheckSign` `vm/src/action/envfunc.rs:82`. + +What we explicitly do **not** do: +- No multi-output/split coinbase — needs a consensus change (org-level). +- No payment-channel streaming — needs registering the modeled channel challenge + action (a node/consensus change). +- No opaque custody. If we can't disclose it and bound it, we don't ship it. + +--- + +## 7. Worker changes (poworker — we own it) + +Today `poworker` talks only to the node's fixed-template API and submits full +solutions (`app/src/poworker.rs`). For the pool it must, additionally: + +1. Pull the pool's template (coinbase = pool treasury) instead of the node's. +2. Mine against the pool's **share target** and submit **shares** (partial proofs) + to the pool over a small pool↔worker protocol (not the node's 2-nonce submit). +3. Report its **payout address** to the pool once at connect. + +This is client software we control; no node change. The GUI adds a "Pool" mode +that already exists — it just points at the pool endpoint (see the pool directory +work in `miner-panel/src/connect.rs`). + +--- + +## 8. Phased roadmap + +| Phase | Deliverable | Custody | Effort | +|-------|-------------|---------|--------| +| **P0** | Honest work-relay / shared node (done) | none | shipped | +| **P1** | Batched PPLNS pool: share validator + accounting + block assembler + batched settlement + modified worker. Frequent, fair, fractional payouts. Layer A trust. | bounded (~1 window) | the real build | +| **P2** | Multisig treasury (Layer B) | bounded + no single key | small, on top of P1 | +| **P3** | P2SH-HTLC per-miner escrow (Layer C) | escapable by miners | larger (lockbox bytecode + audit) | + +Ship P1, run it on the M6 with a few friends, prove demand, then P2/P3. + +--- + +## 9. Parameters (initial guesses — tune on testnet first) + +- Share difficulty factor `D`: start 1/1024, adjust so a mid-GPU emits a few + shares/minute. +- PPLNS window `N`: ≈ 3× (expected shares per found block). +- Settlement cadence `S`: 12 blocks (~1 h) for P1; shorten if treasury feels big. +- `min_payout` dust floor: 0.01 HAC. Pool fee: small, disclosed (e.g. 1%). + +--- + +## 10. Risks & things to verify at implementation + +- **Off-node assembly must be byte-exact.** Serialization, merkle prelude, + `x16rs::block_hash`, and next-difficulty (ASERT) must match the node or blocks + are rejected. Mitigation: the pool is a Rust process **linking the workspace + crates** (`mint`/`protocol`/`x16rs`/`basis`) — reuse, don't reimplement. Verify + exact struct field names/serialization against `protocol/src/block/` and + `mint/src/check/block_build.rs` before coding. +- **Testnet first.** Every gate is bypassed on non-zero `chain_id` + (`protocol/src/upgrade.rs:11`) — build and test the whole flow on testnet with + fake money before a single HAC of real reward is at stake. +- **Fees / mempool.** No mempool-dump RPC exists, so off-node templates are + coinbase-only (no fee txs). Fine while fees ≈ 0; revisit if that changes. +- **Reachability.** A home pool needs a public reachable IP (NAT/CGNAT blocks it) + — orthogonal to this design but required for others to join (see the panel's + reachability caveat). +- **Reorgs / stale.** Standard pool concerns: handle share timing across block + changes, don't credit shares for a stale height. + +--- + +## 11. Bottom line + +On Hacash today the best realizable pool for newcomers is a **transparent PPLNS +pool with frequent batched fractional settlement**, hardened with multisig and +optional P2SH-HTLC escrow. It gives small GPUs smooth, fair, frequent payouts — +the thing whole-block rotation cannot — while keeping custody bounded, guarded, +and (in P3) escapable. Fully trustless smooth payouts would require a consensus +change (multi-output coinbase) or activating trustless channels; both are +node/org-level and out of scope for a community build. diff --git a/docs/COMMUNITY-REQUIREMENTS.md b/docs/COMMUNITY-REQUIREMENTS.md new file mode 100644 index 00000000..5de59623 --- /dev/null +++ b/docs/COMMUNITY-REQUIREMENTS.md @@ -0,0 +1,57 @@ +# Community miner requirements — status + +Target list from community / jojoin discussion: + +1. Stratum and free IP pool +2. Including new versions of CUDA +3. Integration with open source and official libraries +4. Diaworker + Poworker +5. Anyone can broadcast a public pool of content +6. JoJoin rebuilds miner when fullnode updates + +## Status matrix + +| # | Requirement | Status | How | +|---|-------------|--------|-----| +| 1 | Stratum + free IP pool | **Implemented (v1)** | Binary `hac-pool`: HTTP free-IP bind + Stratum TCP | +| 2 | New CUDA versions | **Implemented (validated)** | CUDA 12/13, sm_75/86/89; T4 Colab PASS | +| 3 | Official open-source libs | **Yes (community fork)** | Based on `hacash/fullnodedev`; integration request open | +| 4 | Diaworker + Poworker | **Yes** | Both in packages and builds | +| 5 | Public pool broadcast | **Implemented (v1)** | Anyone runs `hac-pool` on 0.0.0.0; workers connect | +| 6 | JoJoin rebuild | **Process ready** | See [JOJOIN-REBUILD.md](JOJOIN-REBUILD.md); needs org ownership | + +## Quick start public pool (1 + 5) + +```bash +# 1) fullnode with miner API (loopback OK) +# 2) free-IP pool in front of it +cargo build --release -p miner-pool +./target/release/hac-pool \ + --upstream 127.0.0.1:8080 \ + --http-bind 0.0.0.0:3333 \ + --stratum-bind 0.0.0.0:3334 + +# 3) workers (existing poworker) point at the pool IP +# poworker.config.ini: +# connect = YOUR_PUBLIC_IP:3333 +``` + +Optional: `--pool-token SECRET` then workers send `api_token` / Stratum password. + +## CUDA (2) + +- Docs: [MINING-NVIDIA-CUDA.md](MINING-NVIDIA-CUDA.md) +- Colab smoke: `scripts/mining-nvidia/colab_cuda_smoke.sh` +- Evidence: Tesla T4, `cargo test -p x16rs-cuda --features cuda` → 4 passed + +## Official integration (3 + 6) + +- Issues: https://github.com/hacash/fullnodedev/issues/9 +- Rebuild recipe: [JOJOIN-REBUILD.md](JOJOIN-REBUILD.md) + +## Honest limits (v1) + +- Pool is a **work proxy** (official miner RPC + minimal Stratum). +- No share accounting / PPS / wallet payouts yet (can be added later). +- Stratum is **Hacash-oriented** (job carries `block_intro` + height); not a drop-in for every third-party closed miner. +- Existing **poworker** uses HTTP pool port (not Stratum) for zero worker code change. diff --git a/docs/JOJOIN-REBUILD.md b/docs/JOJOIN-REBUILD.md new file mode 100644 index 00000000..d5a508a4 --- /dev/null +++ b/docs/JOJOIN-REBUILD.md @@ -0,0 +1,43 @@ +# JoJoin / official rebuild recipe + +Requirement 6: when a new fullnode version ships, the miner should be rebuildable by JoJoin (or any official maintainer) for compatibility. + +## Reproducible builds + +```bash +git clone https://github.com/hacash/fullnodedev.git # or Moskyera fork until merged +cd fullnodedev +git checkout + +# lockfile required +cargo build --locked --release --features ocl \ + --bin poworker --bin list_opencl --bin diagnose_opencl +cargo build --locked --release --bin hacash --bin diaworker +cargo build --locked --release -p miner-panel +cargo build --locked --release -p miner-pool # public pool (hac-pool) + +# optional NVIDIA CUDA worker +cargo build --locked --release --features cuda --bin poworker +``` + +## Version alignment + +| Component | Must match | +|-----------|------------| +| `protocol` / Istanbul gates | same commit as fullnode | +| `poworker` / `diaworker` | same workspace revision as `hacash` fullnode | +| `hac-pool` | same miner RPC paths as mint API | + +## CI + +`.github/workflows/release.yml` builds OpenCL workers + panel with `--locked`. +CUDA package builds need a GPU runner or offline kernel artifact (see mining-nvidia scripts). + +## Suggested official process + +1. Tag fullnode release `vX.Y.Z`. +2. Rebuild miner bins from the **same tag**. +3. Attach miner artifacts to the same GitHub Release (or linked release notes). +4. List community GPU tools on https://hacash.org/miner when accepted. + +Contact: integration request https://github.com/hacash/fullnodedev/issues/9 diff --git a/docs/MINING-NVIDIA-CUDA.md b/docs/MINING-NVIDIA-CUDA.md index 9d72558b..ddf831da 100644 --- a/docs/MINING-NVIDIA-CUDA.md +++ b/docs/MINING-NVIDIA-CUDA.md @@ -2,66 +2,51 @@ Native CUDA block miner for Hacash, integrated with the existing `poworker` + fullnode RPC stack (same protocol as OpenCL/CPU miners). -> **Work in progress — NOT production ready.** Kernels compile on Windows (CUDA 12.x/13.x + MSVC Build Tools), but **GPU runtime has not been validated** on this dev machine (no local NVIDIA GPU). Genesis test and end-to-end mining require an **RTX tester** — see [HANDOFF-RTX.md](../scripts/mining-nvidia/HANDOFF-RTX.md). +## Status (requirement 2) -**Status:** Kernels compile on Windows (CUDA 12.x/13.x + MSVC Build Tools). GPU runtime validation requires an NVIDIA RTX machine. +| Item | Status | +|------|--------| +| Kernels + host (`x16rs-cuda`) | Yes | +| CUDA Toolkit | **12.x / 13.x** | +| GPU arch fatbin | SASS **sm_75** (T4/20xx), **sm_86** (30xx), **sm_89** (40xx) **+ PTX `compute_89`** | +| Newer GPUs (sm_90 Hopper, sm_120 Blackwell/50xx) | **JIT via embedded PTX** (driver compiles at launch; no source edit) | +| Runtime validation | **PASS on NVIDIA Tesla T4** (Google Colab, 2026-07-22); sm_86/89 via fatbin, sm_90+ via PTX (not yet runtime-verified) | +| Tests | `cargo test -p x16rs-cuda --features cuda` → **4 passed** | -**RTX handoff:** [scripts/mining-nvidia/HANDOFF-RTX.md](../scripts/mining-nvidia/HANDOFF-RTX.md) (Greek + English checklist for testers). +Colab free-tier smoke: [scripts/mining-nvidia/COLAB-T4.md](../scripts/mining-nvidia/COLAB-T4.md). ## Requirements -- NVIDIA GPU (RTX 20xx / 30xx / 40xx — sm_75 / sm_86 / sm_89) +- NVIDIA GPU (T4 / RTX 20xx / 30xx / 40xx) - [CUDA Toolkit](https://developer.nvidia.com/cuda-downloads) 12.x or 13.x -- Windows: [Visual Studio Build Tools](https://visualstudio.microsoft.com/downloads/) with **Desktop development with C++** (`cl.exe` for nvcc) +- Windows: VS Build Tools with C++ (`cl.exe` for nvcc) **or** Linux + nvcc - Rust toolchain (edition 2024) -- Fullnode with `[miner] enable = true` +- Fullnode with miner API, **or** `hac-pool` in front of it -## Build (Windows) +## Build + +### Windows ```bat scripts\mining-nvidia\BUILD-CUDA-MINER.bat ``` -The script auto-detects `CUDA_PATH`, runs `vcvars64.bat`, and builds `target\release\poworker.exe`. - -Manual build: +### Linux / Colab -```bat -call "C:\Program Files (x86)\Microsoft Visual Studio\18\BuildTools\VC\Auxiliary\Build\vcvars64.bat" -set CUDA_PATH=C:\Program Files\NVIDIA GPU Computing Toolkit\CUDA\v13.3 +```bash +export CUDA_PATH=/usr/local/cuda +cargo test -p x16rs-cuda --features cuda cargo build --release --bin poworker --features cuda ``` -Successful kernel build prints: `Using CUDA Toolkit at ...` (no `build without GPU kernels` warning). - -## RTX tester handoff checklist - -Run on a machine **with NVIDIA GPU**: - -```bat -scripts\mining-nvidia\BUILD-CUDA-MINER.bat -scripts\mining-nvidia\TEST-CUDA-GPU.bat -scripts\mining-nvidia\INSTALL-CUDA-CONFIG.bat -scripts\mining-nvidia\START-CUDA-MINING.bat -``` - -1. **Genesis GPU test** must pass: - - Expected hash: `000000077790ba2fcdeaef4a4299d9b667135bac577ce204dee8388f1b97f7e6` -2. **poworker startup** must show: - - `[CUDA] Device #0: ...` - - `[CUDA] Initialized device #0 work_groups=...` -3. **Mining** against a fullnode with pending work — submit a block or report hashrate logs. - -Report back: GPU model, CUDA version, genesis test result, and any `nvcc`/runtime errors. - -Example config: `scripts/mining-nvidia/poworker.cuda.ini.example` +Successful kernel build logs: `Using CUDA Toolkit at ...` ## Configure (`poworker.config.ini`) ```ini [default] connect = 127.0.0.1:8080 -supervene = 4 +; or public pool: connect = POOL_IP:3333 [gpu] use_cuda = true @@ -69,30 +54,16 @@ use_opencl = false cuda_device = 0 work_groups = 131072 unit_size = 8 -cpu_assist = true ``` -CUDA takes priority over OpenCL when `use_cuda = true`. - ## Architecture | Layer | Path | -|-------|------| -| RPC / work loop | `app/src/poworker.rs` (unchanged protocol) | +|------|------| +| RPC / work loop | `app/src/poworker.rs` | | CUDA backend | `app/src/cuda_pow.rs` | | GPU kernels | `x16rs-cuda/cuda/block_miner.cu` | -| OpenCL reuse | `x16rs/opencl/*.cl` via `ocl_compat.cuh` | - -Kernels implement the same x16rs flow as OpenCL: SHA3-256 block intro → x16rs chain → nonce batch search. - -## Tests - -```bat -cargo test -p x16rs-cuda --features cuda -``` - -Genesis vector (`x16rs/tests/test.rs`) is cross-checked on GPU when CUDA is available. -## vs community CUDA miners +## vs third-party CUDA pool miners -This miner uses **official fullnode RPC** (`/query/miner/pending`, `/submit/miner/success`). Third-party CUDA pool miners use a different protocol and are not drop-in replacements. \ No newline at end of file +This miner uses **official fullnode / hac-pool miner RPC**. Closed Stratum-only miners are not drop-in; use `hac-pool` HTTP port with stock `poworker`, or Stratum port with a compatible client. diff --git a/docs/POOL-OPERATOR.md b/docs/POOL-OPERATOR.md new file mode 100644 index 00000000..ab01a4c4 --- /dev/null +++ b/docs/POOL-OPERATOR.md @@ -0,0 +1,501 @@ +# Running HBIT (`hbit-pool-server` + `hbit-pool-payout`) + +HBIT is the mining pool this project builds, and this is the operator runbook +for it: the program that serves work to other people's miners, keeps PPLNS share +accounting, submits found blocks and pays everybody out. It handles **real money +that is not yours**, so read the warnings below before you start it. Each one +changes something an operator can see, and not knowing about it is how people +lose coins or think the pool is broken. + +If you have not read `README-POOL.txt` yet, read that first. It is the short +version of why you would run a pool at all. This file is what you keep open +while you do. + +In a release download both programs are already built and sit next to the miner. +From a source checkout they live in the `hbit-pool` crate and are built with +`cargo build --release -p hbit-pool`. + +Design background: **[COMMUNITY-POOL-DESIGN.md](COMMUNITY-POOL-DESIGN.md)**. +The separate free-IP work relay (`hac-pool`) is **[PUBLIC-POOL.md](PUBLIC-POOL.md)**. + +| Program | What it does | +|---------|--------------| +| `hbit-pool-server` | Serves work, validates shares, submits blocks, settles on a timer | +| `hbit-pool-payout` | Manual settlement, run by hand when the server is stopped | + +``` +hbit-pool-server [settle_secs] +hbit-pool-payout [wallet_file] [reserve_units] [dust_units] [--commit] +``` + +**Both programs print all of that themselves.** `hbit-pool-server --help` and +`hbit-pool-payout --help` describe every argument, with a working example and +the two settings a miner needs; the `hbit-pool.example.ini` worksheet that ships +beside them is the same list with room to write your own answers in. You never +have to guess an argument from this file, and neither program ever asks you a +question: everything is an argument or an environment variable, so both run +unattended under a service manager. + +--- + +## 0. The first ten minutes + +**Before the first start.** Have your own Hacash fullnode running and synced. +The pool talks to its API, whose port is the `[server] listen` value in that +node's `hacash.config.ini`; the config this package ships uses **8080**, so the +node URL is normally `http://127.0.0.1:8080`. Set a wallet passphrase in the +same window you are about to start the pool in (section 1), because the wallet +is created on the first start and it is encrypted only if a passphrase is set +then. + +**The first start.** Bind to loopback while you look around, so nobody can mine +here yet: + +```powershell +$env:HBIT_WALLET_PASSWORD = "a long passphrase you have written down" +.\hbit-pool-server.exe http://127.0.0.1:8080 pool-wallet.key 127.0.0.1:9777 24 mainnet +``` + +```bash +export HBIT_WALLET_PASSWORD='a long passphrase you have written down' +./hbit-pool-server http://127.0.0.1:8080 pool-wallet.key 127.0.0.1:9777 24 mainnet +``` + +It refuses to start, with an explanation and a `What to do:` line, if the node +is not answering, if the chain argument does not match that node, if the listen +address is wrong or its port is taken, if `share_bits` or `settle_secs` is not a +number in range, or if another copy is already running on the same wallet. +Nothing is mined and nothing is paid when it refuses. + +**What a good start looks like.** Just before `listening on` it prints a +readback. Check every line of it: + +``` +---------------------------------------------------------------------- + HBIT pool is up. Read this back before you let anyone mine here. + pays FROM + key file pool-wallet.key, ENCRYPTED at rest + follows http://127.0.0.1:8080 (mainnet, at block ) + terms PPLNS over the last 4096 shares, no pool fee, minimum payout 0.1 HAC + block income payable 16 blocks after this pool finds it, settles every 5m + share 2^24 easier to find than a network block + miners set connect = + pool_worker = + loopback only: no other machine can mine here. Bind 0.0.0.0 when you are ready + check it http:///terms +---------------------------------------------------------------------- +``` + +- **pays FROM** is the wallet every payout comes out of. It is the address you + back up, and the one to check on a block explorer. +- **key file** says what is really on the disk. `PLAINTEXT on disk` means no + passphrase was set: fix that before real money arrives (section 1). +- **follows** must be your own node, and the height must be the real chain tip. +- **terms** is read out of the same constants the payout code uses, so it is + what your miners will actually get. It is the same thing `/terms` serves them. +- **miners set** is the line to paste to a miner. If the pool is bound to + `0.0.0.0` it cannot know your public address, so it says so instead of + inventing one. + +**On the very first start only**, a second block follows it: the pool created +its wallet. Back that file up before you go any further; section 1 is the whole +of why. + +**Then open it up.** Stop the pool, restart it with `0.0.0.0:9777` in place of +`127.0.0.1:9777`, and give a miner the `connect` and `pool_worker` lines above. +Check `http:///terms` and `http:///earnings?worker=` from another machine to confirm it is reachable. + +--- + +## 1. The wallet file can now be encrypted, and the passphrase is half the key + +The pool wallet key file (default `pool-wallet.key`) holds the private key that +controls **every coin the pool has taken in but not yet paid out**. It can now be +stored encrypted with Argon2id + AES-256-GCM. + +Set a passphrase in one of these two environment variables before starting +`hbit-pool-server` or `hbit-pool-payout`: + +| Variable | Meaning | +|----------|---------| +| `HBIT_WALLET_PASSWORD` | The passphrase itself | +| `HBIT_WALLET_PASSWORD_FILE` | Path to a file holding the passphrase (for services that cannot carry secrets in the environment) | + +`HBIT_WALLET_PASSWORD` wins if both are set. The passphrase must be at least 8 +characters; anything shorter is refused at startup rather than silently accepted. + +Windows PowerShell: + +```powershell +$env:HBIT_WALLET_PASSWORD = "a long passphrase you have written down" +.\hbit-pool-server.exe http://127.0.0.1:8080 pool-wallet.key 0.0.0.0:9777 24 mainnet +``` + +Linux: + +```bash +export HBIT_WALLET_PASSWORD='a long passphrase you have written down' +./hbit-pool-server http://127.0.0.1:8080 pool-wallet.key 0.0.0.0:9777 24 mainnet +``` + +A passphrase shorter than 8 characters, or a `HBIT_WALLET_PASSWORD_FILE` that +cannot be read, stops the pool before it touches a key, with a message saying +which one it was. It never falls back to writing the key in the clear because +the passphrase was faulty. + +What happens next: + +- **No wallet file yet:** a new wallet is generated and written **encrypted**. + The pool then prints a block you cannot miss, as the last thing before it + starts serving: the file, the address, and what losing either half costs. +- **An existing plaintext wallet file:** it is **migrated automatically** the + next time the wallet is loaded. The encrypted form is decrypted and compared + against the original key before it replaces the file, so a failed migration can + never cost you the wallet. The pool prints `[wallet] is now ENCRYPTED`. +- **No passphrase set:** the file stays plaintext and the pool prints a loud + warning every time it starts. This still works, it is just not protected. + +Either way the startup readback states which it is, every single start: +`key file pool-wallet.key, ENCRYPTED at rest` or `key file pool-wallet.key, +PLAINTEXT on disk, no passphrase set`. That line is read from the file itself, +not from the environment, so it says what is really on the disk. + +### Back up the passphrase ALONGSIDE the key file + +This is the part that loses money if you skip it. + +- **The key file alone is useless without the passphrase.** There is no reset, no + recovery question and no support address. If you back up the encrypted file and + forget the passphrase, every coin the pool holds is gone for good. +- **The passphrase alone is useless without the key file.** Both halves must + survive whatever kills the machine. Keep them together in the same safe place, + or keep both in two separate safe places. Do not put one on the mining rig and + the other nowhere. +- Test the pair before you trust it: with the pool stopped, restore the backed-up + file to a scratch directory, set the passphrase and start `hbit-pool-payout` in + its default dry-run mode. It prints the wallet address. If that address matches + your live pool address, the backup works. + +### Your OLD plaintext copies are still out there + +Encrypting the file today does **not** reach backwards. The plaintext private key +may still exist in: + +- ordinary file backups taken before the migration, +- Windows shadow copies / VSS snapshots and Linux filesystem or VM snapshots, +- cloud sync folders and their version history, +- old drives, images and machines you no longer use. + +Anything holding one of those copies can spend the pool's funds, passphrase or +not. Treat every pre-migration backup and snapshot as a live secret: destroy the +ones you do not need, and keep the ones you do need under the same protection you +would give cash. + +If you believe a plaintext copy leaked, the only real fix is a new wallet: stop +the pool, run `hbit-pool-payout --commit` to pay everyone out of the old wallet, +move the remainder to your own address, then start the pool with a fresh wallet +file and a fresh passphrase. + +--- + +## 2. `hbit-pool-payout` will not run while `hbit-pool-server` is running + +`hbit-pool-server` now takes an **exclusive OS lock on the wallet** for its whole +run, and `hbit-pool-payout` takes the same lock. So: + +- `hbit-pool-payout` started while the server is up **refuses to run and exits + non-zero**, printing `REFUSING to run: another hbit-pool-server or + hbit-pool-payout already holds `. +- `hbit-pool-server` started while `hbit-pool-payout` is mid-run refuses to start + the same way. + +**This is deliberate and it is protecting your money.** Both programs decide what +to pay from the wallet's *confirmed* balance, and a payout sitting in the mempool +does not reduce that balance. Run them at once and each one sees the full balance, +each one believes it is the only settler, and the same PPLNS window gets paid +**twice** out of the operator's own funds. The lock is what makes that impossible. + +The lock is held by the operating system, so a crash or a kill releases it +immediately. There is nothing to clean up by hand, and **deleting the +`.settle.lock` file does not release anything**: the lock belongs to +the running process, not to the file, so removing it would only let two payers +run at once. Both refusals say so themselves. + +### Correct procedure for a manual payout + +1. **Stop `hbit-pool-server`** and wait for the process to actually exit. +2. Run the tool in its **dry-run** default first and read the planned split: + ```bash + ./hbit-pool-payout http://127.0.0.1:9777 http://127.0.0.1:8080 mainnet pool-wallet.key + ``` + It pays nothing without `--commit`. +3. If the split looks right, run it again with `--commit`. +4. **Restart `hbit-pool-server`.** + +Set `HBIT_WALLET_PASSWORD` in that window too if the wallet is encrypted, and +run it **in the folder that holds the wallet file** (or pass the path as +argument 4). The tool never creates a wallet: if there is no key file where it +was told to look, it refuses and says so, rather than making a fresh empty one +and reporting that you have nothing to pay. + +If the node is not answering it refuses as well, instead of reading an empty +balance as a zero balance. + +While the server is stopped its `/stats` endpoint cannot answer, so +`hbit-pool-payout` reads the share window out of the accounting file the server +left next to the wallet (`.state.json`). Keep that file with the +wallet file; it also carries the shared pending-payout ledger that stops a +re-run, a crash or an overlapping cron job paying the same window twice. + +--- + +## 3. Payouts lag block discovery by about 16 blocks + +When the pool finds a block, that block's coinbase reward is **held back from +settlement until the chain has buried the block 16 blocks deep**. Only then does +it join the distributable balance. + +On mainnet a block is targeted at 5 minutes, so **income from a block you just +found becomes payable roughly 80 minutes later**. While it waits, the pool prints + +``` +[settle] holding back N unit(s) of block income that is not yet buried 16 deep; +nothing matured to pay this cycle +``` + +**Nothing is stuck and nothing is missing.** The reason for the delay: + +- A freshly found block can still be **orphaned** by a reorg. The node itself + treats the last 4 blocks as reorg-able. +- If the pool paid that block's reward out immediately and the block were then + orphaned, the income would vanish from the canonical chain while the payout + transaction that spent it stays perfectly valid. The miners keep the coins, the + chain never delivers the reward, and **the operator eats the whole subsidy out + of their own pocket** with no way to recover it. +- 16 confirmations puts a wide margin over the node's own reorg window, so this + can only happen after a reorg deeper than anything the network has ever seen. + +An orphan is detected and the held-back amount is simply dropped, never paid. +Confirmed blocks and orphans are both counted on `/stats`. + +Practical consequences to tell your miners about: + +- The first payout after the pool's very first block arrives roughly 80 minutes + after that block, not immediately. +- Steady state is unaffected: once the pool is finding blocks regularly, the + 16-block lag is a constant offset, not a growing backlog. +- Payouts below the dust floor (default 0.1 HAC) roll over to the next window + instead of being paid, which is a separate and expected reason a small miner + sees nothing on a given cycle. + +--- + +## 4. `hbit-pool-server` refuses to start on a bad configuration + +The server checks its own configuration before it serves a single piece of work. +Each check below exits with status 2 and an explanation ending in a `What to do:` +line, instead of running in a state that would quietly lose money. **When it +refuses, nothing has been mined and nothing has been paid.** On the checks that +run before the wallet is opened it does not even create a wallet file, so a +mistyped first attempt leaves nothing behind to tidy up. + +### Every argument is required, and none is guessed + +All five positional arguments must be present. A number that is not a number is +**refused, never replaced by the default**: an operator who mistyped `share_bits` +would otherwise mine for weeks on a share size they did not choose and were never +told about. Running the program with no arguments, or with `--help`, prints the +whole usage text. + +### `share_bits` must be between 18 and 40 + +`share_bits` (argument 4) says how many powers of two easier a share is than a +real network block. Outside `18..=40` the server prints +` must be between 18 and 40 (got N)` and exits. + +- **Below 18:** shares get so hard that the 4096-share PPLNS window covers a + meaningful slice of a block interval. A difficulty change landing inside a live + window then splits real payout money by share counts that stand for different + amounts of work. +- **Above 40:** shares get so easy that a whole GPU batch always beats one, so + credit tracks batch cadence rather than hashrate, and the share target + degenerates. + +**20 is the recommended value**, and the shipped `deploy/docker-compose.yml` +uses it. Measured on the live chain on 2026-07-27, a block costs 2^42, so 20 +leaves each share costing 2^22 hashes. That choice is about payout memory as much +as about size: PPLNS pays on the last 4096 shares, so EASIER shares make that +window cover LESS TIME. At `share_bits` 24 an ordinary card produces roughly 26 +shares a second, and a pool with ten miners turns its whole window over in about +fifteen seconds, so a miner that drops off for half a minute loses everything it +was owed. At 20 the same pool keeps about four minutes of history. + +### The live difficulty is checked too, and it is checked twice + +The range above is only about the number you typed. Once the pool has fetched a +template it checks the target it would really serve, and there are TWO ways that +can fail, because a ratio and a cost are different questions. + +- **The ratio.** `share_target = network_target * 2^share_bits` saturates at the + all-0xff ceiling, so on a very easy chain what workers get is not what was + asked for. If the achieved factor falls below 18 the pool refuses. +- **The cost.** A share must be worth at least 2^16 hashes. This is the bound + that matters and it is NOT implied by the first one: a chain whose target has + 22 leading zero bits, served with `share_bits` 24, reports an achieved factor + of 22, clears the ratio bound comfortably, and still hands out a share target + that every hash on earth beats. + +In both cases the reason is the same. When a share costs nothing, PPLNS credit +measures how fast a worker completes an HTTP round trip instead of how much it +hashed, so the fastest submitter takes the window from miners doing more work. +The pool will not distribute real money on that basis. + +The remedies are opposite, so read which one you got. Since +`leading_zero_bits(network) = achieved + cost`, lowering `share_bits` moves work +out of the ratio and into the cost, and the message names the highest value that +would work. Only when the chain cannot support any legal setting, which is where +a fresh testnet always sits, does the pool tell you that nothing here helps and +that the ceiling is the chain's rather than yours. + +### `settle_secs` must be between 30 and 86400 + +`settle_secs` (argument 6, default 300) is the automatic payout interval. Each +settlement is a signed transaction that carries a network fee, so paying out +every few seconds spends the reserve for nothing, and `0` would leave the +settlement thread spinning against the node with no pause at all. Leave it out +unless you have a reason. + +### The node has to be there, and it has to be yours + +Before anything else the pool asks the node for its current block. If nothing +answers it refuses, naming the URL it tried and where the port comes from (the +`[server] listen` value in the node's `hacash.config.ini`, 8080 in the config +this package ships). A node that is still syncing, or that would not hand over a +block template, is refused the same way. + +### The listen address has to be usable + +The pool binds its port **before** it opens the wallet, so the two commonest +first-run mistakes here cost nothing: a `` with no port in it, and a port +something else already holds. Both are refused with the correct form and the +likely cause. + +### The test routes require `worker=` + +The `/work` and `/share` test routes now demand a real, payable HAC address: + +``` +/work?worker= +/share?worker=&height=...&nonce=... +``` + +The old placeholder worker name `w1` (and any other non-address name) is +**rejected** with `set worker= so the pool can pay you`. The +standard `/submit/miner/success` route enforces the same rule via `pool_worker`. + +This is not pedantry. Share credit is keyed by payout address, and the PPLNS +window is a fixed 4096 shares shared by everybody. A share credited to a name the +pool cannot pay is work done for nothing that **also evicts a payable miner's +share** from the window, so small and intermittent miners drop out of the window +before a block is found. Any script or monitoring check still using `worker=w1` +must be updated. + +### The `chain` argument is required and is proved against the node + +`chain` (argument 5) has no default, because a pool running the wrong difficulty +rule mines work the node rejects forever without saying so. Accepted values: + +| Value | Use for | +|-------|---------| +| `mainnet` | Real Hacash mainnet (consensus-fixed 288 blocks / 300s) | +| `testnet` | A testnet running the documented 288 / 10s pair | +| `testnet::` | A testnet configured with any other pair | + +The third form exists because a testnet node reads `difficulty_adjust_blocks` and +`each_block_target_time` from its **own** `hacash.config.ini`, so the label alone +proves nothing. Spell out the pair your node actually uses. + +At startup the server **recomputes the difficulty of the node's own current tip** +and compares it with what the node stored. If they do not match it prints +`REFUSING to start: difficulty rule mismatch at the node's own tip ...` and exits. +An exact match is the only proof that the rule in force here is the one the node +validates with. If you see this error, fix the chain argument; do not work around +it, because every block the pool finds would otherwise be thrown away. + +`hbit-pool-payout` takes the same required `chain` argument in the same three +forms. + +--- + +## 5. Telling miners how to reach your pool + +The miner panel lists **HBIT pool** first in its pool picker, and it ships that +entry with **no address**, because no HBIT address is published in this +repository and an invented one is an address somebody would paste in and point +real hashrate at. You publish yours. + +A miner that is configuring `poworker` by hand needs exactly two settings, and +the pool prints them for you in its startup readback: + +``` +connect = +pool_worker = +``` + +`pool_worker` is where that miner gets paid, so it is theirs, not yours. If your +pool is bound to `0.0.0.0` it cannot know what address the outside world reaches +it on, so the readback says `` instead of +inventing one: that part is for you to fill in. + +For the panel, drop a `pools.json` next to `miner-panel.exe` on the miner's PC: + +```json +[ + {"name": "HBIT pool", "connect": "pool.example:9777"} +] +``` + +The name is matched case insensitively against the built-in entry, so this +replaces the HBIT entry in place rather than adding a second one; it keeps its +first position, and the panel reads the file when the miner presses Refresh next +to the picker, with no rebuild. Two things to know: + +- Overriding an entry replaces the whole entry, so the built-in note disappears + and the panel falls back to its generic pool hint. Set `"note"` yourself if + you want one. A note that promises a payout scheme, a fee level or a minimum + is refused and replaced by the panel: those are claims it cannot check. +- `"verified": true` is the panel's own statement that it reached the endpoint. + Leave it out. + +What the panel shows about your terms does not come from that file. It comes +from your running pool: the dashboard reads `/terms` and `/earnings` from +`hbit-pool-server` and shows the miner your real scheme, fee and minimum payout, +and what you have already paid that miner. That is the whole reason HBIT can be +listed honestly next to pools the panel has never spoken to. + +--- + +## Quick reference + +Every refusal below ends with its own `What to do:` line; this table is for +finding the one you are looking at. + +| Symptom | Cause | Fix | +|---------|-------|-----| +| `REFUSING to run: another hbit-pool-server or hbit-pool-payout already holds ...` | Both settlers running at once | Stop `hbit-pool-server`, run the tool, restart the server. Do not delete the `.settle.lock` file: it frees nothing | +| `REFUSING to start: no Hacash fullnode answered at ...` | Node not running, still starting, or wrong URL/port | Start and sync the node; use the `[server] listen` port from its `hacash.config.ini` (8080 here) | +| `REFUSING to start: the node ... would not give a block template` | Node is up but not ready to be mined on | Let it finish syncing, then start the pool again | +| `REFUSING to start: cannot listen on ...` | `` is not `:`, or the port is taken | Use `0.0.0.0:9777` or `127.0.0.1:9777`; if the form is right, something else holds that port | +| `wallet file ... is encrypted but no passphrase is configured` | Passphrase missing from the environment | Set `HBIT_WALLET_PASSWORD` or `HBIT_WALLET_PASSWORD_FILE` | +| `cannot decrypt wallet file ...` | Wrong passphrase, or a corrupted file | Use the backed-up passphrase; restore the file from backup | +| `the wallet passphrase in ... must be at least 8` | Passphrase too short to protect real money | Set a longer one, written down somewhere physical | +| ` must be between 18 and 40` | Out-of-range or mistyped argument 4 | Use 20 | +| `the network difficulty in force ... is too low to serve a fair share` | The chain is so easy the share target saturates, so the achieved ratio is under 18 | Point the pool at mainnet. Lowering `share_bits` cannot help here | +| `the network difficulty in force ... is too low to serve a share worth counting` | The ratio is fine but a share would cost under 2^16 hashes | Lower `share_bits` to the value the message names; if it says nothing helps, the chain itself is too easy | +| `[settle_secs] must be between 30 and 86400` | Out-of-range or mistyped argument 6 | Leave it out to get 300 | +| `set worker= so the pool can pay you` | Worker name is not a payable address | Pass the miner's real HAC address | +| `REFUSING to start: difficulty rule mismatch ...` | Wrong `chain` argument for this node | Use `mainnet`, or spell out `testnet::` | +| `REFUSING to run: there is no wallet file at ...` | `hbit-pool-payout` run in the wrong folder | Run it where the wallet file is, or pass its path as argument 4 | +| `[settle] holding back N unit(s) ...` | Recently found block not yet 16 deep | Nothing to do, wait about 80 minutes | +| Miners cannot connect at all | Pool bound to `127.0.0.1`, so only this machine can reach it | Restart with `0.0.0.0:`; the readback says which one it is | diff --git a/docs/POOL-README.md b/docs/POOL-README.md new file mode 100644 index 00000000..8a749236 --- /dev/null +++ b/docs/POOL-README.md @@ -0,0 +1,180 @@ +HBIT Pool - RUNNING A POOL (optional) +By Mosky +===================================== + +WHAT THIS IS, AND WHO NEEDS IT +------------------------------ +Most people who downloaded this package do NOT need anything in this file. + +If you want to mine, you are already done: run SETUP.bat (Windows) or +./SETUP-LINUX.sh (Linux), open the miner panel, and point it at a pool or at +your own fullnode. Mining does not require running a pool. + +HBIT is the other side of that. It is a mining pool: a program that serves work +to OTHER PEOPLE'S miners, keeps track of the shares they find, submits the +blocks it finds, and pays everybody out. You need it only if you intend to run +a pool that other people mine on. Two programs do that job: + + hbit-pool-server serves work, counts shares, submits blocks, pays out on a + timer. This is the pool. + hbit-pool-payout pays out by hand, run only while the server is stopped. + +Running a pool means holding other people's money and answering for it. If that +is not what you set out to do, close this file and go mine. + + +WHAT IT WILL DO WITH YOUR MONEY +------------------------------- +The pool mines into a wallet of its own, holds that income, and then pays it +out to the miners who found the shares. Concretely: + + - Every block the pool finds is mined into ITS OWN wallet, not yours. The + money lands in the wallet file described in the next section. + - It splits that balance over the last 4096 accepted shares (this is PPLNS) + and pays each miner their proportion, automatically, every 5 minutes by + default. + - It takes NO fee. Everything it mines goes to the miners. Running the pool + does not pay you anything, and you personally earn only from your own + hashrate, on the same terms as everyone else. + - Each payout transaction costs a 0.01 HAC network fee, funded out of a 0.5 + HAC reserve the pool keeps in its wallet. So running the pool costs you the + fees. That reserve is not skimmed from miners; whatever of it is not needed + is paid out later. + - Income from a block the pool just found is held back for 16 blocks, which + is about 80 minutes on mainnet, before it can be paid out. This is + deliberate. Paying it sooner and then losing the block to a chain reorg + would mean paying out money the chain never delivered, out of your pocket. + - A miner whose share of a cycle rounds below 0.1 HAC is paid nothing that + cycle. Their money is not taken from them: it stays in the wallet and is + part of the next cycle. + +The pool tells miners all of this itself, at http:///terms, read out +of the same constants it settles with. A miner can check what it is owed and +what it was paid at http:///earnings?worker=. + + +THE WALLET FILE: LOSE IT AND THE MONEY IS GONE +---------------------------------------------- +The first time you start hbit-pool-server it CREATES a wallet key file at the +path you gave it (the examples below call it pool-wallet.key, in the same +folder as the program). That file holds the private key to every coin the pool +has taken in and not yet paid out, including money that belongs to your miners. + + - There is no copy of that key anywhere else. Not on a server, not in this + package, not with the author. Nobody can send it to you. + - Delete the file, lose the disk, or reinstall over it, and every coin in the + pool wallet is gone permanently. That includes the miners' money, which you + will still owe them. + - Back it up the day you create it, before any miner connects, and keep the + backup somewhere the mining PC dying does not take with it. + +The pool writes two more files next to it. Keep them in the same backup: + + pool-wallet.key.state.json share accounting and the pending-payout ledger + pool-wallet.key.settle.lock the lock described at the end of this file + +Losing the .state.json does not lose coins, but it loses the record of who is +owed what and what has already been paid. + + +THE PASSPHRASE: HALF OF THE KEY +------------------------------- +Set HBIT_WALLET_PASSWORD before you start the pool and the wallet file is +stored encrypted (Argon2id + AES-256-GCM), so a stolen backup, an old drive or +a disk snapshot is useless to whoever finds it. Without it, the key file is +plaintext and the pool warns you about that every time it starts. + +Windows PowerShell, in the same window you are about to start the pool in: + + $env:HBIT_WALLET_PASSWORD = "a long passphrase you have written down" + +Linux: + + export HBIT_WALLET_PASSWORD='a long passphrase you have written down' + +It must be at least 8 characters. If you would rather not put it in the +environment, put it in a file and set HBIT_WALLET_PASSWORD_FILE to that file's +path instead. + +NEITHER HALF WORKS ALONE. The encrypted key file without the passphrase is +noise, and the passphrase without the key file is a string of words. There is +no reset, no recovery question and no support address. So back up BOTH, and +back them up together: + + - Write the passphrase down somewhere physical. Do not keep it only on the + machine that holds the key file, and do not keep it only in your head. + - Test the pair before you trust it. Stop hbit-pool-server but leave your + fullnode running, copy the backed-up key file into an empty folder, set the + passphrase in that window, and run hbit-pool-payout there with no --commit, + which pays nothing: + + .\hbit-pool-payout.exe http://127.0.0.1:9777 http://127.0.0.1:8080 mainnet pool-wallet.key + + It prints a line starting "wallet =". If that address is your pool's + address, the backup works. The node has to be reachable for this: the tool + reads the balance before it prints the address, and gives up if it cannot. + - Encrypting the file today does not reach backwards. Any backup or snapshot + taken while the file was still plaintext still contains the bare private + key. Treat those as live secrets. + + +STARTING IT +----------- +You need your own Hacash fullnode running and synced first. In this package +that is hacash.exe (Windows) or ./hacash (Linux), with hacash.config.ini set up +by SETUP.bat / SETUP-LINUX.sh, and its RPC on port 8080. + +Windows PowerShell, from the folder you extracted this package into: + + $env:HBIT_WALLET_PASSWORD = "a long passphrase you have written down" + .\hbit-pool-server.exe http://127.0.0.1:8080 pool-wallet.key 0.0.0.0:9777 24 mainnet + +Linux, from the same folder: + + export HBIT_WALLET_PASSWORD='a long passphrase you have written down' + ./hbit-pool-server http://127.0.0.1:8080 pool-wallet.key 0.0.0.0:9777 24 mainnet + +Reading that command left to right: the fullnode to use, the wallet file to +create or open, the address and port to serve miners on, how hard a share is, +and which chain. Every one of them is explained in hbit-pool.example.ini, which +is a worksheet, not a config file: no program reads it. + +Replace 0.0.0.0:9777 with 127.0.0.1:9777 while you are testing. 0.0.0.0 means +any machine that can reach this PC can mine here, which is what a real pool +wants and is also what puts this program on the network. Do not open a port to +it until the wallet is encrypted and backed up. + +The `mainnet` argument is required and has no default. The server proves it +against your node's own current block before serving any work, and refuses to +start if they disagree, because a pool running the wrong rule mines work the +node throws away forever without saying so. + + +NEVER RUN hbit-pool-payout WHILE THE SERVER IS RUNNING +------------------------------------------------------ +hbit-pool-payout is for paying out by hand. It must only run while +hbit-pool-server is STOPPED. + +Both programs decide what to pay from the wallet's confirmed balance, and a +payout still sitting in the mempool does not reduce that balance. Run both at +once and each one sees the full balance, each one believes it is the only payer, +and the same shares get paid TWICE out of your own funds. + +The programs enforce this themselves: each takes an exclusive lock on the wallet +and the second one refuses to start, printing + + REFUSING to run: another hbit-pool-server or hbit-pool-payout already holds ... + +That message is not a bug and there is no lock file to delete. Stop the server, +wait for it to actually exit, then run the tool. + +Run it dry first. Without --commit it pays nothing and just prints the split it +would make. Read that, then run the same command again with --commit on the end, +then start the server back up. + + +THE REST +-------- +POOL-OPERATOR.md, in this same folder, is the full runbook: the payout timing, +what each startup refusal means, how to tell your miners your pool's address, +and what to do if you think the key leaked. Read it before real miners connect. diff --git a/docs/PUBLIC-POOL.md b/docs/PUBLIC-POOL.md new file mode 100644 index 00000000..cb4e94bb --- /dev/null +++ b/docs/PUBLIC-POOL.md @@ -0,0 +1,145 @@ +# Public free-IP pool (`hac-pool`) + +## What it is + +`hac-pool` lets **anyone** run a public mining pool on a free IP: + +1. **HTTP miner RPC** (port default `3333`) — compatible with existing `poworker` +2. **Stratum TCP** (port default `3334`) — minimal JSON-RPC for multi-worker clients +3. **Upstream** = your fullnode `host:port` miner API + +## All-in-one (miner-panel) + +1. Build: `cargo build --release -p miner-pool -p miner-panel` (needs `hac-pool` next to the panel). +2. Open **Settings**. +3. Section **PUBLIC FREE-IP POOL (ALL-IN-ONE)**: + - Enable public pool controls + - Upstream fullnode (default `127.0.0.1:8080`) + - HTTP / Stratum ports + - Optional token + - **Start public pool** +4. With “mine through it” checked, Connect becomes `127.0.0.1:HTTP`. +5. **Start Mining** — if pool hosting is enabled and pool is stopped, the panel auto-starts the pool first. + +## Run (CLI) + +```bash +# fullnode must expose miner API (e.g. listen 8080) +cargo run --release -p miner-pool -- \ + --upstream 127.0.0.1:8080 \ + --http-bind 0.0.0.0:3333 \ + --stratum-bind 0.0.0.0:3334 +``` + +Open free pool (no password): + +```bash +# default: empty --pool-token +``` + +Token-protected: + +```bash +cargo run --release -p miner-pool -- \ + --upstream 127.0.0.1:8080 \ + --pool-token "shared-secret" +``` + +## Workers (poworker) + +```ini +; poworker.config.ini +connect = POOL_PUBLIC_IP:3333 +; if pool token set: +api_token = shared-secret +``` + +Firewall: open TCP 3333 (and 3334 for Stratum). + +## Stratum (minimal) + +Line-delimited JSON-RPC: + +- `mining.subscribe` +- `mining.authorize` with password = pool token (or any if open) +- `mining.notify` push: `[job_id, height, block_intro_hex, target]` +- `mining.submit`: `[worker, job_id, block_nonce, coinbase_nonce]` +- `mining.get_job`: full pending JSON (Hacash-native helper) + +## Connection limits + +| Option | Env var | Default | +|---|---|---| +| `--max-conns-per-ip` | `HAC_POOL_MAX_CONNS_PER_IP` | `128` (`0` = unlimited) | + +Caps how many **Stratum** connections one source IP may hold at once, so a single +peer cannot pin every slot and lock real miners out. Over the cap the pool logs +`stratum per-IP cap (N) reached; dropping ` and closes the new socket; the +worker sees a dropped connection and reconnects. + +Raise it for a **large farm behind one NAT address or one VPN exit**, where every +rig looks like the same IP: the default 128 is generous for a home farm but a +200-rig site needs more. Set it to `0` only on a pool that is not reachable from +the internet. There is a separate hard cap of 1024 Stratum connections in total. + +```bash +cargo run --release -p miner-pool -- \ + --upstream 127.0.0.1:8080 \ + --max-conns-per-ip 400 +``` + +## Job freshness and "upstream stale" + +`hac-pool` mirrors work from the upstream fullnode every `--poll-ms` +(default 2000). A mirrored job is only handed out while it is **fresh**, where +fresh means younger than `max(poll_ms x 4, 15s)`. At the default poll interval +that TTL is 15 seconds; at `--poll-ms 10000` it is 40 seconds. + +The refresh runs on every successful poll even when the height has not changed, +so the TTL only elapses during a real upstream outage. It never elapses just +because the chain is quiet between blocks. + +Once the TTL elapses the pool stops serving that job and says so: + +- HTTP miner RPC answers `{"err":"upstream stale; work is not being refreshed"}` + to `/query/miner/pending` and to a long-poll `/query/miner/notice` that times + out. +- Stratum simply does not push the stale job to a miner. + +**What it means for a worker:** the pool is up, but its fullnode is not +answering, so the work it holds is for a height the network has probably already +passed. Mining it would burn electricity on results that can only be rejected, so +the worker is told to wait instead. It is the operator's fullnode that needs +attention, not the worker. The pool logs `upstream job is stale (no refresh for +>Ns)` once when it starts and `upstream job refresh recovered` once when it comes +back, so an outage is easy to tell apart from a quiet chain. + +## Found-block submit retries + +A found block is the highest-value event in mining and there is no second chance +at it, so submits upstream are not one-shot: + +| Setting | Value | +|---|---| +| Submit attempts | up to 5 | +| Backoff between attempts | 250ms, 500ms, 1s, 2s (about a 3.75s total budget) | +| Per-attempt timeout, submit | 60s | +| Per-attempt timeout, job polling | 30s | + +Submits get the longer 60s budget because a busy fullnode validates the whole +block before answering, and that happens at exactly the moment a solution +arrives. Using the 30s polling timeout there would discard found blocks. + +Only transport failures and upstream 5xx / 408 / 429 replies are retried. An +HTTP 200 body is the node's own verdict and is returned to the worker verbatim +after a single attempt, so a genuine "stale height" rejection is never +re-hammered. Re-sending is safe in any case: the fullnode matches a submit by +height and can only ever include one block per height, so a repeat after a lost +reply is at worst a no-op and can never pay twice. + +## Security notes + +- Public bind without token is intentional for “free IP pool” but risks abuse. +- Prefer `--pool-token` on the internet. +- Upstream fullnode should stay on localhost; only `hac-pool` is public. +- Keep `--max-conns-per-ip` non-zero on a public bind. diff --git a/field/src/ini.rs b/field/src/ini.rs index c8063175..9dec1612 100644 --- a/field/src/ini.rs +++ b/field/src/ini.rs @@ -1,22 +1,176 @@ +// The ini parser already drops inline comments, but config sections are also built by hand +// (tests, embedded defaults, the panel). Money keys therefore clean their own value: an +// operator writing `reward = 1Abc... ; my wallet` must never end up with a mangled address. +fn ini_money_value(raw: &str) -> &str { + raw.split(|c| c == ';' || c == '#').next().unwrap_or("").trim() +} + +fn ini_money_required<'a>( + sec: &'a HashMap>, + sec_name: &str, + key: &str, + what: &str, +) -> &'a str { + let val = sec + .get(key) + .and_then(|v| v.as_deref()) + .map(ini_money_value) + .unwrap_or(""); + if val.is_empty() { + panic!( + "[Config Error] [{}] {} is required: set '{} = <{}>' in the config file. There is no default, a wrong or missing value sends the money somewhere you do not control.", + sec_name, key, key, what + ) + } + val +} -pub fn ini_must_address(sec: &HashMap>, key: &str) -> Address { - let adr = ini_must(sec, key, "1AVRuFXNFi3rdMrPH4hdqSgFrEBnWisWaS"); - let Ok(addr) = Address::from_readable(&adr) else { - panic!("[Config Error] address {} format invalid.", &adr) +// A reward address receives every coinbase this node mines, so it has NO default. +// It used to fall back to a hardcoded example address, which silently paid a node's whole +// mining income to a stranger for as long as the operator failed to notice. +pub fn ini_must_address_required( + sec: &HashMap>, + sec_name: &str, + key: &str, +) -> Address { + let adr = ini_money_required(sec, sec_name, key, "your own wallet address"); + let Ok(addr) = Address::from_readable(adr) else { + panic!("[Config Error] [{}] {} address {} format invalid.", sec_name, key, adr) }; addr } -pub fn ini_must_amount(sec: &HashMap>, key: &str) -> Amount { - let amt = ini_must(sec, key, "1:248"); - let Ok(amount) = Amount::from(&amt) else { - panic!("[Config Error] amount {} format invalid.", &amt) +// Bid and fee amounts have no default either: a placeholder amount is either far too small +// (the node bids and never wins, burning power for nothing) or far too large (it overpays +// from the operator's own wallet). Both are silent, so an unset key must stop startup. +pub fn ini_must_amount_required( + sec: &HashMap>, + sec_name: &str, + key: &str, +) -> Amount { + let amt = ini_money_required(sec, sec_name, key, "amount, e.g. 1 for 1 HAC"); + let Ok(amount) = Amount::from(amt) else { + panic!("[Config Error] [{}] {} amount {} format invalid.", sec_name, key, amt) }; amount } +#[cfg(test)] +mod ini_money_tests { + use super::*; + + fn sec_of(pairs: &[(&str, Option<&str>)]) -> HashMap> { + pairs + .iter() + .map(|(k, v)| (k.to_string(), v.map(|s| s.to_string()))) + .collect() + } + + #[test] + fn missing_reward_address_is_a_hard_failure() { + let sec = sec_of(&[("enable", Some("true"))]); + let res = std::panic::catch_unwind(|| ini_must_address_required(&sec, "miner", "reward")); + assert!(res.is_err(), "a missing reward key must never resolve to a built in address"); + } + + #[test] + fn valueless_reward_address_is_a_hard_failure() { + // `reward` written on its own line, with no '=' at all + let sec = sec_of(&[("reward", None)]); + let res = std::panic::catch_unwind(|| ini_must_address_required(&sec, "miner", "reward")); + assert!(res.is_err(), "a valueless reward key must never resolve to a built in address"); + } + + #[test] + fn blank_reward_address_is_a_hard_failure() { + for blank in ["", " ", "\t", " ; not set yet", "# todo"] { + let sec = sec_of(&[("reward", Some(blank))]); + let res = std::panic::catch_unwind(|| ini_must_address_required(&sec, "miner", "reward")); + assert!(res.is_err(), "a blank reward value {:?} must be rejected", blank); + } + } + + #[test] + fn reward_address_panic_names_the_section_and_the_key() { + let sec = sec_of(&[]); + let res = std::panic::catch_unwind(|| ini_must_address_required(&sec, "diamondminer", "reward")); + let err = res.unwrap_err(); + let msg = err + .downcast_ref::() + .cloned() + .unwrap_or_else(|| err.downcast_ref::<&str>().map(|s| s.to_string()).unwrap_or_default()); + assert!(msg.contains("[diamondminer]"), "message must name the section: {}", msg); + assert!(msg.contains("reward"), "message must name the key: {}", msg); + } + + #[test] + fn no_hardcoded_example_address_can_be_returned() { + // the old default, a well known example address nobody running this node owns + let old_default = "1AVRuFXNFi3rdMrPH4hdqSgFrEBnWisWaS"; + for sec in [sec_of(&[]), sec_of(&[("reward", None)]), sec_of(&[("reward", Some(" "))])] { + let got = std::panic::catch_unwind(|| { + ini_must_address_required(&sec, "miner", "reward").to_readable() + }); + assert!( + got.is_err(), + "an unset reward must fail, it must never fall back to {}", + old_default + ); + } + } + + #[test] + fn valid_reward_address_is_read_with_or_without_an_inline_comment() { + let want = "1MzNY1oA3kfgYi75zquj3SRUPYztzXHzK9"; + for raw in [ + "1MzNY1oA3kfgYi75zquj3SRUPYztzXHzK9", + " 1MzNY1oA3kfgYi75zquj3SRUPYztzXHzK9 ", + "1MzNY1oA3kfgYi75zquj3SRUPYztzXHzK9 ; payout wallet", + "1MzNY1oA3kfgYi75zquj3SRUPYztzXHzK9 # payout wallet", + ] { + let sec = sec_of(&[("reward", Some(raw))]); + let addr = ini_must_address_required(&sec, "miner", "reward"); + assert_eq!(addr.to_readable(), want, "value {:?} must parse to the address", raw); + } + } + + #[test] + fn invalid_reward_address_is_a_hard_failure() { + let sec = sec_of(&[("reward", Some("not-an-address"))]); + let res = std::panic::catch_unwind(|| ini_must_address_required(&sec, "miner", "reward")); + assert!(res.is_err(), "a malformed address must be rejected"); + } + + #[test] + fn missing_or_blank_amount_is_a_hard_failure() { + let cases: [HashMap>; 3] = + [sec_of(&[]), sec_of(&[("bid_min", None)]), sec_of(&[("bid_min", Some(" "))])]; + for sec in cases { + let res = + std::panic::catch_unwind(|| ini_must_amount_required(&sec, "diamondminer", "bid_min")); + assert!(res.is_err(), "an unset bid amount must never fall back to a placeholder"); + } + } + + #[test] + fn valid_amount_is_read_with_or_without_an_inline_comment() { + let want = Amount::from("1:248").unwrap(); + for raw in ["1:248", " 1:248 ", "1:248 ; smallest unit", "1:248 # smallest unit"] { + let sec = sec_of(&[("bid_min", Some(raw))]); + let amt = ini_must_amount_required(&sec, "diamondminer", "bid_min"); + assert_eq!(amt, want, "value {:?} must parse to the amount", raw); + } + } + + #[test] + fn invalid_amount_is_a_hard_failure() { + let sec = sec_of(&[("bid_min", Some("one hac"))]); + let res = std::panic::catch_unwind(|| ini_must_amount_required(&sec, "diamondminer", "bid_min")); + assert!(res.is_err(), "a malformed amount must be rejected"); + } +} diff --git a/hbit-pool/Cargo.toml b/hbit-pool/Cargo.toml new file mode 100644 index 00000000..fc781104 --- /dev/null +++ b/hbit-pool/Cargo.toml @@ -0,0 +1,52 @@ +[package] +name = "hbit-pool" +version = "0.1.0" +edition = "2024" +description = "HBIT: the Hacash mining pool this project ships. Share accounting, PPLNS payout splitting and on-chain settlement, plus the feasibility spikes it grew out of." + +# The pool daemon: serves work, validates shares, submits blocks, settles. +[[bin]] +name = "hbit-pool-server" +path = "src/server.rs" + +# Manual settlement, run by hand while the server is stopped. +[[bin]] +name = "hbit-pool-payout" +path = "src/payout.rs" + +# CPU worker used to soak-test the pool protocol. +[[bin]] +name = "hbit-test-miner" +path = "src/miner.rs" + +# P1 feasibility spikes. These really are spikes, so they keep the word. +[[bin]] +name = "hbit-pool-spike" +path = "src/main.rs" + +[[bin]] +name = "hbit-settle-spike" +path = "src/settle.rs" + +[[bin]] +name = "hbit-asert-check" +path = "src/asert_check.rs" + +[dependencies] +field = { path = "../field" } +basis = { path = "../basis" } +protocol = { path = "../protocol" } +mint = { path = "../mint" } +sys = { path = "../sys" } +x16rs = { path = "../x16rs" } +reqwest = { version = "0.12", default-features = false, features = ["rustls-tls", "blocking"] } +serde_json = "1.0" +hex = "0.4.3" +getrandom = "0.3.2" +num-bigint = "0.4.6" +# Cross-process advisory file lock: exactly one process may settle a wallet. +fs2 = "0.4.3" +# Encryption at rest for the pool wallet key (same primitives as sdk/keystore). +aes-gcm = { version = "0.10.3", features = ["zeroize"] } +argon2 = "0.5.3" +zeroize = "1.9.0" diff --git a/hbit-pool/hbit-pool.example.ini b/hbit-pool/hbit-pool.example.ini new file mode 100644 index 00000000..2299472c --- /dev/null +++ b/hbit-pool/hbit-pool.example.ini @@ -0,0 +1,185 @@ +; ============================================================================ +; HBIT pool - ARGUMENT WORKSHEET +; ---------------------------------------------------------------------------- +; READ THIS FIRST: no program reads this file. +; +; hbit-pool-server and hbit-pool-payout take every setting as an ARGUMENT on +; the command line, in a fixed order. There is no config file. This worksheet +; exists so you can write your answers down once, in one place, and then read +; the command off the bottom of it. +; +; So filling this in and then starting the server with no arguments does NOT +; work. It fails loudly rather than quietly: the server refuses to start +; without the `chain` argument instead of guessing one, so you get an error and +; nothing is lost. +; +; Nothing here is filled in for you. Every blank is a choice only you can make, +; and one of them decides where real money is held. A value shipped by somebody +; else is a value you did not choose. +; +; Never write the wallet passphrase in this file, or in any file next to the +; wallet key. See the PASSPHRASE section at the bottom for where it goes. +; +; The full runbook is POOL-OPERATOR.md, in this same folder. +; ============================================================================ + + +; ============================================================================ +; hbit-pool-server +; hbit-pool-server [settle_secs] +; ============================================================================ +[hbit-pool-server] + +; ARGUMENT 1: node +; Base URL of YOUR OWN synced Hacash fullnode RPC. The pool asks it for block +; templates, submits found blocks to it, and pays miners through it. It has to +; be a node you control: whoever runs the node decides what your pool mines and +; sees every payout before it is signed. +; The fullnode in this package listens on port 8080 (that is `listen` under +; [server] in hacash.config.ini), so on the same PC the answer is +; http://127.0.0.1:8080 +; If you leave the argument off entirely the built-in default is +; http://127.0.0.1:8088, which is NOT the port this package configures. +node = + +; ARGUMENT 2: wallet_file +; Path to the pool's own wallet key file. If the file does not exist the pool +; CREATES a new wallet there on first start and mines into it. That one file +; controls every coin the pool has taken in and not yet paid out. Back it up, +; and read the WALLET section of README-POOL.txt before you start. +; Put it somewhere you back up, not in a temporary folder. A bare name like +; pool-wallet.key +; means "next to the program", which is fine as long as that folder is backed +; up. If you leave the argument off, that bare name is the built-in default. +; The pool also writes two files beside it, and they belong with it in any +; backup: +; .state.json share accounting and the pending-payout ledger +; .settle.lock the lock described in ARGUMENT 3 of the payout +; section below +wallet_file = + +; ARGUMENT 3: listen +; Address and port this pool serves miners on, as :. +; 127.0.0.1:9777 only this PC can mine here (use this to test) +; 0.0.0.0:9777 any machine that can reach this PC can mine here +; 0.0.0.0 is what a real pool needs, and it is also what exposes this program +; to the internet. Do not open a port to it until you have read POOL-OPERATOR.md +; and the wallet is encrypted and backed up. +; Built-in default if you leave the argument off: 127.0.0.1:9777 +listen = + +; ARGUMENT 4: share_bits +; How many powers of two easier a share is than a real network block. Must be +; between 18 and 40; the server exits with an explanation outside that range. +; 24 suits GPU miners and is the right answer unless you have measured +; otherwise, so it is pre-filled here with the same value the program uses when +; the argument is left off. +share_bits = 24 + +; ARGUMENT 5: chain +; REQUIRED. There is no default, because a pool running the wrong difficulty +; rule mines work the node throws away forever without ever saying so. +; mainnet real Hacash mainnet +; testnet a testnet on the documented 288 / 10s pair +; testnet:: a testnet configured with any other pair +; The server proves your answer against the node's own current block before it +; serves any work, and refuses to start if the two disagree. +chain = + +; ARGUMENT 6: settle_secs (optional) +; Seconds between automatic payout runs. Pre-filled with the same value the +; program uses when the argument is left off. +settle_secs = 300 + + +; ============================================================================ +; hbit-pool-payout +; hbit-pool-payout [wallet_file] [reserve_units] [dust_units] [--commit] +; +; Manual settlement, for when the server is STOPPED. Both programs take an +; exclusive lock on the wallet, so this tool refuses to run while the server is +; up. That refusal is protecting your money: see POOL-OPERATOR.md section 2. +; ============================================================================ +[hbit-pool-payout] + +; ARGUMENT 1: pool_base +; Base URL of the pool server's own HTTP port, so the tool can read the share +; window from it. With the server stopped it falls back to the accounting file +; next to the wallet, which is the normal case for a manual payout. +; This is your ARGUMENT 3 above, as a URL, for example http://127.0.0.1:9777 +; Built-in default if left off: http://127.0.0.1:9777 +pool_base = + +; ARGUMENT 2: node +; The same fullnode RPC URL as ARGUMENT 1 of the server section. +node = + +; ARGUMENT 3: chain +; The same value as ARGUMENT 5 of the server section. Required here too. +chain = + +; ARGUMENT 4: wallet_file (optional) +; The same wallet file as ARGUMENT 2 of the server section. Point this at a +; different file and you are paying out of a different wallet. +wallet_file = + +; ARGUMENT 5: reserve_units (optional) +; Kept back in the pool wallet so it can always fund the network fee on a +; payout transaction, in whole units of 0.1 HAC. Not a fee and not skimmed: a +; later cycle pays out whatever of it is no longer needed. Pre-filled with the +; value the program uses when the argument is left off (5 = 0.5 HAC). +reserve_units = 5 + +; ARGUMENT 6: dust_units (optional) +; Smallest payout a cycle will include, in whole units of 0.1 HAC. A miner +; whose share rounds below this is paid nothing that cycle and the money stays +; in the pool wallet for the next one. Pre-filled with the built-in value +; (1 = 0.1 HAC). +dust_units = 1 + +; --commit +; NOT a numbered argument: write it anywhere on the command line. Without it +; the tool is a dry run that prints the planned split and pays nothing. Read +; the dry run first, every time. + + +; ============================================================================ +; PASSPHRASE (an environment variable, never a line in this file) +; ---------------------------------------------------------------------------- +; With a passphrase set, the wallet key file is stored encrypted with Argon2id +; and AES-256-GCM, so a stolen backup or an old disk is inert without it. With +; no passphrase set the key file is plaintext and the pool warns about it on +; every start. +; +; Set ONE of these in the shell you start the pool from, minimum 8 characters: +; HBIT_WALLET_PASSWORD the passphrase itself +; HBIT_WALLET_PASSWORD_FILE path to a file holding it, for services that +; cannot carry secrets in the environment +; HBIT_WALLET_PASSWORD wins if both are set. +; +; Windows PowerShell, in the window you are about to start the pool in: +; $env:HBIT_WALLET_PASSWORD = "the passphrase you wrote down" +; Linux: +; export HBIT_WALLET_PASSWORD='the passphrase you wrote down' +; +; The passphrase and the key file are two halves of one key and NEITHER half +; works alone. Back them up together, or the coins are gone: there is no reset +; and no recovery. + + +; ============================================================================ +; THE COMMAND (fill the blanks above in, then read it off here) +; ---------------------------------------------------------------------------- +; Windows PowerShell, from the folder holding hbit-pool-server.exe: +; $env:HBIT_WALLET_PASSWORD = "" +; .\hbit-pool-server.exe +; +; Linux, from the folder holding hbit-pool-server: +; export HBIT_WALLET_PASSWORD='' +; ./hbit-pool-server +; +; A manual payout, with the server stopped, dry run first: +; .\hbit-pool-payout.exe +; then, only if the printed split is right: +; .\hbit-pool-payout.exe --commit +; ============================================================================ diff --git a/hbit-pool/src/asert_check.rs b/hbit-pool/src/asert_check.rs new file mode 100644 index 00000000..0ba2def4 --- /dev/null +++ b/hbit-pool/src/asert_check.rs @@ -0,0 +1,127 @@ +//! Validate the off-node ASERT reimplementation against REAL chain history. +//! +//! For each of the last N blocks it recomputes the difficulty from that block's +//! own timestamp, its parent's difficulty and the anchor block's timestamp, then +//! compares against what the chain actually stored. A mismatch anywhere means +//! the pool would build blocks the node rejects, so this is the go/no-go check +//! before pointing a pool at mainnet. +//! +//! It checks BOTH quantities the node validates, because they are not +//! interchangeable: the header `difficulty` num, and the exact 32-byte PoW +//! target (each historical block was accepted by the network, so its own hash +//! must satisfy the target we recompute). A num-only check would pass a build +//! whose target hash is too tight, and such a pool silently discards solutions +//! the node would have accepted - lost blocks, lost revenue. +//! +//! Usage: hbit-asert-check [node_base] [count] [chain] +//! chain = mainnet | testnet | testnet:: + +use basis::difficulty::hash_bigger_than; +use hbit_pool::difficulty::{ChainParams, next_difficulty}; +use hbit_pool::{find_str, find_u64, get_json, http_client}; + +/// The block's own PoW hash from a `/query/block/intro` response. +fn block_hash32(b: &serde_json::Value) -> Option<[u8; 32]> { + let v = hex::decode(find_str(b, "hash")?).ok()?; + (v.len() == 32).then(|| { + let mut out = [0u8; 32]; + out.copy_from_slice(&v); + out + }) +} + +fn main() { + let a: Vec = std::env::args().collect(); + let node = a + .get(1) + .cloned() + .unwrap_or_else(|| "http://127.0.0.1:8080".to_string()); + let node = node.trim_end_matches('/').to_string(); + // Clamp to >=1: `count - 1` below would otherwise underflow at count==0. + let count: u64 = a.get(2).and_then(|s| s.parse().ok()).unwrap_or(10).max(1); + let chain = a.get(3).cloned().unwrap_or_else(|| "mainnet".to_string()); + let Some(params) = ChainParams::parse(&chain) else { + eprintln!( + "chain must be `mainnet`, `testnet`, or \ + `testnet::` (got `{chain}`)" + ); + std::process::exit(2); + }; + + let client = http_client(); + let tip = find_u64( + &get_json(&client, &format!("{node}/query/latest")), + "height", + ) + .expect("no chain tip"); + + println!("== HBIT ASERT check =="); + println!("node = {node}"); + println!("chain = {chain} (ASERT anchor at height {})", params.asert_height); + println!("tip = {tip}"); + + let anchor_time = find_u64( + &get_json( + &client, + &format!("{node}/query/block/intro?height={}", params.asert_height), + ), + "timestamp", + ) + .expect("anchor block timestamp (is the node synced past the anchor?)"); + println!("anchor ts = {anchor_time}\n"); + + let first = tip.saturating_sub(count - 1).max(params.asert_height + 1); + let mut ok = 0u64; + let mut bad = 0u64; + for h in first..=tip { + let b = get_json(&client, &format!("{node}/query/block/intro?height={h}")); + let (Some(ts), Some(stored)) = (find_u64(&b, "timestamp"), find_u64(&b, "difficulty")) + else { + println!("h={h} (missing block data, skipped)"); + continue; + }; + let pb = get_json(&client, &format!("{node}/query/block/intro?height={}", h - 1)); + let Some(prev_diff) = find_u64(&pb, "difficulty") else { + println!("h={h} (missing parent, skipped)"); + continue; + }; + let (ours, target) = next_difficulty(¶ms, h, ts, prev_diff as u32, anchor_time); + if ours as u64 != stored { + bad += 1; + println!("h={h} MISMATCH ours={ours} chain={stored}"); + continue; + } + // The node validates a block against BOTH quantities: the header `num` + // above AND the exact 32-byte PoW target, which on the from_big path is + // more precise than u32_to_hash(num). Comparing only the num would pass a + // build whose target hash is wrong, and the pool would then throw away + // solutions the node accepts. This block WAS accepted by the network, so + // its own hash must satisfy the target we just recomputed. + match block_hash32(&b) { + Some(bh) if hash_bigger_than(&bh, &target) => { + bad += 1; + println!( + "h={h} TARGET-TOO-TIGHT the chain's own block hash exceeds our target {}", + hex::encode(target) + ); + } + Some(_) => { + ok += 1; + println!("h={h} OK difficulty={stored} target={}", hex::encode(target)); + } + None => { + // Without the block's hash only half the check ran; do not report + // that as a pass. + bad += 1; + println!("h={h} NO-HASH could not read the block's own hash to verify the target"); + } + } + } + + println!("\n{ok} matched, {bad} mismatched"); + if bad == 0 && ok > 0 { + println!("PASS: the off-node ASERT reproduces real chain difficulty exactly."); + } else { + println!("FAIL: do NOT point a pool at this chain until this matches."); + } +} diff --git a/hbit-pool/src/difficulty.rs b/hbit-pool/src/difficulty.rs new file mode 100644 index 00000000..8918b537 --- /dev/null +++ b/hbit-pool/src/difficulty.rs @@ -0,0 +1,257 @@ +//! Off-node reimplementation of the node's next-block difficulty rule, so the +//! pool can build templates the node accepts at REAL (mainnet) heights. +//! +//! This mirrors mint/src/check/difficulty_asert.rs exactly. Every detail below +//! is load-bearing — a value that is off by one means the node rejects the +//! block: +//! * the exponent uses i128 `/` (truncates TOWARD ZERO, not floor) +//! * num_shifts uses an arithmetic shift (floor) and the fraction is derived +//! from it, so it is always in [0, 65535] +//! * the polynomial adds (1<<47) BEFORE the >>48 truncation (round-half-up) +//! * the target is shifted TWICE, separately (>> -num_shifts, then >> 16); +//! fusing them changes the truncation +//! * clamp order: zero-floor, then the 2x ease cap, then the LOWEST ceiling +//! +//! It returns BOTH representations, which are NOT interchangeable: the block +//! header must carry the u32 `num`, while the PoW comparison uses the exact +//! 32-byte target hash (which, on the from_big path, is more precise than +//! u32_to_hash(num)). + +use basis::difficulty::*; +use num_bigint::BigUint; + +const ASERT_START_TARGET_NUM: u32 = 0xe9cf_ffff; +const ASERT_HALF_LIFE: i128 = 10800; +const ASERT_RADIX: i128 = 1 << 16; +const ASERT_POLY_1: u128 = 195_766_423_245_049; +const ASERT_POLY_2: u128 = 971_821_376; +const ASERT_POLY_3: u128 = 5_127; +const ASERT_POLY_TERM_SHIFT: u32 = 48; +const ASERT_EASING_MAX_SCALE: u32 = 2; + +/// The chain parameters the difficulty rule depends on. +#[derive(Clone, Debug)] +pub struct ChainParams { + /// Height at which ASERT activates and which is also its anchor. + pub asert_height: u64, + /// `[mint] each_block_target_time` (mainnet 300s, testnet 10s). + pub target_time: u64, + /// Heights <= this use the bootstrap LOWEST_DIFFICULTY (testnet only). + pub bootstrap_max: u64, +} + +impl ChainParams { + pub fn mainnet() -> Self { + Self { + asert_height: 738654, + target_time: 300, + bootstrap_max: 0, + } + } + /// Non-mainnet: ASERT anchors at window+2 and heights <= window+1 bootstrap. + pub fn testnet(adjust_blocks: u64, target_time: u64) -> Self { + Self { + asert_height: adjust_blocks + 2, + target_time, + bootstrap_max: adjust_blocks + 1, + } + } + /// Parse a chain selector into real parameters, or None if it names no chain + /// this builder can mine. + /// + /// `mainnet` is consensus-fixed (anchor 738654, 300s). A testnet is NOT: the + /// node reads `difficulty_adjust_blocks` and `each_block_target_time` from + /// its own config file, and a pool that assumes a different pair computes a + /// different target for every block, so the node rejects all of them. Bare + /// `testnet` keeps the documented 288/10 pair; spell the real pair out as + /// `testnet::` for any node configured + /// otherwise. Either way the caller must PROVE the choice against the node + /// (`hbit_pool::verify_chain_params`) instead of trusting the label. + pub fn parse(name: &str) -> Option { + if name == "mainnet" { + return Some(Self::mainnet()); + } + let rest = name.strip_prefix("testnet")?; + if rest.is_empty() { + return Some(Self::testnet(288, 10)); + } + let mut fields = rest.strip_prefix(':')?.split(':'); + let adjust_blocks: u64 = fields.next()?.trim().parse().ok()?; + let target_time: u64 = fields.next()?.trim().parse().ok()?; + if fields.next().is_some() || adjust_blocks == 0 || target_time == 0 { + return None; + } + Some(Self::testnet(adjust_blocks, target_time)) + } + /// [`ChainParams::parse`] for the demo/testnet spikes, which have no operator + /// to report a bad selector to. Never use it on the money path. + pub fn from_name(name: &str) -> Self { + Self::parse(name).unwrap_or_else(|| Self::testnet(288, 10)) + } + /// Does computing this height's difficulty need the anchor block's timestamp? + pub fn needs_anchor(&self, height: u64) -> bool { + height > self.asert_height + } +} + +/// Next block's difficulty as (header `difficulty` u32, PoW target hash). +pub fn next_difficulty( + p: &ChainParams, + height: u64, + timestamp: u64, + prev_difficulty: u32, + anchor_time: u64, +) -> (u32, [u8; 32]) { + if height <= p.bootstrap_max { + let t = DifficultyTarget::from_num(LOWEST_DIFFICULTY); + return (t.num, t.hash); + } + if height == p.asert_height { + // Activation block: fixed start target, no parent cap. + let t = DifficultyTarget::from_num(ASERT_START_TARGET_NUM); + return (t.num, t.hash); + } + assert!( + height > p.asert_height, + "height {height} is in the pre-ASERT (legacy/LWMA) range, which this \ + off-node builder does not implement — a pool only mines at the tip" + ); + + let time_delta = timestamp as i128 - anchor_time as i128; + let height_delta = height as i128 - p.asert_height as i128; + // i128 division truncates toward zero. Multiply by the radix FIRST. + let exponent = + ((time_delta - p.target_time as i128 * height_delta) * ASERT_RADIX) / ASERT_HALF_LIFE; + let num_shifts = exponent >> 16; // arithmetic shift == floor + let frac = (exponent - (num_shifts << 16)) as u128; // always 0..=65535 + let frac2 = frac * frac; + let frac3 = frac2 * frac; + let factor = (((ASERT_POLY_1 * frac + + ASERT_POLY_2 * frac2 + + ASERT_POLY_3 * frac3 + + (1u128 << (ASERT_POLY_TERM_SHIFT - 1))) + >> ASERT_POLY_TERM_SHIFT) + + 65536) as u64; + + let anchor_target = u32_to_biguint(ASERT_START_TARGET_NUM); + let ease_target = u32_to_biguint(prev_difficulty) * BigUint::from(ASERT_EASING_MAX_SCALE); + let max_target = u32_to_biguint(LOWEST_DIFFICULTY); + + let mut next = anchor_target * BigUint::from(factor); + if num_shifts < 0 { + next >>= (-num_shifts) as usize; + } else if num_shifts > 0 { + next <<= num_shifts as usize; + } + next >>= 16usize; + + if next == BigUint::from(0u8) { + let t = DifficultyTarget::from_big(BigUint::from(1u8)); + return (t.num, t.hash); + } + if next > ease_target { + next = ease_target; // never more than 2x easier than the parent + } + if next > max_target { + let t = DifficultyTarget::from_num(LOWEST_DIFFICULTY); + return (t.num, t.hash); + } + let t = DifficultyTarget::from_big(next); + (t.num, t.hash) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn a_testnet_selector_can_carry_the_nodes_real_parameters() { + // The node takes difficulty_adjust_blocks and each_block_target_time from + // its own config; assuming 288/10 against a node using anything else made + // every template wrong, with no way for the operator to say otherwise. + let p = ChainParams::parse("testnet:8:10").expect("explicit testnet params"); + assert_eq!(p.asert_height, 10); + assert_eq!(p.bootstrap_max, 9); + assert_eq!(p.target_time, 10); + // Documented defaults and mainnet still parse. + let d = ChainParams::parse("testnet").expect("bare testnet"); + assert_eq!((d.asert_height, d.target_time), (290, 10)); + assert_eq!(ChainParams::parse("mainnet").expect("mainnet").asert_height, 738654); + // Anything we cannot mine is refused rather than silently guessed. + assert!(ChainParams::parse("regtest").is_none()); + assert!(ChainParams::parse("testnet:8").is_none()); + assert!(ChainParams::parse("testnet:8:10:2").is_none()); + assert!(ChainParams::parse("testnet:0:10").is_none()); + assert!(ChainParams::parse("testnet:8:0").is_none()); + assert!(ChainParams::parse("testnet:eight:10").is_none()); + } + + #[test] + fn bootstrap_heights_use_lowest_difficulty() { + let p = ChainParams::testnet(288, 10); + let (num, hash) = next_difficulty(&p, 1, 1_000, 0, 0); + assert_eq!(num, LOWEST_DIFFICULTY); + assert_eq!(hash, DifficultyTarget::from_num(LOWEST_DIFFICULTY).hash); + // the last bootstrap height is window+1 + assert_eq!(next_difficulty(&p, 289, 1_000, 0, 0).0, LOWEST_DIFFICULTY); + } + + #[test] + fn activation_height_uses_the_fixed_start_target() { + let p = ChainParams::testnet(288, 10); + let (num, hash) = next_difficulty(&p, 290, 9_999, LOWEST_DIFFICULTY, 0); + assert_eq!(num, ASERT_START_TARGET_NUM); + assert_eq!(hash, DifficultyTarget::from_num(ASERT_START_TARGET_NUM).hash); + // mainnet anchors at 738654 + let m = ChainParams::mainnet(); + assert_eq!( + next_difficulty(&m, 738654, 9_999, 0, 0).0, + ASERT_START_TARGET_NUM + ); + } + + #[test] + fn on_schedule_reproduces_the_anchor_target() { + // Exactly on schedule => exponent 0 => factor 65536 => target == anchor. + let p = ChainParams::testnet(288, 10); + let anchor_time = 1_000_000u64; + let height = p.asert_height + 5; + let timestamp = anchor_time + 5 * p.target_time; // perfectly on schedule + let (num, _) = next_difficulty(&p, height, timestamp, ASERT_START_TARGET_NUM, anchor_time); + assert_eq!(num, ASERT_START_TARGET_NUM); + } + + #[test] + fn faster_blocks_make_it_harder_slower_makes_it_easier() { + let p = ChainParams::testnet(288, 10); + let anchor_time = 1_000_000u64; + let height = p.asert_height + 100; + let on_time = anchor_time + 100 * p.target_time; + let base = DifficultyTarget::from_num( + next_difficulty(&p, height, on_time, ASERT_START_TARGET_NUM, anchor_time).0, + ); + // ahead of schedule (mined too fast) -> smaller target (harder) + let fast = DifficultyTarget::from_num( + next_difficulty(&p, height, on_time - 600, ASERT_START_TARGET_NUM, anchor_time).0, + ); + // behind schedule -> larger target (easier), capped at 2x the parent + let slow = DifficultyTarget::from_num( + next_difficulty(&p, height, on_time + 600, ASERT_START_TARGET_NUM, anchor_time).0, + ); + assert!(fast.big < base.big, "faster blocks must tighten the target"); + assert!(slow.big > base.big, "slower blocks must ease the target"); + } + + #[test] + fn easing_is_capped_at_twice_the_parent_target() { + let p = ChainParams::testnet(288, 10); + let anchor_time = 1_000_000u64; + let height = p.asert_height + 10; + // absurdly far behind schedule -> would explode, must clamp to 2x parent + let prev = ASERT_START_TARGET_NUM; + let (_, hash) = next_difficulty(&p, height, anchor_time + 10_000_000, prev, anchor_time); + let cap = u32_to_biguint(prev) * BigUint::from(2u32); + let got = DifficultyTarget::from_num(hash_to_u32(&hash)).big; + assert!(got <= cap, "must never ease past 2x the parent target"); + } +} diff --git a/hbit-pool/src/lib.rs b/hbit-pool/src/lib.rs new file mode 100644 index 00000000..b55f0f24 --- /dev/null +++ b/hbit-pool/src/lib.rs @@ -0,0 +1,2694 @@ +//! Shared helpers for the pool spikes: HTTP glue + off-node block assembly that +//! mirrors the node's `impl_packing_next_block` for a block containing a +//! coinbase plus optional extra transactions. Targets a fresh local testnet +//! (bootstrap LOWEST_DIFFICULTY); does not reproduce mainnet ASERT difficulty. + +pub mod difficulty; +pub mod pool_core; + +use difficulty::ChainParams; + +use std::collections::{HashMap, HashSet}; +use std::sync::{Arc, LazyLock, Mutex}; + +use basis::difficulty::*; +use basis::interface::*; +use field::*; +use protocol::block::*; +use protocol::transaction::*; +use sys::*; + +use serde_json::Value; +use zeroize::Zeroizing; + +pub fn http_client() -> reqwest::blocking::Client { + reqwest::blocking::Client::builder() + .timeout(std::time::Duration::from_secs(20)) + .build() + .expect("http client") +} + +pub fn get_json(client: &reqwest::blocking::Client, url: &str) -> Value { + let text = client + .get(url) + .send() + .and_then(|r| r.text()) + .unwrap_or_else(|e| format!("{{\"http_error\":\"{e}\"}}")); + serde_json::from_str(&text).unwrap_or_else(|_| Value::String(text)) +} + +pub fn post_hex(client: &reqwest::blocking::Client, url: &str, body: &str) -> String { + client + .post(url) + .header("content-type", "text/plain") + .body(body.to_string()) + .send() + .and_then(|r| r.text()) + .unwrap_or_else(|e| format!("http_error: {e}")) +} + +pub fn find_u64(v: &Value, key: &str) -> Option { + find_value(v, key).and_then(|x| { + x.as_u64() + .or_else(|| x.as_str().and_then(|s| s.trim().parse().ok())) + }) +} + +pub fn find_str(v: &Value, key: &str) -> Option { + find_value(v, key).and_then(|x| x.as_str().map(|s| s.to_string())) +} + +pub fn find_value<'a>(v: &'a Value, key: &str) -> Option<&'a Value> { + match v { + Value::Object(map) => map + .get(key) + .or_else(|| map.values().find_map(|child| find_value(child, key))), + Value::Array(arr) => arr.iter().find_map(|child| find_value(child, key)), + _ => None, + } +} + +/// The recipient's "hacash" balance string (e.g. "1:248"), or "" if none. +pub fn balance(client: &reqwest::blocking::Client, base: &str, addr: &str) -> String { + let j = get_json(client, &format!("{base}/query/balance?address={addr}")); + find_str(&j, "hacash").unwrap_or_default() +} + +/// The largest balance the pool will act on, in units of 0.1 HAC. Hacash's whole +/// coin supply is tens of millions of HAC, so anything past 100 billion HAC is a +/// corrupt or hostile answer, not a wallet. Refusing it keeps a bad number out of +/// the payout split instead of turning it into a maximal payout plan. +pub const MAX_PLAUSIBLE_UNITS: u64 = 1_000_000_000_000; + +/// A node "mantissa:unit" balance expressed in whole units of 0.1 HAC (unit 247). +/// +/// Hacash stores amounts normalized (trailing zeros stripped, unit raised), so a +/// balance like 4.9 HAC comes back as "49:246", not "490:247". FLOOR to 0.1-HAC +/// granularity, keeping the whole part, rather than discarding a balance just +/// because it is finer than 0.1 HAC. Shared by the pool server and the payout +/// tool so both value a balance identically. +/// +/// `None` means the node's answer was missing a separator, unparseable, or +/// larger than any real wallet: the caller must SKIP settlement rather than pay +/// out on it. Saturating to u64::MAX here (as this used to) means "infinite +/// money" to `distributable_units` and `split_payout`, which then plan a payout +/// of the whole u64 range off one malformed response. An EMPTY string is not an +/// error: the node simply omits the field for an address holding nothing. +pub fn balance_units(bal: &str) -> Option { + if bal.trim().is_empty() { + return Some(0); + } + let (m, u) = bal.split_once(':')?; + let (Ok(m), Ok(u)) = (m.trim().parse::(), u.trim().parse::()) else { + return None; + }; + let units = if u >= 247 { + let exp = u - 247; + if exp > 18 { + return None; // beyond any representable wallet, not a big balance + } + m.checked_mul(10u64.pow(exp as u32))? + } else { + let exp = 247 - u; + if exp > 18 { + return Some(0); // finer than 0.1 HAC: floors to nothing payable + } + m / 10u64.pow(exp as u32) + }; + (units <= MAX_PLAUSIBLE_UNITS).then_some(units) +} + +/// The coinbase subsidy of the block at `height`, in units of 0.1 HAC. The pool +/// mines coinbase-only blocks, so this is the entire income a found block brings +/// into the wallet (`block_reward` is a whole number of HAC = unit 248). +pub fn block_reward_units(height: u64) -> u64 { + mint::genesis::block_reward_number(height) as u64 * 10 +} + +/// How deep a payout transaction must be buried before the pool stops tracking +/// it. The node keeps up to `unstable_block` (4) blocks reorg-able, so a payout +/// that is only 1-3 confirmations deep can still come back to the mempool; +/// forgetting it that early lets the next cycle pay the same PPLNS window a +/// second time. 6 keeps a margin over the node's own window. +pub const PAYOUT_MATURITY_DEPTH: u64 = 6; + +/// What the node says about a payout transaction we previously submitted. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum PayoutTxState { + /// Still waiting in the mempool. + Pending, + /// Mined, but shallower than [`PAYOUT_MATURITY_DEPTH`] - a reorg could still + /// put it back in the mempool, so it is not finished with. + Confirming(u64), + /// Mined and buried deep enough that a reorg cannot undo it. + Buried(u64), + /// The node definitively does not know this hash: it was rejected, never + /// relayed, or dropped from the mempool. Settling again is the right move. + Gone, + /// We could not reach the node, or could not understand its answer. This is + /// NOT a resolution: treating it as one is exactly what opens a double-payout + /// window, so the caller must keep the hash and skip the cycle. + Unknown, +} + +/// Classify a `/query/transaction?hash=...` response. Fails SAFE: anything that +/// is not an unambiguous verdict from the node comes back as `Unknown`, and a +/// shallow confirmation counts as still in flight. +pub fn classify_payout_tx(j: &Value) -> PayoutTxState { + // get_json encodes a transport failure as {"http_error": "..."} and a + // non-JSON body as a bare string. Neither is the node speaking. + if !j.is_object() || j.get("http_error").is_some() { + return PayoutTxState::Unknown; + } + let Some(ret) = find_u64(j, "ret") else { + return PayoutTxState::Unknown; + }; + if ret != 0 { + return PayoutTxState::Gone; // the node answered "transaction not found" + } + let is_pending = j + .get("data") + .and_then(|d| d.get("pending")) + .and_then(|v| v.as_bool()) + .or_else(|| j.get("pending").and_then(|v| v.as_bool())) + .unwrap_or(false); + if is_pending { + return PayoutTxState::Pending; + } + // ret=0 and not pending means mined; the node reports the burial depth. + match find_u64(j, "confirm") { + Some(d) if d >= PAYOUT_MATURITY_DEPTH => PayoutTxState::Buried(d), + Some(d) => PayoutTxState::Confirming(d), + // ret=0 with neither `pending` nor `confirm` is a shape we do not + // recognise; unresolved is the safe reading. + None => PayoutTxState::Unknown, + } +} + +/// How many times to ask the node whether it really holds a payout we just +/// submitted, and how long to wait between asks. `/submit/transaction` answers +/// ret=0 the moment the API has validated the transaction and handed it to a +/// background task, so the node needs a moment before its own view of the hash +/// means anything. +pub const ADMIT_POLL_TRIES: u32 = 10; +pub const ADMIT_POLL_DELAY: std::time::Duration = std::time::Duration::from_millis(500); + +/// What the NODE says it holds after we submitted a payout transaction. +/// +/// `/submit/transaction` returning ret=0 is NOT this answer. The node validates +/// the transaction synchronously and then performs the mempool insert on a +/// background task whose result it DISCARDS, so the API reports "ok" for a +/// transaction the mempool went on to refuse - and the pool then reports a +/// payout that does not exist. The node's own view of the hash is the only +/// evidence that a payout is really in flight. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum Admission { + /// The node reports the transaction: waiting in its mempool, or already mined. + Held, + /// The node answered definitively that it does not know this transaction. + /// It was never inserted, so it was never relayed either. + Missing, + /// No usable answer (node unreachable, unparseable reply). NOT a resolution: + /// the payout may well be in flight, so it must stay tracked. + Unresolved, +} + +/// Map the node's `/query/transaction` answer to an admission verdict. +pub fn admission_of(j: &Value) -> Admission { + match classify_payout_tx(j) { + PayoutTxState::Pending | PayoutTxState::Confirming(_) | PayoutTxState::Buried(_) => { + Admission::Held + } + PayoutTxState::Gone => Admission::Missing, + PayoutTxState::Unknown => Admission::Unresolved, + } +} + +/// Ask the node whether it really holds `txhash`, retrying while it has not made +/// up its mind. The insert runs on a background task, so an immediate "not +/// found" only becomes a verdict once the node has had time to do it. +pub fn verify_admitted( + client: &reqwest::blocking::Client, + node: &str, + txhash: &str, +) -> Admission { + let mut last = Admission::Unresolved; + for attempt in 0..ADMIT_POLL_TRIES { + let j = get_json(client, &format!("{node}/query/transaction?hash={txhash}")); + match admission_of(&j) { + Admission::Held => return Admission::Held, + other => last = other, + } + if attempt + 1 < ADMIT_POLL_TRIES { + std::thread::sleep(ADMIT_POLL_DELAY); + } + } + last +} + +/// What a settlement may actually pay out: the wallet balance MINUS income a +/// reorg could still take back, MINUS the fee reserve. `None` means "nothing +/// spendable, do not settle this cycle". +/// +/// `immature_units` is the coinbase of blocks the pool found that are not yet +/// buried deep enough to be final. Distributing that and then losing the block +/// to a reorg is an unrecoverable operator loss: the income disappears from the +/// canonical chain while the payout transaction that spent it stays valid. +/// +/// All arithmetic saturates, so an out-of-range reserve can never wrap the +/// guard open the way `reserve + 1` used to. +pub fn distributable_units( + balance_units: u64, + immature_units: u64, + reserve_units: u64, +) -> Option { + let matured = balance_units.saturating_sub(immature_units); + if matured <= reserve_units.saturating_add(1) { + return None; + } + Some(matured - reserve_units) +} + +/// Atomic file write (temp + optional fsync + rename) so a crash or a full disk +/// mid-write can never leave a truncated or corrupt file behind. `durable` +/// fsyncs before the rename. +pub fn atomic_write(path: &str, body: &[u8], durable: bool) -> std::io::Result<()> { + use std::io::Write; + let tmp = format!("{path}.tmp.{}", std::process::id()); + { + let mut f = std::fs::File::create(&tmp)?; + f.write_all(body)?; + if durable { + let _ = f.sync_all(); + } + } + std::fs::rename(&tmp, path) +} + +/// The pool's accounting file for `wallet_file`. The auto-settle server and the +/// manual payout tool MUST agree on this path: it carries the ONE pending-payout +/// ledger that stops the two of them paying the same PPLNS window twice. +pub fn pool_state_path(wallet_file: &str) -> String { + format!("{wallet_file}.state.json") +} + +fn read_state_json(state_file: &str) -> Option { + let txt = std::fs::read_to_string(state_file).ok()?; + let j: Value = serde_json::from_str(&txt).ok()?; + j.is_object().then_some(j) +} + +/// The shared pending-payout ledger. A missing or corrupt file reads as an empty +/// ledger (the server rewrites that file wholesale and reports the corruption). +pub fn load_pending_payout_txs(state_file: &str) -> Vec { + let Some(j) = read_state_json(state_file) else { + return Vec::new(); + }; + j.get("settle_pending_txs") + .and_then(|v| v.as_array()) + .map(|a| { + a.iter() + .filter_map(|x| x.as_str().map(|s| s.to_string())) + .collect() + }) + .unwrap_or_default() +} + +/// Rolling PPLNS window: the last N accepted shares decide the payout split. +pub const PPLNS_WINDOW: usize = 4096; + +/// Rebuild the PPLNS share counts from the pool's own accounting file. +/// +/// The manual payout tool needs this because the server holds the wallet's +/// settlement lock for its whole run: if the tool is able to settle at all then +/// the server is stopped, so its `/stats` endpoint cannot answer and the file it +/// left behind is the authority on who is owed what. +pub fn load_pplns_counts(state_file: &str) -> Vec<(String, u64)> { + let Some(j) = read_state_json(state_file) else { + return Vec::new(); + }; + let window = j + .get("window") + .and_then(|v| v.as_u64()) + .unwrap_or(PPLNS_WINDOW as u64) as usize; + let order: Vec = j + .get("order") + .and_then(|v| v.as_array()) + .map(|a| { + a.iter() + .filter_map(|x| x.as_str().map(|s| s.to_string())) + .collect() + }) + .unwrap_or_default(); + if order.is_empty() { + return Vec::new(); + } + pool_core::Pplns::restore(window, order).counts() +} + +/// Total held-back (not yet final) block income recorded by the pool server, in +/// units of 0.1 HAC. The manual payout tool reads it so it applies the SAME +/// maturity gate as the automatic settlement instead of paying at the tip. +pub fn load_immature_units(state_file: &str) -> u64 { + let Some(j) = read_state_json(state_file) else { + return 0; + }; + j.get("immature") + .and_then(|v| v.as_array()) + .map(|a| { + a.iter() + .filter_map(|x| x.get("units").and_then(|v| v.as_u64())) + .sum() + }) + .unwrap_or(0) +} + +/// Replace `settle_pending_txs` in the pool state file, preserving every other +/// field the server keeps there (share window, counters, immature income). +pub fn save_pending_payout_txs(state_file: &str, hashes: &[String]) -> std::io::Result<()> { + let mut j = read_state_json(state_file).unwrap_or_else(|| serde_json::json!({})); + j["settle_pending_txs"] = serde_json::json!(hashes); + atomic_write(state_file, j.to_string().as_bytes(), true) +} + +/* --------------------------------------------------------------------------- + * The pool's money terms, in ONE place. + * + * `/terms` reads these same constants and `settle_once` / `hbit-pool-payout` apply + * them, so what the pool advertises cannot drift from what it does. Change a + * number here and every place that states it changes with it. + * ------------------------------------------------------------------------- */ + +/// The amount unit the pool accounts in: 0.1 HAC (Hacash amount unit 247). Every +/// payout it plans, submits and reports is a whole number of these. +pub const PAYOUT_UNIT: u8 = 247; + +/// `units` of 0.1 HAC as the chain's OWN money type. The pool never renders money +/// as a float or a hand-rolled decimal: what a miner is shown is exactly what the +/// transaction carries. +pub fn payout_amount(units: u64) -> Amount { + Amount::coin(units, PAYOUT_UNIT) +} + +/// The network fee ONE settlement transaction carries: 0.01 HAC. It comes out of +/// the reserve below, never out of a miner's share. +pub fn chunk_tx_fee() -> Amount { + Amount::coin(1, 246) +} + +/// The pool's own fee, in units of 0.1 HAC, taken off the top of a settlement +/// before it is split. It is ZERO: this pool skims nothing. +pub const POOL_FEE_UNITS: u64 = 0; + +/// Held back from every settlement so the wallet can always fund the per-chunk +/// network fee above. This is NOT a fee: it stays in the pool wallet and a later +/// cycle distributes whatever of it is no longer needed. +pub const SETTLE_RESERVE_UNITS: u64 = 5; + +/// The smallest payout a settlement will include, in units of 0.1 HAC. A worker +/// whose share of a cycle rounds below this is paid nothing THAT CYCLE; the money +/// is never taken from anyone - it stays in the pool wallet and is part of the +/// next cycle's distributable balance. +pub const PAYOUT_DUST_UNITS: u64 = 1; + +/// Recipients per settlement transaction. The node enforces TX_ACTIONS_MAX = 200 +/// actions, so stay safely under it: a large payout is chunked, never rejected. +pub const PAYOUT_CHUNK: usize = 190; + +/* --------------------------------------------------------------------------- + * Per-worker settlement ledger. + * + * `settle_pending_txs` alone can only say that SOME payout is in flight. A miner + * needs to know what is in flight FOR IT, what it has actually been paid, and + * when - so the pool keeps the exact per-recipient rows of every payout it + * submits, and folds them into a paid ledger when, and only when, the node + * reports that transaction buried. + * ------------------------------------------------------------------------- */ + +/// One settlement transaction this pool submitted, with the exact amounts it +/// carries for each recipient. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct PayoutRecord { + pub hash: String, + /// Unix seconds when the pool submitted it. + pub at: u64, + /// Did the NODE confirm it holds this transaction? `false` means it was + /// submitted but the node's verdict could not be read: it may well be in + /// flight, so it stays tracked, but nothing about it is claimed. + pub node_holds: bool, + /// (worker address, units of 0.1 HAC) exactly as the transaction pays them. + pub rows: Vec<(String, u64)>, +} + +impl PayoutRecord { + /// Total this transaction pays, in units of 0.1 HAC. + pub fn units(&self) -> u64 { + self.rows.iter().map(|(_, u)| *u).fold(0u64, |a, b| a.saturating_add(b)) + } + + /// What this transaction pays ONE worker. + pub fn units_for(&self, worker: &str) -> u64 { + self.rows + .iter() + .filter(|(w, _)| w == worker) + .map(|(_, u)| *u) + .fold(0u64, |a, b| a.saturating_add(b)) + } + + pub fn to_json(&self) -> Value { + serde_json::json!({ + "hash": self.hash, + "at": self.at, + "node_holds": self.node_holds, + "rows": self.rows.iter() + .map(|(w, u)| serde_json::json!([w, u])) + .collect::>(), + }) + } + + /// Rebuild one record. A row the file cannot describe is DROPPED rather than + /// guessed at: an unreadable amount must never become a number a miner is + /// shown. + pub fn from_json(v: &Value) -> Option { + let hash = v.get("hash")?.as_str()?.to_string(); + if hash.is_empty() { + return None; + } + let rows = v + .get("rows")? + .as_array()? + .iter() + .filter_map(|r| { + let a = r.as_array()?; + Some((a.first()?.as_str()?.to_string(), a.get(1)?.as_u64()?)) + }) + .collect(); + Some(Self { + hash, + at: v.get("at").and_then(|x| x.as_u64()).unwrap_or(0), + node_holds: v + .get("node_holds") + .and_then(|x| x.as_bool()) + .unwrap_or(false), + rows, + }) + } +} + +/// What this pool's ledger has actually paid ONE worker. +#[derive(Debug, Clone, Default, PartialEq, Eq)] +pub struct PaidRow { + /// Total confirmed paid, in units of 0.1 HAC. Only ever grows. + pub units: u64, + /// The most recent confirmed payout: amount, transaction, and when the pool + /// saw the node bury it. + pub last_units: u64, + pub last_hash: String, + pub last_at: u64, +} + +/// Confirmed payouts per worker. +/// +/// `since` is reported next to every total on purpose: this is what the pool has +/// paid SINCE THIS LEDGER EXISTED, not what a worker has ever earned. A pool +/// whose state file was lost starts a new ledger, and a miner must be able to see +/// that rather than read a total that silently means something else. +#[derive(Debug, Clone, Default, PartialEq, Eq)] +pub struct PaidLedger { + pub since: u64, + rows: HashMap, +} + +impl PaidLedger { + /// A fresh ledger that starts counting now. + pub fn started(at: u64) -> Self { + Self { + since: at, + rows: HashMap::new(), + } + } + + pub fn get(&self, worker: &str) -> Option<&PaidRow> { + self.rows.get(worker) + } + + pub fn workers(&self) -> usize { + self.rows.len() + } + + pub fn total_units(&self) -> u64 { + self.rows + .values() + .map(|r| r.units) + .fold(0u64, |a, b| a.saturating_add(b)) + } + + /// Fold a payout the node has BURIED into the ledger. + /// + /// Only ever ADDS: a worker's paid total can never go down, and it moves only + /// on a node confirmation - never when a payout is merely submitted, and + /// never when one is only shallowly mined and a reorg could still undo it. + pub fn credit(&mut self, rec: &PayoutRecord, confirmed_at: u64) { + for (worker, units) in &rec.rows { + if *units == 0 { + continue; + } + let row = self.rows.entry(worker.clone()).or_default(); + row.units = row.units.saturating_add(*units); + row.last_units = *units; + row.last_hash = rec.hash.clone(); + row.last_at = confirmed_at; + } + } + + pub fn to_json(&self) -> Value { + let mut rows: Vec<(&String, &PaidRow)> = self.rows.iter().collect(); + rows.sort_by(|a, b| a.0.cmp(b.0)); // stable file, stable diffs + serde_json::json!({ + "since": self.since, + "rows": rows.iter().map(|(w, r)| serde_json::json!({ + "worker": w, + "units": r.units, + "last_units": r.last_units, + "last_hash": r.last_hash, + "last_at": r.last_at, + })).collect::>(), + }) + } + + pub fn from_json(v: &Value) -> Self { + let since = v.get("since").and_then(|x| x.as_u64()).unwrap_or(0); + let mut rows: HashMap = HashMap::new(); + if let Some(a) = v.get("rows").and_then(|x| x.as_array()) { + for r in a { + let Some(w) = r.get("worker").and_then(|x| x.as_str()) else { + continue; + }; + let Some(units) = r.get("units").and_then(|x| x.as_u64()) else { + continue; // an unreadable total is not a zero total + }; + rows.insert( + w.to_string(), + PaidRow { + units, + last_units: r.get("last_units").and_then(|x| x.as_u64()).unwrap_or(0), + last_hash: r + .get("last_hash") + .and_then(|x| x.as_str()) + .unwrap_or_default() + .to_string(), + last_at: r.get("last_at").and_then(|x| x.as_u64()).unwrap_or(0), + }, + ); + } + } + Self { since, rows } + } +} + +/// Move a payout the node reports BURIED out of the in-flight list and into the +/// paid ledger, in one step so a unit can never be counted in both. +/// +/// Returns the record it credited, or `None` if this pool has no rows for that +/// hash (a payout submitted by an older build, or by a tool that did not record +/// its rows). `None` is why the caller must still drop the hash from the pending +/// ledger: the money moved, this pool just cannot attribute it. +pub fn confirm_payout( + records: &mut Vec, + paid: &mut PaidLedger, + hash: &str, + confirmed_at: u64, +) -> Option { + let i = records.iter().position(|r| r.hash == hash)?; + let rec = records.remove(i); + paid.credit(&rec, confirmed_at); + Some(rec) +} + +/// Drop a payout the node definitively does NOT hold. Nothing was paid, so +/// nothing is credited: that money is still owed and goes back to `pending`. +pub fn drop_payout(records: &mut Vec, hash: &str) -> Option { + let i = records.iter().position(|r| r.hash == hash)?; + Some(records.remove(i)) +} + +/// The per-transaction rows of every payout this pool has in flight. +pub fn load_payout_records(state_file: &str) -> Vec { + let Some(j) = read_state_json(state_file) else { + return Vec::new(); + }; + parse_payout_records(&j) +} + +/// Read the in-flight payout rows out of an already-parsed state document. +pub fn parse_payout_records(j: &Value) -> Vec { + j.get("payouts_inflight") + .and_then(|v| v.as_array()) + .map(|a| a.iter().filter_map(PayoutRecord::from_json).collect()) + .unwrap_or_default() +} + +/// The confirmed-payout ledger. +pub fn load_paid_ledger(state_file: &str) -> PaidLedger { + let Some(j) = read_state_json(state_file) else { + return PaidLedger::default(); + }; + parse_paid_ledger(&j) +} + +/// Read the confirmed-payout ledger out of an already-parsed state document. +pub fn parse_paid_ledger(j: &Value) -> PaidLedger { + j.get("paid").map(PaidLedger::from_json).unwrap_or_default() +} + +/// Replace the WHOLE settlement ledger (pending hashes, per-transaction rows and +/// confirmed totals) in the pool state file, preserving every other field. +/// +/// One write, because the three move together: a payout leaves the in-flight +/// rows at the same instant it enters the paid totals, and a crash between the +/// two would either lose a payment or count it twice. +pub fn save_settlement_ledger( + state_file: &str, + hashes: &[String], + records: &[PayoutRecord], + paid: &PaidLedger, +) -> std::io::Result<()> { + let mut j = read_state_json(state_file).unwrap_or_else(|| serde_json::json!({})); + j["settle_pending_txs"] = serde_json::json!(hashes); + j["payouts_inflight"] = Value::Array(records.iter().map(|r| r.to_json()).collect()); + j["paid"] = paid.to_json(); + atomic_write(state_file, j.to_string().as_bytes(), true) +} + +/// The lock file guarding one wallet's settlement. +pub fn settle_lock_path(wallet_file: &str) -> String { + format!("{wallet_file}.settle.lock") +} + +/// An exclusive, cross-process claim on one wallet's settlement, held for as +/// long as the value lives. The OS releases it if the holder dies, so a crash +/// can never wedge payouts the way a hand-rolled PID file would. +pub struct SettleLock { + _file: std::fs::File, +} + +/// Take the wallet's settlement lock, or fail if another process holds it. +/// +/// The pool server takes this for its whole run and `hbit-pool-payout` takes it for +/// its whole run. Without it the two paths each see the full CONFIRMED balance +/// (a payout sitting in the mempool does not reduce it) and each pays the same +/// PPLNS window - a real double payout of the pool's distributable balance. +pub fn acquire_settle_lock(wallet_file: &str) -> std::io::Result { + let path = settle_lock_path(wallet_file); + let file = std::fs::OpenOptions::new() + .read(true) + .write(true) + .create(true) + .truncate(false) + .open(&path)?; + // Call it as a trait function so it can never be confused with a same-named + // inherent method on File. + fs2::FileExt::try_lock_exclusive(&file)?; + Ok(SettleLock { _file: file }) +} + +/// Is this string a payable Hacash address (normal single-key PRIVAKEY)? +/// Workers announce one as `&worker=
`; the pool then uses the address +/// itself as the share-accounting key, so payouts need no name->address map. +pub fn is_payout_address(s: &str) -> bool { + Address::from_readable(s) + .map(|a| a.is_privakey()) + .unwrap_or(false) +} + +/// Environment variable holding the pool wallet passphrase. When it is set the +/// key file is stored ENCRYPTED (Argon2id + AES-256-GCM), so a stolen backup, a +/// VSS/disk snapshot or a decommissioned drive is inert without the passphrase. +pub const WALLET_PASSWORD_ENV: &str = "HBIT_WALLET_PASSWORD"; +/// Alternative source for that passphrase: a file holding it, for services that +/// cannot carry secrets in the environment. +pub const WALLET_PASSWORD_FILE_ENV: &str = "HBIT_WALLET_PASSWORD_FILE"; + +/// The shortest passphrase the pool will protect a real wallet with. Anything +/// shorter is refused rather than silently accepted: it is the only thing +/// standing between a stolen backup and every coin the pool holds. +pub const WALLET_PASSWORD_MIN: usize = 8; + +/// Characters in an unencrypted wallet file: a 32-byte private key in hex. +const WALLET_KEY_HEX_LEN: usize = 64; + +const WALLET_ENVELOPE_VERSION: u64 = 1; +const WALLET_KDF_M_COST_KB: u32 = 19456; +const WALLET_KDF_T_COST: u32 = 2; +const WALLET_KDF_P_COST: u32 = 1; +/// Upper bounds on the KDF parameters read back from a file, so a tampered +/// envelope cannot turn a startup into an out-of-memory or an endless grind. +const WALLET_KDF_MAX_M_COST_KB: u32 = 256 * 1024; +const WALLET_KDF_MAX_T_COST: u32 = 16; +const WALLET_KDF_MAX_P_COST: u32 = 16; + +/// The configured wallet passphrase, or None for the (loudly warned about) +/// plaintext mode. +/// +/// Never prompts for anything: a service manager gives this process no terminal, +/// so every input comes from the environment or a file and anything missing is a +/// refusal that names what to set. +fn wallet_password() -> Result>, String> { + wallet_password_from( + std::env::var(WALLET_PASSWORD_ENV).ok(), + std::env::var(WALLET_PASSWORD_FILE_ENV).ok(), + ) +} + +/// The passphrase rule itself, with the two configured values passed in so the +/// whole of it can be exercised without touching this process's environment. +/// +/// The variable wins over the file, and an EMPTY variable means "not set" so a +/// service unit that always exports it can still leave it blank. A file that +/// exists but holds nothing is NOT the same as no passphrase at all: read that +/// way, an encrypted wallet would be reported as "no passphrase is configured" +/// to an operator who had configured one, and the fix they were told to apply is +/// the one they had already applied. +/// +/// No message built here ever carries the passphrase, or its length: these lines +/// go to a journal that outlives the mistake. +fn wallet_password_from( + direct: Option, + file: Option, +) -> Result>, String> { + let direct = Zeroizing::new(direct.unwrap_or_default()); + if !direct.is_empty() { + if direct.len() < WALLET_PASSWORD_MIN { + return Err(format!( + "REFUSING to touch the pool wallet: the passphrase in {WALLET_PASSWORD_ENV} is \ + shorter than {WALLET_PASSWORD_MIN} characters.\n\ + Nothing was read and nothing was written.\n\ + What to do: set {WALLET_PASSWORD_ENV} to a longer passphrase, one you have \ + written down somewhere physical, and start again. If a wallet file already \ + exists, it must be the passphrase that wallet was created with." + )); + } + return Ok(Some(direct)); + } + // No passphrase in the environment: fall back to the file, if one is named. + let Some(file) = file.filter(|f| !f.trim().is_empty()) else { + return Ok(None); + }; + if std::path::Path::new(&file).is_dir() { + return Err(format!( + "REFUSING to touch the pool wallet: {WALLET_PASSWORD_FILE_ENV} names {file}, which is \ + a directory, not a file.\n\ + What to do: point {WALLET_PASSWORD_FILE_ENV} at the FILE that holds the passphrase \ + and nothing else, or unset it and put the passphrase in {WALLET_PASSWORD_ENV}." + )); + } + let text = match std::fs::read_to_string(&file) { + Ok(t) => Zeroizing::new(t), + Err(e) => { + let why = match e.kind() { + std::io::ErrorKind::NotFound => "there is no file at that path".to_string(), + std::io::ErrorKind::PermissionDenied => { + "this account is not allowed to read it".to_string() + } + std::io::ErrorKind::InvalidData => "it is not text".to_string(), + _ => format!("the operating system said: {e}"), + }; + return Err(format!( + "REFUSING to touch the pool wallet: {WALLET_PASSWORD_FILE_ENV} names {file}, \ + which cannot be read ({why}).\n\ + Nothing was read and nothing was written.\n\ + What to do: point {WALLET_PASSWORD_FILE_ENV} at a readable file holding just the \ + passphrase, and make sure the account this service runs as can read it. Or unset \ + it and use {WALLET_PASSWORD_ENV} instead." + )); + } + }; + // Trimmed, exactly as it always has been: a passphrase file written by an + // editor ends in a newline, and the wallets already on disk were encrypted + // with the trimmed form. + let pass = Zeroizing::new(text.trim().to_string()); + if pass.is_empty() { + return Err(format!( + "REFUSING to touch the pool wallet: {WALLET_PASSWORD_FILE_ENV} names {file}, which is \ + empty, so there is no passphrase to use.\n\ + Nothing was read and nothing was written. This is NOT being treated as \"no \ + passphrase\": if the wallet is encrypted, running on without one would look like a \ + configuration you never made.\n\ + What to do: put the wallet's passphrase in {file}, or unset \ + {WALLET_PASSWORD_FILE_ENV} and use {WALLET_PASSWORD_ENV}." + )); + } + if pass.len() < WALLET_PASSWORD_MIN { + return Err(format!( + "REFUSING to touch the pool wallet: the passphrase in {file} is shorter than \ + {WALLET_PASSWORD_MIN} characters.\n\ + Nothing was read and nothing was written.\n\ + What to do: put a longer passphrase in that file, one you have written down \ + somewhere physical, and start again. If a wallet file already exists, it must be the \ + passphrase that wallet was created with." + )); + } + Ok(Some(pass)) +} + +fn wallet_derive_key( + pass: &str, + salt: &[u8], + m_cost_kb: u32, + t_cost: u32, + p_cost: u32, +) -> Result, String> { + use argon2::{Algorithm, Argon2, Params, Version}; + let params = Params::new(m_cost_kb, t_cost, p_cost, Some(32)) + .map_err(|e: argon2::Error| e.to_string())?; + let argon = Argon2::new(Algorithm::Argon2id, Version::V0x13, params); + let mut key = Zeroizing::new([0u8; 32]); + argon + .hash_password_into(pass.as_bytes(), salt, &mut *key) + .map_err(|e: argon2::Error| e.to_string())?; + Ok(key) +} + +/// Wrap a 64-hex private key in a versioned Argon2id + AES-256-GCM envelope. +fn encrypt_key_hex(key_hex: &str, pass: &str) -> Result { + use aes_gcm::aead::{Aead, KeyInit}; + use aes_gcm::{Aes256Gcm, Nonce}; + let mut salt = [0u8; 16]; + let mut nonce = [0u8; 12]; + getrandom::fill(&mut salt).map_err(|e| e.to_string())?; + getrandom::fill(&mut nonce).map_err(|e| e.to_string())?; + let key = wallet_derive_key( + pass, + &salt, + WALLET_KDF_M_COST_KB, + WALLET_KDF_T_COST, + WALLET_KDF_P_COST, + )?; + let cipher = Aes256Gcm::new_from_slice(&*key).map_err(|e| e.to_string())?; + let ciphertext = cipher + .encrypt(Nonce::from_slice(&nonce), key_hex.as_bytes()) + .map_err(|e: aes_gcm::Error| e.to_string())?; + Ok(serde_json::json!({ + "hbit_wallet": WALLET_ENVELOPE_VERSION, + "kdf": "argon2id", + "kdf_salt": hex::encode(salt), + "kdf_m_cost_kb": WALLET_KDF_M_COST_KB, + "kdf_t_cost": WALLET_KDF_T_COST, + "kdf_p_cost": WALLET_KDF_P_COST, + "cipher": "aes-256-gcm", + "cipher_nonce": hex::encode(nonce), + "ciphertext": hex::encode(ciphertext), + }) + .to_string()) +} + +fn envelope_u32(j: &Value, key: &str, default: u32, max: u32) -> Result { + let v = j.get(key).and_then(|v| v.as_u64()).unwrap_or(default as u64); + if v == 0 || v > max as u64 { + return Err(format!("its `{key}` is outside the range this build accepts")); + } + Ok(v as u32) +} + +/// Why an encrypted wallet file would not open. +/// +/// The three are kept apart on purpose, and the ONE distinction that costs money +/// is `Shape` versus `Undecryptable`. Telling an operator their file is corrupt +/// when they merely mistyped the passphrase invites them to restore a backup +/// over the real wallet, or to delete it and let the pool create a fresh empty +/// one, abandoning the funds in the file they still had. +#[derive(Debug, Clone, PartialEq, Eq)] +enum EnvelopeError { + /// The file is not an envelope this build can read. Decided from the file's + /// structure alone, BEFORE any passphrase is tried, so it is certain and it + /// is never a passphrase problem. + Shape(String), + /// The authentication tag did not verify. That is a wrong passphrase OR + /// damaged ciphertext, and AES-GCM cannot tell those apart: it is the same + /// check that fails either way. Neither can this pool, so neither may the + /// message it prints. + Undecryptable, + /// The tag DID verify, so the passphrase is right and the file is intact, + /// but what came out of it is not a private key. + Content(String), +} + +/// Unwrap an envelope written by [`encrypt_key_hex`]. +/// +/// No error carries any part of the passphrase, the ciphertext or the key: the +/// caller turns these into lines that end up in a journal. +fn decrypt_key_hex(body: &str, pass: &str) -> Result, EnvelopeError> { + use aes_gcm::aead::{Aead, KeyInit}; + use aes_gcm::{Aes256Gcm, Nonce}; + let shape = |s: String| EnvelopeError::Shape(s); + // Everything down to the decrypt itself is a property of the FILE. None of + // it depends on the passphrase, which is what makes `Shape` certain. + let j: Value = serde_json::from_str(body) + .map_err(|_| shape("it is not the JSON this pool writes".to_string()))?; + let ver = j.get("hbit_wallet").and_then(|v| v.as_u64()).unwrap_or(0); + if ver != WALLET_ENVELOPE_VERSION { + return Err(shape(format!( + "it declares wallet format {ver} and this build reads format {WALLET_ENVELOPE_VERSION}" + ))); + } + let hex_field = |k: &str| -> Result, EnvelopeError> { + let s = j + .get(k) + .and_then(|v| v.as_str()) + .ok_or_else(|| shape(format!("it is missing the `{k}` field")))?; + hex::decode(s).map_err(|_| shape(format!("its `{k}` field is not hex"))) + }; + let salt = hex_field("kdf_salt")?; + let nonce = hex_field("cipher_nonce")?; + let ciphertext = hex_field("ciphertext")?; + if nonce.len() != 12 { + return Err(shape(format!( + "its nonce is {} bytes and must be 12", + nonce.len() + ))); + } + let key = wallet_derive_key( + pass, + &salt, + envelope_u32(&j, "kdf_m_cost_kb", WALLET_KDF_M_COST_KB, WALLET_KDF_MAX_M_COST_KB) + .map_err(EnvelopeError::Shape)?, + envelope_u32(&j, "kdf_t_cost", WALLET_KDF_T_COST, WALLET_KDF_MAX_T_COST) + .map_err(EnvelopeError::Shape)?, + envelope_u32(&j, "kdf_p_cost", WALLET_KDF_P_COST, WALLET_KDF_MAX_P_COST) + .map_err(EnvelopeError::Shape)?, + ) + .map_err(|e| shape(format!("its key-derivation settings cannot be used here ({e})")))?; + let cipher = Aes256Gcm::new_from_slice(&*key) + .map_err(|e| shape(format!("its cipher key could not be set up ({e})")))?; + // The one check that cannot say WHY it failed. + let plain = cipher + .decrypt(Nonce::from_slice(&nonce), ciphertext.as_ref()) + .map(Zeroizing::new) + .map_err(|_e: aes_gcm::Error| EnvelopeError::Undecryptable)?; + let txt = String::from_utf8(plain.to_vec()) + .map(Zeroizing::new) + .map_err(|_| EnvelopeError::Content("what is inside it is not text".to_string()))?; + Ok(txt) +} + +/// The refusal for an envelope that would not open, said so that an operator +/// cannot mistake one cause for the other. +fn envelope_refusal(path: &str, e: &EnvelopeError) -> String { + match e { + EnvelopeError::Shape(why) => format!( + "REFUSING to open the pool wallet: {path} is not an encrypted wallet file this build \ + can read ({why}).\n\ + This is NOT about your passphrase. The file's structure is checked before any \ + passphrase is tried, so a different passphrase would not change this.\n\ + Nothing has been written to it and no new wallet has been created.\n\ + What to do: check that {path} really is the pool's key file and not another file \ + that ended up at that path, then restore it from your backup. Do not delete it \ + first: nothing here says the key inside is gone." + ), + EnvelopeError::Undecryptable => format!( + "REFUSING to open the pool wallet: {path} did not open.\n\ + This is EITHER the wrong passphrase OR a damaged file, and there is no way to tell \ + which: the check that failed is the same one in both cases. Do not act as though you \ + knew which it was.\n\ + What to do, in this order:\n\ + 1. Check the passphrase. It is taken from {WALLET_PASSWORD_ENV} when that is set, \ + otherwise from the file named in {WALLET_PASSWORD_FILE_ENV}. Look for a trailing \ + space, a different keyboard layout, or a service unit still exporting an old value.\n\ + 2. Only once you are certain the passphrase is right, restore {path} from your \ + backup.\n\ + DO NOT delete, move or overwrite {path}, and do not let the pool create a new wallet \ + in its place. If the passphrase is simply wrong, that file still holds the key to \ + every coin this pool has mined; a new wallet is a new address, and the old money \ + would be out of reach for good. Nothing has been written to it.", + ), + EnvelopeError::Content(why) => format!( + "REFUSING to open the pool wallet: {path} decrypted correctly, so the passphrase is \ + right and the file is intact, but {why}.\n\ + What to do: this is not a file this pool wrote. Check that {path} is the right file \ + and restore it from your backup. Nothing has been written to it and no new wallet \ + has been created; do not delete it." + ), + } +} + +/// True the FIRST time this (tag, path) pair comes up in this process. The +/// settlement loop reloads the wallet on every cycle, so once-per-wallet work +/// and warnings must not repeat with it. +fn first_time_for(tag: &str, path: &str) -> bool { + static SEEN: LazyLock>> = LazyLock::new(|| Mutex::new(HashSet::new())); + let mut seen = SEEN.lock().unwrap_or_else(|e| e.into_inner()); + seen.insert(format!("{tag}:{path}")) +} + +/// Say plainly what an unencrypted key file costs. Printed once per path per +/// process so it cannot be lost in the settlement loop's output. +fn warn_plaintext_wallet(path: &str) { + if !first_time_for("plaintext-warned", path) { + return; + } + eprintln!( + "[wallet] WARNING: {path} holds the pool's private key in PLAINTEXT. Anything that can\n\ + [wallet] read those bytes - a backup, a VSS or disk snapshot, an old drive - can spend\n\ + [wallet] every coin the pool holds. Set {WALLET_PASSWORD_ENV} (or {WALLET_PASSWORD_FILE_ENV})\n\ + [wallet] and restart: the file is then re-written encrypted (Argon2id + AES-256-GCM)." + ); +} + +/// Load the pool wallet, or print the refusal and stop. +/// +/// Kept for callers that have no way to report a failure of their own; prefer +/// [`try_load_or_create_wallet`], which hands the same text back for the caller +/// to print in its own house style. A refusal exits with status 2, the same +/// status the server uses for a configuration it will not start on, rather than +/// panicking: a panic buries the one line that says what to fix under a +/// backtrace note, in a journal, at 3am, on a machine nobody is logged in to. +pub fn load_or_create_wallet(path: &str) -> Account { + match try_load_or_create_wallet(path) { + Ok(acc) => acc, + Err(why) => { + eprintln!("{why}"); + std::process::exit(2) + } + } +} + +/// Load the pool wallet from `path`, creating a fresh random one if, and only +/// if, there is no file there at all. The file holds either a 64-hex secp256k1 +/// private key or, when a passphrase is configured, an encrypted envelope. The +/// private key only ever lives in that file: it is never printed or logged, and +/// no refusal below names anything but the FILE that failed. +/// +/// Every failure is an `Err` an operator can act on, and NO failure creates, +/// truncates or overwrites `path`. That is the money-critical property: the file +/// on disk is the only copy of the key to the address the pool's income lands +/// in, so a pool that "recovered" from a mistyped passphrase by making a new +/// wallet would abandon every coin in the old one and start owing miners from an +/// address with nothing in it. +pub fn try_load_or_create_wallet(path: &str) -> Result { + // Resolve the passphrase FIRST, before anything opens the key file, so a + // passphrase that is missing, empty or too short is refused with the wallet + // untouched. Resolved once, and passed down, so the settlement loop does not + // re-read the passphrase file on every cycle either. + let pass = wallet_password()?; + load_or_create_wallet_with(path, pass.as_ref().map(|p| p.as_str())) +} + +/// The wallet loader with the passphrase already resolved, so the whole of it can +/// be exercised without this process's environment. +fn load_or_create_wallet_with(path: &str, pass: Option<&str>) -> Result { + // A directory at the wallet path reads back as a different OS error on every + // platform (and as "permission denied" on Windows), so name it here rather + // than leave an operator chasing an ACL that is not the problem. + if std::path::Path::new(path).is_dir() { + return Err(format!( + "REFUSING to open the pool wallet: {path} is a directory, not a wallet file.\n\ + Nothing was created inside it and nothing was written.\n\ + What to do: pass the path of the key FILE as . If you meant a file of \ + that name inside a folder, spell out the whole path, for example \ + {path}{sep}pool-wallet.key", + sep = std::path::MAIN_SEPARATOR, + )); + } + match std::fs::read_to_string(path) { + Ok(txt) => { + let acc = account_from_wallet_file(path, &txt, pass)?; + // Re-apply and re-verify the owner-only permissions on the LOAD path + // too, not only at creation: a key that lost its ACL (restored from a + // backup, copied by hand) must not keep serving funds. + secure_existing_key_file(path)?; + println!("pool wallet {} (from {path})", acc.readable()); + Ok(acc) + } + // The ONE branch that may write a key: there is no file here at all. + // Nothing above can reach it, so no failure to read or decrypt an + // existing wallet can ever fall through into creating a new one. + Err(e) if e.kind() == std::io::ErrorKind::NotFound => create_wallet(path, pass), + // Never generate-and-overwrite on any other error: a locked or + // transiently-unreadable key file must not be silently replaced. + Err(e) => Err(unreadable_wallet_refusal(path, &e)), + } +} + +/// The refusal for a wallet file that exists but could not be read. +/// +/// Split out from the loader so each cause can be tested, and so each one gets +/// the fix that actually applies to it. +fn unreadable_wallet_refusal(path: &str, e: &std::io::Error) -> String { + let (why, fix) = match e.kind() { + std::io::ErrorKind::PermissionDenied => ( + "this account is not allowed to read it".to_string(), + format!( + "run the pool as the account that owns {path}, or give that account read access \ + to it (Windows: icacls, Linux: chown then chmod 600). The pool locks this file \ + down to one account on purpose, so the usual cause is that the service now runs \ + as somebody else." + ), + ), + std::io::ErrorKind::InvalidData => ( + "it is not text".to_string(), + format!( + "a wallet file is either 64 hex characters or the JSON envelope this pool writes, \ + and both are plain ASCII. Check that {path} is really the pool's key file and \ + not some other file that ended up at that path, then restore it from your backup." + ), + ), + _ => ( + format!("the operating system said: {e}"), + format!( + "make {path} readable by the account this pool runs as, then start it again. If \ + the file is on a network or removable drive, check that the drive is mounted." + ), + ), + }; + format!( + "REFUSING to open the pool wallet: {path} exists but cannot be read ({why}).\n\ + No new wallet has been created in its place, and nothing has been written to it. A new \ + wallet would be a new address, and every coin already mined into this one would be left \ + behind with no way back.\n\ + What to do: {fix}" + ) +} + +/// Create the pool's wallet. The ONE path in this file that writes a key. +fn create_wallet(path: &str, pass: Option<&str>) -> Result { + let acc = new_random_account()?; + let key_hex = Zeroizing::new(hex::encode(acc.secret_key().serialize())); + let body = match pass { + Some(p) => Zeroizing::new(encrypt_key_hex(&key_hex, p).map_err(|e| { + format!( + "REFUSING to create the pool wallet: the new key could not be encrypted ({e}).\n\ + Nothing has been written to {path}.\n\ + What to do: this is the machine's own crypto or randomness failing, not your \ + configuration. Try again; if it repeats, the pool must not run here, because a \ + wallet it cannot protect is a wallet anyone with the disk can empty." + ) + })?), + None => Zeroizing::new(key_hex.to_string()), + }; + if let Err(e) = write_key_file(path, &body) { + // A key we could not protect must never be left lying around. This runs + // only here, where a moment ago there was no file at that path, so + // nothing an operator already had is removed. + let _ = std::fs::remove_file(path); + return Err(format!( + "REFUSING to run with an unprotected pool wallet: {path} could not be written and \ + locked down to this account only ({e}).\n\ + The key that was generated has been discarded, so nothing was left half-written.\n\ + What to do: check that the folder holding {path} exists and that this account may \ + create files in it, then start again. That file is the private key to every coin the \ + pool will hold, so the pool will not settle for leaving it readable by others." + )); + } + println!("CREATED A NEW POOL WALLET -> {path}"); + println!(" address: {}", acc.readable()); + if pass.is_some() { + println!(" the file is ENCRYPTED with {WALLET_PASSWORD_ENV}."); + println!(" BACK UP THAT FILE **AND** THAT PASSPHRASE: neither one alone can spend,"); + println!(" and losing either one loses the pool's funds for good."); + } else { + println!(" BACK UP THAT FILE. Whoever holds it controls the pool's funds."); + warn_plaintext_wallet(path); + } + Ok(acc) +} + +/// A fresh random account, or a refusal. +/// +/// Bounded, so a broken generator that keeps returning a value the curve rejects +/// is a refusal rather than a daemon spinning at 100% forever with nothing in the +/// log. A wallet is only ever as good as the randomness behind it, so a failing +/// RNG must stop the pool rather than be retried around. +fn new_random_account() -> Result { + for _ in 0..64 { + let mut key = Zeroizing::new([0u8; 32]); + if let Err(e) = getrandom::fill(&mut *key) { + return Err(format!( + "REFUSING to create the pool wallet: this machine's random number generator \ + failed ({e}). Nothing has been written.\n\ + What to do: a key made without real randomness could be guessed and the wallet \ + emptied, so the pool will not make one. Fix the system entropy source (on Linux \ + that is getrandom on /dev/urandom) and start again." + )); + } + if let Ok(a) = Account::create_by_secret_key_value(*key) { + return Ok(a); + } + } + Err( + "REFUSING to create the pool wallet: 64 random keys in a row were all rejected by the \ + curve, which cannot happen with a working random number generator. Nothing has been \ + written.\n\ + What to do: this machine's randomness is broken. Fix it and start again." + .to_string(), + ) +} + +/// Turn 64 hex characters into an Account, and refuse anything else. +/// +/// Never `Account::create_by`: when its argument is not exactly 64 hex +/// characters that function falls back to treating the text as a PASSPHRASE and +/// returns the account of its sha2. A key file with one altered character would +/// then load a completely different wallet, and the pool would quietly mine into +/// an address whose key nobody holds while the operator's real funds sat in a +/// file it had stopped using. A refusal is recoverable; a wrong wallet is not. +/// +/// The reason it returns is a fragment for the caller's message and never +/// contains any part of the key. +fn account_from_key_hex(key_hex: &str) -> Result { + let raw = Zeroizing::new(hex::decode(key_hex).map_err(|_| "it is not hexadecimal")?); + if raw.len() != 32 { + return Err(format!( + "it is {} bytes of hex and a private key is 32", + raw.len() + )); + } + let mut key = Zeroizing::new([0u8; 32]); + key.copy_from_slice(&raw); + // The underlying reason is dropped on purpose: nothing derived from key + // material goes into a message. + Account::create_by_secret_key_value(*key) + .map_err(|_| "it is not a private key this curve accepts".to_string()) +} + +/// Turn the wallet file's contents into an Account, transparently handling both +/// the encrypted envelope and the legacy plaintext-hex form. +/// +/// Reads only. Every refusal here leaves `path` exactly as it was found. +fn account_from_wallet_file(path: &str, txt: &str, pass: Option<&str>) -> Result { + let body = txt.trim(); + if body.is_empty() { + return Err(format!( + "REFUSING to open the pool wallet: {path} is empty, so it holds no private key.\n\ + The pool has NOT written a new key into it. It creates a wallet only when there is \ + no file at all, because a new wallet is a new address and any coins mined into the \ + old one would be left behind.\n\ + What to do: if this file is meant to be your pool wallet, restore it from your \ + backup. If this really is a first run and the empty file was made by accident (a \ + shell redirect, an editor), delete the empty file and start the pool again." + )); + } + // A JSON envelope is the encrypted form; anything else is the legacy + // plaintext key. + if body.starts_with('{') { + let Some(pass) = pass else { + return Err(format!( + "REFUSING to open the pool wallet: {path} is encrypted and no passphrase is \ + configured.\n\ + Nothing has been written to it and no new wallet has been created.\n\ + What to do: set {WALLET_PASSWORD_ENV} to the passphrase this wallet was created \ + with, or put that passphrase in a file of its own and name that file in \ + {WALLET_PASSWORD_FILE_ENV}. Then start again. Do NOT delete or move {path}: it \ + holds the only key to the address the pool's income is paid into." + )); + }; + let key_hex = decrypt_key_hex(body, pass).map_err(|e| envelope_refusal(path, &e))?; + // The tag verified, so the passphrase is right and the bytes are intact: + // anything wrong from here is the CONTENT, which is a different thing to + // tell the operator. + return account_from_key_hex(key_hex.trim()) + .map_err(|why| envelope_refusal(path, &EnvelopeError::Content(why))); + } + if body.len() != WALLET_KEY_HEX_LEN { + let shape = if body.len() < WALLET_KEY_HEX_LEN { + "a truncated copy: a transfer that stopped early, or a backup taken while the file was \ + still being written" + } else { + "a file with something extra in it: a second key, a note, or two files run together" + }; + // Said only when it can help. A passphrase being set is exactly when an + // operator expects to be looking at an encrypted file. + let hint = if pass.is_some() { + "\nA passphrase IS configured, so if you expected this wallet to be encrypted: the \ + encrypted form is JSON and starts with `{`. This file does not, so it is not that." + } else { + "" + }; + return Err(format!( + "REFUSING to open the pool wallet: {path} holds {n} characters. An unencrypted wallet \ + file is exactly {WALLET_KEY_HEX_LEN} (a 32-byte private key written in hex), and an \ + encrypted one is JSON starting with `{{`.\n\ + That looks like {shape}.{hint}\n\ + Nothing has been written to it and no new wallet has been created.\n\ + What to do: restore {path} from your backup. Keep the file you have until the \ + restored one opens: deleting it is the one step that cannot be undone.", + n = body.len() + )); + } + let acc = account_from_key_hex(body).map_err(|why| { + format!( + "REFUSING to open the pool wallet: {path} holds {WALLET_KEY_HEX_LEN} characters, but \ + they are not a private key ({why}).\n\ + Nothing has been written to it and no new wallet has been created, and this pool will \ + not guess: text that is not a key can be turned into SOME wallet, and it would not be \ + yours.\n\ + What to do: restore {path} from your backup. Do not delete the file you have." + ) + })?; + // Plaintext on disk: move it into an encrypted envelope as soon as a + // passphrase is configured, otherwise say plainly what is at risk. + match pass { + Some(pass) => migrate_key_file_to_encrypted(path, body, pass)?, + None => warn_plaintext_wallet(path), + } + Ok(acc) +} + +/// Re-write a legacy plaintext key file as an encrypted envelope. The envelope +/// is decrypted back and compared BEFORE it replaces the only copy of the key, +/// so a bad envelope can never cost the pool its wallet. +/// +/// Failing to encrypt is a warning, not a refusal: the pool ran yesterday with +/// that plaintext file and can run today. Failing to WRITE what was already +/// verified is a refusal, because at that point the state of the file on disk is +/// the thing in doubt. +fn migrate_key_file_to_encrypted(path: &str, key_hex: &str, pass: &str) -> Result<(), String> { + let body = match encrypt_key_hex(key_hex, pass) { + Ok(b) => Zeroizing::new(b), + Err(e) => { + eprintln!("[wallet] WARNING: could not encrypt {path} ({e}); it stays plaintext."); + return Ok(()); + } + }; + match decrypt_key_hex(&body, pass) { + Ok(back) if back.trim().eq_ignore_ascii_case(key_hex) => {} + _ => { + eprintln!("[wallet] WARNING: the encrypted form of {path} did not verify; leaving it as-is."); + return Ok(()); + } + } + // Only now does anything replace the file, and what replaces it has already + // been decrypted back to the very key that is in it. + if let Err(e) = write_key_file(path, &body) { + return Err(format!( + "REFUSING to run with an unprotected pool wallet: {path} could not be re-written \ + encrypted and owner-only ({e}).\n\ + The key controls ALL pool funds. Either the file is still the plaintext key it was, \ + or it is the encrypted form that was verified before it was written; in both cases \ + the key in it is yours and nothing was lost.\n\ + What to do: check that this account may write in that folder and that no backup tool \ + is holding the file open, then start again. To carry on without encryption for now, \ + unset {WALLET_PASSWORD_ENV} and {WALLET_PASSWORD_FILE_ENV}, knowing that the key is \ + then plaintext on this disk." + )); + } + println!("[wallet] {path} is now ENCRYPTED with {WALLET_PASSWORD_ENV}."); + println!("[wallet] KEEP THAT PASSPHRASE: without it the file cannot be decrypted and the"); + println!("[wallet] pool's funds are unrecoverable. The previous plaintext copy may still"); + println!("[wallet] exist in backups and snapshots - treat those as sensitive."); + Ok(()) +} + +/// Re-apply and verify the key file's owner-only permissions, once per path per +/// process. Settlement reloads the wallet every cycle, and spawning icacls each +/// time would be pointless work that a transient hiccup could turn into a +/// skipped payout. +fn secure_existing_key_file(path: &str) -> Result<(), String> { + if !first_time_for("secured", path) { + return Ok(()); + } + restrict_key_file_permissions(path).map_err(|e| { + format!( + "REFUSING to run with an unprotected pool wallet: the owner-only permissions on \ + {path} could not be applied and verified ({e}).\n\ + The contents of the wallet were not changed.\n\ + What to do: that file is the private key to every coin the pool holds, so it must not \ + be readable by other accounts on this machine. Make sure this account owns it \ + (Windows: icacls {path} /inheritance:r /grant:r \"%USERNAME%\":F, Linux: chown then \ + chmod 600), then start again." + ) + }) +} + +/// Write the wallet file owner-only via a temp file + atomic rename, so a +/// concurrent reader never sees an empty or half-written key. The file controls +/// ALL pool funds, so securing it is MANDATORY: if the owner-only permissions +/// cannot be applied AND verified this returns Err, and the caller must not keep +/// running with an unprotected key. +/// +/// Every failure removes the temp file before it returns. That file holds the +/// private key, and a refusal must not leave a copy of it lying next to the +/// wallet under a name nobody will think to look at. +fn write_key_file(path: &str, body: &str) -> std::io::Result<()> { + let tmp = format!("{path}.tmp.{}", std::process::id()); + if let Err(e) = fill_key_tmp(&tmp, body) { + let _ = std::fs::remove_file(&tmp); + return Err(e); + } + if let Err(e) = std::fs::rename(&tmp, path) { + let _ = std::fs::remove_file(&tmp); + return Err(e); + } + restrict_key_file_permissions(path) +} + +/// Create the temp file, harden it, then put the key in it. +fn fill_key_tmp(tmp: &str, body: &str) -> std::io::Result<()> { + use std::io::Write; + let mut opts = std::fs::OpenOptions::new(); + opts.write(true).create(true).truncate(true); + #[cfg(unix)] + { + use std::os::unix::fs::OpenOptionsExt; + opts.mode(0o600); + } + let mut f = opts.open(tmp)?; + // Harden the (still empty) temp file BEFORE the secret reaches it. On + // Windows it is created with the directory's inherited ACL and NTFS carries + // that ACL across the rename, so hardening only the final path would leave a + // window in which the key is readable by other accounts. + #[cfg(windows)] + restrict_key_file_permissions(tmp)?; + writeln!(f, "{body}")?; + let _ = f.sync_all(); + Ok(()) +} + +/// Lock the wallet key down to the current user only, and prove it worked. +/// On Windows the default ACL is inherited and readable by other local accounts; +/// without this the key controlling the pool balance is exposed to any local +/// user or process. Every failure is fatal to the caller: "could not secure the +/// private key" is not a warning for a daemon that moves real money. +fn restrict_key_file_permissions(path: &str) -> std::io::Result<()> { + #[cfg(windows)] + { + // Resolve the principal from the process token, NEVER from USERNAME: + // that variable is empty under a service or a scheduled task, and + // `/inheritance:r` with no matching `/grant` leaves an EMPTY DACL that + // locks the pool out of its own wallet on the very next start. + let (name, sid) = windows_current_principal()?; + let out = std::process::Command::new("icacls") + .arg(path) + .arg("/inheritance:r") + .arg("/grant:r") + .arg(format!("*{sid}:F")) + .output()?; + if !out.status.success() { + return Err(std::io::Error::other(format!( + "icacls failed: {}", + String::from_utf8_lossy(&out.stderr).trim() + ))); + } + // The owner must still be able to READ the key, or every later start + // dies on "cannot read wallet file". + drop(std::fs::File::open(path)?); + windows_verify_owner_only(path, &name, &sid)?; + } + #[cfg(unix)] + { + use std::os::unix::fs::PermissionsExt; + std::fs::set_permissions(path, std::fs::Permissions::from_mode(0o600))?; + let mode = std::fs::metadata(path)?.permissions().mode() & 0o777; + if mode != 0o600 { + return Err(std::io::Error::other(format!( + "wallet file mode is {mode:o}, expected 600" + ))); + } + } + #[cfg(not(any(unix, windows)))] + { + let _ = path; + return Err(std::io::Error::other( + "cannot restrict wallet file permissions on this platform", + )); + } + Ok(()) +} + +/// The current account's `DOMAIN\name` and SID, read from the process token via +/// `whoami /user`. Granting by SID keeps the ACL correct even where the display +/// name is ambiguous, and it never depends on the USERNAME variable. +#[cfg(windows)] +fn windows_current_principal() -> std::io::Result<(String, String)> { + let out = std::process::Command::new("whoami") + .args(["/user", "/fo", "csv", "/nh"]) + .output()?; + if !out.status.success() { + return Err(std::io::Error::other("`whoami /user` failed")); + } + let txt = String::from_utf8_lossy(&out.stdout); + let line = txt.lines().find(|l| !l.trim().is_empty()).unwrap_or(""); + // CSV is `"DOMAIN\user","S-1-5-..."`; a Windows account name cannot contain + // a comma, so a plain split is safe. + let mut cols = line.split(',').map(|c| c.trim().trim_matches('"')); + let name = cols.next().unwrap_or("").to_string(); + let sid = cols.next().unwrap_or("").to_string(); + if name.is_empty() || !sid.starts_with("S-1-") { + return Err(std::io::Error::other( + "could not resolve the current account SID from `whoami /user`", + )); + } + Ok((name, sid)) +} + +/// Principals it is not meaningful to lock a file against on Windows: LocalSystem +/// and the local Administrators group. +/// +/// Excluding these buys nothing. Anything running as either can take ownership of +/// the file, read this process's memory, or load a driver, so a key they cannot +/// read through the DACL is a key they can read another way five seconds later. +/// What the DACL genuinely protects against is OTHER ordinary accounts on a +/// shared machine, and that protection is unaffected by leaving these two in. +/// +/// Refusing them instead made the pool fail CLOSED on machines where the OS keeps +/// its own ACE, which for a pool means it will not write its wallet and therefore +/// pays nobody. Availability lost, security unchanged. +#[cfg(windows)] +const WINDOWS_OS_PRINCIPAL_SIDS: [&str; 2] = [ + "S-1-5-18", // NT AUTHORITY\SYSTEM + "S-1-5-32-544", // BUILTIN\Administrators +]; + +/// Resolve an account name printed by icacls to its SID. +/// +/// By SID and not by name on purpose: icacls prints LOCALISED names, so matching +/// the strings "NT AUTHORITY\SYSTEM" or "BUILTIN\Administrators" would silently +/// stop working on a German or Greek Windows and start rejecting the very +/// principals this is meant to accept. +#[cfg(windows)] +fn windows_sid_of(principal: &str) -> Option { + let script = format!( + "try {{ ([System.Security.Principal.NTAccount]'{}')\ + .Translate([System.Security.Principal.SecurityIdentifier]).Value }} catch {{ '' }}", + principal.replace('\'', "''") + ); + let out = std::process::Command::new("powershell") + .args(["-NoProfile", "-NonInteractive", "-Command", &script]) + .output() + .ok()?; + let sid = String::from_utf8_lossy(&out.stdout).trim().to_string(); + sid.starts_with("S-1-").then_some(sid) +} + +/// Read the DACL back and refuse to continue if any principal other than the +/// current account, or the operating system itself, is listed. Parsing icacls +/// output is best-effort, so an unreadable listing only warns: the mandatory +/// checks in the caller (icacls exit status plus the file still being readable) +/// already rule out the empty-DACL and silent-failure cases this guards against. +#[cfg(windows)] +fn windows_verify_owner_only(path: &str, name: &str, sid: &str) -> std::io::Result<()> { + let unverified = |why: &str| { + eprintln!("[wallet] WARNING: could not verify the ACL of {path} ({why}); check it manually."); + }; + let Ok(out) = std::process::Command::new("icacls").arg(path).output() else { + unverified("icacls did not run"); + return Ok(()); + }; + if !out.status.success() { + unverified("icacls reported an error"); + return Ok(()); + } + let txt = String::from_utf8_lossy(&out.stdout); + let mut aces = 0usize; + for (i, raw) in txt.lines().enumerate() { + if raw.trim().is_empty() { + break; // a blank line ends the ACE list + } + let entry = if i == 0 { + // The first line echoes the path we passed, then the first ACE. + match raw.trim_start().strip_prefix(path) { + Some(rest) => rest.trim(), + None => { + unverified("unexpected output layout"); + return Ok(()); + } + } + } else { + raw.trim() + }; + let Some((principal, _)) = entry.split_once(":(") else { + continue; + }; + aces += 1; + if !principal.eq_ignore_ascii_case(name) && !principal.eq_ignore_ascii_case(sid) { + if let Some(resolved) = windows_sid_of(principal) { + if WINDOWS_OS_PRINCIPAL_SIDS.contains(&resolved.as_str()) { + // Said out loud rather than passed over silently: the + // operator should know exactly who else can read the key. + eprintln!( + "[wallet] NOTE: {path} is also readable by `{principal}` ({resolved}). \ + That is the operating system itself and cannot be excluded; anything \ + able to act as it already controls this machine. No other account can \ + read the file." + ); + continue; + } + } + return Err(std::io::Error::other(format!( + "{path} is still accessible to `{principal}`" + ))); + } + } + if aces == 0 { + unverified("no access entries parsed"); + } + Ok(()) +} + +/// The transaction set the NODE has packed for the height a template extends, +/// carried exactly as the node serialized it. +/// +/// The whole point of a pool is that it pays its OWN wallet, so it cannot reuse +/// the node's coinbase. It does not have to. `mrklrts` is the node's merkle +/// "prelude modify" list: the sibling hashes on the path from transaction 0 up +/// to the root. Not one of those siblings is derived from transaction 0 (at +/// every level the list takes element 1, which covers original leaves 2 and +/// above), so folding a DIFFERENT coinbase hash through the same list yields +/// exactly the merkle root of "our coinbase + the node's transactions". That is +/// the same arithmetic the node's own `miner_success` performs, and it is what +/// lets this pool keep the node's transactions while still paying itself. +/// +/// `bodies` EXCLUDES the node's coinbase: slot 0 of the block is always ours. +#[derive(Clone, Default, Debug)] +pub struct PackedTxs { + /// Serialized transaction bodies, in the node's packing order, starting at + /// the node's transaction 1. Kept as raw bytes on purpose: a block is + /// serialized as `intro || tx0 || tx1 || ...` with the count living in the + /// intro, so these can be appended verbatim. The pool therefore never has to + /// decode a transaction, and cannot drop one whose type it does not know. + pub bodies: Vec>, + /// The node's merkle prelude modify list for this transaction set. + pub mrklrts: Vec, +} + +impl PackedTxs { + /// Transactions in the block this set produces, coinbase included. + pub fn block_tx_count(&self) -> u32 { + // A block can hold at most `max_block_txs` (1000 by default), so this + // cannot truncate in practice; saturate rather than wrap if it ever did. + self.bodies.len().saturating_add(1).min(u32::MAX as usize) as u32 + } +} + +/// Everything the pool needs to build and verify blocks for the current tip. +/// The pool serves one template to all workers; each worker gets its own +/// extranonce (the coinbase `miner_nonce`), which changes the merkle root and +/// therefore gives every worker a private search space. +#[derive(Clone)] +pub struct Template { + pub height: u64, + pub prevhash: Hash, + pub timestamp: u64, + /// Header `difficulty` field (u32) — must equal what the node recomputes. + pub difficulty: u32, + /// The exact PoW target for this block. NOT interchangeable with + /// u32_to_hash(difficulty): on the from_big path it is more precise. + pub target: [u8; 32], + pub coinbase_addr: Address, + /// The transactions the node packed for this height, empty when the node + /// would not tell us. Behind an `Arc` because the pool clones a whole + /// template on every single share submission while holding its global lock, + /// and a full block of transaction bodies is up to a megabyte. + pub txs: Arc, +} + +/// Read the chain tip and build a template for the next block, computing the +/// next difficulty off-node with the same rule the node will validate against. +/// +/// Returns `None` on any transient node/HTTP problem instead of panicking, so a +/// caller holding a lock (the pool server) can skip the cycle and retry rather +/// than poisoning its mutex and taking the whole pool down. +pub fn fetch_template( + client: &reqwest::blocking::Client, + base: &str, + coinbase_addr: &str, + params: &ChainParams, +) -> Option