diff --git a/bottleneck-cluster/README.md b/bottleneck-cluster/README.md new file mode 100644 index 0000000..c5b46e4 --- /dev/null +++ b/bottleneck-cluster/README.md @@ -0,0 +1,72 @@ +# Does Bottleneck's rate limit hold across Node `cluster` workers? + +This repo's own rate limiter had a bug: each process enforced the delay +independently, so the limiter reported a perfect interval while the host on the +other end received every request at once. The fix was an atomic claim in +MongoDB, and the test that caught it asserts on **what the server received** +rather than on what the limiter reported. + +That test generalises. This directory points the same idea at +[Bottleneck](https://github.com/SGrondin/bottleneck), which is the most widely +used rate limiter on npm. + +## Result + +Bottleneck's default is `datastore: "local"`, and `LocalDatastore` keeps +`_nextRequest` as an instance field. There is no IPC in the library: every +occurrence of "cluster" in its source refers to *Redis* Cluster, not Node's +`cluster` module. So under `cluster` with default options, each worker holds an +independent limiter. + +Configured `minTime: 500` (2 requests/second), 6 workers, 6 requests each, +measured at the server: + +| arm | min gap | gaps under minTime | rate | limiter's self-report | +|---|---|---|---|---| +| 1 worker, `local` (control) | 487.6 ms | 0/5 | 2.01/s | 498.4 ms | +| **6 workers, `local` (default)** | **0.1 ms** | **30/35** | **14.04/s** | **495.8 ms** | +| 6 workers, `ioredis` (control) | 483.7 ms | 0/35 | 2.00/s | n/a | + +The aggregate rate tracks the worker count: + +| workers | rate | vs intended | +|---|---|---| +| 1 | 2.01/s | 1.00x | +| 2 | 4.42/s | 2.21x | +| 4 | 9.22/s | 4.61x | +| 6 | 14.04/s | 7.02x | +| 8 | 18.61/s | 9.30x | + +**This is not a bug in Bottleneck.** `local` means local, and clustering via +Redis is documented. The finding is that the default configuration degrades +silently under `cluster`: the limiter's own self-report stays at ~497 ms in +every arm, including the ones where the server is being hit 7x too fast. + +Both controls matter. The 1-worker arm shows the harness measures spacing +correctly when spacing is happening; the `ioredis` arm shows it passes a +correctly coordinated limiter. Without them, "0.1 ms" is indistinguishable from +a broken test. + +## Run it + +```bash +npm install +npm run control # 1 worker, local -> expect 0 violations +npm test # 6 workers, local -> expect ~30/35 violations +docker run -d --rm -p 6379:6379 redis:7-alpine +npm run redis # 6 workers, ioredis -> expect 0 violations +``` + +Any arm: `node harness.js --workers N --datastore local|ioredis --min-time MS --reqs K` + +## Method + +The primary process starts an HTTP server on a random port and records +`process.hrtime.bigint()` for every arrival. It forks N workers; each builds one +`Bottleneck` with `minTime` and `maxConcurrent: 1`, and schedules K GETs through +it. Each worker also reports the gaps between its *own* job starts, which is the +limiter's view, so the two can be compared in the same run. + +Measured on Node v24.11.1, Bottleneck 2.19.5, macOS. Numbers above are from +three repeats of each arm; the `local` 6-worker arm gave 30, 30 and 31 +violations out of 35. diff --git a/bottleneck-cluster/harness.js b/bottleneck-cluster/harness.js new file mode 100644 index 0000000..c2a1fc6 --- /dev/null +++ b/bottleneck-cluster/harness.js @@ -0,0 +1,93 @@ +// Does Bottleneck's rate limit hold across Node `cluster` workers? +// +// Measured from the SERVER's side: a real HTTP server records the arrival time +// of every request, and the verdict is the smallest gap between consecutive +// arrivals. The limiter's own opinion is recorded separately, so the two can be +// compared. +// +// node harness.js --workers 6 --datastore local --min-time 500 --reqs 5 +// +// Arms: 1 worker (control, must pass), N workers local, N workers ioredis. +const cluster = require('node:cluster'); +const http = require('node:http'); + +const arg = (n, d) => { const i = process.argv.indexOf('--' + n); return i === -1 ? d : process.argv[i + 1]; }; +const WORKERS = Number(arg('workers', 6)); +const DATASTORE = arg('datastore', 'local'); +const MIN_TIME = Number(arg('min-time', 500)); +const REQS = Number(arg('reqs', 5)); +const REDIS = arg('redis', '127.0.0.1:6379'); + +if (cluster.isPrimary) { + const arrivals = []; + const server = http.createServer((req, res) => { + arrivals.push({ t: process.hrtime.bigint(), w: req.headers['x-worker'] }); + res.writeHead(200); res.end('ok'); + }); + + server.listen(0, '127.0.0.1', () => { + const port = server.address().port; + const selfReports = []; + let done = 0; + + for (let i = 0; i < WORKERS; i++) { + const w = cluster.fork({ PORT: port, WID: String(i), DATASTORE, MIN_TIME, REQS, REDIS }); + w.on('message', (m) => selfReports.push(m)); + w.on('exit', () => { + if (++done === WORKERS) { server.close(); report(arrivals, selfReports); } + }); + } + }); + + function report(arrivals, selfReports) { + arrivals.sort((a, b) => (a.t < b.t ? -1 : 1)); + const gaps = []; + for (let i = 1; i < arrivals.length; i++) gaps.push(Number(arrivals[i].t - arrivals[i - 1].t) / 1e6); + const span = Number(arrivals.at(-1).t - arrivals[0].t) / 1e6; + const violations = gaps.filter((g) => g < MIN_TIME * 0.9).length; + + // What each worker's own limiter thought its spacing was. + const selfGaps = selfReports.flatMap((r) => r.gaps); + const selfMin = Math.min(...selfGaps); + + const out = { + arm: `${WORKERS} worker(s), datastore=${DATASTORE}`, + configured_min_time_ms: MIN_TIME, + requests: arrivals.length, + server_min_gap_ms: +Math.min(...gaps).toFixed(1), + server_median_gap_ms: +gaps.sort((a, b) => a - b)[Math.floor(gaps.length / 2)].toFixed(1), + gaps_under_min_time: `${violations}/${gaps.length}`, + effective_rate_per_sec: +((arrivals.length - 1) / (span / 1000)).toFixed(2), + intended_rate_per_sec: +(1000 / MIN_TIME).toFixed(2), + limiter_self_reported_min_gap_ms: +selfMin.toFixed(1), + }; + console.log(JSON.stringify(out, null, 2)); + } +} else { + const Bottleneck = require('bottleneck'); + const opts = { minTime: Number(process.env.MIN_TIME), maxConcurrent: 1 }; + if (process.env.DATASTORE === 'ioredis') { + const [host, port] = process.env.REDIS.split(':'); + Object.assign(opts, { datastore: 'ioredis', clearDatastore: false, + clientOptions: { host, port: Number(port) }, id: 'politeness-test' }); + } + const limiter = new Bottleneck(opts); + const starts = []; + const get = () => new Promise((resolve, reject) => { + const req = http.get({ host: '127.0.0.1', port: Number(process.env.PORT), path: '/p', + headers: { 'x-worker': process.env.WID } }, (res) => { res.resume(); res.on('end', resolve); }); + req.on('error', reject); + }); + + (async () => { + const n = Number(process.env.REQS); + await Promise.all(Array.from({ length: n }, () => limiter.schedule(() => { + starts.push(Number(process.hrtime.bigint()) / 1e6); + return get(); + }))); + const gaps = starts.sort((a, b) => a - b).slice(1).map((t, i) => t - starts[i]); + process.send({ gaps }); + await limiter.disconnect?.(); + process.exit(0); + })().catch((e) => { console.error('worker error', e.message); process.exit(1); }); +} diff --git a/bottleneck-cluster/package.json b/bottleneck-cluster/package.json new file mode 100644 index 0000000..5567b07 --- /dev/null +++ b/bottleneck-cluster/package.json @@ -0,0 +1,13 @@ +{ + "name": "bottleneck-cluster-harness", + "version": "1.0.0", + "private": true, + "description": "Does Bottleneck's rate limit hold across Node cluster workers? Measured from the server's side.", + "scripts": { + "control": "node harness.js --workers 1 --datastore local --min-time 500 --reqs 6", + "test": "node harness.js --workers 6 --datastore local --min-time 500 --reqs 6", + "redis": "node harness.js --workers 6 --datastore ioredis --min-time 500 --reqs 6" + }, + "dependencies": { "bottleneck": "^2.19.5", "ioredis": "^5.4.1" }, + "license": "MIT" +}