From 1d003350715dda02c812b6121bcf86fbebfdbe51 Mon Sep 17 00:00:00 2001 From: Albert Bausili Date: Sat, 26 Sep 2026 14:22:18 +0200 Subject: [PATCH 1/2] refactor(h1): delete the unused SIMD header scan and stop advertising a SIMD parser (celeris#424) findHeaderEnd (protocol/h1/findheader_*: SSE2/AVX2 on amd64, NEON on arm64, a scalar loop elsewhere) was the only assembly in the tree, and it has had no production caller since 96581bc (#359, 2026-06-17), which dropped the parser's upfront header-block scan as double work. #424, filed a month later, asked CI to exercise the arm64 path of a routine nothing calls. What remained was the routine, its own tests and benchmarks, and a README feature bullet ("SIMD HTTP parser -- SSE2 (amd64) and NEON (arm64) with a generic SWAR fallback") describing code that no request runs. At 9f4d89b, a search for the name followed by "(" in non-test Go files finds only the three declarations. The same query finds parser.go:118's live parseHeaders call, and at 96581bc~1 it finds the removed `if findHeaderEnd(remaining) < 0`. No go:linkname names it. This deletes the six findheader files, the five findHeaderEnd tests and two benchmarks, and a comment that named it. The README bullet now describes what the parser does: header and body slices alias the read buffer (the package doc's own claim). After the deletion, `go build ./...` and `go vet ./protocol/...` pass for linux/{amd64,arm64,386,riscv64} and darwin/{amd64,arm64}, which covers every architecture that had an implementation. `go test ./protocol/h1` passes 140/140. go.mod is unchanged (tidy -diff clean). Control: a reintroduced call fails to build ("undefined: findHeaderEnd") on amd64 and arm64. --- README.md | 2 +- protocol/h1/bench_test.go | 35 ------- protocol/h1/findheader.go | 1 - protocol/h1/findheader_amd64.go | 16 --- protocol/h1/findheader_amd64.s | 158 ------------------------------ protocol/h1/findheader_arm64.go | 6 -- protocol/h1/findheader_arm64.s | 85 ---------------- protocol/h1/findheader_generic.go | 16 --- protocol/h1/parser.go | 2 +- protocol/h1/parser_test.go | 113 --------------------- 10 files changed, 2 insertions(+), 432 deletions(-) delete mode 100644 protocol/h1/findheader.go delete mode 100644 protocol/h1/findheader_amd64.go delete mode 100644 protocol/h1/findheader_amd64.s delete mode 100644 protocol/h1/findheader_arm64.go delete mode 100644 protocol/h1/findheader_arm64.s delete mode 100644 protocol/h1/findheader_generic.go diff --git a/README.md b/README.md index 6162aa4d..a0fee093 100644 --- a/README.md +++ b/README.md @@ -32,7 +32,7 @@ An engine-correctness release, driven by what the [probatorium](https://github.c - **Edge-triggered epoll** — per-core event loops with CPU pinning. - **Adaptive meta-engine** — transplants between io_uring and epoll at runtime based on telemetry. - **First-party database drivers** — native [`driver/postgres`](driver/postgres), [`driver/redis`](driver/redis), and [`driver/memcached`](driver/memcached) run on the celeris event loop (see [Database drivers](#database-drivers)). -- **SIMD HTTP parser** — SSE2 (amd64) and NEON (arm64) with a generic SWAR fallback. +- **Zero-copy HTTP/1.1 parser** — header and body slices alias the connection's read buffer instead of being copied out of it. - **HTTP/2 cleartext (h2c)** — full stream multiplexing, flow control, HPACK, inline handler execution, zero-alloc HEADERS fast path. - **Auto-detect** — protocol negotiation from the first bytes on the wire. - **Error-returning handlers** — `HandlerFunc` returns `error`; structured `*HTTPError` carries status codes. diff --git a/protocol/h1/bench_test.go b/protocol/h1/bench_test.go index a7212ee2..0cf467d4 100644 --- a/protocol/h1/bench_test.go +++ b/protocol/h1/bench_test.go @@ -84,38 +84,3 @@ func BenchmarkParseRequest_Pipelined(b *testing.B) { } } } - -func BenchmarkFindHeaderEnd(b *testing.B) { - sizes := []int{64, 256, 1024, 4096} - for _, size := range sizes { - b.Run(fmt.Sprintf("size=%d", size), func(b *testing.B) { - // Place \r\n\r\n at the end - buf := make([]byte, size) - for i := range buf { - buf[i] = 'A' - } - copy(buf[size-4:], "\r\n\r\n") - b.SetBytes(int64(size)) - b.ReportAllocs() - b.ResetTimer() - for i := 0; i < b.N; i++ { - findHeaderEnd(buf) - } - }) - } -} - -func BenchmarkFindHeaderEnd_8K(b *testing.B) { - size := 8192 - buf := make([]byte, size) - for i := range buf { - buf[i] = 'A' - } - copy(buf[size-4:], "\r\n\r\n") - b.SetBytes(int64(size)) - b.ReportAllocs() - b.ResetTimer() - for i := 0; i < b.N; i++ { - findHeaderEnd(buf) - } -} diff --git a/protocol/h1/findheader.go b/protocol/h1/findheader.go deleted file mode 100644 index 8375a33d..00000000 --- a/protocol/h1/findheader.go +++ /dev/null @@ -1 +0,0 @@ -package h1 diff --git a/protocol/h1/findheader_amd64.go b/protocol/h1/findheader_amd64.go deleted file mode 100644 index ec1761fd..00000000 --- a/protocol/h1/findheader_amd64.go +++ /dev/null @@ -1,16 +0,0 @@ -//go:build amd64 - -package h1 - -import "golang.org/x/sys/cpu" - -// useAVX2 is set at init time. The assembly (findheader_amd64.s) reads this -// to branch between the AVX2 (32-byte) and SSE2 (16-byte) scan loops. -var useAVX2 bool //nolint:unused // referenced by assembly via ·useAVX2(SB) - -func init() { - useAVX2 = cpu.X86.HasAVX2 -} - -//go:noescape -func findHeaderEnd(buf []byte) int diff --git a/protocol/h1/findheader_amd64.s b/protocol/h1/findheader_amd64.s deleted file mode 100644 index 25543c71..00000000 --- a/protocol/h1/findheader_amd64.s +++ /dev/null @@ -1,158 +0,0 @@ -#include "textflag.h" - -// func findHeaderEnd(buf []byte) int -// -// Scans buf for the HTTP header terminator \r\n\r\n (0x0D 0x0A 0x0D 0x0A). -// Returns the index one past the terminator (i.e., the start of the body), -// or -1 if not found. -// -// At entry the function checks the package-level useAVX2 flag and dispatches -// to either a 32-byte (AVX2) or 16-byte (SSE2) vectorized loop. Both fall -// through to the same scalar tail. -TEXT ·findHeaderEnd(SB), NOSPLIT, $0-32 - MOVQ buf_base+0(FP), SI // SI = &buf[0] - MOVQ buf_len+8(FP), CX // CX = len(buf) - CMPQ CX, $4 - JL not_found - - XORQ DI, DI // DI = current offset - - // Check runtime AVX2 flag. - CMPB ·useAVX2(SB), $0 - JE sse2_init - -// ----------------------------------------------------------------------- -// AVX2 path — 32 bytes per iteration -// ----------------------------------------------------------------------- -avx2_init: - // Broadcast '\r' (0x0D) into Y0 (32 bytes). - MOVQ $0x0D0D0D0D0D0D0D0D, AX - MOVQ AX, X0 - VPBROADCASTQ X0, Y0 - - MOVQ CX, DX - SUBQ $35, DX // last safe AVX2 start = len - 32 - 3 - JL avx2_to_sse2 // buffer too small for even one AVX2 iteration - -avx2_loop: - CMPQ DI, DX - JG avx2_to_sse2 - - // Load 32 unaligned bytes, compare each byte against '\r'. - VMOVDQU (SI)(DI*1), Y1 - VPCMPEQB Y0, Y1, Y2 - VPMOVMSKB Y2, AX // AX = 32-bit mask of '\r' positions - TESTL AX, AX - JZ avx2_next - -avx2_check_bits: - BSFL AX, BX // BX = index of first set bit - LEAQ (DI)(BX*1), R8 // R8 = absolute position in buf - LEAQ 4(R8), R9 - CMPQ R9, CX - JG avx2_clear_bit - MOVL (SI)(R8*1), R10 - CMPL R10, $0x0A0D0A0D // \r\n\r\n in little-endian - JE avx2_found - BTRL BX, AX // clear this bit, check next '\r' - TESTL AX, AX - JNZ avx2_check_bits - -avx2_next: - ADDQ $32, DI - JMP avx2_loop - -avx2_clear_bit: - BTRL BX, AX - TESTL AX, AX - JNZ avx2_check_bits - ADDQ $32, DI - JMP avx2_loop - -avx2_found: - VZEROUPPER - LEAQ 4(R8), AX - MOVQ AX, ret+24(FP) - RET - - // Transition: remaining bytes too few for 32-byte loads. - // Fall through to SSE2 for 16-byte chunks, then scalar tail. -avx2_to_sse2: - VZEROUPPER - -// ----------------------------------------------------------------------- -// SSE2 path — 16 bytes per iteration -// ----------------------------------------------------------------------- -sse2_init: - // Broadcast '\r' (0x0D) into X0 (16 bytes). - MOVQ $0x0D0D0D0D0D0D0D0D, AX - MOVQ AX, X0 - PUNPCKLQDQ X0, X0 - - MOVQ CX, DX - SUBQ $19, DX // last safe SSE2 start = len - 16 - 3 - JL scalar_init - -sse2_loop: - CMPQ DI, DX - JG scalar_init - - MOVOU (SI)(DI*1), X1 - PCMPEQB X0, X1 - PMOVMSKB X1, AX - TESTL AX, AX - JZ sse2_next - -sse2_check_bits: - BSFL AX, BX - LEAQ (DI)(BX*1), R8 - LEAQ 4(R8), R9 - CMPQ R9, CX - JG sse2_clear_bit - MOVL (SI)(R8*1), R10 - CMPL R10, $0x0A0D0A0D - JE found_at_r8 - BTRL BX, AX - TESTL AX, AX - JNZ sse2_check_bits - -sse2_next: - ADDQ $16, DI - JMP sse2_loop - -sse2_clear_bit: - BTRL BX, AX - TESTL AX, AX - JNZ sse2_check_bits - ADDQ $16, DI - JMP sse2_loop - -// ----------------------------------------------------------------------- -// Scalar tail — one byte at a time for remaining bytes -// ----------------------------------------------------------------------- -scalar_init: - MOVQ CX, DX - SUBQ $3, DX - -scalar_loop: - CMPQ DI, DX - JGE not_found - MOVL (SI)(DI*1), AX - CMPL AX, $0x0A0D0A0D - JE scalar_found - INCQ DI - JMP scalar_loop - -scalar_found: - LEAQ 4(DI), AX - MOVQ AX, ret+24(FP) - RET - -found_at_r8: - LEAQ 4(R8), AX - MOVQ AX, ret+24(FP) - RET - -not_found: - MOVQ $-1, ret+24(FP) - RET diff --git a/protocol/h1/findheader_arm64.go b/protocol/h1/findheader_arm64.go deleted file mode 100644 index 5078c190..00000000 --- a/protocol/h1/findheader_arm64.go +++ /dev/null @@ -1,6 +0,0 @@ -//go:build arm64 - -package h1 - -//go:noescape -func findHeaderEnd(buf []byte) int diff --git a/protocol/h1/findheader_arm64.s b/protocol/h1/findheader_arm64.s deleted file mode 100644 index e63c6f07..00000000 --- a/protocol/h1/findheader_arm64.s +++ /dev/null @@ -1,85 +0,0 @@ -#include "textflag.h" - -// func findHeaderEnd(buf []byte) int -TEXT ·findHeaderEnd(SB), NOSPLIT, $0-32 - MOVD buf_base+0(FP), R0 - MOVD buf_len+8(FP), R1 - CMP $4, R1 - BLT not_found - - MOVD $0x0D0D0D0D0D0D0D0D, R2 - VDUP R2, V0.D2 - - MOVD $0, R3 - - MOVD R1, R4 - SUB $19, R4, R4 - - CMP R4, R3 - BGT scalar_init - -simd_loop: - ADD R0, R3, R6 - VLD1 (R6), [V1.B16] - VCMEQ V0.B16, V1.B16, V2.B16 - VUADDLV V2.B16, V3 - VMOV V3.D[0], R5 - CBZ R5, simd_next - - MOVD R3, R7 - ADD $16, R3, R8 - SUB $3, R1, R9 - CMP R9, R8 - BLE block_scan - MOVD R9, R8 - -block_scan: - CMP R8, R7 - BGE simd_next_from_r3 - ADD R0, R7, R6 - MOVWU (R6), R10 - MOVD $0x0A0D0A0D, R11 // 0x0A0D0A0D = \r\n\r\n in little-endian byte order - CMP R11, R10 - BEQ found_at_r7 - ADD $1, R7 - B block_scan - -simd_next_from_r3: - ADD $16, R3 - CMP R4, R3 - BLE simd_loop - B scalar_init - -found_at_r7: - ADD $4, R7 - MOVD R7, ret+24(FP) - RET - -simd_next: - ADD $16, R3 - CMP R4, R3 - BLE simd_loop - -scalar_init: - SUB $3, R1, R4 - -scalar_loop: - CMP R4, R3 - BGE not_found - ADD R0, R3, R6 - MOVWU (R6), R5 - MOVD $0x0A0D0A0D, R7 // 0x0A0D0A0D = \r\n\r\n in little-endian byte order - CMP R7, R5 - BEQ scalar_found - ADD $1, R3 - B scalar_loop - -scalar_found: - ADD $4, R3 - MOVD R3, ret+24(FP) - RET - -not_found: - MOVD $-1, R0 - MOVD R0, ret+24(FP) - RET diff --git a/protocol/h1/findheader_generic.go b/protocol/h1/findheader_generic.go deleted file mode 100644 index ee79be48..00000000 --- a/protocol/h1/findheader_generic.go +++ /dev/null @@ -1,16 +0,0 @@ -//go:build !amd64 && !arm64 - -package h1 - -func findHeaderEnd(buf []byte) int { - n := len(buf) - if n < 4 { - return -1 - } - for i := 0; i <= n-4; i++ { - if buf[i] == '\r' && buf[i+1] == '\n' && buf[i+2] == '\r' && buf[i+3] == '\n' { - return i + 4 - } - } - return -1 -} diff --git a/protocol/h1/parser.go b/protocol/h1/parser.go index 36e4bf53..753bdf30 100644 --- a/protocol/h1/parser.go +++ b/protocol/h1/parser.go @@ -86,7 +86,7 @@ func (p *Parser) ParseRequest(req *Request) (int, error) { return 0, nil } - // No upfront whole-block findHeaderEnd scan: parseHeaders detects an + // No upfront whole-block CRLFCRLF scan: parseHeaders detects an // incomplete block itself (a final line with no CRLF yields lineEnd==-1 → // (false,nil)), and every caller Reset()s parser+req before re-parsing, so // a partial parse is always retried cleanly. Scanning here first would diff --git a/protocol/h1/parser_test.go b/protocol/h1/parser_test.go index f5873831..36991a02 100644 --- a/protocol/h1/parser_test.go +++ b/protocol/h1/parser_test.go @@ -439,97 +439,6 @@ func TestParseRequest_DefaultContentLengthMinusOne(t *testing.T) { } } -// TestFindHeaderEnd_AllPositions places \r\n\r\n at every possible offset in -// buffers of varying sizes. This exercises AVX2 (32-byte), SSE2 (16-byte), -// and scalar code paths, including transition boundaries. -func TestFindHeaderEnd_AllPositions(t *testing.T) { - // Sizes chosen to stress boundaries: - // 4 — minimum, scalar only - // 15 — just under one SSE2 block - // 16 — exactly one SSE2 block (no room for \r\n\r\n verification) - // 19 — first size that enters the SSE2 loop - // 31 — just under one AVX2 block - // 32 — exactly one AVX2 block - // 35 — first size that enters the AVX2 loop - // 48 — AVX2 + SSE2 tail - // 64 — two full AVX2 iterations - // 100 — AVX2 + SSE2 + scalar tail - // 256 — multiple AVX2 iterations - // 1024 — large buffer - sizes := []int{4, 5, 15, 16, 17, 19, 20, 31, 32, 33, 35, 36, 48, 63, 64, 65, 100, 256, 1024} - - for _, size := range sizes { - t.Run(fmt.Sprintf("size=%d", size), func(t *testing.T) { - for pos := 0; pos <= size-4; pos++ { - buf := make([]byte, size) - for i := range buf { - buf[i] = 'A' - } - buf[pos] = '\r' - buf[pos+1] = '\n' - buf[pos+2] = '\r' - buf[pos+3] = '\n' - - got := findHeaderEnd(buf) - want := pos + 4 - if got != want { - t.Fatalf("size=%d pos=%d: got %d, want %d", size, pos, got, want) - } - } - }) - } -} - -// TestFindHeaderEnd_NotFound verifies -1 is returned when no \r\n\r\n exists. -func TestFindHeaderEnd_NotFound(t *testing.T) { - sizes := []int{0, 1, 2, 3, 4, 16, 32, 64, 128, 256} - for _, size := range sizes { - buf := make([]byte, size) - for i := range buf { - buf[i] = 'A' - } - got := findHeaderEnd(buf) - if got != -1 { - t.Fatalf("size=%d: got %d, want -1 (no \\r\\n\\r\\n present)", size, got) - } - } -} - -// TestFindHeaderEnd_FalsePositiveCR verifies that lone \r bytes do not -// cause false positives. The buffer is filled with \r but no \r\n\r\n exists. -func TestFindHeaderEnd_FalsePositiveCR(t *testing.T) { - sizes := []int{16, 32, 64, 128} - for _, size := range sizes { - buf := make([]byte, size) - for i := range buf { - buf[i] = '\r' - } - // No \n follows any \r, so no valid \r\n\r\n sequence. - got := findHeaderEnd(buf) - if got != -1 { - t.Fatalf("size=%d (all \\r): got %d, want -1", size, got) - } - } -} - -// TestFindHeaderEnd_MultipleCR verifies correct behavior when multiple \r -// bytes exist before the actual \r\n\r\n sequence. -func TestFindHeaderEnd_MultipleCR(t *testing.T) { - buf := make([]byte, 128) - for i := range buf { - buf[i] = '\r' - } - // Place the real terminator at offset 100. - buf[100] = '\r' - buf[101] = '\n' - buf[102] = '\r' - buf[103] = '\n' - got := findHeaderEnd(buf) - if got != 104 { - t.Fatalf("got %d, want 104", got) - } -} - func TestParseRequest_DuplicateContentLength_Conflicting(t *testing.T) { for _, zeroCopy := range []bool{false, true} { name := "standard" @@ -605,28 +514,6 @@ func TestParseRequest_ContentLengthAndChunked_Rejected(t *testing.T) { } } -// TestFindHeaderEnd_PartialSequence ensures partial \r\n sequences -// (without the full \r\n\r\n) are not mistakenly matched. -func TestFindHeaderEnd_PartialSequence(t *testing.T) { - tests := []struct { - name string - data string - }{ - {"cr_only", "AAAA\rAAAA"}, - {"crlf_only", "AAAA\r\nAAAA"}, - {"crlf_cr", "AAAA\r\n\rAAAA"}, - {"lf_crlf", "AAAA\n\r\nAAAA"}, - } - for _, tc := range tests { - t.Run(tc.name, func(t *testing.T) { - got := findHeaderEnd([]byte(tc.data)) - if got != -1 { - t.Fatalf("got %d, want -1 for partial sequence", got) - } - }) - } -} - func TestParseRequest_H2CUpgrade(t *testing.T) { cases := []struct { name string From 5ca6f6e13b1410cf76ff89ef1fff55bdb4d6c074 Mon Sep 17 00:00:00 2001 From: Albert Bausili Date: Sat, 26 Sep 2026 18:17:37 +0200 Subject: [PATCH 2/2] docs(h1): say exactly what the parser's slices alias (celeris#424) The README bullet and the protocol/h1 package doc said header and body slices alias "the connection's read buffer". That is true of the parser's return values relative to the buffer it is given, not of the engine path: epoll and io_uring hand the parser their read buffer only for a request that arrives whole in one read, has no body or a Content-Length body, and runs inline. A request that spans reads (its headers gathered in the connection's H1 buffer, a fixed-length body in bodyBuf, internal/conn/h1.go), a chunked body, or a request for an async handler (asyncInBuf) is served from a per-connection buffer. The package doc also listed the std engine as a consumer; std uses net/http's parser. --- README.md | 2 +- protocol/h1/doc.go | 12 +++++++----- 2 files changed, 8 insertions(+), 6 deletions(-) diff --git a/README.md b/README.md index a0fee093..c7f21528 100644 --- a/README.md +++ b/README.md @@ -32,7 +32,7 @@ An engine-correctness release, driven by what the [probatorium](https://github.c - **Edge-triggered epoll** — per-core event loops with CPU pinning. - **Adaptive meta-engine** — transplants between io_uring and epoll at runtime based on telemetry. - **First-party database drivers** — native [`driver/postgres`](driver/postgres), [`driver/redis`](driver/redis), and [`driver/memcached`](driver/memcached) run on the celeris event loop (see [Database drivers](#database-drivers)). -- **Zero-copy HTTP/1.1 parser** — header and body slices alias the connection's read buffer instead of being copied out of it. +- **Zero-copy HTTP/1.1 parser** — the parser returns header and body slices that alias the bytes it parses instead of copying them. On epoll and io_uring, a request that arrives whole in one read, with no body or a `Content-Length` body, and runs inline is parsed in place in the engine's read buffer; a request that spans reads, has a chunked body, or runs on an async handler is served from a per-connection buffer instead. - **HTTP/2 cleartext (h2c)** — full stream multiplexing, flow control, HPACK, inline handler execution, zero-alloc HEADERS fast path. - **Auto-detect** — protocol negotiation from the first bytes on the wire. - **Error-returning handlers** — `HandlerFunc` returns `error`; structured `*HTTPError` carries status codes. diff --git a/protocol/h1/doc.go b/protocol/h1/doc.go index b65b35d9..59c8a178 100644 --- a/protocol/h1/doc.go +++ b/protocol/h1/doc.go @@ -1,11 +1,13 @@ // Package h1 implements a zero-copy HTTP/1.1 parser. // // This is a low-level parser package consumed by the celeris engine layer -// (engine/iouring, engine/epoll, engine/std bridge); application code -// should not import it directly. The parser returns header and body slices -// that alias the connection's read buffer — callers must materialize -// (clone) any value they retain past the next [ParseRequest] call on the -// same connection. +// (engine/epoll and engine/iouring, through internal/conn; the std engine +// uses net/http's parser); application code should not import it directly. +// The parser returns header and body slices that alias the buffer passed to +// [Parser.Reset]: an engine's read buffer, or a per-connection buffer the +// engine gathered the request into. Callers must materialize (clone) any +// value they retain past the next [Parser.ParseRequest] call on the same +// connection. // // # Documentation //