diff --git a/README.md b/README.md index eeab7621..814c1884 100644 --- a/README.md +++ b/README.md @@ -32,7 +32,7 @@ An engine-correctness release, driven by what the [probatorium](https://github.c - **Edge-triggered epoll** — per-core event loops with CPU pinning. - **Adaptive meta-engine** — transplants between io_uring and epoll at runtime based on telemetry. - **First-party database drivers** — native [`driver/postgres`](driver/postgres), [`driver/redis`](driver/redis), and [`driver/memcached`](driver/memcached) run on the celeris event loop (see [Database drivers](#database-drivers)). -- **SIMD HTTP parser** — SSE2 (amd64) and NEON (arm64) with a generic SWAR fallback. +- **Zero-copy HTTP/1.1 parser** — the parser returns header and body slices that alias the bytes it parses instead of copying them. On epoll and io_uring, a request that arrives whole in one read, with no body or a `Content-Length` body, and runs inline is parsed in place in the engine's read buffer; a request that spans reads, has a chunked body, or runs on an async handler is served from a per-connection buffer instead. - **HTTP/2 cleartext (h2c)** — full stream multiplexing, flow control, HPACK, inline handler execution, zero-alloc HEADERS fast path. - **Auto-detect** — protocol negotiation from the first bytes on the wire. - **Error-returning handlers** — `HandlerFunc` returns `error`; structured `*HTTPError` carries status codes. diff --git a/protocol/h1/bench_test.go b/protocol/h1/bench_test.go index a7212ee2..0cf467d4 100644 --- a/protocol/h1/bench_test.go +++ b/protocol/h1/bench_test.go @@ -84,38 +84,3 @@ func BenchmarkParseRequest_Pipelined(b *testing.B) { } } } - -func BenchmarkFindHeaderEnd(b *testing.B) { - sizes := []int{64, 256, 1024, 4096} - for _, size := range sizes { - b.Run(fmt.Sprintf("size=%d", size), func(b *testing.B) { - // Place \r\n\r\n at the end - buf := make([]byte, size) - for i := range buf { - buf[i] = 'A' - } - copy(buf[size-4:], "\r\n\r\n") - b.SetBytes(int64(size)) - b.ReportAllocs() - b.ResetTimer() - for i := 0; i < b.N; i++ { - findHeaderEnd(buf) - } - }) - } -} - -func BenchmarkFindHeaderEnd_8K(b *testing.B) { - size := 8192 - buf := make([]byte, size) - for i := range buf { - buf[i] = 'A' - } - copy(buf[size-4:], "\r\n\r\n") - b.SetBytes(int64(size)) - b.ReportAllocs() - b.ResetTimer() - for i := 0; i < b.N; i++ { - findHeaderEnd(buf) - } -} diff --git a/protocol/h1/doc.go b/protocol/h1/doc.go index b65b35d9..59c8a178 100644 --- a/protocol/h1/doc.go +++ b/protocol/h1/doc.go @@ -1,11 +1,13 @@ // Package h1 implements a zero-copy HTTP/1.1 parser. // // This is a low-level parser package consumed by the celeris engine layer -// (engine/iouring, engine/epoll, engine/std bridge); application code -// should not import it directly. The parser returns header and body slices -// that alias the connection's read buffer — callers must materialize -// (clone) any value they retain past the next [ParseRequest] call on the -// same connection. +// (engine/epoll and engine/iouring, through internal/conn; the std engine +// uses net/http's parser); application code should not import it directly. +// The parser returns header and body slices that alias the buffer passed to +// [Parser.Reset]: an engine's read buffer, or a per-connection buffer the +// engine gathered the request into. Callers must materialize (clone) any +// value they retain past the next [Parser.ParseRequest] call on the same +// connection. // // # Documentation // diff --git a/protocol/h1/findheader.go b/protocol/h1/findheader.go deleted file mode 100644 index 8375a33d..00000000 --- a/protocol/h1/findheader.go +++ /dev/null @@ -1 +0,0 @@ -package h1 diff --git a/protocol/h1/findheader_amd64.go b/protocol/h1/findheader_amd64.go deleted file mode 100644 index ec1761fd..00000000 --- a/protocol/h1/findheader_amd64.go +++ /dev/null @@ -1,16 +0,0 @@ -//go:build amd64 - -package h1 - -import "golang.org/x/sys/cpu" - -// useAVX2 is set at init time. The assembly (findheader_amd64.s) reads this -// to branch between the AVX2 (32-byte) and SSE2 (16-byte) scan loops. -var useAVX2 bool //nolint:unused // referenced by assembly via ·useAVX2(SB) - -func init() { - useAVX2 = cpu.X86.HasAVX2 -} - -//go:noescape -func findHeaderEnd(buf []byte) int diff --git a/protocol/h1/findheader_amd64.s b/protocol/h1/findheader_amd64.s deleted file mode 100644 index 25543c71..00000000 --- a/protocol/h1/findheader_amd64.s +++ /dev/null @@ -1,158 +0,0 @@ -#include "textflag.h" - -// func findHeaderEnd(buf []byte) int -// -// Scans buf for the HTTP header terminator \r\n\r\n (0x0D 0x0A 0x0D 0x0A). -// Returns the index one past the terminator (i.e., the start of the body), -// or -1 if not found. -// -// At entry the function checks the package-level useAVX2 flag and dispatches -// to either a 32-byte (AVX2) or 16-byte (SSE2) vectorized loop. Both fall -// through to the same scalar tail. -TEXT ·findHeaderEnd(SB), NOSPLIT, $0-32 - MOVQ buf_base+0(FP), SI // SI = &buf[0] - MOVQ buf_len+8(FP), CX // CX = len(buf) - CMPQ CX, $4 - JL not_found - - XORQ DI, DI // DI = current offset - - // Check runtime AVX2 flag. - CMPB ·useAVX2(SB), $0 - JE sse2_init - -// ----------------------------------------------------------------------- -// AVX2 path — 32 bytes per iteration -// ----------------------------------------------------------------------- -avx2_init: - // Broadcast '\r' (0x0D) into Y0 (32 bytes). - MOVQ $0x0D0D0D0D0D0D0D0D, AX - MOVQ AX, X0 - VPBROADCASTQ X0, Y0 - - MOVQ CX, DX - SUBQ $35, DX // last safe AVX2 start = len - 32 - 3 - JL avx2_to_sse2 // buffer too small for even one AVX2 iteration - -avx2_loop: - CMPQ DI, DX - JG avx2_to_sse2 - - // Load 32 unaligned bytes, compare each byte against '\r'. - VMOVDQU (SI)(DI*1), Y1 - VPCMPEQB Y0, Y1, Y2 - VPMOVMSKB Y2, AX // AX = 32-bit mask of '\r' positions - TESTL AX, AX - JZ avx2_next - -avx2_check_bits: - BSFL AX, BX // BX = index of first set bit - LEAQ (DI)(BX*1), R8 // R8 = absolute position in buf - LEAQ 4(R8), R9 - CMPQ R9, CX - JG avx2_clear_bit - MOVL (SI)(R8*1), R10 - CMPL R10, $0x0A0D0A0D // \r\n\r\n in little-endian - JE avx2_found - BTRL BX, AX // clear this bit, check next '\r' - TESTL AX, AX - JNZ avx2_check_bits - -avx2_next: - ADDQ $32, DI - JMP avx2_loop - -avx2_clear_bit: - BTRL BX, AX - TESTL AX, AX - JNZ avx2_check_bits - ADDQ $32, DI - JMP avx2_loop - -avx2_found: - VZEROUPPER - LEAQ 4(R8), AX - MOVQ AX, ret+24(FP) - RET - - // Transition: remaining bytes too few for 32-byte loads. - // Fall through to SSE2 for 16-byte chunks, then scalar tail. -avx2_to_sse2: - VZEROUPPER - -// ----------------------------------------------------------------------- -// SSE2 path — 16 bytes per iteration -// ----------------------------------------------------------------------- -sse2_init: - // Broadcast '\r' (0x0D) into X0 (16 bytes). - MOVQ $0x0D0D0D0D0D0D0D0D, AX - MOVQ AX, X0 - PUNPCKLQDQ X0, X0 - - MOVQ CX, DX - SUBQ $19, DX // last safe SSE2 start = len - 16 - 3 - JL scalar_init - -sse2_loop: - CMPQ DI, DX - JG scalar_init - - MOVOU (SI)(DI*1), X1 - PCMPEQB X0, X1 - PMOVMSKB X1, AX - TESTL AX, AX - JZ sse2_next - -sse2_check_bits: - BSFL AX, BX - LEAQ (DI)(BX*1), R8 - LEAQ 4(R8), R9 - CMPQ R9, CX - JG sse2_clear_bit - MOVL (SI)(R8*1), R10 - CMPL R10, $0x0A0D0A0D - JE found_at_r8 - BTRL BX, AX - TESTL AX, AX - JNZ sse2_check_bits - -sse2_next: - ADDQ $16, DI - JMP sse2_loop - -sse2_clear_bit: - BTRL BX, AX - TESTL AX, AX - JNZ sse2_check_bits - ADDQ $16, DI - JMP sse2_loop - -// ----------------------------------------------------------------------- -// Scalar tail — one byte at a time for remaining bytes -// ----------------------------------------------------------------------- -scalar_init: - MOVQ CX, DX - SUBQ $3, DX - -scalar_loop: - CMPQ DI, DX - JGE not_found - MOVL (SI)(DI*1), AX - CMPL AX, $0x0A0D0A0D - JE scalar_found - INCQ DI - JMP scalar_loop - -scalar_found: - LEAQ 4(DI), AX - MOVQ AX, ret+24(FP) - RET - -found_at_r8: - LEAQ 4(R8), AX - MOVQ AX, ret+24(FP) - RET - -not_found: - MOVQ $-1, ret+24(FP) - RET diff --git a/protocol/h1/findheader_arm64.go b/protocol/h1/findheader_arm64.go deleted file mode 100644 index 5078c190..00000000 --- a/protocol/h1/findheader_arm64.go +++ /dev/null @@ -1,6 +0,0 @@ -//go:build arm64 - -package h1 - -//go:noescape -func findHeaderEnd(buf []byte) int diff --git a/protocol/h1/findheader_arm64.s b/protocol/h1/findheader_arm64.s deleted file mode 100644 index e63c6f07..00000000 --- a/protocol/h1/findheader_arm64.s +++ /dev/null @@ -1,85 +0,0 @@ -#include "textflag.h" - -// func findHeaderEnd(buf []byte) int -TEXT ·findHeaderEnd(SB), NOSPLIT, $0-32 - MOVD buf_base+0(FP), R0 - MOVD buf_len+8(FP), R1 - CMP $4, R1 - BLT not_found - - MOVD $0x0D0D0D0D0D0D0D0D, R2 - VDUP R2, V0.D2 - - MOVD $0, R3 - - MOVD R1, R4 - SUB $19, R4, R4 - - CMP R4, R3 - BGT scalar_init - -simd_loop: - ADD R0, R3, R6 - VLD1 (R6), [V1.B16] - VCMEQ V0.B16, V1.B16, V2.B16 - VUADDLV V2.B16, V3 - VMOV V3.D[0], R5 - CBZ R5, simd_next - - MOVD R3, R7 - ADD $16, R3, R8 - SUB $3, R1, R9 - CMP R9, R8 - BLE block_scan - MOVD R9, R8 - -block_scan: - CMP R8, R7 - BGE simd_next_from_r3 - ADD R0, R7, R6 - MOVWU (R6), R10 - MOVD $0x0A0D0A0D, R11 // 0x0A0D0A0D = \r\n\r\n in little-endian byte order - CMP R11, R10 - BEQ found_at_r7 - ADD $1, R7 - B block_scan - -simd_next_from_r3: - ADD $16, R3 - CMP R4, R3 - BLE simd_loop - B scalar_init - -found_at_r7: - ADD $4, R7 - MOVD R7, ret+24(FP) - RET - -simd_next: - ADD $16, R3 - CMP R4, R3 - BLE simd_loop - -scalar_init: - SUB $3, R1, R4 - -scalar_loop: - CMP R4, R3 - BGE not_found - ADD R0, R3, R6 - MOVWU (R6), R5 - MOVD $0x0A0D0A0D, R7 // 0x0A0D0A0D = \r\n\r\n in little-endian byte order - CMP R7, R5 - BEQ scalar_found - ADD $1, R3 - B scalar_loop - -scalar_found: - ADD $4, R3 - MOVD R3, ret+24(FP) - RET - -not_found: - MOVD $-1, R0 - MOVD R0, ret+24(FP) - RET diff --git a/protocol/h1/findheader_generic.go b/protocol/h1/findheader_generic.go deleted file mode 100644 index ee79be48..00000000 --- a/protocol/h1/findheader_generic.go +++ /dev/null @@ -1,16 +0,0 @@ -//go:build !amd64 && !arm64 - -package h1 - -func findHeaderEnd(buf []byte) int { - n := len(buf) - if n < 4 { - return -1 - } - for i := 0; i <= n-4; i++ { - if buf[i] == '\r' && buf[i+1] == '\n' && buf[i+2] == '\r' && buf[i+3] == '\n' { - return i + 4 - } - } - return -1 -} diff --git a/protocol/h1/parser.go b/protocol/h1/parser.go index 36e4bf53..753bdf30 100644 --- a/protocol/h1/parser.go +++ b/protocol/h1/parser.go @@ -86,7 +86,7 @@ func (p *Parser) ParseRequest(req *Request) (int, error) { return 0, nil } - // No upfront whole-block findHeaderEnd scan: parseHeaders detects an + // No upfront whole-block CRLFCRLF scan: parseHeaders detects an // incomplete block itself (a final line with no CRLF yields lineEnd==-1 → // (false,nil)), and every caller Reset()s parser+req before re-parsing, so // a partial parse is always retried cleanly. Scanning here first would diff --git a/protocol/h1/parser_test.go b/protocol/h1/parser_test.go index f5873831..36991a02 100644 --- a/protocol/h1/parser_test.go +++ b/protocol/h1/parser_test.go @@ -439,97 +439,6 @@ func TestParseRequest_DefaultContentLengthMinusOne(t *testing.T) { } } -// TestFindHeaderEnd_AllPositions places \r\n\r\n at every possible offset in -// buffers of varying sizes. This exercises AVX2 (32-byte), SSE2 (16-byte), -// and scalar code paths, including transition boundaries. -func TestFindHeaderEnd_AllPositions(t *testing.T) { - // Sizes chosen to stress boundaries: - // 4 — minimum, scalar only - // 15 — just under one SSE2 block - // 16 — exactly one SSE2 block (no room for \r\n\r\n verification) - // 19 — first size that enters the SSE2 loop - // 31 — just under one AVX2 block - // 32 — exactly one AVX2 block - // 35 — first size that enters the AVX2 loop - // 48 — AVX2 + SSE2 tail - // 64 — two full AVX2 iterations - // 100 — AVX2 + SSE2 + scalar tail - // 256 — multiple AVX2 iterations - // 1024 — large buffer - sizes := []int{4, 5, 15, 16, 17, 19, 20, 31, 32, 33, 35, 36, 48, 63, 64, 65, 100, 256, 1024} - - for _, size := range sizes { - t.Run(fmt.Sprintf("size=%d", size), func(t *testing.T) { - for pos := 0; pos <= size-4; pos++ { - buf := make([]byte, size) - for i := range buf { - buf[i] = 'A' - } - buf[pos] = '\r' - buf[pos+1] = '\n' - buf[pos+2] = '\r' - buf[pos+3] = '\n' - - got := findHeaderEnd(buf) - want := pos + 4 - if got != want { - t.Fatalf("size=%d pos=%d: got %d, want %d", size, pos, got, want) - } - } - }) - } -} - -// TestFindHeaderEnd_NotFound verifies -1 is returned when no \r\n\r\n exists. -func TestFindHeaderEnd_NotFound(t *testing.T) { - sizes := []int{0, 1, 2, 3, 4, 16, 32, 64, 128, 256} - for _, size := range sizes { - buf := make([]byte, size) - for i := range buf { - buf[i] = 'A' - } - got := findHeaderEnd(buf) - if got != -1 { - t.Fatalf("size=%d: got %d, want -1 (no \\r\\n\\r\\n present)", size, got) - } - } -} - -// TestFindHeaderEnd_FalsePositiveCR verifies that lone \r bytes do not -// cause false positives. The buffer is filled with \r but no \r\n\r\n exists. -func TestFindHeaderEnd_FalsePositiveCR(t *testing.T) { - sizes := []int{16, 32, 64, 128} - for _, size := range sizes { - buf := make([]byte, size) - for i := range buf { - buf[i] = '\r' - } - // No \n follows any \r, so no valid \r\n\r\n sequence. - got := findHeaderEnd(buf) - if got != -1 { - t.Fatalf("size=%d (all \\r): got %d, want -1", size, got) - } - } -} - -// TestFindHeaderEnd_MultipleCR verifies correct behavior when multiple \r -// bytes exist before the actual \r\n\r\n sequence. -func TestFindHeaderEnd_MultipleCR(t *testing.T) { - buf := make([]byte, 128) - for i := range buf { - buf[i] = '\r' - } - // Place the real terminator at offset 100. - buf[100] = '\r' - buf[101] = '\n' - buf[102] = '\r' - buf[103] = '\n' - got := findHeaderEnd(buf) - if got != 104 { - t.Fatalf("got %d, want 104", got) - } -} - func TestParseRequest_DuplicateContentLength_Conflicting(t *testing.T) { for _, zeroCopy := range []bool{false, true} { name := "standard" @@ -605,28 +514,6 @@ func TestParseRequest_ContentLengthAndChunked_Rejected(t *testing.T) { } } -// TestFindHeaderEnd_PartialSequence ensures partial \r\n sequences -// (without the full \r\n\r\n) are not mistakenly matched. -func TestFindHeaderEnd_PartialSequence(t *testing.T) { - tests := []struct { - name string - data string - }{ - {"cr_only", "AAAA\rAAAA"}, - {"crlf_only", "AAAA\r\nAAAA"}, - {"crlf_cr", "AAAA\r\n\rAAAA"}, - {"lf_crlf", "AAAA\n\r\nAAAA"}, - } - for _, tc := range tests { - t.Run(tc.name, func(t *testing.T) { - got := findHeaderEnd([]byte(tc.data)) - if got != -1 { - t.Fatalf("got %d, want -1 for partial sequence", got) - } - }) - } -} - func TestParseRequest_H2CUpgrade(t *testing.T) { cases := []struct { name string