From 9f022bf37549ee4c1ee0c019f17a158a9436b380 Mon Sep 17 00:00:00 2001 From: ZhouGuangyuan Date: Fri, 2 Oct 2026 23:56:16 +0800 Subject: [PATCH 01/20] simd: lower SIMD128 array memory and broadcast operations --- cl/compile.go | 9 +- cl/simd.go | 70 ++++++++++++---- internal/build/simd_test.go | 12 ++- ssa/simd.go | 36 +++++++- test/simd/README.md | 12 ++- test/simd/memory_test.go | 121 +++++++++++++++++++++++++++ test/simd/unimplemented_llgo_test.go | 2 - 7 files changed, 233 insertions(+), 29 deletions(-) create mode 100644 test/simd/memory_test.go diff --git a/cl/compile.go b/cl/compile.go index 20176da7ae..58a5340cdb 100644 --- a/cl/compile.go +++ b/cl/compile.go @@ -708,7 +708,7 @@ func (p *context) compileFuncDecl(pkg llssa.Package, f *ssa.Function) (llssa.Fun if simdDecl { b := fn.MakeBody(1) if simd.op == llssa.SIMDUnimplemented { - b.SIMD(simd.op, b.Str(name)) + b.SIMD(simd.op, p.prog.Void(), b.Str(name)) b.Unreachable() } else { n := sig.Params().Len() @@ -719,7 +719,12 @@ func (p *context) compileFuncDecl(pkg llssa.Package, f *ssa.Function) (llssa.Fun for i := range args { args[i] = fn.Param(i) } - b.Return(b.SIMD(simd.op, args...)) + result := b.SIMD(simd.op, p.simdResultType(f.Signature), args...) + if sig.Results().Len() == 0 { + b.Return() + } else { + b.Return(result) + } } b.EndBuild() b.Dispose() diff --git a/cl/simd.go b/cl/simd.go index dfd9855df7..ee7e2a78b7 100644 --- a/cl/simd.go +++ b/cl/simd.go @@ -1,6 +1,7 @@ package cl import ( + "fmt" "go/ast" "go/types" @@ -18,6 +19,10 @@ const ( simdExtract simdInsert simdUnsupported + simdUnary + simdLoad + simdStore + simdBroadcast ) type simdOperation struct { @@ -30,14 +35,27 @@ type simdOperation struct { // recoverable failure. Adding an implementation replaces this fallback for that // operation; functions with Go bodies continue through normal compilation. var simdOperations = map[simdKey]simdOperation{ - {"*", "*"}: {llssa.SIMDUnimplemented, simdUnsupported, false}, - {"numeric", "Add"}: {llssa.SIMDAdd, simdBinary, false}, - {"numeric", "Sub"}: {llssa.SIMDSub, simdBinary, false}, - {"numeric", "And"}: {llssa.SIMDAnd, simdBinary, true}, - {"numeric", "Or"}: {llssa.SIMDOr, simdBinary, true}, - {"numeric", "Xor"}: {llssa.SIMDXor, simdBinary, true}, - {"numeric", "GetElem"}: {llssa.SIMDExtractLane, simdExtract, false}, - {"numeric", "SetElem"}: {llssa.SIMDInsertLane, simdInsert, false}, + {"*", "*"}: {llssa.SIMDUnimplemented, simdUnsupported, false}, + {"numeric", "Add"}: {llssa.SIMDAdd, simdBinary, false}, + {"numeric", "Sub"}: {llssa.SIMDSub, simdBinary, false}, + {"numeric", "And"}: {llssa.SIMDAnd, simdBinary, true}, + {"numeric", "Or"}: {llssa.SIMDOr, simdBinary, true}, + {"numeric", "Xor"}: {llssa.SIMDXor, simdBinary, true}, + {"numeric", "GetElem"}: {llssa.SIMDExtractLane, simdExtract, false}, + {"numeric", "SetElem"}: {llssa.SIMDInsertLane, simdInsert, false}, + {"numeric", "StoreArray"}: {llssa.SIMDStore, simdStore, false}, +} + +// These registrations share lowering but retain exact declaration names and +// signature checks. Source Go slice helpers keep their own bounds checks. +func init() { + for _, name := range []string{"Int8x16", "Uint8x16", "Int16x8", "Uint16x8", "Int32x4", "Uint32x4", "Int64x2", "Uint64x2", "Float32x4", "Float64x2"} { + simdOperations[simdKey{"", "Load" + name + "Array"}] = simdOperation{llssa.SIMDLoad, simdLoad, false} + simdOperations[simdKey{"", "Broadcast" + name}] = simdOperation{llssa.SIMDBroadcast, simdBroadcast, false} + } + for _, lanes := range []int{2, 4, 8, 16} { + simdOperations[simdKey{"numeric", fmt.Sprintf("broadcast1To%d", lanes)}] = simdOperation{llssa.SIMDSplatLane0, simdUnary, false} + } } // Resolve only declared official operations, never synthetic wrapper names. @@ -56,15 +74,16 @@ func lookupSIMD(fn *ssa.Function, arch string) (simdOperation, bool) { if !types.Identical(sig, obj.Type()) { return simdOperation{}, false } + decl, isDecl := fn.Syntax().(*ast.FuncDecl) fallback := func() (simdOperation, bool) { // Imported declarations and synthetic wrappers are not definitions. Emit // the fallback only for a source intrinsic declaration in archsimd. - if decl, ok := fn.Syntax().(*ast.FuncDecl); ok && decl.Body == nil { + if isDecl && decl.Body == nil { return simdOperations[simdKey{"*", "*"}], true } return simdOperation{}, false } - if decl, ok := fn.Syntax().(*ast.FuncDecl); ok && decl.Body != nil { + if isDecl && decl.Body != nil { return simdOperation{}, false } key := simdKey{name: obj.Name()} @@ -86,7 +105,7 @@ func lookupSIMD(fn *ssa.Function, arch string) (simdOperation, bool) { } func (d simdOperation) matches(sig *types.Signature, vector types.Type) bool { - if vector == nil || sig.Variadic() || sig.Results().Len() != 1 { + if vector == nil || sig.Variadic() { return false } lanes, ok := llssa.SIMDNumericShape(vector) @@ -99,6 +118,15 @@ func (d simdOperation) matches(sig *types.Signature, vector types.Type) bool { var params []types.Type result := vector switch d.signature { + case simdUnary: + // Receiver only. + case simdLoad, simdStore: + params = []types.Type{types.NewPointer(lanes)} + if d.signature == simdStore { + result = nil + } + case simdBroadcast: + params = []types.Type{lanes.Elem()} case simdBinary: params = []types.Type{vector} case simdExtract: @@ -108,10 +136,17 @@ func (d simdOperation) matches(sig *types.Signature, vector types.Type) bool { default: return false } - if sig.Recv() == nil { + if sig.Recv() == nil && d.signature != simdLoad && d.signature != simdBroadcast { params = append([]types.Type{vector}, params...) } - if sig.Params().Len() != len(params) || !types.Identical(sig.Results().At(0).Type(), result) { + if sig.Params().Len() != len(params) { + return false + } + if result == nil { + if sig.Results().Len() != 0 { + return false + } + } else if sig.Results().Len() != 1 || !types.Identical(sig.Results().At(0).Type(), result) { return false } for i, typ := range params { @@ -132,5 +167,12 @@ func (p *context) simdCall(b llssa.Builder, fn *ssa.Function, args []ssa.Value) if !ok { panic("invalid SIMD intrinsic") } - return b.SIMD(desc.op, p.compileValues(b, args, fnNormal)...) + return b.SIMD(desc.op, p.simdResultType(fn.Signature), p.compileValues(b, args, fnNormal)...) +} + +func (p *context) simdResultType(sig *types.Signature) llssa.Type { + if sig.Results().Len() == 0 { + return p.prog.Void() + } + return p.prog.Type(sig.Results().At(0).Type(), llssa.InGo) } diff --git a/internal/build/simd_test.go b/internal/build/simd_test.go index 2ea9d0a53c..b61a352c70 100644 --- a/internal/build/simd_test.go +++ b/internal/build/simd_test.go @@ -31,6 +31,8 @@ func loop(x archsimd.Float32x4, n int) (archsimd.Float32x4, int) { for i := 0; i < n; i++ { x = x.Add(x) } return x, n } +func load(p *[4]float32) archsimd.Float32x4 { return archsimd.LoadFloat32x4Array(p) } +func store(p *[4]float32, x archsimd.Float32x4) { x.StoreArray(p) } func fixed(x archsimd.Float32x4) float32 { return x.GetElem(1) } func boxed(x any) archsimd.Float32x4 { return x.(archsimd.Float32x4) } func invoke(x, y archsimd.Float32x4) { defer x.Add(y); go x.Sub(y) } @@ -84,6 +86,12 @@ func TestSIMD128LLVM(t *testing.T) { if fixed := mod.NamedFunction("main.fixed").String(); strings.Contains(fixed, "PanicSIMDImmediate") || strings.Contains(fixed, "br ") { t.Fatalf("valid constant lane retained a bounds branch:\n%s", fixed) } + for _, name := range []string{"load", "store"} { + ir := mod.NamedFunction("main." + name).String() + if !strings.Contains(ir, name+" <4 x float>") || !strings.Contains(ir, "align 4") { + t.Fatalf("%s does not use element-aligned vector memory:\n%s", name, ir) + } + } identity := mod.NamedFunction("main.identity") if identity.GlobalValueType().ReturnType().TypeKind() != llvm.VectorTypeKind || identity.GlobalValueType().ParamTypes()[0].TypeKind() != llvm.VectorTypeKind { t.Fatalf("identity does not use a vector ABI:\n%s", identity.String()) @@ -195,8 +203,8 @@ func main() { println(broadcast(1).GetElem(0)) } t.Fatal("linkname target body was discarded") } callee := mod.NamedFunction("simd/archsimd.Float32x4.broadcast1To4") - if callee.IsNil() || callee.IsDeclaration() || !strings.Contains(callee.String(), "PanicSIMDUnimplemented") { - t.Fatal("transitive unsupported intrinsic lacks a panic implementation") + if callee.IsNil() || callee.IsDeclaration() || !strings.Contains(callee.String(), "shufflevector") { + t.Fatal("transitive broadcast intrinsic lacks a vector implementation") } return } diff --git a/ssa/simd.go b/ssa/simd.go index 2a9a6c0048..3ab610046a 100644 --- a/ssa/simd.go +++ b/ssa/simd.go @@ -20,6 +20,10 @@ const ( SIMDExtractLane SIMDInsertLane SIMDUnimplemented + SIMDLoad + SIMDStore + SIMDBroadcast + SIMDSplatLane0 ) // SIMDNumericShape validates the official numeric aggregate representation. @@ -80,12 +84,32 @@ func simdLanes(typ types.Type) *types.Array { // SIMD applies the selected operation's feature requirements before lowering. // These seven implementations need only baseline native instructions; wasm // requires SIMD128. Future feature-specific implementations extend this entry. -func (b Builder) SIMD(op SIMDOp, args ...Expr) Expr { +func (b Builder) SIMD(op SIMDOp, result Type, args ...Expr) Expr { if op == SIMDUnimplemented { + if len(args) != 1 || !types.Identical(args[0].RawType(), types.Typ[types.String]) { + panic("SIMD fallback requires an intrinsic name string") + } return b.Call(b.Pkg.rtFunc("PanicSIMDUnimplemented"), args...) } b.simdFeatures(op) switch op { + case SIMDLoad: + ptr := args[0] + b.AssertNilDeref(ptr) + v := llvm.CreateLoad(b.impl, result.ll, ptr.impl) + v.SetAlignment(int(b.Prog.AlignOf(b.Prog.Elem(ptr.Type)))) + return Expr{v, result} + case SIMDStore: + ptr := args[1] + b.AssertNilDeref(ptr) + v := b.impl.CreateStore(args[0].impl, ptr.impl) + v.SetAlignment(int(b.Prog.AlignOf(b.Prog.Elem(ptr.Type)))) + return Expr{v, b.Prog.Void()} + case SIMDBroadcast: + v := b.impl.CreateInsertElement(llvm.Undef(result.ll), args[0].impl, llvm.ConstInt(b.Prog.tyInt32(), 0, false), "") + return b.simdSplatLane0(Expr{v, result}) + case SIMDSplatLane0: + return b.simdSplatLane0(args[0]) case SIMDExtractLane: return b.simdGetElem(args[0], args[1]) case SIMDInsertLane: @@ -97,7 +121,7 @@ func (b Builder) SIMD(op SIMDOp, args ...Expr) Expr { func (b Builder) simdFeatures(op SIMDOp) { switch op { - case SIMDAdd, SIMDSub, SIMDAnd, SIMDOr, SIMDXor, SIMDExtractLane, SIMDInsertLane: + case SIMDAdd, SIMDSub, SIMDAnd, SIMDOr, SIMDXor, SIMDExtractLane, SIMDInsertLane, SIMDLoad, SIMDStore, SIMDBroadcast, SIMDSplatLane0: default: panic("unsupported SIMD operation") } @@ -257,3 +281,11 @@ func llvmTypeHasVector(t llvm.Type) bool { } return false } + +func (b Builder) simdSplatLane0(x Expr) Expr { + mask := make([]llvm.Value, simdLanes(x.RawType()).Len()) + for i := range mask { + mask[i] = llvm.ConstInt(b.Prog.tyInt32(), 0, false) + } + return Expr{b.impl.CreateShuffleVector(x.impl, llvm.Undef(x.ll), llvm.ConstVector(mask, false), ""), x.Type} +} diff --git a/test/simd/README.md b/test/simd/README.md index 0eae7afca2..930209e859 100644 --- a/test/simd/README.md +++ b/test/simd/README.md @@ -14,16 +14,14 @@ GOEXPERIMENT=simd llgo test -O0 -count=1 ./test/simd/... GOEXPERIMENT=simd llgo test -O2 -count=1 ./test/simd/... GOEXPERIMENT=simd GOOS=wasip1 GOARCH=wasm go test -exec=wasmtime -count=1 ./test/simd/... -GOEXPERIMENT=simd llgo test -O0 -target wasi -emulator -count=1 -timeout=2m ./test/simd/... GOEXPERIMENT=simd llgo test -O2 -target wasi -emulator -count=1 -timeout=2m ./test/simd/... ``` -WebAssembly execution requires Wasmtime for Go and Wasmer 7.5.0 for LLGo. -The LLGo WASI runner supports SIMD, shared-memory threads and standard Wasm -exception handling, with Wasmer choosing an available backend automatically. -CI runs the native suite on amd64/arm64 and the WebAssembly suite in the existing -wasm test-command job. LLGo runs at O0 and O2 on both native and WebAssembly targets. -The shared suite covers implemented SIMD128 operations. +WASI execution requires Wasmtime and LLGo's supported Binaryen (`WASMOPT`). +CI runs the native suite on amd64/arm64 and the WASI suite in the existing wasm +test-command job. Native LLGo runs at O0 and O2; WASI runs at O2 because the unoptimized +standard testing framework exceeds Wasmtime's local-variable limit. The shared suite covers implemented SIMD128 operations, array/slice memory access, +and broadcast, including element-aligned addresses, short slices, and nil arrays. `unimplemented_llgo_test.go` checks that remaining intrinsic declarations panic with their symbol name, including indirect, deferred, and linkname calls. SIMD reflection is outside this stage's scope, matching the Go 1.27 support diff --git a/test/simd/memory_test.go b/test/simd/memory_test.go new file mode 100644 index 0000000000..308495f032 --- /dev/null +++ b/test/simd/memory_test.go @@ -0,0 +1,121 @@ +//go:build goexperiment.simd && (amd64 || arm64 || wasm) + +package simd_test + +import ( + "math" + "simd/archsimd" + "testing" +) + +// Package initialization must be able to call both Go broadcast helpers and +// their bodyless implementation, before any user function runs. +var memoryResult archsimd.Float32x4 + +var broadcastAtInit = archsimd.BroadcastFloat32x4(3.5) + +//go:noinline +func indirectLoad(f func(*[4]float32) archsimd.Float32x4, p *[4]float32) archsimd.Float32x4 { + return f(p) +} + +func TestSIMDMemory(t *testing.T) { + values := [4]float32{math.Float32frombits(0x80000000), math.Float32frombits(0x7fc12345), 3.25, -7.5} + for offset := 0; offset < 4; offset++ { + // Exercise element-aligned addresses at every offset within a vector. + src := make([]float32, 8) + dst := make([]float32, 8) + copy(src[offset:], values[:]) + for i := range dst { + dst[i] = 123 + } + v := archsimd.LoadFloat32x4(src[offset:]) + store := v.Store + store(dst[offset:]) + for i := range dst { + want := float32(123) + if i >= offset && i < offset+4 { + want = values[i-offset] + } + if math.Float32bits(dst[i]) != math.Float32bits(want) { + t.Fatalf("offset %d lane %d: got %08x want %08x", offset, i, math.Float32bits(dst[i]), math.Float32bits(want)) + } + } + } + v := indirectLoad(archsimd.LoadFloat32x4Array, &values) + var copied [4]float32 + v.StoreArray(&copied) + for i := range values { + if math.Float32bits(copied[i]) != math.Float32bits(values[i]) { + t.Fatalf("indirect load lane %d", i) + } + } + + bytes := [16]byte{0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 255} + var byteCopy [16]byte + archsimd.LoadUint8x16Array(&bytes).StoreArray(&byteCopy) + if byteCopy != bytes { + t.Fatal("byte memory roundtrip") + } + words := [2]int64{math.MinInt64, math.MaxInt64} + var wordCopy [2]int64 + archsimd.LoadInt64x2Array(&words).StoreArray(&wordCopy) + if wordCopy != words { + t.Fatal("integer memory roundtrip") + } + doubles := [2]float64{math.Inf(-1), math.SmallestNonzeroFloat64} + var doubleCopy [2]float64 + archsimd.LoadFloat64x2Array(&doubles).StoreArray(&doubleCopy) + if doubleCopy != doubles { + t.Fatal("double memory roundtrip") + } +} + +func TestSIMDBroadcast(t *testing.T) { + var floats [4]float32 + broadcastAtInit.StoreArray(&floats) + if floats != [4]float32{3.5, 3.5, 3.5, 3.5} { + t.Fatal(floats) + } + var bytes [16]int8 + archsimd.BroadcastInt8x16(-37).StoreArray(&bytes) + for i, x := range bytes { + if x != -37 { + t.Fatalf("byte lane %d = %d", i, x) + } + } + var words [2]uint64 + archsimd.BroadcastUint64x2(0xfedcba9876543210).StoreArray(&words) + if words != [2]uint64{0xfedcba9876543210, 0xfedcba9876543210} { + t.Fatal(words) + } + var doubles [2]float64 + archsimd.BroadcastFloat64x2(math.Copysign(0, -1)).StoreArray(&doubles) + for _, x := range doubles { + if math.Float64bits(x) != 1<<63 { + t.Fatal("broadcast lost signed zero") + } + } +} + +func TestSIMDMemoryBounds(t *testing.T) { + var v archsimd.Float32x4 + for _, tc := range []struct { + name string + f func() + }{ + {"short load", func() { archsimd.LoadFloat32x4(make([]float32, 3)) }}, + {"short store", func() { v.Store(make([]float32, 3)) }}, + {"nil array load", func() { memoryResult = archsimd.LoadFloat32x4Array(nil) }}, + {"nil array store", func() { v.StoreArray(nil) }}, + } { + t.Run(tc.name, func(t *testing.T) { + defer func() { + if recover() == nil { + t.Fatal("missing memory bounds panic") + } + }() + tc.f() + }) + } +} diff --git a/test/simd/unimplemented_llgo_test.go b/test/simd/unimplemented_llgo_test.go index cd1267bef9..ab8a25793a 100644 --- a/test/simd/unimplemented_llgo_test.go +++ b/test/simd/unimplemented_llgo_test.go @@ -26,8 +26,6 @@ func TestUnimplementedSIMD(t *testing.T) { {"deferred", func() { defer x.Div(x) }, "simd/archsimd.Float32x4.Div"}, {"linkname", func() { simdDiv(x, x) }, "simd/archsimd.Float32x4.Div"}, {"mask result", func() { x.Equal(x) }, "simd/archsimd.Float32x4.Equal"}, - {"package function", func() { archsimd.LoadFloat32x4Array(new([4]float32)) }, "simd/archsimd.LoadFloat32x4Array"}, - {"helper", func() { archsimd.BroadcastFloat32x4(1) }, "simd/archsimd."}, } { t.Run(tc.name, func(t *testing.T) { defer func() { From 159c54d63618b98221cd8acea70659ec3d40c5a3 Mon Sep 17 00:00:00 2001 From: ZhouGuangyuan Date: Sat, 3 Oct 2026 00:14:30 +0800 Subject: [PATCH 02/20] simd: lower vector arithmetic rounding and bitcasts --- cl/simd.go | 56 +++++++++---- cl/simd_test.go | 2 +- internal/build/simd_test.go | 25 +++++- ssa/simd.go | 118 +++++++++++++++++++++++++-- test/simd/arithmetic_arm64_test.go | 19 +++++ test/simd/arithmetic_test.go | 107 ++++++++++++++++++++++++ test/simd/unimplemented_llgo_test.go | 16 ++-- 7 files changed, 308 insertions(+), 35 deletions(-) create mode 100644 test/simd/arithmetic_arm64_test.go create mode 100644 test/simd/arithmetic_test.go diff --git a/cl/simd.go b/cl/simd.go index ee7e2a78b7..3facb026eb 100644 --- a/cl/simd.go +++ b/cl/simd.go @@ -23,38 +23,54 @@ const ( simdLoad simdStore simdBroadcast + simdBitcast ) type simdOperation struct { - op llssa.SIMDOp - signature simdSignature - integerOnly bool + op llssa.SIMDOp + signature simdSignature + elements types.BasicInfo } // The default entry gives not-yet-implemented intrinsic declarations a defined, // recoverable failure. Adding an implementation replaces this fallback for that // operation; functions with Go bodies continue through normal compilation. var simdOperations = map[simdKey]simdOperation{ - {"*", "*"}: {llssa.SIMDUnimplemented, simdUnsupported, false}, - {"numeric", "Add"}: {llssa.SIMDAdd, simdBinary, false}, - {"numeric", "Sub"}: {llssa.SIMDSub, simdBinary, false}, - {"numeric", "And"}: {llssa.SIMDAnd, simdBinary, true}, - {"numeric", "Or"}: {llssa.SIMDOr, simdBinary, true}, - {"numeric", "Xor"}: {llssa.SIMDXor, simdBinary, true}, - {"numeric", "GetElem"}: {llssa.SIMDExtractLane, simdExtract, false}, - {"numeric", "SetElem"}: {llssa.SIMDInsertLane, simdInsert, false}, - {"numeric", "StoreArray"}: {llssa.SIMDStore, simdStore, false}, + {"*", "*"}: {llssa.SIMDUnimplemented, simdUnsupported, 0}, + {"numeric", "Add"}: {llssa.SIMDAdd, simdBinary, 0}, + {"numeric", "Sub"}: {llssa.SIMDSub, simdBinary, 0}, + {"numeric", "And"}: {llssa.SIMDAnd, simdBinary, types.IsInteger}, + {"numeric", "Or"}: {llssa.SIMDOr, simdBinary, types.IsInteger}, + {"numeric", "Xor"}: {llssa.SIMDXor, simdBinary, types.IsInteger}, + {"numeric", "GetElem"}: {llssa.SIMDExtractLane, simdExtract, 0}, + {"numeric", "SetElem"}: {llssa.SIMDInsertLane, simdInsert, 0}, + {"numeric", "StoreArray"}: {llssa.SIMDStore, simdStore, 0}, + {"numeric", "Mul"}: {llssa.SIMDMul, simdBinary, 0}, + {"numeric", "Div"}: {llssa.SIMDDiv, simdBinary, types.IsFloat}, + {"numeric", "AndNot"}: {llssa.SIMDAndNot, simdBinary, types.IsInteger}, + {"numeric", "OrNot"}: {llssa.SIMDOrNot, simdBinary, types.IsInteger}, + {"numeric", "Not"}: {llssa.SIMDNot, simdUnary, types.IsInteger}, + {"numeric", "Neg"}: {llssa.SIMDNeg, simdUnary, 0}, + {"numeric", "Abs"}: {llssa.SIMDAbs, simdUnary, 0}, + {"numeric", "Sqrt"}: {llssa.SIMDSqrt, simdUnary, types.IsFloat}, + {"numeric", "Ceil"}: {llssa.SIMDCeil, simdUnary, types.IsFloat}, + {"numeric", "Floor"}: {llssa.SIMDFloor, simdUnary, types.IsFloat}, + {"numeric", "Trunc"}: {llssa.SIMDTrunc, simdUnary, types.IsFloat}, + {"numeric", "Round"}: {llssa.SIMDRound, simdUnary, types.IsFloat}, } // These registrations share lowering but retain exact declaration names and // signature checks. Source Go slice helpers keep their own bounds checks. func init() { + for _, name := range []string{"ToBits", "BitsToInt8", "BitsToInt16", "BitsToInt32", "BitsToInt64", "BitsToFloat32", "BitsToFloat64", "ReshapeToUint8s", "ReshapeToUint16s", "ReshapeToUint32s", "ReshapeToUint64s"} { + simdOperations[simdKey{"numeric", name}] = simdOperation{llssa.SIMDBitcast, simdBitcast, 0} + } for _, name := range []string{"Int8x16", "Uint8x16", "Int16x8", "Uint16x8", "Int32x4", "Uint32x4", "Int64x2", "Uint64x2", "Float32x4", "Float64x2"} { - simdOperations[simdKey{"", "Load" + name + "Array"}] = simdOperation{llssa.SIMDLoad, simdLoad, false} - simdOperations[simdKey{"", "Broadcast" + name}] = simdOperation{llssa.SIMDBroadcast, simdBroadcast, false} + simdOperations[simdKey{"", "Load" + name + "Array"}] = simdOperation{llssa.SIMDLoad, simdLoad, 0} + simdOperations[simdKey{"", "Broadcast" + name}] = simdOperation{llssa.SIMDBroadcast, simdBroadcast, 0} } for _, lanes := range []int{2, 4, 8, 16} { - simdOperations[simdKey{"numeric", fmt.Sprintf("broadcast1To%d", lanes)}] = simdOperation{llssa.SIMDSplatLane0, simdUnary, false} + simdOperations[simdKey{"numeric", fmt.Sprintf("broadcast1To%d", lanes)}] = simdOperation{llssa.SIMDSplatLane0, simdUnary, 0} } } @@ -112,12 +128,20 @@ func (d simdOperation) matches(sig *types.Signature, vector types.Type) bool { if !ok { return false } - if d.integerOnly && lanes.Elem().Underlying().(*types.Basic).Info()&types.IsInteger == 0 { + if d.elements != 0 && lanes.Elem().Underlying().(*types.Basic).Info()&d.elements == 0 { return false } var params []types.Type result := vector switch d.signature { + case simdBitcast: + if sig.Results().Len() != 1 { + return false + } + result = sig.Results().At(0).Type() + if _, ok := llssa.SIMDNumericShape(result); !ok { + return false + } case simdUnary: // Receiver only. case simdLoad, simdStore: diff --git a/cl/simd_test.go b/cl/simd_test.go index 35ae2ea729..5ad8934f44 100644 --- a/cl/simd_test.go +++ b/cl/simd_test.go @@ -23,7 +23,7 @@ func TestSIMDOperationIdentity(t *testing.T) { // A test-only package-function registration exercises the same signature // family without adding an operation to the supported official API. key := simdKey{"", "Sum"} - simdOperations[key] = simdOperation{llssa.SIMDAdd, simdBinary, false} + simdOperations[key] = simdOperation{llssa.SIMDAdd, simdBinary, 0} defer delete(simdOperations, key) for _, tc := range []struct { name, path, arch, source string diff --git a/internal/build/simd_test.go b/internal/build/simd_test.go index b61a352c70..eda25a6f90 100644 --- a/internal/build/simd_test.go +++ b/internal/build/simd_test.go @@ -33,6 +33,11 @@ func loop(x archsimd.Float32x4, n int) (archsimd.Float32x4, int) { } func load(p *[4]float32) archsimd.Float32x4 { return archsimd.LoadFloat32x4Array(p) } func store(p *[4]float32, x archsimd.Float32x4) { x.StoreArray(p) } +func arithmetic(x, y archsimd.Float32x4) archsimd.Float32x4 { return x.Mul(y).Div(y).Sqrt().Round() } +func bitcast(x archsimd.Uint32x4) archsimd.Float32x4 { return x.BitsToFloat32() } +func abs(x archsimd.Int32x4) archsimd.Int32x4 { return x.Abs() } +func round32(x archsimd.Float32x4) archsimd.Float32x4 { return x.Round() } +func round64(x archsimd.Float64x2) archsimd.Float64x2 { return x.Round() } func fixed(x archsimd.Float32x4) float32 { return x.GetElem(1) } func boxed(x any) archsimd.Float32x4 { return x.(archsimd.Float32x4) } func invoke(x, y archsimd.Float32x4) { defer x.Add(y); go x.Sub(y) } @@ -92,6 +97,21 @@ func TestSIMD128LLVM(t *testing.T) { t.Fatalf("%s does not use element-aligned vector memory:\n%s", name, ir) } } + for name, instructions := range map[string][]string{ + "arithmetic": {"fmul <4 x float>", "fdiv <4 x float>", "@llvm.sqrt.v4f32"}, + "bitcast": {"bitcast <4 x i32>", "to <4 x float>"}, + "abs": {"@llvm.abs.v4i32", "i1 false"}, + } { + ir := mod.NamedFunction("main." + name).String() + for _, instruction := range instructions { + if !strings.Contains(ir, instruction) { + t.Fatalf("%s missing %s:\n%s", name, instruction, ir) + } + } + } + if target.arch != "amd64" && !strings.Contains(mod.NamedFunction("main.round32").String(), "@llvm.roundeven.v4f32") { + t.Fatal("missing native vector roundeven") + } identity := mod.NamedFunction("main.identity") if identity.GlobalValueType().ReturnType().TypeKind() != llvm.VectorTypeKind || identity.GlobalValueType().ParamTypes()[0].TypeKind() != llvm.VectorTypeKind { t.Fatalf("identity does not use a vector ABI:\n%s", identity.String()) @@ -129,6 +149,9 @@ func TestSIMD128LLVM(t *testing.T) { if !strings.Contains(string(asm.Bytes()), want) { t.Fatalf("missing %s in assembly", want) } + if target.arch == "amd64" && strings.Contains(string(asm.Bytes()), "roundeven") { + t.Fatal("baseline rounding requires nonportable libm roundeven") + } }) } @@ -155,7 +178,7 @@ func TestSIMDIntrinsicDefinitions(t *testing.T) { if err := llvm.VerifyModule(mod, llvm.ReturnStatusAction); err != nil { t.Fatal(err) } - if fn := mod.NamedFunction("simd/archsimd.Float32x4.Div"); fn.IsNil() || !strings.Contains(fn.String(), "PanicSIMDUnimplemented") { + if fn := mod.NamedFunction("simd/archsimd.Float32x4.Min"); fn.IsNil() || !strings.Contains(fn.String(), "PanicSIMDUnimplemented") { t.Fatal("missing explicit unsupported implementation") } if fn := mod.NamedFunction("simd/archsimd.Float32x4.Add"); fn.IsNil() || !strings.Contains(fn.String(), "fadd <4 x float>") { diff --git a/ssa/simd.go b/ssa/simd.go index 3ab610046a..b7c84801e2 100644 --- a/ssa/simd.go +++ b/ssa/simd.go @@ -24,6 +24,19 @@ const ( SIMDStore SIMDBroadcast SIMDSplatLane0 + SIMDMul + SIMDDiv + SIMDAndNot + SIMDOrNot + SIMDNot + SIMDNeg + SIMDAbs + SIMDSqrt + SIMDCeil + SIMDFloor + SIMDTrunc + SIMDRound + SIMDBitcast ) // SIMDNumericShape validates the official numeric aggregate representation. @@ -81,9 +94,9 @@ func simdLanes(typ types.Type) *types.Array { return lanes } -// SIMD applies the selected operation's feature requirements before lowering. -// These seven implementations need only baseline native instructions; wasm -// requires SIMD128. Future feature-specific implementations extend this entry. +// SIMD lowers a validated source operation. LLVM legalizes these operations +// for the native baseline; wasm needs the SIMD128 feature. The result type is +// explicit for loads and conversions whose result differs from their operands. func (b Builder) SIMD(op SIMDOp, result Type, args ...Expr) Expr { if op == SIMDUnimplemented { if len(args) != 1 || !types.Identical(args[0].RawType(), types.Typ[types.String]) { @@ -92,7 +105,18 @@ func (b Builder) SIMD(op SIMDOp, result Type, args ...Expr) Expr { return b.Call(b.Pkg.rtFunc("PanicSIMDUnimplemented"), args...) } b.simdFeatures(op) + if op == SIMDRound && b.Prog.Target().GOARCH == "amd64" { + return b.simdRoundEven(args[0]) + } + if name, ok := simdFloatUnary[op]; ok { + v := b.impl.CreateIntrinsic(result.ll, llvm.LookupIntrinsicID(name), []llvm.Value{args[0].impl}, "") + return Expr{v, result} + } switch op { + case SIMDBitcast: + return Expr{b.impl.CreateBitCast(args[0].impl, result.ll, ""), result} + case SIMDNeg, SIMDNot, SIMDAbs: + return b.simdUnary(op, args[0]) case SIMDLoad: ptr := args[0] b.AssertNilDeref(ptr) @@ -114,17 +138,19 @@ func (b Builder) SIMD(op SIMDOp, result Type, args ...Expr) Expr { return b.simdGetElem(args[0], args[1]) case SIMDInsertLane: return b.simdSetElem(args[0], args[1], args[2]) - default: + case SIMDAdd, SIMDSub, SIMDMul, SIMDDiv, SIMDAnd, SIMDOr, SIMDXor, SIMDAndNot, SIMDOrNot: return b.simdBinary(op, args[0], args[1]) + default: + panic("unsupported SIMD operation") } } +var simdFloatUnary = map[SIMDOp]string{ + SIMDSqrt: "llvm.sqrt", SIMDCeil: "llvm.ceil", SIMDFloor: "llvm.floor", + SIMDTrunc: "llvm.trunc", SIMDRound: "llvm.roundeven", +} + func (b Builder) simdFeatures(op SIMDOp) { - switch op { - case SIMDAdd, SIMDSub, SIMDAnd, SIMDOr, SIMDXor, SIMDExtractLane, SIMDInsertLane, SIMDLoad, SIMDStore, SIMDBroadcast, SIMDSplatLane0: - default: - panic("unsupported SIMD operation") - } b.requireSIMDFeatures() } @@ -185,6 +211,18 @@ func (b Builder) simdBinary(op SIMDOp, x, y Expr) Expr { } else { v = b.impl.CreateSub(a, c, "") } + case SIMDMul: + if floating { + v = b.impl.CreateFMul(a, c, "") + } else { + v = b.impl.CreateMul(a, c, "") + } + case SIMDDiv: + v = b.impl.CreateFDiv(a, c, "") + case SIMDAndNot: + v = b.impl.CreateAnd(a, b.impl.CreateNot(c, ""), "") + case SIMDOrNot: + v = b.impl.CreateOr(a, b.impl.CreateNot(c, ""), "") case SIMDAnd: v = b.impl.CreateAnd(a, c, "") case SIMDOr: @@ -289,3 +327,65 @@ func (b Builder) simdSplatLane0(x Expr) Expr { } return Expr{b.impl.CreateShuffleVector(x.impl, llvm.Undef(x.ll), llvm.ConstVector(mask, false), ""), x.Type} } + +func (b Builder) simdUnary(op SIMDOp, x Expr) Expr { + floating := simdLanes(x.RawType()).Elem().Underlying().(*types.Basic).Info()&types.IsFloat != 0 + var v llvm.Value + switch op { + case SIMDNeg: + if floating { + v = llvm.CreateFNeg(b.impl, x.impl) + } else { + v = llvm.CreateNeg(b.impl, x.impl) + } + case SIMDNot: + v = b.impl.CreateNot(x.impl, "") + case SIMDAbs: + if floating { + v = b.impl.CreateIntrinsic(x.ll, llvm.LookupIntrinsicID("llvm.fabs"), []llvm.Value{x.impl}, "") + } else { + // Signed minimum keeps its two's-complement bit pattern, never poison. + v = b.impl.CreateIntrinsic(x.ll, llvm.LookupIntrinsicID("llvm.abs"), []llvm.Value{x.impl, llvm.ConstInt(b.Prog.Bool().ll, 0, false)}, "") + } + default: + panic("invalid SIMD unary operation") + } + return Expr{v, x.Type} +} + +// SSE2 has no rounding instruction. LLVM's roundeven fallback calls a C23 +// libm function that is unavailable on some supported systems. Round the +// significand with integer vectors instead, independently of the FP mode. +func (b Builder) simdRoundEven(x Expr) Expr { + lanes := simdLanes(x.RawType()) + width, fraction, bias := 32, uint64(23), uint64(127) + if lanes.Elem().Underlying().(*types.Basic).Kind() == types.Float64 { + width, fraction, bias = 64, 52, 1023 + } + integer := b.Prog.ctx.IntType(width) + vector := llvm.VectorType(integer, int(lanes.Len())) + constant := func(value uint64) llvm.Value { + values := make([]llvm.Value, lanes.Len()) + for i := range values { + values[i] = llvm.ConstInt(integer, value, false) + } + return llvm.ConstVector(values, false) + } + bits := b.impl.CreateBitCast(x.impl, vector, "") + sign := b.impl.CreateAnd(bits, constant(uint64(1)<<(width-1)), "") + magnitude := b.impl.CreateAnd(bits, constant((uint64(1)<<(width-1))-1), "") + exponent := b.impl.CreateLShr(magnitude, constant(fraction), "") + small := llvm.CreateICmp(b.impl, llvm.IntULT, exponent, constant(bias)) + hasFraction := llvm.CreateICmp(b.impl, llvm.IntULT, exponent, constant(bias+fraction)) + valid := b.impl.CreateAnd(b.impl.CreateNot(small, ""), hasFraction, "") + // Every shift is in range even in lanes discarded by the final select. + e := b.impl.CreateSelect(valid, b.impl.CreateSub(exponent, constant(bias), ""), constant(0), "") + odd := b.impl.CreateAnd(b.impl.CreateLShr(bits, b.impl.CreateSub(constant(fraction), e, ""), ""), constant(1), "") + increment := b.impl.CreateLShr(b.impl.CreateAdd(constant((uint64(1)<<(fraction-1))-1), odd, ""), e, "") + mask := b.impl.CreateLShr(constant((uint64(1)<>(8*(i%4))) { + t.Fatalf("reshape byte %d = %x", i, b) + } + } +} diff --git a/test/simd/unimplemented_llgo_test.go b/test/simd/unimplemented_llgo_test.go index ab8a25793a..4c2ff03dc9 100644 --- a/test/simd/unimplemented_llgo_test.go +++ b/test/simd/unimplemented_llgo_test.go @@ -9,22 +9,22 @@ import ( _ "unsafe" ) -//go:linkname simdDiv simd/archsimd.Float32x4.Div -func simdDiv(x, y archsimd.Float32x4) archsimd.Float32x4 +//go:linkname simdMin simd/archsimd.Float32x4.Min +func simdMin(x, y archsimd.Float32x4) archsimd.Float32x4 func TestUnimplementedSIMD(t *testing.T) { var x archsimd.Float32x4 - method := x.Div + method := x.Min for _, tc := range []struct { name string call func() symbol string }{ - {"direct", func() { x.Div(x) }, "simd/archsimd.Float32x4.Div"}, - {"method value", func() { method(x) }, "simd/archsimd.Float32x4.Div"}, - {"method expression", func() { indirect(archsimd.Float32x4.Div, x, x) }, "simd/archsimd.Float32x4.Div"}, - {"deferred", func() { defer x.Div(x) }, "simd/archsimd.Float32x4.Div"}, - {"linkname", func() { simdDiv(x, x) }, "simd/archsimd.Float32x4.Div"}, + {"direct", func() { x.Min(x) }, "simd/archsimd.Float32x4.Min"}, + {"method value", func() { method(x) }, "simd/archsimd.Float32x4.Min"}, + {"method expression", func() { indirect(archsimd.Float32x4.Min, x, x) }, "simd/archsimd.Float32x4.Min"}, + {"deferred", func() { defer x.Min(x) }, "simd/archsimd.Float32x4.Min"}, + {"linkname", func() { simdMin(x, x) }, "simd/archsimd.Float32x4.Min"}, {"mask result", func() { x.Equal(x) }, "simd/archsimd.Float32x4.Equal"}, } { t.Run(tc.name, func(t *testing.T) { From 7e65ff54bd1af5a11aa18b851445586604c9a072 Mon Sep 17 00:00:00 2001 From: ZhouGuangyuan Date: Sat, 3 Oct 2026 00:27:33 +0800 Subject: [PATCH 03/20] simd: add vector masks comparisons and conditional selection --- cl/simd.go | 109 +++++++++++++++++------ internal/build/simd_test.go | 33 ++++++- ssa/simd.go | 89 ++++++++++++++++++- ssa/type.go | 4 +- test/simd/mask_test.go | 124 +++++++++++++++++++++++++++ test/simd/unimplemented_llgo_test.go | 1 - 6 files changed, 330 insertions(+), 30 deletions(-) create mode 100644 test/simd/mask_test.go diff --git a/cl/simd.go b/cl/simd.go index 3facb026eb..09174d88c7 100644 --- a/cl/simd.go +++ b/cl/simd.go @@ -24,6 +24,11 @@ const ( simdStore simdBroadcast simdBitcast + simdCompare + simdToMask + simdTernary + simdMaskFromBits + simdMaskToBits ) type simdOperation struct { @@ -36,32 +41,56 @@ type simdOperation struct { // recoverable failure. Adding an implementation replaces this fallback for that // operation; functions with Go bodies continue through normal compilation. var simdOperations = map[simdKey]simdOperation{ - {"*", "*"}: {llssa.SIMDUnimplemented, simdUnsupported, 0}, - {"numeric", "Add"}: {llssa.SIMDAdd, simdBinary, 0}, - {"numeric", "Sub"}: {llssa.SIMDSub, simdBinary, 0}, - {"numeric", "And"}: {llssa.SIMDAnd, simdBinary, types.IsInteger}, - {"numeric", "Or"}: {llssa.SIMDOr, simdBinary, types.IsInteger}, - {"numeric", "Xor"}: {llssa.SIMDXor, simdBinary, types.IsInteger}, - {"numeric", "GetElem"}: {llssa.SIMDExtractLane, simdExtract, 0}, - {"numeric", "SetElem"}: {llssa.SIMDInsertLane, simdInsert, 0}, - {"numeric", "StoreArray"}: {llssa.SIMDStore, simdStore, 0}, - {"numeric", "Mul"}: {llssa.SIMDMul, simdBinary, 0}, - {"numeric", "Div"}: {llssa.SIMDDiv, simdBinary, types.IsFloat}, - {"numeric", "AndNot"}: {llssa.SIMDAndNot, simdBinary, types.IsInteger}, - {"numeric", "OrNot"}: {llssa.SIMDOrNot, simdBinary, types.IsInteger}, - {"numeric", "Not"}: {llssa.SIMDNot, simdUnary, types.IsInteger}, - {"numeric", "Neg"}: {llssa.SIMDNeg, simdUnary, 0}, - {"numeric", "Abs"}: {llssa.SIMDAbs, simdUnary, 0}, - {"numeric", "Sqrt"}: {llssa.SIMDSqrt, simdUnary, types.IsFloat}, - {"numeric", "Ceil"}: {llssa.SIMDCeil, simdUnary, types.IsFloat}, - {"numeric", "Floor"}: {llssa.SIMDFloor, simdUnary, types.IsFloat}, - {"numeric", "Trunc"}: {llssa.SIMDTrunc, simdUnary, types.IsFloat}, - {"numeric", "Round"}: {llssa.SIMDRound, simdUnary, types.IsFloat}, + {"*", "*"}: {llssa.SIMDUnimplemented, simdUnsupported, 0}, + {"numeric", "Add"}: {llssa.SIMDAdd, simdBinary, 0}, + {"numeric", "Sub"}: {llssa.SIMDSub, simdBinary, 0}, + {"numeric", "And"}: {llssa.SIMDAnd, simdBinary, types.IsInteger}, + {"numeric", "Or"}: {llssa.SIMDOr, simdBinary, types.IsInteger}, + {"numeric", "Xor"}: {llssa.SIMDXor, simdBinary, types.IsInteger}, + {"numeric", "GetElem"}: {llssa.SIMDExtractLane, simdExtract, 0}, + {"numeric", "SetElem"}: {llssa.SIMDInsertLane, simdInsert, 0}, + {"numeric", "StoreArray"}: {llssa.SIMDStore, simdStore, 0}, + {"numeric", "Mul"}: {llssa.SIMDMul, simdBinary, 0}, + {"numeric", "Div"}: {llssa.SIMDDiv, simdBinary, types.IsFloat}, + {"numeric", "AndNot"}: {llssa.SIMDAndNot, simdBinary, types.IsInteger}, + {"numeric", "OrNot"}: {llssa.SIMDOrNot, simdBinary, types.IsInteger}, + {"numeric", "Not"}: {llssa.SIMDNot, simdUnary, types.IsInteger}, + {"numeric", "Neg"}: {llssa.SIMDNeg, simdUnary, 0}, + {"numeric", "Abs"}: {llssa.SIMDAbs, simdUnary, 0}, + {"numeric", "Sqrt"}: {llssa.SIMDSqrt, simdUnary, types.IsFloat}, + {"numeric", "Ceil"}: {llssa.SIMDCeil, simdUnary, types.IsFloat}, + {"numeric", "Floor"}: {llssa.SIMDFloor, simdUnary, types.IsFloat}, + {"numeric", "Trunc"}: {llssa.SIMDTrunc, simdUnary, types.IsFloat}, + {"numeric", "Round"}: {llssa.SIMDRound, simdUnary, types.IsFloat}, + {"numeric", "Equal"}: {llssa.SIMDEqual, simdCompare, 0}, + {"numeric", "NotEqual"}: {llssa.SIMDNotEqual, simdCompare, 0}, + {"numeric", "Less"}: {llssa.SIMDLess, simdCompare, 0}, + {"numeric", "LessEqual"}: {llssa.SIMDLessEqual, simdCompare, 0}, + {"numeric", "Greater"}: {llssa.SIMDGreater, simdCompare, 0}, + {"numeric", "GreaterEqual"}: {llssa.SIMDGreaterEqual, simdCompare, 0}, + {"numeric", "ToMask"}: {llssa.SIMDToMask, simdToMask, types.IsInteger}, + {"numeric", "asMask"}: {llssa.SIMDBitcast, simdToMask, types.IsInteger}, + {"numeric", "bitSelect"}: {llssa.SIMDBitSelect, simdTernary, types.IsInteger}, + {"numeric", "BitSelect"}: {llssa.SIMDBitSelect, simdTernary, types.IsInteger}, + {"numeric", "bitSelectNot"}: {llssa.SIMDBitSelectNot, simdTernary, types.IsInteger}, + {"numeric", "blend"}: {llssa.SIMDBlend, simdTernary, types.IsInteger}, + {"mask", "And"}: {llssa.SIMDAnd, simdBinary, 0}, + {"mask", "Or"}: {llssa.SIMDOr, simdBinary, 0}, + {"mask", "Xor"}: {llssa.SIMDXor, simdBinary, 0}, + {"mask", "AndNot"}: {llssa.SIMDAndNot, simdBinary, 0}, + {"mask", "Not"}: {llssa.SIMDNot, simdUnary, 0}, } // These registrations share lowering but retain exact declaration names and // signature checks. Source Go slice helpers keep their own bounds checks. func init() { + for _, name := range []string{"Mask8x16", "Mask16x8", "Mask32x4", "Mask64x2"} { + simdOperations[simdKey{"", name + "FromBits"}] = simdOperation{llssa.SIMDMaskFromBits, simdMaskFromBits, 0} + } + simdOperations[simdKey{"mask", "ToBits"}] = simdOperation{llssa.SIMDMaskToBits, simdMaskToBits, 0} + for _, name := range []string{"ToInt8x16", "ToInt16x8", "ToInt32x4", "ToInt64x2"} { + simdOperations[simdKey{"mask", name}] = simdOperation{llssa.SIMDBitcast, simdBitcast, 0} + } for _, name := range []string{"ToBits", "BitsToInt8", "BitsToInt16", "BitsToInt32", "BitsToInt64", "BitsToFloat32", "BitsToFloat64", "ReshapeToUint8s", "ReshapeToUint16s", "ReshapeToUint32s", "ReshapeToUint64s"} { simdOperations[simdKey{"numeric", name}] = simdOperation{llssa.SIMDBitcast, simdBitcast, 0} } @@ -106,10 +135,13 @@ func lookupSIMD(fn *ssa.Function, arch string) (simdOperation, bool) { var vector types.Type if recv := sig.Recv(); recv != nil { vector = recv.Type() - if _, ok := llssa.SIMDNumericShape(vector); !ok { + if _, ok := llssa.SIMDNumericShape(vector); ok { + key.receiver = "numeric" + } else if _, ok := llssa.SIMDMaskShape(vector); ok { + key.receiver = "mask" + } else { return fallback() } - key.receiver = "numeric" } else if sig.Results().Len() == 1 { vector = sig.Results().At(0).Type() } @@ -124,7 +156,7 @@ func (d simdOperation) matches(sig *types.Signature, vector types.Type) bool { if vector == nil || sig.Variadic() { return false } - lanes, ok := llssa.SIMDNumericShape(vector) + lanes, ok := llssa.SIMDVectorShape(vector) if !ok { return false } @@ -134,6 +166,33 @@ func (d simdOperation) matches(sig *types.Signature, vector types.Type) bool { var params []types.Type result := vector switch d.signature { + case simdMaskFromBits, simdMaskToBits: + if _, ok := llssa.SIMDMaskShape(vector); !ok { + return false + } + bits := types.Typ[types.Uint8] + if lanes.Len() == 16 { + bits = types.Typ[types.Uint16] + } + if d.signature == simdMaskFromBits { + params = []types.Type{bits} + } else { + result = bits + } + case simdCompare, simdToMask: + if sig.Results().Len() != 1 { + return false + } + result = sig.Results().At(0).Type() + mask, ok := llssa.SIMDMaskShape(result) + if !ok || mask.Len() != lanes.Len() { + return false + } + if d.signature == simdCompare { + params = []types.Type{vector} + } + case simdTernary: + params = []types.Type{vector, vector} case simdBitcast: if sig.Results().Len() != 1 { return false @@ -160,7 +219,7 @@ func (d simdOperation) matches(sig *types.Signature, vector types.Type) bool { default: return false } - if sig.Recv() == nil && d.signature != simdLoad && d.signature != simdBroadcast { + if sig.Recv() == nil && d.signature != simdLoad && d.signature != simdBroadcast && d.signature != simdMaskFromBits { params = append([]types.Type{vector}, params...) } if sig.Params().Len() != len(params) { diff --git a/internal/build/simd_test.go b/internal/build/simd_test.go index eda25a6f90..69f3fb8824 100644 --- a/internal/build/simd_test.go +++ b/internal/build/simd_test.go @@ -38,6 +38,9 @@ func bitcast(x archsimd.Uint32x4) archsimd.Float32x4 { return x.BitsToFloat32() func abs(x archsimd.Int32x4) archsimd.Int32x4 { return x.Abs() } func round32(x archsimd.Float32x4) archsimd.Float32x4 { return x.Round() } func round64(x archsimd.Float64x2) archsimd.Float64x2 { return x.Round() } +func compare(x, y archsimd.Float32x4) archsimd.Mask32x4 { return x.Equal(y) } +func maskpass(x archsimd.Mask32x4) (archsimd.Mask32x4, int) { return x, 1 } +func maskbits(x archsimd.Mask32x4) archsimd.Int32x4 { return x.ToInt32x4() } func fixed(x archsimd.Float32x4) float32 { return x.GetElem(1) } func boxed(x any) archsimd.Float32x4 { return x.(archsimd.Float32x4) } func invoke(x, y archsimd.Float32x4) { defer x.Add(y); go x.Sub(y) } @@ -52,10 +55,22 @@ func main() { } ` +const simdMaskBitmapSource = `package main +import "simd/archsimd" +func maskFrom8(x uint16) archsimd.Mask8x16 { return archsimd.Mask8x16FromBits(x) } +func maskTo8(x archsimd.Mask8x16) uint16 { return x.ToBits() } +func maskFrom16(x uint8) archsimd.Mask16x8 { return archsimd.Mask16x8FromBits(x) } +func maskTo16(x archsimd.Mask16x8) uint8 { return x.ToBits() } +func maskFrom32(x uint8) archsimd.Mask32x4 { return archsimd.Mask32x4FromBits(x) } +func maskTo32(x archsimd.Mask32x4) uint8 { return x.ToBits() } +func maskFrom64(x uint8) archsimd.Mask64x2 { return archsimd.Mask64x2FromBits(x) } +func maskTo64(x archsimd.Mask64x2) uint8 { return x.ToBits() } +` + func simdTestDir(t *testing.T) string { t.Helper() dir := t.TempDir() - for name, text := range map[string]string{"go.mod": "module simdtest\n\ngo 1.27\n", "main.go": simd128Source} { + for name, text := range map[string]string{"go.mod": "module simdtest\n\ngo 1.27\n", "main.go": simd128Source, "bitmap_amd64.go": simdMaskBitmapSource} { if err := os.WriteFile(filepath.Join(dir, name), []byte(text), 0600); err != nil { t.Fatal(err) } @@ -112,6 +127,22 @@ func TestSIMD128LLVM(t *testing.T) { if target.arch != "amd64" && !strings.Contains(mod.NamedFunction("main.round32").String(), "@llvm.roundeven.v4f32") { t.Fatal("missing native vector roundeven") } + if ir := mod.NamedFunction("main.compare").String(); !strings.Contains(ir, "fcmp oeq <4 x float>") || !strings.Contains(ir, "sext <4 x i1>") { + t.Fatalf("comparison does not produce canonical vector mask:\n%s", ir) + } + maskpass := mod.NamedFunction("main.maskpass") + params := maskpass.GlobalValueType().ParamTypes() + if params[len(params)-1].TypeKind() != llvm.VectorTypeKind || !strings.Contains(maskpass.String(), "sret({ <4 x i32>, i64 })") { + t.Fatalf("mask loses vector ABI across multiple-result calls:\n%s", maskpass.String()) + } + if target.arch == "amd64" { + for _, name := range []string{"maskFrom8", "maskFrom16", "maskFrom32", "maskFrom64", "maskTo8", "maskTo16", "maskTo32", "maskTo64"} { + ir := mod.NamedFunction("main." + name).String() + if strings.Contains(ir, "call ") || !strings.Contains(ir, "bitcast") { + t.Fatalf("mask bitmap conversion uses an external call:\n%s", ir) + } + } + } identity := mod.NamedFunction("main.identity") if identity.GlobalValueType().ReturnType().TypeKind() != llvm.VectorTypeKind || identity.GlobalValueType().ParamTypes()[0].TypeKind() != llvm.VectorTypeKind { t.Fatalf("identity does not use a vector ABI:\n%s", identity.String()) diff --git a/ssa/simd.go b/ssa/simd.go index b7c84801e2..2a35d04768 100644 --- a/ssa/simd.go +++ b/ssa/simd.go @@ -37,11 +37,36 @@ const ( SIMDTrunc SIMDRound SIMDBitcast + SIMDEqual + SIMDNotEqual + SIMDLess + SIMDLessEqual + SIMDGreater + SIMDGreaterEqual + SIMDToMask + SIMDBitSelect + SIMDBitSelectNot + SIMDBlend + SIMDMaskFromBits + SIMDMaskToBits ) // SIMDNumericShape validates the official numeric aggregate representation. // Keep storage knowledge here, separate from operation selection and features. func SIMDNumericShape(typ types.Type) (*types.Array, bool) { + lanes, ok := SIMDVectorShape(typ) + return lanes, ok && !strings.HasPrefix(types.Unalias(typ).(*types.Named).Obj().Name(), "Mask") +} + +// SIMDMaskShape recognizes the four SIMD128 masks. Masks use canonical zero +// or all-one integer lanes, preserving their public Go storage representation. +func SIMDMaskShape(typ types.Type) (*types.Array, bool) { + lanes, ok := SIMDVectorShape(typ) + return lanes, ok && strings.HasPrefix(types.Unalias(typ).(*types.Named).Obj().Name(), "Mask") +} + +// SIMDVectorShape validates numeric and mask SIMD128 storage. +func SIMDVectorShape(typ types.Type) (*types.Array, bool) { named, ok := types.Unalias(typ).(*types.Named) if !ok || named.Obj().Pkg() == nil || named.Obj().Pkg().Path() != "simd/archsimd" { return nil, false @@ -73,6 +98,9 @@ func SIMDNumericShape(typ types.Type) (*types.Array, bool) { } name := elem.Name() expected := fmt.Sprintf("%s%sx%d", strings.ToUpper(name[:1]), name[1:], lanes.Len()) + if strings.HasPrefix(named.Obj().Name(), "Mask") && elem.Info()&types.IsInteger != 0 && elem.Info()&types.IsUnsigned == 0 { + expected = fmt.Sprintf("Mask%dx%d", bits, lanes.Len()) + } if named.Obj().Name() != expected || lanes.Len()*bits != 128 { return nil, false } @@ -87,7 +115,7 @@ func SIMDNumericShape(typ types.Type) (*types.Array, bool) { } func simdLanes(typ types.Type) *types.Array { - lanes, ok := SIMDNumericShape(typ) + lanes, ok := SIMDVectorShape(typ) if !ok { panic("unsupported SIMD numeric storage: " + typ.String()) } @@ -113,6 +141,32 @@ func (b Builder) SIMD(op SIMDOp, result Type, args ...Expr) Expr { return Expr{v, result} } switch op { + case SIMDMaskFromBits: + n := int(simdLanes(result.RawType()).Len()) + bits := b.impl.CreateTrunc(args[0].impl, b.Prog.ctx.IntType(n), "") + mask := b.impl.CreateBitCast(bits, llvm.VectorType(b.Prog.Bool().ll, n), "") + return Expr{b.impl.CreateSExt(mask, result.ll, ""), result} + case SIMDMaskToBits: + n := int(simdLanes(args[0].RawType()).Len()) + mask := llvm.CreateICmp(b.impl, llvm.IntNE, args[0].impl, llvm.ConstNull(args[0].ll)) + bits := b.impl.CreateBitCast(mask, b.Prog.ctx.IntType(n), "") + return Expr{b.impl.CreateZExt(bits, result.ll, ""), result} + case SIMDEqual, SIMDNotEqual, SIMDLess, SIMDLessEqual, SIMDGreater, SIMDGreaterEqual: + return b.simdCompare(op, result, args[0], args[1]) + case SIMDToMask: + cond := llvm.CreateICmp(b.impl, llvm.IntNE, args[0].impl, llvm.ConstNull(args[0].ll)) + return Expr{b.impl.CreateSExt(cond, result.ll, ""), result} + case SIMDBitSelect, SIMDBitSelectNot, SIMDBlend: + x, y, mask := args[0].impl, args[1].impl, args[2].impl + if op == SIMDBlend { + cond := llvm.CreateICmp(b.impl, llvm.IntSLT, mask, llvm.ConstNull(mask.Type())) + return Expr{b.impl.CreateSelect(cond, y, x, ""), result} + } + if op == SIMDBitSelectNot { + x, y = y, x + } + v := b.impl.CreateOr(b.impl.CreateAnd(x, mask, ""), b.impl.CreateAnd(y, b.impl.CreateNot(mask, ""), ""), "") + return Expr{v, result} case SIMDBitcast: return Expr{b.impl.CreateBitCast(args[0].impl, result.ll, ""), result} case SIMDNeg, SIMDNot, SIMDAbs: @@ -145,6 +199,39 @@ func (b Builder) SIMD(op SIMDOp, result Type, args ...Expr) Expr { } } +func (b Builder) simdCompare(op SIMDOp, result Type, x, y Expr) Expr { + info := simdLanes(x.RawType()).Elem().Underlying().(*types.Basic).Info() + var cond llvm.Value + if info&types.IsFloat != 0 { + pred := map[SIMDOp]llvm.FloatPredicate{ + SIMDEqual: llvm.FloatOEQ, SIMDNotEqual: llvm.FloatUNE, + SIMDLess: llvm.FloatOLT, SIMDLessEqual: llvm.FloatOLE, + SIMDGreater: llvm.FloatOGT, SIMDGreaterEqual: llvm.FloatOGE, + }[op] + cond = b.impl.CreateFCmp(pred, x.impl, y.impl, "") + } else { + pred := map[SIMDOp]llvm.IntPredicate{ + SIMDEqual: llvm.IntEQ, SIMDNotEqual: llvm.IntNE, + SIMDLess: llvm.IntSLT, SIMDLessEqual: llvm.IntSLE, + SIMDGreater: llvm.IntSGT, SIMDGreaterEqual: llvm.IntSGE, + }[op] + if info&types.IsUnsigned != 0 { + switch pred { + case llvm.IntSLT: + pred = llvm.IntULT + case llvm.IntSLE: + pred = llvm.IntULE + case llvm.IntSGT: + pred = llvm.IntUGT + case llvm.IntSGE: + pred = llvm.IntUGE + } + } + cond = llvm.CreateICmp(b.impl, pred, x.impl, y.impl) + } + return Expr{b.impl.CreateSExt(cond, result.ll, ""), result} +} + var simdFloatUnary = map[SIMDOp]string{ SIMDSqrt: "llvm.sqrt", SIMDCeil: "llvm.ceil", SIMDFloor: "llvm.floor", SIMDTrunc: "llvm.trunc", SIMDRound: "llvm.roundeven", diff --git a/ssa/type.go b/ssa/type.go index fdbfb4f317..c97798b53c 100644 --- a/ssa/type.go +++ b/ssa/type.go @@ -531,7 +531,7 @@ func (p Program) toLLVMTuple(t *types.Tuple) llvm.Type { // Tuples containing vectors are compiler values, not source Go structs. // Preserve their vector components across multiple-result calls as well. for i := 0; i < t.Len(); i++ { - if _, ok := SIMDNumericShape(t.At(i).Type()); ok { + if _, ok := SIMDVectorShape(t.At(i).Type()); ok { return p.ctx.StructType(p.toLLVMTypes(t, t.Len()), false) } } @@ -678,7 +678,7 @@ func (p Program) toNamed(raw *types.Named) Type { break } } - if lanes, ok := SIMDNumericShape(raw); ok { + if lanes, ok := SIMDVectorShape(raw); ok { typ := &aType{llvm.VectorType(p.rawType(lanes.Elem()).ll, int(lanes.Len())), rawType{raw}, vkSIMD} p.named[name] = typ return typ diff --git a/test/simd/mask_test.go b/test/simd/mask_test.go new file mode 100644 index 0000000000..9efe41995b --- /dev/null +++ b/test/simd/mask_test.go @@ -0,0 +1,124 @@ +//go:build goexperiment.simd && (amd64 || arm64 || wasm) + +package simd_test + +import ( + "math" + "simd/archsimd" + "testing" +) + +func checkMask32(t *testing.T, name string, mask archsimd.Mask32x4, want [4]bool) { + t.Helper() + var got [4]int32 + mask.ToInt32x4().StoreArray(&got) + for i, yes := range want { + value := int32(0) + if yes { + value = -1 + } + if got[i] != value { + t.Fatalf("%s lane %d = %x want %x", name, i, got[i], value) + } + } +} + +func TestSIMDCompare(t *testing.T) { + x, y := [4]int32{math.MinInt32, -1, 7, math.MaxInt32}, [4]int32{math.MaxInt32, -1, -8, math.MinInt32} + a, b := archsimd.LoadInt32x4Array(&x), archsimd.LoadInt32x4Array(&y) + checkMask32(t, "signed equal", a.Equal(b), [4]bool{false, true, false, false}) + checkMask32(t, "signed not equal", a.NotEqual(b), [4]bool{true, false, true, true}) + checkMask32(t, "signed less", a.Less(b), [4]bool{true, false, false, false}) + checkMask32(t, "signed less equal", a.LessEqual(b), [4]bool{true, true, false, false}) + checkMask32(t, "signed greater", a.Greater(b), [4]bool{false, false, true, true}) + checkMask32(t, "signed greater equal", a.GreaterEqual(b), [4]bool{false, true, true, true}) + checkMask32(t, "unsigned less", a.ToBits().Less(b.ToBits()), [4]bool{false, false, true, true}) + checkMask32(t, "unsigned greater equal", a.ToBits().GreaterEqual(b.ToBits()), [4]bool{true, true, false, false}) + fx, fy := [4]float32{float32(math.NaN()), float32(math.Inf(1)), math.Float32frombits(1 << 31), 3}, [4]float32{1, float32(math.Inf(1)), 0, -4} + fa, fb := archsimd.LoadFloat32x4Array(&fx), archsimd.LoadFloat32x4Array(&fy) + checkMask32(t, "float equal", fa.Equal(fb), [4]bool{false, true, true, false}) + checkMask32(t, "float not equal", fa.NotEqual(fb), [4]bool{true, false, false, true}) + checkMask32(t, "float less", fa.Less(fb), [4]bool{}) + checkMask32(t, "float less equal", fa.LessEqual(fb), [4]bool{false, true, true, false}) + checkMask32(t, "float greater", fa.Greater(fb), [4]bool{false, false, false, true}) + checkMask32(t, "float greater equal", fa.GreaterEqual(fb), [4]bool{false, true, true, true}) + checkMask32(t, "NaN rhs", fb.LessEqual(fa), [4]bool{false, true, true, true}) +} + +//go:noinline +func passMask(m archsimd.Mask32x4) (archsimd.Mask32x4, int) { return m, 37 } + +func TestSIMDMaskStorageAndSelect(t *testing.T) { + values := [4]int32{0, 1, -1, math.MinInt32} + m := archsimd.LoadInt32x4Array(&values).ToMask() + checkMask32(t, "nonzero", m, [4]bool{false, true, true, true}) + type record struct { + before byte + masks [2]archsimd.Mask32x4 + after byte + } + r := record{before: 19, masks: [2]archsimd.Mask32x4{m, {}}, after: 23} + var boxed any = r.masks[0] + f := passMask + got, n := f(boxed.(archsimd.Mask32x4)) + if r.before != 19 || r.after != 23 || n != 37 { + t.Fatal("mask storage damaged adjacent data") + } + checkMask32(t, "indirect storage", got.And(m).Or(r.masks[1]), [4]bool{false, true, true, true}) + x := [4]float32{5, math.Float32frombits(0x7fc12345), math.Float32frombits(1 << 31), -7} + y := [4]float32{-2, 99, 99, 99} + a, b := archsimd.LoadFloat32x4Array(&x), archsimd.LoadFloat32x4Array(&y) + var selected, masked [4]float32 + a.IfElse(m, b).StoreArray(&selected) + a.Masked(m).StoreArray(&masked) + for i := range x { + want, wantMasked := x[i], x[i] + if i == 0 { + want, wantMasked = y[i], 0 + } + if math.Float32bits(selected[i]) != math.Float32bits(want) || math.Float32bits(masked[i]) != math.Float32bits(wantMasked) { + t.Fatalf("selection lane %d: %x %x", i, math.Float32bits(selected[i]), math.Float32bits(masked[i])) + } + } + var zero archsimd.Mask32x4 + checkMask32(t, "zero mask", zero, [4]bool{}) +} + +func TestSIMDMaskWidths(t *testing.T) { + var bytes [16]int8 + for i := range bytes { + bytes[i] = int8(i) - 8 + } + a := archsimd.LoadInt8x16Array(&bytes) + var selected [16]int8 + a.IfElse(a.Greater(archsimd.BroadcastInt8x16(0)), archsimd.BroadcastInt8x16(42)).StoreArray(&selected) + for i, x := range bytes { + want := x + if x <= 0 { + want = 42 + } + if selected[i] != want { + t.Fatalf("byte lane %d", i) + } + } + words := [8]uint16{0, 1, 32767, 32768, 65535, 9, 123, 1000} + u := archsimd.LoadUint16x8Array(&words) + var mask16 [8]int16 + u.Greater(archsimd.BroadcastUint16x8(32767)).ToInt16x8().StoreArray(&mask16) + for i, x := range words { + want := int16(0) + if x > 32767 { + want = -1 + } + if mask16[i] != want { + t.Fatalf("uint16 lane %d", i) + } + } + longs := [2]int64{math.MinInt64, math.MaxInt64} + v := archsimd.LoadInt64x2Array(&longs) + var mask64 [2]int64 + v.Less(archsimd.BroadcastInt64x2(0)).ToInt64x2().StoreArray(&mask64) + if mask64 != [2]int64{-1, 0} { + t.Fatal(mask64) + } +} diff --git a/test/simd/unimplemented_llgo_test.go b/test/simd/unimplemented_llgo_test.go index 4c2ff03dc9..c498d0ecbd 100644 --- a/test/simd/unimplemented_llgo_test.go +++ b/test/simd/unimplemented_llgo_test.go @@ -25,7 +25,6 @@ func TestUnimplementedSIMD(t *testing.T) { {"method expression", func() { indirect(archsimd.Float32x4.Min, x, x) }, "simd/archsimd.Float32x4.Min"}, {"deferred", func() { defer x.Min(x) }, "simd/archsimd.Float32x4.Min"}, {"linkname", func() { simdMin(x, x) }, "simd/archsimd.Float32x4.Min"}, - {"mask result", func() { x.Equal(x) }, "simd/archsimd.Float32x4.Equal"}, } { t.Run(tc.name, func(t *testing.T) { defer func() { From e743f5740e6880cabda2d5a277b02027a78eef46 Mon Sep 17 00:00:00 2001 From: ZhouGuangyuan Date: Sat, 3 Oct 2026 00:49:21 +0800 Subject: [PATCH 04/20] packages: attach source patches to each test package variant --- internal/packages/load.go | 8 ++++---- internal/packages/load_test.go | 5 ++++- 2 files changed, 8 insertions(+), 5 deletions(-) diff --git a/internal/packages/load.go b/internal/packages/load.go index 5f5c77fa50..665f649450 100644 --- a/internal/packages/load.go +++ b/internal/packages/load.go @@ -354,10 +354,10 @@ func (tc *typecheckContext) typecheckPackage(pkg *Package) { if tc.dedup.setpath != nil { pkg.PkgPath = tc.dedup.setpath(pkg.PkgPath, pkg.Name) } - // Source-patch files are selected per import path and intentionally - // shared by the ordinary and test-augmented variants. - if _, ok := tc.dedup.checked.Load(pkg.PkgPath); !ok { - tc.dedup.checked.Store(pkg.PkgPath, struct{}{}) + // Source patches are selected by import path, but each package variant + // needs the files. A test-augmented package has a distinct file list. + if _, ok := tc.dedup.checked.Load(pkg.ID); !ok { + tc.dedup.checked.Store(pkg.ID, struct{}{}) if files, ok := tc.dedup.llgoFiles[pkg.PkgPath]; ok { pkg.CompiledGoFiles = append(pkg.CompiledGoFiles, files...) } diff --git a/internal/packages/load_test.go b/internal/packages/load_test.go index 516bc5a0b1..829823c10a 100644 --- a/internal/packages/load_test.go +++ b/internal/packages/load_test.go @@ -53,13 +53,16 @@ func TestDeduperKeepsTestPackageIdentities(t *testing.T) { dir := t.TempDir() baseFile := filepath.Join(dir, "helper.go") testFile := filepath.Join(dir, "helper_test.go") - writeLoadTestFile(t, baseFile, "package helper\nconst Value = 1\n") + patchFile := filepath.Join(dir, "patch.go") + writeLoadTestFile(t, baseFile, "package helper\nconst Value = Patched\n") writeLoadTestFile(t, testFile, "package helper\nconst TestOnly = 2\n") + writeLoadTestFile(t, patchFile, "package helper\nconst Patched = 1\n") const path = "example.com/helper" const testID = path + " [" + path + ".test]" for _, order := range [][]string{{path, testID}, {testID, path}} { t.Run(order[0], func(t *testing.T) { dedup := NewDeduper() + dedup.SetLLGoFiles(map[string][]string{path: {patchFile}}) tc := &typecheckContext{dedup: dedup, cfg: loadTestConfig(dir), fset: token.NewFileSet(), origMode: NeedTypes | NeedTypesInfo} makePackage := func(id string) *Package { pkg := &Package{ID: id, PkgPath: path, Name: "helper", CompiledGoFiles: []string{baseFile}} From 1cb590c7aaa7c081a90117ff7e705f0695e2639d Mon Sep 17 00:00:00 2001 From: ZhouGuangyuan Date: Sat, 3 Oct 2026 00:57:58 +0800 Subject: [PATCH 05/20] runtime: initialize effective CPU features before their consumers --- internal/build/cpu_init_test.go | 152 ++++++++++++++++++ internal/build/plan9asm.go | 2 +- internal/build/plan9asm_altpkg_test.go | 4 + internal/plan9asm/translate.go | 11 ++ internal/plan9asm/translate_helpers_test.go | 38 +++++ .../internal/cpu/features_windows_llgo.go | 13 ++ .../internal/cpu/init_linux_arm64_llgo.go | 15 ++ runtime/_patch/internal/cpu/init_llgo.go | 16 ++ .../_patch/internal/cpu/init_other_llgo.go | 5 + .../_patch/internal/cpu/sysctl_darwin_llgo.go | 28 ++++ runtime/build.go | 1 + .../internal/lib/runtime/link_windows_llgo.go | 12 -- runtime/internal/lib/runtime/os_darwin.go | 47 ------ runtime/internal/runtime/cpu_env.go | 20 +++ 14 files changed, 304 insertions(+), 60 deletions(-) create mode 100644 internal/build/cpu_init_test.go create mode 100644 runtime/_patch/internal/cpu/features_windows_llgo.go create mode 100644 runtime/_patch/internal/cpu/init_linux_arm64_llgo.go create mode 100644 runtime/_patch/internal/cpu/init_llgo.go create mode 100644 runtime/_patch/internal/cpu/init_other_llgo.go create mode 100644 runtime/_patch/internal/cpu/sysctl_darwin_llgo.go delete mode 100644 runtime/internal/lib/runtime/os_darwin.go create mode 100644 runtime/internal/runtime/cpu_env.go diff --git a/internal/build/cpu_init_test.go b/internal/build/cpu_init_test.go new file mode 100644 index 0000000000..c88e2cb270 --- /dev/null +++ b/internal/build/cpu_init_test.go @@ -0,0 +1,152 @@ +//go:build !llgo + +package build + +import ( + "os" + "os/exec" + "path/filepath" + "runtime" + "strings" + "testing" + + "github.com/xgo-dev/llgo/internal/lto" + "github.com/xgo-dev/llgo/internal/optlevel" + "github.com/xgo-dev/llvm" +) + +// This fixture deliberately has no public runtime import. CPU detection must +// also work when archsimd is the only standard-library dependency. +const cpuInitAMD64Source = `package main +import "simd/archsimd" +var earlyAVX = archsimd.X86.AVX() +var earlyAVX2 = archsimd.X86.AVX2() +func main() { + println(earlyAVX, earlyAVX2, archsimd.X86.AVX(), archsimd.X86.AVX2(), archsimd.X86.FMA(), archsimd.X86.AVXAES()) +} + +` + +const cpuInitARM64Source = `package main +import _ "simd/archsimd" +import _ "unsafe" +// Match the official internal/cpu.ARM64 prefix. That variable explicitly +// supports linkname users; no private initialization function is called here. +//go:linkname flags internal/cpu.ARM64 +var flags struct { + _ [128]byte + AES, PMULL, SHA1, SHA2, SHA512, SHA3, CRC32, ATOMICS, CPUID, DIT, SB, Neoverse bool + _ [128]byte +} +var earlyAES = flags.AES +func main() { + println(earlyAES, flags.AES, flags.PMULL, flags.SHA1, flags.SHA2, flags.CRC32) +} +` + +func TestCPUInitializationMatchesGo(t *testing.T) { + if runtime.GOARCH != "amd64" && runtime.GOARCH != "arm64" { + t.Skip("CPU fixture requires amd64 or arm64") + } + dir := t.TempDir() + source := cpuInitAMD64Source + if runtime.GOARCH == "arm64" { + source = cpuInitARM64Source + } + for name, text := range map[string]string{"go.mod": "module cpuprobe\n\ngo 1.27\n", "main.go": source} { + if err := os.WriteFile(filepath.Join(dir, name), []byte(text), 0600); err != nil { + t.Fatal(err) + } + } + reference := filepath.Join(dir, "official") + if runtime.GOOS == "windows" { + reference += ".exe" + } + cmd := exec.Command("go", "build", "-o", reference, ".") + cmd.Dir = dir + cmd.Env = withEnv(os.Environ(), "GOEXPERIMENT=simd") + if out, err := cmd.CombinedOutput(); err != nil { + t.Fatalf("official build: %v\n%s", err, out) + } + run := func(bin, debug string) string { + t.Helper() + cmd := exec.Command(bin) + cmd.Env = withEnv(os.Environ(), "GODEBUG="+debug) + out, err := cmd.CombinedOutput() + if err != nil { + t.Fatalf("run with GODEBUG=%q: %v\n%s", debug, err, out) + } + return strings.TrimSpace(string(out)) + } + debugOptions := []string{"", "cpu.all=off", "cpu.aes=off", "cpu.all=off,cpu.aes=on", "cpu.avx=off"} + if runtime.GOARCH == "arm64" { + debugOptions = debugOptions[:4] + } + for _, level := range []optlevel.Level{optlevel.O0, optlevel.O2} { + t.Run(level.String(), func(t *testing.T) { + conf := NewDefaultConf(ModeBuild) + conf.GOEXPERIMENT, conf.OptLevel, conf.LTO = "simd", level, lto.Off + conf.OutFile = filepath.Join(dir, "llgo-"+level.String()) + if runtime.GOOS == "windows" { + conf.OutFile += ".exe" + } + if _, err := Build(Invocation{Args: []string{"."}, Config: conf, Dir: dir}); err != nil { + t.Fatal(err) + } + for _, debug := range debugOptions { + if got, want := run(conf.OutFile, debug), run(reference, debug); got != want { + t.Fatalf("GODEBUG=%q: LLGo %q, official Go %q", debug, got, want) + } + } + }) + } +} + +func TestCPUInitializationTargetHooks(t *testing.T) { + for _, target := range []struct{ os, arch string }{ + {"darwin", "amd64"}, {"darwin", "arm64"}, {"linux", "amd64"}, {"linux", "arm64"}, + {"windows", "amd64"}, {"windows", "arm64"}, {"wasip1", "wasm"}, + } { + t.Run(target.os+"/"+target.arch, func(t *testing.T) { + conf := NewDefaultConf(ModeGen) + conf.Goos, conf.Goarch = target.os, target.arch + pkgs, err := Do([]string{"internal/cpu"}, conf) + if err != nil { + t.Fatal(err) + } + defer pkgs[0].LPkg.Prog.Dispose() + mod := pkgs[0].LPkg.Module() + if err := llvm.VerifyModule(mod, llvm.ReturnStatusAction); err != nil { + t.Fatal(err) + } + initialized := false + for fn := mod.FirstFunction(); !fn.IsNil(); fn = llvm.NextFunction(fn) { + if strings.HasPrefix(fn.Name(), "internal/cpu.init") && strings.Contains(fn.String(), "internal/cpu.Initialize") { + initialized = strings.Contains(fn.String(), "CPUEnvironment") + } + } + if !initialized { + t.Fatal("CPU initialization does not apply process overrides") + } + var bridge string + switch target.os { + case "darwin": + bridge = "internal/cpu.sysctlbynameInt32" + case "windows": + bridge = "internal/cpu.isProcessorFeaturePresent" + case "linux": + if target.arch == "arm64" { + if !strings.Contains(mod.NamedFunction("internal/cpu.llgoPrepareCPU").String(), "@getauxval") { + t.Fatal("missing AT_HWCAP initialization") + } + } + } + if bridge != "" { + fn := mod.NamedFunction(bridge) + if fn.IsNil() || fn.IsDeclaration() { + t.Fatalf("CPU bridge %s depends on public runtime", bridge) + } + } + }) + } +} diff --git a/internal/build/plan9asm.go b/internal/build/plan9asm.go index c2edf19d5a..c5719e8735 100644 --- a/internal/build/plan9asm.go +++ b/internal/build/plan9asm.go @@ -15,7 +15,7 @@ import ( ) func plan9asmTranslateOptions(conf *Config) llplan9asm.TranslateOptions { - opt := llplan9asm.TranslateOptions{GOARM: conf.GOARM} + opt := llplan9asm.TranslateOptions{GOARM: conf.GOARM, GOAMD64: conf.GOAMD64} if conf.Goarch == "386" && conf.GO386 == "softfloat" { opt.X87Mode = extplan9asm.X87Software } diff --git a/internal/build/plan9asm_altpkg_test.go b/internal/build/plan9asm_altpkg_test.go index 50ffb9409e..ec18d012e7 100644 --- a/internal/build/plan9asm_altpkg_test.go +++ b/internal/build/plan9asm_altpkg_test.go @@ -17,6 +17,7 @@ func TestPlan9AsmTranslateOptions(t *testing.T) { {name: "386 sse2", conf: Config{Goarch: "386", GO386: "sse2"}, want: extplan9asm.X87Auto}, {name: "386 softfloat", conf: Config{Goarch: "386", GO386: "softfloat"}, want: extplan9asm.X87Software}, {name: "other architecture", conf: Config{Goarch: "amd64", GO386: "softfloat"}, want: extplan9asm.X87Auto}, + {name: "amd64 v3", conf: Config{Goarch: "amd64", GOAMD64: "v3"}, want: extplan9asm.X87Auto}, } for _, test := range tests { t.Run(test.name, func(t *testing.T) { @@ -24,6 +25,9 @@ func TestPlan9AsmTranslateOptions(t *testing.T) { if got.GOARM != test.conf.GOARM { t.Fatalf("GOARM = %q, want %q", got.GOARM, test.conf.GOARM) } + if got.GOAMD64 != test.conf.GOAMD64 { + t.Fatalf("GOAMD64 = %q, want %q", got.GOAMD64, test.conf.GOAMD64) + } if got.X87Mode != test.want { t.Fatalf("X87Mode = %v, want %v", got.X87Mode, test.want) } diff --git a/internal/plan9asm/translate.go b/internal/plan9asm/translate.go index 373952800c..e95524d2b1 100644 --- a/internal/plan9asm/translate.go +++ b/internal/plan9asm/translate.go @@ -7,6 +7,7 @@ import ( "regexp" "strings" + archcfg "github.com/xgo-dev/llgo/internal/goarch" "github.com/xgo-dev/llgo/internal/packages" intllvm "github.com/xgo-dev/llgo/internal/xtool/llvm" llssaabi "github.com/xgo-dev/llgo/ssa/abi" @@ -34,6 +35,7 @@ type ModuleTranslation struct { type TranslateOptions struct { AnnotateSource bool GOARM string + GOAMD64 string // X87Mode controls explicit 386 x87 assembly lowering. The zero value uses // the Go-compatible hardware lowering. X87Mode extplan9asm.X87Mode @@ -104,6 +106,15 @@ func TranslateSourceModuleForPkgWithOptions(pkg *packages.Package, sfile string, imports[path] = imp.Types } } + if goarch == "amd64" { + level, err := archcfg.ResolveAMD64(opt.GOAMD64) + if err != nil { + return nil, err + } + // Match cmd/asm's feature macro, including getGOAMD64level used by + // internal/cpu to decide which baseline features GODEBUG may disable. + src = append([]byte("#define GOAMD64_"+level+"\n"), src...) + } tr, err := extplan9asm.TranslateGoModule(extplan9asm.GoPackage{ Path: symbolPkgPath, diff --git a/internal/plan9asm/translate_helpers_test.go b/internal/plan9asm/translate_helpers_test.go index 16efc795b3..06ae213824 100644 --- a/internal/plan9asm/translate_helpers_test.go +++ b/internal/plan9asm/translate_helpers_test.go @@ -1,6 +1,7 @@ package plan9asm import ( + "fmt" "go/ast" "go/importer" "go/parser" @@ -15,6 +16,43 @@ import ( extplan9asm "github.com/xgo-dev/plan9asm" ) +func TestTranslateGOAMD64CPUDetectionLevel(t *testing.T) { + pkg := mustTestPackage(t, "internal/cpu", "package cpu\nfunc getGOAMD64level() int32\n") + asm := []byte(`TEXT ·getGOAMD64level(SB),NOSPLIT,$0-4 +#ifdef GOAMD64_v4 + MOVL $4, ret+0(FP) +#else +#ifdef GOAMD64_v3 + MOVL $3, ret+0(FP) +#else +#ifdef GOAMD64_v2 + MOVL $2, ret+0(FP) +#else + MOVL $1, ret+0(FP) +#endif +#endif +#endif + RET +`) + for _, level := range []string{"", "v1", "v2", "v3", "v4"} { + t.Run(level, func(t *testing.T) { + tr, err := TranslateSourceModuleForPkgWithOptions(pkg, "cpu_x86.s", asm, "linux", "amd64", TranslateOptions{GOAMD64: level}) + if err != nil { + t.Fatal(err) + } + defer tr.Module.Dispose() + want := byte('1') + if level != "" { + want = level[1] + } + ir := tr.Module.NamedFunction("internal/cpu.getGOAMD64level").String() + if !strings.Contains(ir, fmt.Sprintf("ret i32 %c", want)) { + t.Fatalf("wrong CPU baseline for %q:\n%s", level, ir) + } + }) + } +} + func mustTestPackage(t *testing.T, pkgPath, src string) *llpackages.Package { t.Helper() fset := token.NewFileSet() diff --git a/runtime/_patch/internal/cpu/features_windows_llgo.go b/runtime/_patch/internal/cpu/features_windows_llgo.go new file mode 100644 index 0000000000..3020a947bb --- /dev/null +++ b/runtime/_patch/internal/cpu/features_windows_llgo.go @@ -0,0 +1,13 @@ +//go:build windows && !baremetal + +package cpu + +import _ "unsafe" + +// Preserve the Go bool / Win32 BOOL boundary without a public-runtime import. +func isProcessorFeaturePresent(feature uint32) bool { + return llgoProcessorFeaturePresent(feature) != 0 +} + +//go:linkname llgoProcessorFeaturePresent stdcall.IsProcessorFeaturePresent +func llgoProcessorFeaturePresent(feature uint32) int32 diff --git a/runtime/_patch/internal/cpu/init_linux_arm64_llgo.go b/runtime/_patch/internal/cpu/init_linux_arm64_llgo.go new file mode 100644 index 0000000000..58e8a2b1e8 --- /dev/null +++ b/runtime/_patch/internal/cpu/init_linux_arm64_llgo.go @@ -0,0 +1,15 @@ +//go:build linux && arm64 && !baremetal + +package cpu + +import _ "unsafe" + +// Linux supplies the effective user-space CPU capabilities in AT_HWCAP. +// Populate the official variable before its platform detection reads it. +func llgoPrepareCPU() { + const atHWCAP = 16 + HWCap = uint(llgoGetauxval(atHWCAP)) +} + +//go:linkname llgoGetauxval C.getauxval +func llgoGetauxval(uintptr) uintptr diff --git a/runtime/_patch/internal/cpu/init_llgo.go b/runtime/_patch/internal/cpu/init_llgo.go new file mode 100644 index 0000000000..a2980df5bd --- /dev/null +++ b/runtime/_patch/internal/cpu/init_llgo.go @@ -0,0 +1,16 @@ +//go:build (amd64 || arm64 || wasm) && !baremetal + +package cpu + +import _ "unsafe" + +// The standard runtime calls Initialize before package initialization. LLGo's +// hosted runtime uses the normal import graph: initialize this package before +// any consumer can read its feature flags, including archsimd and user inits. +func init() { + llgoPrepareCPU() + Initialize(llgoCPUEnvironment()) +} + +//go:linkname llgoCPUEnvironment github.com/xgo-dev/llgo/runtime/internal/runtime.CPUEnvironment +func llgoCPUEnvironment() string diff --git a/runtime/_patch/internal/cpu/init_other_llgo.go b/runtime/_patch/internal/cpu/init_other_llgo.go new file mode 100644 index 0000000000..3e05f9034c --- /dev/null +++ b/runtime/_patch/internal/cpu/init_other_llgo.go @@ -0,0 +1,5 @@ +//go:build (amd64 || arm64 || wasm) && !(linux && arm64) && !baremetal + +package cpu + +func llgoPrepareCPU() {} diff --git a/runtime/_patch/internal/cpu/sysctl_darwin_llgo.go b/runtime/_patch/internal/cpu/sysctl_darwin_llgo.go new file mode 100644 index 0000000000..040a7ddb30 --- /dev/null +++ b/runtime/_patch/internal/cpu/sysctl_darwin_llgo.go @@ -0,0 +1,28 @@ +//go:build darwin && !ios && !baremetal + +package cpu + +import "unsafe" + +// Keep CPU detection usable in programs that do not import the public runtime +// package. These hooks belong to internal/cpu's own initialization path. +func sysctlbynameInt32(name []byte) (int32, int32) { + var value int32 + size := unsafe.Sizeof(value) + status := llgoSysctlbyname(&name[0], unsafe.Pointer(&value), &size, nil, 0) + return status, value +} + +func sysctlbynameBytes(name, out []byte) int32 { + size := uintptr(len(out)) + return llgoSysctlbyname(&name[0], unsafe.Pointer(&out[0]), &size, nil, 0) +} + +// Older supported Go sources use this name for the same query. +func getsysctlbyname(name []byte) (int32, int32) { + return sysctlbynameInt32(name) +} + +//go:noescape +//go:linkname llgoSysctlbyname C.sysctlbyname +func llgoSysctlbyname(name *byte, out unsafe.Pointer, size *uintptr, new unsafe.Pointer, newSize uintptr) int32 diff --git a/runtime/build.go b/runtime/build.go index 41dd7288be..33e107e799 100644 --- a/runtime/build.go +++ b/runtime/build.go @@ -82,6 +82,7 @@ func SourcePatchReplacesAsmForGOARCH(path, goarch string) bool { var sourcePatchPkgs = map[string]struct{}{ "crypto/internal/constanttime": {}, "crypto/internal/sysrand": {}, + "internal/cpu": {}, "internal/runtime/atomic": {}, "internal/runtime/maps": {}, "internal/runtime/sys": {}, diff --git a/runtime/internal/lib/runtime/link_windows_llgo.go b/runtime/internal/lib/runtime/link_windows_llgo.go index b6e85b950c..1dd9f0d3a7 100644 --- a/runtime/internal/lib/runtime/link_windows_llgo.go +++ b/runtime/internal/lib/runtime/link_windows_llgo.go @@ -66,18 +66,6 @@ func windows_QueryPerformanceFrequency() int64 { return c_queryPerformanceFrequency() } -// Go 1.27 moved the Windows processor-feature query behind this runtime hook. -// Keep the same thin boundary as the official runtime implementation: the -// standard library sees a Go bool, while the native call uses Win32 BOOL. - -//go:linkname c_isProcessorFeaturePresent stdcall.IsProcessorFeaturePresent -func c_isProcessorFeaturePresent(processorFeature uint32) int32 - -//go:linkname cpu_isProcessorFeaturePresent internal/cpu.isProcessorFeaturePresent -func cpu_isProcessorFeaturePresent(processorFeature uint32) bool { - return c_isProcessorFeaturePresent(processorFeature) != 0 -} - // syscall.Setenv and syscall.Unsetenv have already updated the Win32 // environment before calling these hooks. LLGo only needs to propagate the // runtime-observed GODEBUG change. diff --git a/runtime/internal/lib/runtime/os_darwin.go b/runtime/internal/lib/runtime/os_darwin.go deleted file mode 100644 index d4a8cc0f97..0000000000 --- a/runtime/internal/lib/runtime/os_darwin.go +++ /dev/null @@ -1,47 +0,0 @@ -package runtime - -import "unsafe" - -func sysctlbynameInt32(name []byte) (int32, int32) { - out := int32(0) - nout := unsafe.Sizeof(out) - ret := sysctlbyname(&name[0], (*byte)(unsafe.Pointer(&out)), &nout, nil, 0) - return ret, out -} - -func sysctlbynameBytes(name, out []byte) int32 { - nout := uintptr(len(out)) - return sysctlbyname(&name[0], &out[0], &nout, nil, 0) -} - -//go:linkname internal_cpu_getsysctlbyname internal/cpu.getsysctlbyname -func internal_cpu_getsysctlbyname(name []byte) (int32, int32) { - return sysctlbynameInt32(name) -} - -//go:linkname internal_cpu_sysctlbynameInt32 internal/cpu.sysctlbynameInt32 -func internal_cpu_sysctlbynameInt32(name []byte) (int32, int32) { - return sysctlbynameInt32(name) -} - -//go:linkname internal_cpu_sysctlbynameBytes internal/cpu.sysctlbynameBytes -func internal_cpu_sysctlbynameBytes(name, out []byte) int32 { - return sysctlbynameBytes(name, out) -} - -//go:cgo_import_dynamic libc_sysctlbyname sysctlbyname "/usr/lib/libSystem.B.dylib" -var libc_sysctlbyname_trampoline_addr uintptr - -// adapted from runtime/sys_darwin.go in the pattern of sysctl() above, as defined in x/sys/unix -func sysctlbyname(name *byte, old *byte, oldlen *uintptr, new *byte, newlen uintptr) int32 { - r, _, _ := llgo_rawSyscall6( - libc_sysctlbyname_trampoline_addr, - uintptr(unsafe.Pointer(name)), - uintptr(unsafe.Pointer(old)), - uintptr(unsafe.Pointer(oldlen)), - uintptr(unsafe.Pointer(new)), - uintptr(newlen), - 0, - ) - return int32(r) -} diff --git a/runtime/internal/runtime/cpu_env.go b/runtime/internal/runtime/cpu_env.go new file mode 100644 index 0000000000..cc8806be52 --- /dev/null +++ b/runtime/internal/runtime/cpu_env.go @@ -0,0 +1,20 @@ +//go:build !baremetal + +package runtime + +import ( + c "github.com/xgo-dev/llgo/runtime/internal/clite" + _ "unsafe" +) + +//go:linkname cpuGetenv C.getenv +func cpuGetenv(*c.Char) *c.Char + +// CPUEnvironment returns the process-start CPU overrides. The internal/cpu +// initialization hook applies the official feature and GODEBUG policy once. +func CPUEnvironment() string { + if env := cpuGetenv(c.Str("GODEBUG")); env != nil { + return c.GoString(env) + } + return "" +} From aa3bd1314766673f1e0e8af79452eb20c0656633 Mon Sep 17 00:00:00 2001 From: ZhouGuangyuan Date: Sat, 3 Oct 2026 01:07:13 +0800 Subject: [PATCH 06/20] simd: lower bounded shifts and saturating integer operations --- cl/simd.go | 104 +++++++++++++++++++++------------- internal/build/simd_test.go | 4 ++ ssa/simd.go | 13 +++++ ssa/simd_integer.go | 67 ++++++++++++++++++++++ test/simd/integer_test.go | 85 +++++++++++++++++++++++++++ test/simd/shift_amd64_test.go | 26 +++++++++ test/simd/shift_arm64_test.go | 35 ++++++++++++ test/simd/shift_byte_test.go | 25 ++++++++ 8 files changed, 321 insertions(+), 38 deletions(-) create mode 100644 ssa/simd_integer.go create mode 100644 test/simd/integer_test.go create mode 100644 test/simd/shift_amd64_test.go create mode 100644 test/simd/shift_arm64_test.go create mode 100644 test/simd/shift_byte_test.go diff --git a/cl/simd.go b/cl/simd.go index 09174d88c7..0e7f06ab3c 100644 --- a/cl/simd.go +++ b/cl/simd.go @@ -29,6 +29,9 @@ const ( simdTernary simdMaskFromBits simdMaskToBits + simdScalarShift + simdVectorShift + simdSignedShift ) type simdOperation struct { @@ -41,44 +44,53 @@ type simdOperation struct { // recoverable failure. Adding an implementation replaces this fallback for that // operation; functions with Go bodies continue through normal compilation. var simdOperations = map[simdKey]simdOperation{ - {"*", "*"}: {llssa.SIMDUnimplemented, simdUnsupported, 0}, - {"numeric", "Add"}: {llssa.SIMDAdd, simdBinary, 0}, - {"numeric", "Sub"}: {llssa.SIMDSub, simdBinary, 0}, - {"numeric", "And"}: {llssa.SIMDAnd, simdBinary, types.IsInteger}, - {"numeric", "Or"}: {llssa.SIMDOr, simdBinary, types.IsInteger}, - {"numeric", "Xor"}: {llssa.SIMDXor, simdBinary, types.IsInteger}, - {"numeric", "GetElem"}: {llssa.SIMDExtractLane, simdExtract, 0}, - {"numeric", "SetElem"}: {llssa.SIMDInsertLane, simdInsert, 0}, - {"numeric", "StoreArray"}: {llssa.SIMDStore, simdStore, 0}, - {"numeric", "Mul"}: {llssa.SIMDMul, simdBinary, 0}, - {"numeric", "Div"}: {llssa.SIMDDiv, simdBinary, types.IsFloat}, - {"numeric", "AndNot"}: {llssa.SIMDAndNot, simdBinary, types.IsInteger}, - {"numeric", "OrNot"}: {llssa.SIMDOrNot, simdBinary, types.IsInteger}, - {"numeric", "Not"}: {llssa.SIMDNot, simdUnary, types.IsInteger}, - {"numeric", "Neg"}: {llssa.SIMDNeg, simdUnary, 0}, - {"numeric", "Abs"}: {llssa.SIMDAbs, simdUnary, 0}, - {"numeric", "Sqrt"}: {llssa.SIMDSqrt, simdUnary, types.IsFloat}, - {"numeric", "Ceil"}: {llssa.SIMDCeil, simdUnary, types.IsFloat}, - {"numeric", "Floor"}: {llssa.SIMDFloor, simdUnary, types.IsFloat}, - {"numeric", "Trunc"}: {llssa.SIMDTrunc, simdUnary, types.IsFloat}, - {"numeric", "Round"}: {llssa.SIMDRound, simdUnary, types.IsFloat}, - {"numeric", "Equal"}: {llssa.SIMDEqual, simdCompare, 0}, - {"numeric", "NotEqual"}: {llssa.SIMDNotEqual, simdCompare, 0}, - {"numeric", "Less"}: {llssa.SIMDLess, simdCompare, 0}, - {"numeric", "LessEqual"}: {llssa.SIMDLessEqual, simdCompare, 0}, - {"numeric", "Greater"}: {llssa.SIMDGreater, simdCompare, 0}, - {"numeric", "GreaterEqual"}: {llssa.SIMDGreaterEqual, simdCompare, 0}, - {"numeric", "ToMask"}: {llssa.SIMDToMask, simdToMask, types.IsInteger}, - {"numeric", "asMask"}: {llssa.SIMDBitcast, simdToMask, types.IsInteger}, - {"numeric", "bitSelect"}: {llssa.SIMDBitSelect, simdTernary, types.IsInteger}, - {"numeric", "BitSelect"}: {llssa.SIMDBitSelect, simdTernary, types.IsInteger}, - {"numeric", "bitSelectNot"}: {llssa.SIMDBitSelectNot, simdTernary, types.IsInteger}, - {"numeric", "blend"}: {llssa.SIMDBlend, simdTernary, types.IsInteger}, - {"mask", "And"}: {llssa.SIMDAnd, simdBinary, 0}, - {"mask", "Or"}: {llssa.SIMDOr, simdBinary, 0}, - {"mask", "Xor"}: {llssa.SIMDXor, simdBinary, 0}, - {"mask", "AndNot"}: {llssa.SIMDAndNot, simdBinary, 0}, - {"mask", "Not"}: {llssa.SIMDNot, simdUnary, 0}, + {"*", "*"}: {llssa.SIMDUnimplemented, simdUnsupported, 0}, + {"numeric", "Add"}: {llssa.SIMDAdd, simdBinary, 0}, + {"numeric", "Sub"}: {llssa.SIMDSub, simdBinary, 0}, + {"numeric", "And"}: {llssa.SIMDAnd, simdBinary, types.IsInteger}, + {"numeric", "Or"}: {llssa.SIMDOr, simdBinary, types.IsInteger}, + {"numeric", "Xor"}: {llssa.SIMDXor, simdBinary, types.IsInteger}, + {"numeric", "GetElem"}: {llssa.SIMDExtractLane, simdExtract, 0}, + {"numeric", "SetElem"}: {llssa.SIMDInsertLane, simdInsert, 0}, + {"numeric", "StoreArray"}: {llssa.SIMDStore, simdStore, 0}, + {"numeric", "Mul"}: {llssa.SIMDMul, simdBinary, 0}, + {"numeric", "Div"}: {llssa.SIMDDiv, simdBinary, types.IsFloat}, + {"numeric", "AndNot"}: {llssa.SIMDAndNot, simdBinary, types.IsInteger}, + {"numeric", "OrNot"}: {llssa.SIMDOrNot, simdBinary, types.IsInteger}, + {"numeric", "Not"}: {llssa.SIMDNot, simdUnary, types.IsInteger}, + {"numeric", "Neg"}: {llssa.SIMDNeg, simdUnary, 0}, + {"numeric", "Abs"}: {llssa.SIMDAbs, simdUnary, 0}, + {"numeric", "Sqrt"}: {llssa.SIMDSqrt, simdUnary, types.IsFloat}, + {"numeric", "Ceil"}: {llssa.SIMDCeil, simdUnary, types.IsFloat}, + {"numeric", "Floor"}: {llssa.SIMDFloor, simdUnary, types.IsFloat}, + {"numeric", "Trunc"}: {llssa.SIMDTrunc, simdUnary, types.IsFloat}, + {"numeric", "Round"}: {llssa.SIMDRound, simdUnary, types.IsFloat}, + {"numeric", "Equal"}: {llssa.SIMDEqual, simdCompare, 0}, + {"numeric", "NotEqual"}: {llssa.SIMDNotEqual, simdCompare, 0}, + {"numeric", "Less"}: {llssa.SIMDLess, simdCompare, 0}, + {"numeric", "LessEqual"}: {llssa.SIMDLessEqual, simdCompare, 0}, + {"numeric", "Greater"}: {llssa.SIMDGreater, simdCompare, 0}, + {"numeric", "GreaterEqual"}: {llssa.SIMDGreaterEqual, simdCompare, 0}, + {"numeric", "ToMask"}: {llssa.SIMDToMask, simdToMask, types.IsInteger}, + {"numeric", "asMask"}: {llssa.SIMDBitcast, simdToMask, types.IsInteger}, + {"numeric", "bitSelect"}: {llssa.SIMDBitSelect, simdTernary, types.IsInteger}, + {"numeric", "BitSelect"}: {llssa.SIMDBitSelect, simdTernary, types.IsInteger}, + {"numeric", "bitSelectNot"}: {llssa.SIMDBitSelectNot, simdTernary, types.IsInteger}, + {"numeric", "blend"}: {llssa.SIMDBlend, simdTernary, types.IsInteger}, + {"mask", "And"}: {llssa.SIMDAnd, simdBinary, 0}, + {"mask", "Or"}: {llssa.SIMDOr, simdBinary, 0}, + {"mask", "Xor"}: {llssa.SIMDXor, simdBinary, 0}, + {"mask", "AndNot"}: {llssa.SIMDAndNot, simdBinary, 0}, + {"mask", "Not"}: {llssa.SIMDNot, simdUnary, 0}, + {"numeric", "ShiftAllLeft"}: {llssa.SIMDShiftAllLeft, simdScalarShift, types.IsInteger}, + {"numeric", "ShiftAllRight"}: {llssa.SIMDShiftAllRight, simdScalarShift, types.IsInteger}, + {"numeric", "ShiftLeft"}: {llssa.SIMDShiftLeft, simdVectorShift, types.IsInteger}, + {"numeric", "ShiftRight"}: {llssa.SIMDShiftRight, simdVectorShift, types.IsInteger}, + {"numeric", "Shift"}: {llssa.SIMDShift, simdSignedShift, types.IsInteger}, + {"numeric", "AddSaturated"}: {llssa.SIMDAddSaturated, simdBinary, types.IsInteger}, + {"numeric", "SubSaturated"}: {llssa.SIMDSubSaturated, simdBinary, types.IsInteger}, + {"numeric", "Min"}: {llssa.SIMDMin, simdBinary, types.IsInteger}, + {"numeric", "Max"}: {llssa.SIMDMax, simdBinary, types.IsInteger}, } // These registrations share lowering but retain exact declaration names and @@ -166,6 +178,22 @@ func (d simdOperation) matches(sig *types.Signature, vector types.Type) bool { var params []types.Type result := vector switch d.signature { + case simdScalarShift: + params = []types.Type{types.Typ[types.Uint64]} + case simdVectorShift, simdSignedShift: + if sig.Params().Len() != 1 { + return false + } + counts := sig.Params().At(0).Type() + shape, ok := llssa.SIMDNumericShape(counts) + if !ok || shape.Len() != lanes.Len() { + return false + } + info := shape.Elem().Underlying().(*types.Basic).Info() + if info&types.IsInteger == 0 || (info&types.IsUnsigned != 0) != (d.signature == simdVectorShift) { + return false + } + params = []types.Type{counts} case simdMaskFromBits, simdMaskToBits: if _, ok := llssa.SIMDMaskShape(vector); !ok { return false diff --git a/internal/build/simd_test.go b/internal/build/simd_test.go index 69f3fb8824..b997a9d618 100644 --- a/internal/build/simd_test.go +++ b/internal/build/simd_test.go @@ -41,6 +41,8 @@ func round64(x archsimd.Float64x2) archsimd.Float64x2 { return x.Round() } func compare(x, y archsimd.Float32x4) archsimd.Mask32x4 { return x.Equal(y) } func maskpass(x archsimd.Mask32x4) (archsimd.Mask32x4, int) { return x, 1 } func maskbits(x archsimd.Mask32x4) archsimd.Int32x4 { return x.ToInt32x4() } +func shift(x archsimd.Int32x4, n uint64) archsimd.Int32x4 { return x.ShiftAllLeft(n).ShiftAllRight(n) } +func saturated(x, y archsimd.Int8x16) archsimd.Int8x16 { return x.AddSaturated(y).SubSaturated(y).Min(y).Max(y) } func fixed(x archsimd.Float32x4) float32 { return x.GetElem(1) } func boxed(x any) archsimd.Float32x4 { return x.(archsimd.Float32x4) } func invoke(x, y archsimd.Float32x4) { defer x.Add(y); go x.Sub(y) } @@ -116,6 +118,8 @@ func TestSIMD128LLVM(t *testing.T) { "arithmetic": {"fmul <4 x float>", "fdiv <4 x float>", "@llvm.sqrt.v4f32"}, "bitcast": {"bitcast <4 x i32>", "to <4 x float>"}, "abs": {"@llvm.abs.v4i32", "i1 false"}, + "shift": {"icmp uge i64", "shl <4 x i32>", "ashr <4 x i32>"}, + "saturated": {"@llvm.sadd.sat.v16i8", "@llvm.ssub.sat.v16i8", "@llvm.smin.v16i8", "@llvm.smax.v16i8"}, } { ir := mod.NamedFunction("main." + name).String() for _, instruction := range instructions { diff --git a/ssa/simd.go b/ssa/simd.go index 2a35d04768..828ad1afca 100644 --- a/ssa/simd.go +++ b/ssa/simd.go @@ -49,6 +49,15 @@ const ( SIMDBlend SIMDMaskFromBits SIMDMaskToBits + SIMDShiftAllLeft + SIMDShiftAllRight + SIMDShiftLeft + SIMDShiftRight + SIMDShift + SIMDAddSaturated + SIMDSubSaturated + SIMDMin + SIMDMax ) // SIMDNumericShape validates the official numeric aggregate representation. @@ -141,6 +150,10 @@ func (b Builder) SIMD(op SIMDOp, result Type, args ...Expr) Expr { return Expr{v, result} } switch op { + case SIMDShiftAllLeft, SIMDShiftAllRight, SIMDShiftLeft, SIMDShiftRight, SIMDShift: + return b.simdShift(op, args[0], args[1]) + case SIMDAddSaturated, SIMDSubSaturated, SIMDMin, SIMDMax: + return b.simdIntegerIntrinsic(op, args[0], args[1]) case SIMDMaskFromBits: n := int(simdLanes(result.RawType()).Len()) bits := b.impl.CreateTrunc(args[0].impl, b.Prog.ctx.IntType(n), "") diff --git a/ssa/simd_integer.go b/ssa/simd_integer.go new file mode 100644 index 0000000000..ad6433516e --- /dev/null +++ b/ssa/simd_integer.go @@ -0,0 +1,67 @@ +package ssa + +import ( + "go/types" + + "github.com/xgo-dev/llvm" +) + +func simdIntegerConstant(typ llvm.Type, value uint64) llvm.Value { + if typ.TypeKind() != llvm.VectorTypeKind { + return llvm.ConstInt(typ, value, false) + } + values := make([]llvm.Value, typ.VectorSize()) + for i := range values { + values[i] = llvm.ConstInt(typ.ElementType(), value, false) + } + return llvm.ConstVector(values, false) +} + +func (b Builder) simdShift(op SIMDOp, x, count Expr) Expr { + distance := count.impl + if op == SIMDShift { + // ARM64 variable shifts use only the signed low byte of each count. + bytes := llvm.VectorType(b.Prog.ctx.Int8Type(), x.ll.VectorSize()) + distance = b.impl.CreateSExt(b.impl.CreateTrunc(distance, bytes, ""), x.ll, "") + negative := llvm.CreateICmp(b.impl, llvm.IntSLT, distance, llvm.ConstNull(x.ll)) + magnitude := b.impl.CreateSelect(negative, llvm.CreateNeg(b.impl, distance), distance, "") + left := b.simdShiftValue(x, magnitude, false) + right := b.simdShiftValue(x, magnitude, true) + return Expr{b.impl.CreateSelect(negative, right, left, ""), x.Type} + } + right := op == SIMDShiftRight || op == SIMDShiftAllRight + return Expr{b.simdShiftValue(x, distance, right), x.Type} +} + +func (b Builder) simdShiftValue(x Expr, count llvm.Value, right bool) llvm.Value { + width := uint64(x.ll.ElementType().IntTypeWidth()) + large := llvm.CreateICmp(b.impl, llvm.IntUGE, count, simdIntegerConstant(count.Type(), width)) + // Clamp before truncation and shifting so no lane can produce poison. + safe := b.impl.CreateSelect(large, simdIntegerConstant(count.Type(), width-1), count, "") + if count.Type().TypeKind() != llvm.VectorTypeKind { + safe = b.impl.CreateTrunc(safe, x.ll.ElementType(), "") + v := b.impl.CreateInsertElement(llvm.Undef(x.ll), safe, llvm.ConstInt(b.Prog.tyInt32(), 0, false), "") + safe = b.simdSplatLane0(Expr{v, x.Type}).impl + } + unsigned := simdLanes(x.RawType()).Elem().Underlying().(*types.Basic).Info()&types.IsUnsigned != 0 + if right && !unsigned { + return b.impl.CreateAShr(x.impl, safe, "") + } + var shifted llvm.Value + if right { + shifted = b.impl.CreateLShr(x.impl, safe, "") + } else { + shifted = b.impl.CreateShl(x.impl, safe, "") + } + return b.impl.CreateSelect(large, llvm.ConstNull(x.ll), shifted, "") +} + +func (b Builder) simdIntegerIntrinsic(op SIMDOp, x, y Expr) Expr { + name := map[SIMDOp]string{SIMDAddSaturated: "add.sat", SIMDSubSaturated: "sub.sat", SIMDMin: "min", SIMDMax: "max"}[op] + prefix := "llvm.s" + if simdLanes(x.RawType()).Elem().Underlying().(*types.Basic).Info()&types.IsUnsigned != 0 { + prefix = "llvm.u" + } + value := b.impl.CreateIntrinsic(x.ll, llvm.LookupIntrinsicID(prefix+name), []llvm.Value{x.impl, y.impl}, "") + return Expr{value, x.Type} +} diff --git a/test/simd/integer_test.go b/test/simd/integer_test.go new file mode 100644 index 0000000000..987684c7a7 --- /dev/null +++ b/test/simd/integer_test.go @@ -0,0 +1,85 @@ +//go:build goexperiment.simd && (amd64 || arm64 || wasm) + +package simd_test + +import ( + "math" + "simd/archsimd" + "testing" +) + +func TestSIMDShiftAll(t *testing.T) { + counts := []uint64{0, 1, 7, 8, 15, 16, 31, 32, 33, 63, 64, 65, 127, 128, 255, 256, 1 << 32, math.MaxUint64} + x16 := [8]uint16{0, 1, 32767, 32768, 65535, 123, 9, 40000} + x32 := [4]int32{math.MinInt32, math.MaxInt32, -1, 1} + x64 := [2]uint64{math.MaxUint64, 1 << 63} + for _, n := range counts { + var left16, right16 [8]uint16 + v16 := archsimd.LoadUint16x8Array(&x16) + v16.ShiftAllLeft(n).StoreArray(&left16) + v16.ShiftAllRight(n).StoreArray(&right16) + for i, x := range x16 { + if left16[i] != x<>n { + t.Fatalf("uint16 shift %d lane %d", n, i) + } + } + var left32, right32 [4]int32 + v32 := archsimd.LoadInt32x4Array(&x32) + v32.ShiftAllLeft(n).StoreArray(&left32) + v32.ShiftAllRight(n).StoreArray(&right32) + for i, x := range x32 { + if left32[i] != x<>n { + t.Fatalf("int32 shift %d lane %d", n, i) + } + } + var left64, right64 [2]uint64 + v64 := archsimd.LoadUint64x2Array(&x64) + v64.ShiftAllLeft(n).StoreArray(&left64) + v64.ShiftAllRight(n).StoreArray(&right64) + for i, x := range x64 { + if left64[i] != x<>n { + t.Fatalf("uint64 shift %d lane %d", n, i) + } + } + } +} + +func TestSIMDSaturatedAndMinMax(t *testing.T) { + x := [16]int8{-128, -127, -100, -1, 0, 1, 100, 127, -128, 127, -1, 1, 64, -64, 0, 42} + y := [16]int8{-1, -127, -100, -128, 127, 127, 100, 1, 127, -128, 1, -1, 64, -64, 0, 42} + a, b := archsimd.LoadInt8x16Array(&x), archsimd.LoadInt8x16Array(&y) + var add, sub, lo, hi [16]int8 + a.AddSaturated(b).StoreArray(&add) + a.SubSaturated(b).StoreArray(&sub) + a.Min(b).StoreArray(&lo) + a.Max(b).StoreArray(&hi) + clamp := func(v int) int8 { + if v < -128 { + return -128 + } + if v > 127 { + return 127 + } + return int8(v) + } + for i := range x { + if add[i] != clamp(int(x[i])+int(y[i])) || sub[i] != clamp(int(x[i])-int(y[i])) || lo[i] != min(x[i], y[i]) || hi[i] != max(x[i], y[i]) { + t.Fatalf("signed saturated/minmax lane %d: %d %d %d %d", i, add[i], sub[i], lo[i], hi[i]) + } + } + ux := [8]uint16{0, 1, 65535, 65534, 32768, 123, 9, 40000} + uy := [8]uint16{1, 2, 1, 65535, 32768, 100, 10, 30000} + ua, ub := archsimd.LoadUint16x8Array(&ux), archsimd.LoadUint16x8Array(&uy) + var uadd, usub, ulo, uhi [8]uint16 + ua.AddSaturated(ub).StoreArray(&uadd) + ua.SubSaturated(ub).StoreArray(&usub) + ua.Min(ub).StoreArray(&ulo) + ua.Max(ub).StoreArray(&uhi) + for i := range ux { + wantAdd := uint16(min(uint32(ux[i])+uint32(uy[i]), 65535)) + wantSub := uint16(max(int(ux[i])-int(uy[i]), 0)) + if uadd[i] != wantAdd || usub[i] != wantSub || ulo[i] != min(ux[i], uy[i]) || uhi[i] != max(ux[i], uy[i]) { + t.Fatalf("unsigned saturated/minmax lane %d", i) + } + } +} diff --git a/test/simd/shift_amd64_test.go b/test/simd/shift_amd64_test.go new file mode 100644 index 0000000000..5b6cafa1a8 --- /dev/null +++ b/test/simd/shift_amd64_test.go @@ -0,0 +1,26 @@ +//go:build goexperiment.simd && amd64 + +package simd_test + +import ( + "simd/archsimd" + "testing" +) + +func TestSIMDVariableShift(t *testing.T) { + x := [4]int32{-1, -123, 0x12345678, -2147483648} + a := archsimd.LoadInt32x4Array(&x) + for _, counts := range [][4]uint32{{0, 31, 32, 33}, {255, 256, 1 << 31, 0xffffffff}} { + c := archsimd.LoadUint32x4Array(&counts) + var left, right [4]int32 + var unsigned [4]uint32 + a.ShiftLeft(c).StoreArray(&left) + a.ShiftRight(c).StoreArray(&right) + a.ToBits().ShiftRight(c).StoreArray(&unsigned) + for i, n := range counts { + if left[i] != x[i]<>n || unsigned[i] != uint32(x[i])>>n { + t.Fatalf("count %d lane %d", n, i) + } + } + } +} diff --git a/test/simd/shift_arm64_test.go b/test/simd/shift_arm64_test.go new file mode 100644 index 0000000000..b40a75ab4c --- /dev/null +++ b/test/simd/shift_arm64_test.go @@ -0,0 +1,35 @@ +//go:build goexperiment.simd && arm64 + +package simd_test + +import ( + "simd/archsimd" + "testing" +) + +func TestSIMDSignedLowByteShift(t *testing.T) { + x := [4]int32{-1, -123, 0x12345678, -2147483648} + a := archsimd.LoadInt32x4Array(&x) + for _, counts := range [][4]int32{{0, 31, -32, 257}, {-1, 255, 128, 127}, {-129, 256, -256, -255}} { + c := archsimd.LoadInt32x4Array(&counts) + var signed [4]int32 + var unsigned [4]uint32 + a.Shift(c).StoreArray(&signed) + a.ToBits().Shift(c).StoreArray(&unsigned) + for i, raw := range counts { + n := int(int8(raw)) + var want int32 + var wantU uint32 + if n < 0 { + want = x[i] >> uint(-n) + wantU = uint32(x[i]) >> uint(-n) + } else { + want = x[i] << uint(n) + wantU = uint32(x[i]) << uint(n) + } + if signed[i] != want || unsigned[i] != wantU { + t.Fatalf("count %d lane %d: %x %x want %x %x", raw, i, signed[i], unsigned[i], want, wantU) + } + } + } +} diff --git a/test/simd/shift_byte_test.go b/test/simd/shift_byte_test.go new file mode 100644 index 0000000000..fa9fd89c46 --- /dev/null +++ b/test/simd/shift_byte_test.go @@ -0,0 +1,25 @@ +//go:build goexperiment.simd && (arm64 || wasm) + +package simd_test + +import ( + "math" + "simd/archsimd" + "testing" +) + +func TestSIMDByteShift(t *testing.T) { + x8 := [16]int8{math.MinInt8, math.MaxInt8, -1, 0, 1, -2, 2, 3, -3, 4, -4, 5, -5, 6, -6, 7} + for _, n := range []uint64{0, 1, 7, 8, 9, 127, 128, 255, 256, math.MaxUint64} { + var left8, right8 [16]int8 + v8 := archsimd.LoadInt8x16Array(&x8) + v8.ShiftAllLeft(n).StoreArray(&left8) + v8.ShiftAllRight(n).StoreArray(&right8) + for i, x := range x8 { + if left8[i] != x<>n { + t.Fatalf("int8 shift %d lane %d: %d %d", n, i, left8[i], right8[i]) + } + } + + } +} From bd7e5a13f30f5a8a07301f73963ade61d809e7e9 Mon Sep 17 00:00:00 2001 From: ZhouGuangyuan Date: Sat, 3 Oct 2026 01:11:58 +0800 Subject: [PATCH 07/20] simd: preserve target floating min and max semantics --- cl/simd.go | 4 +- internal/build/simd_test.go | 9 ++- ssa/simd.go | 26 ++++++++- test/simd/minmax_test.go | 83 ++++++++++++++++++++++++++++ test/simd/unimplemented_llgo_test.go | 21 ++++--- 5 files changed, 131 insertions(+), 12 deletions(-) create mode 100644 test/simd/minmax_test.go diff --git a/cl/simd.go b/cl/simd.go index 0e7f06ab3c..5f5c7f089a 100644 --- a/cl/simd.go +++ b/cl/simd.go @@ -89,8 +89,8 @@ var simdOperations = map[simdKey]simdOperation{ {"numeric", "Shift"}: {llssa.SIMDShift, simdSignedShift, types.IsInteger}, {"numeric", "AddSaturated"}: {llssa.SIMDAddSaturated, simdBinary, types.IsInteger}, {"numeric", "SubSaturated"}: {llssa.SIMDSubSaturated, simdBinary, types.IsInteger}, - {"numeric", "Min"}: {llssa.SIMDMin, simdBinary, types.IsInteger}, - {"numeric", "Max"}: {llssa.SIMDMax, simdBinary, types.IsInteger}, + {"numeric", "Min"}: {llssa.SIMDMin, simdBinary, 0}, + {"numeric", "Max"}: {llssa.SIMDMax, simdBinary, 0}, } // These registrations share lowering but retain exact declaration names and diff --git a/internal/build/simd_test.go b/internal/build/simd_test.go index b997a9d618..432981440b 100644 --- a/internal/build/simd_test.go +++ b/internal/build/simd_test.go @@ -36,6 +36,7 @@ func store(p *[4]float32, x archsimd.Float32x4) { x.StoreArray(p) } func arithmetic(x, y archsimd.Float32x4) archsimd.Float32x4 { return x.Mul(y).Div(y).Sqrt().Round() } func bitcast(x archsimd.Uint32x4) archsimd.Float32x4 { return x.BitsToFloat32() } func abs(x archsimd.Int32x4) archsimd.Int32x4 { return x.Abs() } +func minmax(x, y archsimd.Float32x4) archsimd.Float32x4 { return x.Min(y).Max(y) } func round32(x archsimd.Float32x4) archsimd.Float32x4 { return x.Round() } func round64(x archsimd.Float64x2) archsimd.Float64x2 { return x.Round() } func compare(x, y archsimd.Float32x4) archsimd.Mask32x4 { return x.Equal(y) } @@ -128,6 +129,12 @@ func TestSIMD128LLVM(t *testing.T) { } } } + if target.arch != "amd64" { + ir := mod.NamedFunction("main.minmax").String() + if !strings.Contains(ir, "@llvm.minimum.v4f32") || !strings.Contains(ir, "@llvm.maximum.v4f32") { + t.Fatalf("missing IEEE vector min/max:\n%s", ir) + } + } if target.arch != "amd64" && !strings.Contains(mod.NamedFunction("main.round32").String(), "@llvm.roundeven.v4f32") { t.Fatal("missing native vector roundeven") } @@ -213,7 +220,7 @@ func TestSIMDIntrinsicDefinitions(t *testing.T) { if err := llvm.VerifyModule(mod, llvm.ReturnStatusAction); err != nil { t.Fatal(err) } - if fn := mod.NamedFunction("simd/archsimd.Float32x4.Min"); fn.IsNil() || !strings.Contains(fn.String(), "PanicSIMDUnimplemented") { + if fn := mod.NamedFunction("simd/archsimd.Float32x4.ConvertToInt32"); fn.IsNil() || !strings.Contains(fn.String(), "PanicSIMDUnimplemented") { t.Fatal("missing explicit unsupported implementation") } if fn := mod.NamedFunction("simd/archsimd.Float32x4.Add"); fn.IsNil() || !strings.Contains(fn.String(), "fadd <4 x float>") { diff --git a/ssa/simd.go b/ssa/simd.go index 828ad1afca..1a070108b1 100644 --- a/ssa/simd.go +++ b/ssa/simd.go @@ -152,7 +152,12 @@ func (b Builder) SIMD(op SIMDOp, result Type, args ...Expr) Expr { switch op { case SIMDShiftAllLeft, SIMDShiftAllRight, SIMDShiftLeft, SIMDShiftRight, SIMDShift: return b.simdShift(op, args[0], args[1]) - case SIMDAddSaturated, SIMDSubSaturated, SIMDMin, SIMDMax: + case SIMDMin, SIMDMax: + if simdLanes(args[0].RawType()).Elem().Underlying().(*types.Basic).Info()&types.IsFloat != 0 { + return b.simdFloatMinMax(op, args[0], args[1]) + } + return b.simdIntegerIntrinsic(op, args[0], args[1]) + case SIMDAddSaturated, SIMDSubSaturated: return b.simdIntegerIntrinsic(op, args[0], args[1]) case SIMDMaskFromBits: n := int(simdLanes(result.RawType()).Len()) @@ -489,3 +494,22 @@ func (b Builder) simdRoundEven(x Expr) Expr { result := b.impl.CreateSelect(small, smallResult, b.impl.CreateSelect(valid, rounded, bits, ""), "") return Expr{b.impl.CreateBitCast(result, x.ll, ""), x.Type} } + +func (b Builder) simdFloatMinMax(op SIMDOp, x, y Expr) Expr { + if b.Prog.Target().GOARCH == "amd64" { + // MINPS/MAXPS return the second operand for unordered or equal lanes, + // including opposite signed zeroes. Keep the operand order intact. + pred := llvm.FloatOLT + if op == SIMDMax { + pred = llvm.FloatOGT + } + cond := b.impl.CreateFCmp(pred, x.impl, y.impl, "") + return Expr{b.impl.CreateSelect(cond, x.impl, y.impl, ""), x.Type} + } + name := "llvm.minimum" + if op == SIMDMax { + name = "llvm.maximum" + } + value := b.impl.CreateIntrinsic(x.ll, llvm.LookupIntrinsicID(name), []llvm.Value{x.impl, y.impl}, "") + return Expr{value, x.Type} +} diff --git a/test/simd/minmax_test.go b/test/simd/minmax_test.go new file mode 100644 index 0000000000..c0b9c58b9c --- /dev/null +++ b/test/simd/minmax_test.go @@ -0,0 +1,83 @@ +//go:build goexperiment.simd && (amd64 || arm64 || wasm) + +package simd_test + +import ( + "math" + "runtime" + "simd/archsimd" + "testing" +) + +func simdMinMaxReference(x, y float64, largest bool) float64 { + if runtime.GOARCH == "amd64" { + if largest && x > y || !largest && x < y { + return x + } + return y + } + if math.IsNaN(x) || math.IsNaN(y) { + return math.NaN() + } + if largest { + return math.Max(x, y) + } + return math.Min(x, y) +} + +func TestSIMDFloatMinMax(t *testing.T) { + bits32 := []uint32{0, 1 << 31, 1, 1<<31 | 1, 0x3f800000, 0xbf800000, 0x7f800000, 0xff800000, 0x7fc00001, 0xffc00002, 0x7f800001} + for _, xb := range bits32 { + for _, yb := range bits32 { + x, y := math.Float32frombits(xb), math.Float32frombits(yb) + a, b := archsimd.BroadcastFloat32x4(x), archsimd.BroadcastFloat32x4(y) + for _, largest := range []bool{false, true} { + v := a.Min(b) + if largest { + v = a.Max(b) + } + var out [4]float32 + v.StoreArray(&out) + for i, got := range out { + want := float32(simdMinMaxReference(float64(x), float64(y), largest)) + if runtime.GOARCH == "amd64" { + // Select the original bits to avoid quieting a signaling NaN in the reference conversion. + wb := yb + if largest && x > y || !largest && x < y { + wb = xb + } + if math.Float32bits(got) != wb { + t.Fatalf("f32 max=%v x=%x y=%x lane=%d got=%x want=%x", largest, xb, yb, i, math.Float32bits(got), wb) + } + } else if !(math.IsNaN(float64(got)) && math.IsNaN(float64(want))) && math.Float32bits(got) != math.Float32bits(want) { + t.Fatalf("f32 max=%v x=%x y=%x lane=%d got=%x want=%x", largest, xb, yb, i, math.Float32bits(got), math.Float32bits(want)) + } + } + } + } + } + bits64 := []uint64{0, 1 << 63, 1, 1<<63 | 1, 0x3ff0000000000000, 0xbff0000000000000, 0x7ff0000000000000, 0xfff0000000000000, 0x7ff8000000000001, 0xfff8000000000002, 0x7ff0000000000001} + for _, xb := range bits64 { + for _, yb := range bits64 { + x, y := math.Float64frombits(xb), math.Float64frombits(yb) + a, b := archsimd.BroadcastFloat64x2(x), archsimd.BroadcastFloat64x2(y) + for _, largest := range []bool{false, true} { + v := a.Min(b) + if largest { + v = a.Max(b) + } + var out [2]float64 + v.StoreArray(&out) + want := simdMinMaxReference(x, y, largest) + for i, got := range out { + if runtime.GOARCH != "amd64" && math.IsNaN(got) && math.IsNaN(want) { + continue + } + if math.Float64bits(got) != math.Float64bits(want) { + t.Fatalf("f64 max=%v x=%x y=%x lane=%d got=%x want=%x", largest, xb, yb, i, math.Float64bits(got), math.Float64bits(want)) + } + } + } + } + } +} diff --git a/test/simd/unimplemented_llgo_test.go b/test/simd/unimplemented_llgo_test.go index c498d0ecbd..c7e15cd016 100644 --- a/test/simd/unimplemented_llgo_test.go +++ b/test/simd/unimplemented_llgo_test.go @@ -9,22 +9,22 @@ import ( _ "unsafe" ) -//go:linkname simdMin simd/archsimd.Float32x4.Min -func simdMin(x, y archsimd.Float32x4) archsimd.Float32x4 +//go:linkname simdConvert simd/archsimd.Float32x4.ConvertToInt32 +func simdConvert(x archsimd.Float32x4) archsimd.Int32x4 func TestUnimplementedSIMD(t *testing.T) { var x archsimd.Float32x4 - method := x.Min + method := x.ConvertToInt32 for _, tc := range []struct { name string call func() symbol string }{ - {"direct", func() { x.Min(x) }, "simd/archsimd.Float32x4.Min"}, - {"method value", func() { method(x) }, "simd/archsimd.Float32x4.Min"}, - {"method expression", func() { indirect(archsimd.Float32x4.Min, x, x) }, "simd/archsimd.Float32x4.Min"}, - {"deferred", func() { defer x.Min(x) }, "simd/archsimd.Float32x4.Min"}, - {"linkname", func() { simdMin(x, x) }, "simd/archsimd.Float32x4.Min"}, + {"direct", func() { x.ConvertToInt32() }, "simd/archsimd.Float32x4.ConvertToInt32"}, + {"method value", func() { method() }, "simd/archsimd.Float32x4.ConvertToInt32"}, + {"method expression", func() { indirectConvert(archsimd.Float32x4.ConvertToInt32, x) }, "simd/archsimd.Float32x4.ConvertToInt32"}, + {"deferred", func() { defer x.ConvertToInt32() }, "simd/archsimd.Float32x4.ConvertToInt32"}, + {"linkname", func() { simdConvert(x) }, "simd/archsimd.Float32x4.ConvertToInt32"}, } { t.Run(tc.name, func(t *testing.T) { defer func() { @@ -42,3 +42,8 @@ func TestUnimplementedSIMD(t *testing.T) { t.Fatal("Go helper was replaced by the fallback") } } + +//go:noinline +func indirectConvert(f func(archsimd.Float32x4) archsimd.Int32x4, x archsimd.Float32x4) archsimd.Int32x4 { + return f(x) +} From 23bd3eed2dfc79b51a2c8bdbaa4a9660bc0b9a42 Mon Sep 17 00:00:00 2001 From: ZhouGuangyuan Date: Sat, 3 Oct 2026 01:24:02 +0800 Subject: [PATCH 08/20] simd: lower numeric conversions with defined overflow results --- cl/simd.go | 12 ++++- internal/build/simd_test.go | 4 +- ssa/simd.go | 3 ++ ssa/simd_convert.go | 74 +++++++++++++++++++++++++++ test/simd/conversion64_test.go | 63 +++++++++++++++++++++++ test/simd/conversion_native_test.go | 28 ++++++++++ test/simd/conversion_test.go | 48 +++++++++++++++++ test/simd/conversion_unsigned_test.go | 44 ++++++++++++++++ test/simd/unimplemented_llgo_test.go | 24 ++++----- 9 files changed, 285 insertions(+), 15 deletions(-) create mode 100644 ssa/simd_convert.go create mode 100644 test/simd/conversion64_test.go create mode 100644 test/simd/conversion_native_test.go create mode 100644 test/simd/conversion_test.go create mode 100644 test/simd/conversion_unsigned_test.go diff --git a/cl/simd.go b/cl/simd.go index 5f5c7f089a..f2d05dba10 100644 --- a/cl/simd.go +++ b/cl/simd.go @@ -32,6 +32,7 @@ const ( simdScalarShift simdVectorShift simdSignedShift + simdConvert ) type simdOperation struct { @@ -96,6 +97,9 @@ var simdOperations = map[simdKey]simdOperation{ // These registrations share lowering but retain exact declaration names and // signature checks. Source Go slice helpers keep their own bounds checks. func init() { + for _, name := range []string{"ConvertToInt8", "ConvertToUint8", "ConvertToInt16", "ConvertToUint16", "ConvertToInt32", "ConvertToUint32", "ConvertToInt64", "ConvertToUint64", "ConvertToFloat32", "ConvertToFloat64"} { + simdOperations[simdKey{"numeric", name}] = simdOperation{llssa.SIMDConvert, simdConvert, 0} + } for _, name := range []string{"Mask8x16", "Mask16x8", "Mask32x4", "Mask64x2"} { simdOperations[simdKey{"", name + "FromBits"}] = simdOperation{llssa.SIMDMaskFromBits, simdMaskFromBits, 0} } @@ -221,12 +225,16 @@ func (d simdOperation) matches(sig *types.Signature, vector types.Type) bool { } case simdTernary: params = []types.Type{vector, vector} - case simdBitcast: + case simdBitcast, simdConvert: if sig.Results().Len() != 1 { return false } result = sig.Results().At(0).Type() - if _, ok := llssa.SIMDNumericShape(result); !ok { + shape, ok := llssa.SIMDNumericShape(result) + if !ok || d.signature == simdConvert && shape.Len() < lanes.Len() { + return false + } + if d.signature == simdConvert && shape.Len() != lanes.Len() && lanes.Elem().Underlying().(*types.Basic).Info()&types.IsFloat == 0 && shape.Elem().Underlying().(*types.Basic).Info()&types.IsFloat == 0 { return false } case simdUnary: diff --git a/internal/build/simd_test.go b/internal/build/simd_test.go index 432981440b..c7f3a4b04f 100644 --- a/internal/build/simd_test.go +++ b/internal/build/simd_test.go @@ -36,6 +36,7 @@ func store(p *[4]float32, x archsimd.Float32x4) { x.StoreArray(p) } func arithmetic(x, y archsimd.Float32x4) archsimd.Float32x4 { return x.Mul(y).Div(y).Sqrt().Round() } func bitcast(x archsimd.Uint32x4) archsimd.Float32x4 { return x.BitsToFloat32() } func abs(x archsimd.Int32x4) archsimd.Int32x4 { return x.Abs() } +func conversion(x archsimd.Float32x4) archsimd.Float32x4 { return x.ConvertToInt32().ConvertToFloat32() } func minmax(x, y archsimd.Float32x4) archsimd.Float32x4 { return x.Min(y).Max(y) } func round32(x archsimd.Float32x4) archsimd.Float32x4 { return x.Round() } func round64(x archsimd.Float64x2) archsimd.Float64x2 { return x.Round() } @@ -116,6 +117,7 @@ func TestSIMD128LLVM(t *testing.T) { } } for name, instructions := range map[string][]string{ + "conversion": {"@llvm.fptosi.sat.v4i32.v4f32", "sitofp <4 x i32>"}, "arithmetic": {"fmul <4 x float>", "fdiv <4 x float>", "@llvm.sqrt.v4f32"}, "bitcast": {"bitcast <4 x i32>", "to <4 x float>"}, "abs": {"@llvm.abs.v4i32", "i1 false"}, @@ -220,7 +222,7 @@ func TestSIMDIntrinsicDefinitions(t *testing.T) { if err := llvm.VerifyModule(mod, llvm.ReturnStatusAction); err != nil { t.Fatal(err) } - if fn := mod.NamedFunction("simd/archsimd.Float32x4.ConvertToInt32"); fn.IsNil() || !strings.Contains(fn.String(), "PanicSIMDUnimplemented") { + if fn := mod.NamedFunction("simd/archsimd.Uint8x16.Average"); fn.IsNil() || !strings.Contains(fn.String(), "PanicSIMDUnimplemented") { t.Fatal("missing explicit unsupported implementation") } if fn := mod.NamedFunction("simd/archsimd.Float32x4.Add"); fn.IsNil() || !strings.Contains(fn.String(), "fadd <4 x float>") { diff --git a/ssa/simd.go b/ssa/simd.go index 1a070108b1..3327de63f1 100644 --- a/ssa/simd.go +++ b/ssa/simd.go @@ -58,6 +58,7 @@ const ( SIMDSubSaturated SIMDMin SIMDMax + SIMDConvert ) // SIMDNumericShape validates the official numeric aggregate representation. @@ -150,6 +151,8 @@ func (b Builder) SIMD(op SIMDOp, result Type, args ...Expr) Expr { return Expr{v, result} } switch op { + case SIMDConvert: + return b.simdConvert(result, args[0]) case SIMDShiftAllLeft, SIMDShiftAllRight, SIMDShiftLeft, SIMDShiftRight, SIMDShift: return b.simdShift(op, args[0], args[1]) case SIMDMin, SIMDMax: diff --git a/ssa/simd_convert.go b/ssa/simd_convert.go new file mode 100644 index 0000000000..668661279e --- /dev/null +++ b/ssa/simd_convert.go @@ -0,0 +1,74 @@ +package ssa + +import ( + "go/types" + "math" + + "github.com/xgo-dev/llvm" +) + +// simdConvert keeps invalid floating conversions defined. LLVM's plain fptosi +// and fptoui would produce poison for NaNs and out-of-range lanes. +func (b Builder) simdConvert(result Type, x Expr) Expr { + from := simdLanes(x.RawType()).Elem().Underlying().(*types.Basic).Info() + to := simdLanes(result.RawType()).Elem().Underlying().(*types.Basic).Info() + n := x.ll.VectorSize() + typ := llvm.VectorType(result.ll.ElementType(), n) + var v llvm.Value + switch { + case from&types.IsFloat != 0 && to&types.IsInteger != 0: + name := "llvm.fptosi.sat" + unsigned := to&types.IsUnsigned != 0 + if unsigned { + name = "llvm.fptoui.sat" + } + v = b.impl.CreateIntrinsic(typ, llvm.LookupIntrinsicID(name), []llvm.Value{x.impl}, "") + if b.Prog.Target().GOARCH == "amd64" { + width := typ.ElementType().IntTypeWidth() + upper := math.Ldexp(1, width-1) + lower := -upper + invalidResult := uint64(1) << (width - 1) + if unsigned { + upper, lower, invalidResult = math.Ldexp(1, width), -1, ^uint64(0) + } + splat := func(f float64) llvm.Value { + vals := make([]llvm.Value, n) + for i := range vals { + vals[i] = llvm.ConstFloat(x.ll.ElementType(), f) + } + return llvm.ConstVector(vals, false) + } + above := b.impl.CreateFCmp(llvm.FloatUGE, x.impl, splat(upper), "") + pred := llvm.FloatOLT + if unsigned { + pred = llvm.FloatOLE + } + below := b.impl.CreateFCmp(pred, x.impl, splat(lower), "") + invalid := b.impl.CreateOr(above, below, "") + v = b.impl.CreateSelect(invalid, simdIntegerConstant(typ, invalidResult), v, "") + } + case from&types.IsFloat != 0 && to&types.IsFloat != 0: + v = b.impl.CreateFPTrunc(x.impl, typ, "") + case to&types.IsFloat != 0: + if from&types.IsUnsigned != 0 { + v = b.impl.CreateUIToFP(x.impl, typ, "") + } else { + v = b.impl.CreateSIToFP(x.impl, typ, "") + } + default: + v = b.impl.CreateBitCast(x.impl, typ, "") + } + if n != result.ll.VectorSize() { + // Narrowing conversions leave the unused upper lanes zero on these targets. + mask := make([]llvm.Value, result.ll.VectorSize()) + for i := range mask { + index := i + if i >= n { + index = n + } + mask[i] = llvm.ConstInt(b.Prog.tyInt32(), uint64(index), false) + } + v = b.impl.CreateShuffleVector(v, llvm.ConstNull(typ), llvm.ConstVector(mask, false), "") + } + return Expr{v, result} +} diff --git a/test/simd/conversion64_test.go b/test/simd/conversion64_test.go new file mode 100644 index 0000000000..b2b2d350a3 --- /dev/null +++ b/test/simd/conversion64_test.go @@ -0,0 +1,63 @@ +//go:build goexperiment.simd && (arm64 || (llgo && amd64)) + +package simd_test + +import ( + "math" + "runtime" + "simd/archsimd" + "testing" +) + +func TestSIMDConvert64(t *testing.T) { + for _, x := range []float64{0, -0.9, -1, -1.9, 1.9, math.Ldexp(1, 63), -math.Ldexp(1, 63), math.Nextafter(math.Ldexp(1, 63), 0), math.Nextafter(math.Ldexp(1, 64), 0), math.Ldexp(1, 64), math.NaN(), math.Inf(1), math.Inf(-1)} { + a := archsimd.BroadcastFloat64x2(x) + var gotS [2]int64 + var gotU [2]uint64 + a.ConvertToInt64().StoreArray(&gotS) + a.ConvertToUint64().StoreArray(&gotU) + var wantS int64 + var wantU uint64 + switch { + case math.IsNaN(x): + if runtime.GOARCH == "amd64" { + wantS = math.MinInt64 + } + case x >= math.Ldexp(1, 63): + wantS = math.MaxInt64 + if runtime.GOARCH == "amd64" { + wantS = math.MinInt64 + } + case x < -math.Ldexp(1, 63): + wantS = math.MinInt64 + default: + wantS = int64(x) + } + switch { + case math.IsNaN(x) || x <= -1: + if runtime.GOARCH == "amd64" { + wantU = math.MaxUint64 + } + case x >= math.Ldexp(1, 64): + wantU = math.MaxUint64 + case x > 0: + wantU = uint64(x) + } + for i := range gotS { + if gotS[i] != wantS || gotU[i] != wantU { + t.Fatalf("convert64(%g): %d/%d want %d/%d", x, gotS[i], gotU[i], wantS, wantU) + } + } + } + for _, x := range []uint64{0, 1, 1<<53 | 1, 1 << 63, math.MaxUint64} { + var got [2]float64 + archsimd.BroadcastUint64x2(x).ConvertToFloat64().StoreArray(&got) + if got[0] != float64(x) || got[1] != float64(x) { + t.Fatalf("uint64 to float: %v", got) + } + archsimd.BroadcastInt64x2(int64(x)).ConvertToFloat64().StoreArray(&got) + if got[0] != float64(int64(x)) || got[1] != float64(int64(x)) { + t.Fatalf("int64 to float: %v", got) + } + } +} diff --git a/test/simd/conversion_native_test.go b/test/simd/conversion_native_test.go new file mode 100644 index 0000000000..211bd7d243 --- /dev/null +++ b/test/simd/conversion_native_test.go @@ -0,0 +1,28 @@ +//go:build goexperiment.simd && (amd64 || arm64) + +package simd_test + +import ( + "math" + "simd/archsimd" + "testing" +) + +func TestSIMDDemoteFloat64(t *testing.T) { + for _, pair := range [][2]float64{{1.25, -1.25}, {math.Copysign(0, -1), 0}, {math.MaxFloat64, math.SmallestNonzeroFloat64}, {math.NaN(), math.Inf(-1)}} { + var got [4]float32 + archsimd.LoadFloat64x2Array(&pair).ConvertToFloat32().StoreArray(&got) + for i := 0; i < 2; i++ { + want := float32(pair[i]) + if math.IsNaN(float64(want)) && math.IsNaN(float64(got[i])) { + continue + } + if math.Float32bits(got[i]) != math.Float32bits(want) { + t.Fatalf("lane %d: %g want %g", i, got[i], want) + } + } + if math.Float32bits(got[2]) != 0 || math.Float32bits(got[3]) != 0 { + t.Fatalf("nonzero upper lanes: %v", got) + } + } +} diff --git a/test/simd/conversion_test.go b/test/simd/conversion_test.go new file mode 100644 index 0000000000..971cf1fe9f --- /dev/null +++ b/test/simd/conversion_test.go @@ -0,0 +1,48 @@ +//go:build goexperiment.simd && (amd64 || arm64 || wasm) + +package simd_test + +import ( + "math" + "runtime" + "simd/archsimd" + "testing" +) + +func TestSIMDConvertInt32(t *testing.T) { + values := []float32{0, math.Float32frombits(1 << 31), 1.9, -1.9, 123456.75, -123456.75, 2147483520, -2147483648, 2147483648, -2147483904, float32(math.NaN()), float32(math.Inf(1)), float32(math.Inf(-1))} + for _, x := range values { + var got [4]int32 + archsimd.BroadcastFloat32x4(x).ConvertToInt32().StoreArray(&got) + var want int32 + switch { + case math.IsNaN(float64(x)): + if runtime.GOARCH == "amd64" { + want = math.MinInt32 + } + case x >= 2147483648: + want = math.MaxInt32 + if runtime.GOARCH == "amd64" { + want = math.MinInt32 + } + case x < -2147483648: + want = math.MinInt32 + default: + want = int32(x) + } + for _, v := range got { + if v != want { + t.Fatalf("int32(%g): %d want %d", x, v, want) + } + } + } + for _, x := range []int32{math.MinInt32, math.MaxInt32, -16777217, 16777217, -1, 0, 1, 123456789} { + var got [4]float32 + archsimd.BroadcastInt32x4(x).ConvertToFloat32().StoreArray(&got) + for _, v := range got { + if v != float32(x) { + t.Fatalf("float32(%d): %g", x, v) + } + } + } +} diff --git a/test/simd/conversion_unsigned_test.go b/test/simd/conversion_unsigned_test.go new file mode 100644 index 0000000000..f5f543058a --- /dev/null +++ b/test/simd/conversion_unsigned_test.go @@ -0,0 +1,44 @@ +//go:build goexperiment.simd && (arm64 || wasm || (llgo && amd64)) + +package simd_test + +import ( + "math" + "runtime" + "simd/archsimd" + "testing" +) + +// Official amd64 requires AVX512 for these conversions. LLGo's baseline-safe +// LLVM legalization is checked against a scalar reference even without AVX512. +func TestSIMDConvertUint32(t *testing.T) { + for _, x := range []float32{0, -0.9, -1, -1.9, 1.9, 2147483648, 4294967040, 4294967296, float32(math.NaN()), float32(math.Inf(1)), float32(math.Inf(-1))} { + var got [4]uint32 + archsimd.BroadcastFloat32x4(x).ConvertToUint32().StoreArray(&got) + var want uint32 + switch { + case math.IsNaN(float64(x)) || x <= -1: + if runtime.GOARCH == "amd64" { + want = math.MaxUint32 + } + case x >= 4294967296: + want = math.MaxUint32 + case x > 0: + want = uint32(x) + } + for _, v := range got { + if v != want { + t.Fatalf("uint32(%g): %d want %d", x, v, want) + } + } + } + for _, x := range []uint32{0, 1, 16777217, 2147483648, math.MaxUint32} { + var got [4]float32 + archsimd.BroadcastUint32x4(x).ConvertToFloat32().StoreArray(&got) + for _, v := range got { + if v != float32(x) { + t.Fatalf("float32(%d): %g", x, v) + } + } + } +} diff --git a/test/simd/unimplemented_llgo_test.go b/test/simd/unimplemented_llgo_test.go index c7e15cd016..7ace1bfb26 100644 --- a/test/simd/unimplemented_llgo_test.go +++ b/test/simd/unimplemented_llgo_test.go @@ -9,22 +9,22 @@ import ( _ "unsafe" ) -//go:linkname simdConvert simd/archsimd.Float32x4.ConvertToInt32 -func simdConvert(x archsimd.Float32x4) archsimd.Int32x4 +//go:linkname simdAverage simd/archsimd.Uint8x16.Average +func simdAverage(x, y archsimd.Uint8x16) archsimd.Uint8x16 func TestUnimplementedSIMD(t *testing.T) { - var x archsimd.Float32x4 - method := x.ConvertToInt32 + var x archsimd.Uint8x16 + method := x.Average for _, tc := range []struct { name string call func() symbol string }{ - {"direct", func() { x.ConvertToInt32() }, "simd/archsimd.Float32x4.ConvertToInt32"}, - {"method value", func() { method() }, "simd/archsimd.Float32x4.ConvertToInt32"}, - {"method expression", func() { indirectConvert(archsimd.Float32x4.ConvertToInt32, x) }, "simd/archsimd.Float32x4.ConvertToInt32"}, - {"deferred", func() { defer x.ConvertToInt32() }, "simd/archsimd.Float32x4.ConvertToInt32"}, - {"linkname", func() { simdConvert(x) }, "simd/archsimd.Float32x4.ConvertToInt32"}, + {"direct", func() { x.Average(x) }, "simd/archsimd.Uint8x16.Average"}, + {"method value", func() { method(x) }, "simd/archsimd.Uint8x16.Average"}, + {"method expression", func() { indirectAverage(archsimd.Uint8x16.Average, x, x) }, "simd/archsimd.Uint8x16.Average"}, + {"deferred", func() { defer x.Average(x) }, "simd/archsimd.Uint8x16.Average"}, + {"linkname", func() { simdAverage(x, x) }, "simd/archsimd.Uint8x16.Average"}, } { t.Run(tc.name, func(t *testing.T) { defer func() { @@ -38,12 +38,12 @@ func TestUnimplementedSIMD(t *testing.T) { }) } // Go helper bodies remain executable; the fallback applies to declarations. - if x.Len() != 4 { + if x.Len() != 16 { t.Fatal("Go helper was replaced by the fallback") } } //go:noinline -func indirectConvert(f func(archsimd.Float32x4) archsimd.Int32x4, x archsimd.Float32x4) archsimd.Int32x4 { - return f(x) +func indirectAverage(f func(archsimd.Uint8x16, archsimd.Uint8x16) archsimd.Uint8x16, x, y archsimd.Uint8x16) archsimd.Uint8x16 { + return f(x, y) } From 6494ab2516a4c088e74de86bd46fa8eb711a1336 Mon Sep 17 00:00:00 2001 From: ZhouGuangyuan Date: Sat, 3 Oct 2026 01:36:10 +0800 Subject: [PATCH 09/20] wasm: bridge SIMD calls across Emscripten exception boundaries --- internal/build/build.go | 2 + internal/build/wasm_simd_calls.go | 138 ++++++++++++++++++++++++ internal/build/wasm_simd_calls_test.go | 91 ++++++++++++++++ test/simd/call_boundary_test.go | 78 ++++++++++++++ test/simd/internal/vectorcall/vector.go | 19 ++++ test/simd/testdata/boundary/main.go | 62 +++++++++++ 6 files changed, 390 insertions(+) create mode 100644 internal/build/wasm_simd_calls.go create mode 100644 internal/build/wasm_simd_calls_test.go create mode 100644 test/simd/call_boundary_test.go create mode 100644 test/simd/internal/vectorcall/vector.go create mode 100644 test/simd/testdata/boundary/main.go diff --git a/internal/build/build.go b/internal/build/build.go index e4ad1debb1..d75461372f 100644 --- a/internal/build/build.go +++ b/internal/build/build.go @@ -3229,6 +3229,7 @@ func lowerMainCExportModule(ctx *context, pkg llssa.Package, exports []cExport) if err := optimizeLLVMModule(ctx, pkg.Path(), mod); err != nil { return true, err } + lowerEmscriptenSIMDCalls(string(ctx.crossCompile.WasmProvider), mod) localizeWasmStackAddresses(ctx.buildConf.Goarch, mod) return true, nil } @@ -3297,6 +3298,7 @@ func compilePackageModule(ctx *context, aPkg *aPackage, externs []string, verbos if err := optimizeLLVMModule(ctx, pkgPath, ret.Module()); err != nil { return err } + lowerEmscriptenSIMDCalls(string(ctx.crossCompile.WasmProvider), ret.Module()) localizeWasmStackAddresses(ctx.buildConf.Goarch, ret.Module()) dropUnusedWindowsTestMain(ctx, aPkg, ret.Module()) emitFuncInfoEntrySites(ctx, ret) diff --git a/internal/build/wasm_simd_calls.go b/internal/build/wasm_simd_calls.go new file mode 100644 index 0000000000..dcddf1d660 --- /dev/null +++ b/internal/build/wasm_simd_calls.go @@ -0,0 +1,138 @@ +package build + +import ( + "fmt" + "strings" + + "github.com/xgo-dev/llvm" +) + +// lowerEmscriptenSIMDCalls keeps vectors inside Wasm when Emscripten's JS SjLj +// lowering wraps a potentially throwing call. JavaScript cannot carry v128. +// Only calls in setjmp functions need a memory bridge; ordinary vector calls +// retain their vector ABI. Run after optimization so inlined calls are covered. +// Remaining vector-call bodies cannot be inlined later into a setjmp function. +func lowerEmscriptenSIMDCalls(provider string, mod llvm.Module) int { + if provider != "emscripten" { + return 0 + } + setjmp := mod.NamedFunction("setjmp") + functions := make(map[llvm.Value]bool) + if !setjmp.IsNil() { + for use := setjmp.FirstUse(); !use.IsNil(); use = use.NextUse() { + call := use.User().IsACallInst() + if !call.IsNil() && call.CalledValue() == setjmp { + functions[call.InstructionParent().Parent()] = true + } + } + } + var calls []llvm.Value + for fn := mod.FirstFunction(); !fn.IsNil(); fn = llvm.NextFunction(fn) { + for block := fn.FirstBasicBlock(); !block.IsNil(); block = llvm.NextBasicBlock(block) { + for inst := block.FirstInstruction(); !inst.IsNil(); inst = llvm.NextInstruction(inst) { + if inst.IsACallInst().IsNil() || strings.HasPrefix(inst.CalledValue().Name(), "llvm.") { + continue + } + typ := inst.CalledFunctionType() + vector := typ.ReturnType().TypeKind() == llvm.VectorTypeKind + for i := 0; i < inst.OperandsCount()-1; i++ { + vector = vector || inst.Operand(i).Type().TypeKind() == llvm.VectorTypeKind + } + if vector { + if functions[fn] { + calls = append(calls, inst) + } else { + // A later backend/LTO inliner must not transplant these + // unbridged calls into a function containing setjmp. + // The body has already received the selected optimization. + fn.RemoveEnumFunctionAttribute(llvm.AttributeKindID("alwaysinline")) + fn.AddFunctionAttr(mod.Context().CreateEnumAttribute(llvm.AttributeKindID("noinline"), 0)) + } + } + } + } + } + for i, call := range calls { + bridgeEmscriptenSIMDCall(mod, call, i) + } + return len(calls) +} + +func bridgeEmscriptenSIMDCall(mod llvm.Module, call llvm.Value, id int) { + ctx := mod.Context() + b := ctx.NewBuilder() + defer b.Dispose() + oldType := call.CalledFunctionType() + retType := oldType.ReturnType() + vectorResult := retType.TypeKind() == llvm.VectorTypeKind + pointer := llvm.PointerType(ctx.Int8Type(), 0) + types := []llvm.Type{pointer} + args := []llvm.Value{call.CalledValue()} + // New stack slots contain only vector bits, never Go pointers. + allocate := func(typ llvm.Type) llvm.Value { + b.SetInsertPointBefore(call.InstructionParent().Parent().FirstBasicBlock().FirstInstruction()) + return b.CreateAlloca(typ, "simd.sjlj.slot") + } + var resultSlot llvm.Value + if vectorResult { + resultSlot = allocate(retType) + types = append(types, pointer) + args = append(args, resultSlot) + retType = ctx.VoidType() + } + paramOffset := len(types) + // Include any already-promoted variadic operands as fixed bridge arguments. + // LLGo emits no operand bundles; the last call operand is the callee. + oldParams := make([]llvm.Type, call.OperandsCount()-1) + for i := range oldParams { + oldParams[i] = call.Operand(i).Type() + } + for i, typ := range oldParams { + arg := call.Operand(i) + if typ.TypeKind() == llvm.VectorTypeKind { + slot := allocate(typ) + b.SetInsertPointBefore(call) + b.CreateStore(arg, slot) + arg, typ = slot, pointer + } + types, args = append(types, typ), append(args, arg) + } + // Each bridge retains the original call and its complete ABI attributes. + // noinline/optnone keep late optimization from exposing a v128 JS call again. + bridgeType := llvm.FunctionType(retType, types, false) + bridge := llvm.AddFunction(mod, fmt.Sprintf("__llgo_simd_sjlj.%d", id), bridgeType) + bridge.SetLinkage(llvm.InternalLinkage) + for _, name := range []string{"noinline", "optnone"} { + bridge.AddFunctionAttr(ctx.CreateEnumAttribute(llvm.AttributeKindID(name), 0)) + } + bridge.AddTargetDependentFunctionAttr("target-features", "+simd128") + b.SetInsertPointBefore(call) + replacement := b.CreateCall(bridgeType, bridge, args, "") + replacement.InstructionSetDebugLoc(call.InstructionDebugLoc()) + value := replacement + if vectorResult { + value = b.CreateLoad(oldType.ReturnType(), resultSlot, "") + } + call.ReplaceAllUsesWith(value) + block := ctx.AddBasicBlock(bridge, "entry") + b.SetInsertPointAtEnd(block) + for i, typ := range oldParams { + arg := bridge.Param(paramOffset + i) + if typ.TypeKind() == llvm.VectorTypeKind { + arg = b.CreateLoad(typ, arg, "") + } + call.SetOperand(i, arg) + } + call.SetOperand(call.OperandsCount()-1, bridge.Param(0)) + call.RemoveFromParentAsInstruction() + b.Insert(call) + call.SetTailCall(false) + if vectorResult { + b.CreateStore(call, bridge.Param(1)) + } + if retType.TypeKind() == llvm.VoidTypeKind { + b.CreateRetVoid() + } else { + b.CreateRet(call) + } +} diff --git a/internal/build/wasm_simd_calls_test.go b/internal/build/wasm_simd_calls_test.go new file mode 100644 index 0000000000..0813d07680 --- /dev/null +++ b/internal/build/wasm_simd_calls_test.go @@ -0,0 +1,91 @@ +package build + +import ( + "strings" + "testing" + + "github.com/xgo-dev/llvm" +) + +func TestEmscriptenSIMDCallBridge(t *testing.T) { + const source = ` +declare i32 @setjmp(ptr) returns_twice +declare <4 x float> @callee(<4 x float>, ptr) +declare float @scalar(float) +declare float @consumevec(<4 x float>) +declare void @sink(<4 x float>) +declare <4 x float> @llvm.sqrt.v4f32(<4 x float>) +define <4 x float> @withjmp(ptr %jmp, ptr %fn, ptr %env, <4 x float> %x) { + %saved = call i32 @setjmp(ptr %jmp) + %a = call <4 x float> @callee(<4 x float> %x, ptr %env) + %b = call <4 x float> %fn(<4 x float> %a, ptr %env) + %s = call float @scalar(float 1.0) + %late = call float @late() + call void @sink(<4 x float> %a) + %u = call float @consumevec(<4 x float> %a) + %c = call <4 x float> @llvm.sqrt.v4f32(<4 x float> %b) + ret <4 x float> %c +} +define float @late() alwaysinline { + %v = call <4 x float> @callee(<4 x float> zeroinitializer, ptr null) + %f = extractelement <4 x float> %v, i32 0 + ret float %f +} +define <4 x float> @ordinary(<4 x float> %x, ptr %env) { + %v = call <4 x float> @callee(<4 x float> %x, ptr %env) + ret <4 x float> %v +} +` + for _, provider := range []string{"emscripten", "wasi", "gojs"} { + t.Run(provider, func(t *testing.T) { + mod := parseWasmAggregateIR(t, source) + before := mod.NamedFunction("ordinary").String() + count := lowerEmscriptenSIMDCalls(provider, mod) + want := 0 + if provider == "emscripten" { + want = 4 + } + if count != want { + t.Fatalf("bridges=%d want %d", count, want) + } + if err := llvm.VerifyModule(mod, llvm.ReturnStatusAction); err != nil { + t.Fatalf("%v\n%s", err, mod.String()) + } + if provider != "emscripten" && before != mod.NamedFunction("ordinary").String() { + t.Fatal("changed other provider") + } + if !strings.Contains(mod.NamedFunction("ordinary").String(), "call <4 x float> @callee") { + t.Fatal("changed ordinary vector ABI") + } + if provider == "emscripten" { + ir := mod.NamedFunction("withjmp").String() + if strings.Contains(ir, "call <4 x float> @callee") || strings.Contains(ir, "call <4 x float> %fn") { + t.Fatalf("v128 crosses JS SjLj boundary:\n%s", ir) + } + for _, name := range []string{"__llgo_simd_sjlj.0", "__llgo_simd_sjlj.1"} { + fn := mod.NamedFunction(name) + if fn.IsNil() || fn.GlobalValueType().ReturnType().TypeKind() != llvm.VoidTypeKind { + t.Fatal("missing memory bridge") + } + if strings.Contains(fn.String(), "setjmp") || !strings.Contains(fn.String(), "call <4 x float>") { + t.Fatalf("bad bridge:\n%s", fn.String()) + } + } + opts := llvm.NewPassBuilderOptions() + defer opts.Dispose() + if err := mod.RunPasses("default", llvm.TargetMachine{}, opts); err != nil { + t.Fatal(err) + } + if !strings.Contains(mod.NamedFunction("withjmp").String(), "call float @late") { + t.Fatalf("late optimization imported an unbridged vector call:\n%s", mod.String()) + } + if err := llvm.VerifyModule(mod, llvm.ReturnStatusAction); err != nil { + t.Fatal(err) + } + if second := lowerEmscriptenSIMDCalls(provider, mod); second != 0 { + t.Fatalf("pass not idempotent: %d", second) + } + } + }) + } +} diff --git a/test/simd/call_boundary_test.go b/test/simd/call_boundary_test.go new file mode 100644 index 0000000000..e75ea4a870 --- /dev/null +++ b/test/simd/call_boundary_test.go @@ -0,0 +1,78 @@ +//go:build goexperiment.simd && (amd64 || arm64 || wasm) + +package simd_test + +import ( + "runtime" + "simd/archsimd" + "testing" + + "github.com/xgo-dev/llgo/test/simd/internal/vectorcall" +) + +//go:noinline +func vectorDirectRecover(x archsimd.Float32x4, fail bool) (result archsimd.Float32x4) { + live := x.Add(archsimd.BroadcastFloat32x4(3)) + defer func() { + if r := recover(); r != nil { + if r != "vector call" { + panic(r) + } + result = live + } + }() + return vectorcall.Echo(x, fail) +} + +//go:noinline +func vectorIndirectRecover(f func(archsimd.Float32x4, bool) archsimd.Float32x4, x archsimd.Float32x4, fail bool) (result archsimd.Float32x4) { + live := x.Add(archsimd.BroadcastFloat32x4(4)) + defer func() { + if r := recover(); r != nil { + if r != "vector call" { + panic(r) + } + result = live + } + }() + return f(x, fail) +} + +func TestSIMDRecoverAcrossPackageCalls(t *testing.T) { + x := archsimd.BroadcastFloat32x4(10) + for _, tc := range []struct { + value archsimd.Float32x4 + want float32 + }{ + {vectorDirectRecover(x, false), 12}, + {vectorDirectRecover(x, true), 13}, + {vectorIndirectRecover(vectorcall.Echo, x, false), 12}, + {vectorIndirectRecover(vectorcall.Echo, x, true), 14}, + } { + var got [4]float32 + tc.value.StoreArray(&got) + for _, v := range got { + if v != tc.want { + t.Fatalf("got %v want %g", got, tc.want) + } + } + } +} + +func TestSIMDLiveAcrossScheduling(t *testing.T) { + x := archsimd.BroadcastFloat32x4(7) + done := make(chan archsimd.Float32x4, 1) + go func(v archsimd.Float32x4) { + runtime.Gosched() + done <- vectorcall.Echo(v, false) + }(x) + runtime.Gosched() + got := (<-done).Add(x) + var lanes [4]float32 + got.StoreArray(&lanes) + for _, v := range lanes { + if v != 16 { + t.Fatalf("vector across scheduling: %v", lanes) + } + } +} diff --git a/test/simd/internal/vectorcall/vector.go b/test/simd/internal/vectorcall/vector.go new file mode 100644 index 0000000000..40d5e3751d --- /dev/null +++ b/test/simd/internal/vectorcall/vector.go @@ -0,0 +1,19 @@ +//go:build goexperiment.simd && (amd64 || arm64 || wasm) + +package vectorcall + +import "simd/archsimd" + +//go:noinline +func Echo(x archsimd.Float32x4, fail bool) archsimd.Float32x4 { + if fail { + panic("vector call") + } + return x.Add(archsimd.BroadcastFloat32x4(2)) +} + +// Scalar intentionally allows inlining. The Emscripten backend must prevent a +// late inliner from moving its vector call into a caller with a recovery point. +func Scalar(x float32, fail bool) float32 { + return Echo(archsimd.BroadcastFloat32x4(x), fail).GetElem(0) +} diff --git a/test/simd/testdata/boundary/main.go b/test/simd/testdata/boundary/main.go new file mode 100644 index 0000000000..dd9c2f6867 --- /dev/null +++ b/test/simd/testdata/boundary/main.go @@ -0,0 +1,62 @@ +//go:build goexperiment.simd && (amd64 || arm64 || wasm) + +// A small executable also exercises O0 Wasm without the standard testing +// package, whose unoptimized functions exceed engine local-variable limits. +package main + +import ( + "simd/archsimd" + + "github.com/xgo-dev/llgo/test/simd/internal/vectorcall" +) + +var initial = archsimd.BroadcastFloat32x4(10) + +//go:noinline +func recoverVector(f func(archsimd.Float32x4, bool) archsimd.Float32x4, x archsimd.Float32x4, fail bool) (out archsimd.Float32x4) { + live := x.Add(archsimd.BroadcastFloat32x4(3)) + defer func() { + if r := recover(); r != nil { + if r != "vector call" { + panic(r) + } + out = live + } + }() + return f(x, fail) +} + +//go:noinline +func check(x archsimd.Float32x4, want float32) { + var lanes [4]float32 + x.StoreArray(&lanes) + for _, value := range lanes { + if value != want { + panic("vector boundary mismatch") + } + } +} + +func scalarRecover(fail bool) (out float32) { + defer func() { + if r := recover(); r != nil { + if r != "vector call" { + panic(r) + } + out = 42 + } + }() + return vectorcall.Scalar(10, fail) +} + +func main() { + if scalarRecover(false) != 12 || scalarRecover(true) != 42 { + panic("late inline boundary mismatch") + } + check(recoverVector(vectorcall.Echo, initial, false), 12) + check(recoverVector(vectorcall.Echo, initial, true), 13) + done := make(chan archsimd.Float32x4) + go func(x archsimd.Float32x4) { done <- vectorcall.Echo(x, false) }(initial) + check((<-done).Add(initial), 22) + println("SIMD call boundaries PASS") +} From cbf35ad59a00c1dee602dddf0cf4ebdc65081d10 Mon Sep 17 00:00:00 2001 From: ZhouGuangyuan Date: Sat, 3 Oct 2026 01:45:10 +0800 Subject: [PATCH 10/20] simd: add byte lookups and baseline-safe permutations --- cl/simd.go | 20 ++++++---- internal/build/simd_test.go | 23 ++++++++++- ssa/simd.go | 5 +++ ssa/simd_permute.go | 31 +++++++++++++++ test/simd/lookup_test.go | 37 ++++++++++++++++++ test/simd/permute_amd64_test.go | 37 ++++++++++++++++++ test/simd/permute_llgo_amd64_test.go | 40 +++++++++++++++++++ test/simd/workload_test.go | 57 ++++++++++++++++++++++++++++ 8 files changed, 242 insertions(+), 8 deletions(-) create mode 100644 ssa/simd_permute.go create mode 100644 test/simd/lookup_test.go create mode 100644 test/simd/permute_amd64_test.go create mode 100644 test/simd/permute_llgo_amd64_test.go create mode 100644 test/simd/workload_test.go diff --git a/cl/simd.go b/cl/simd.go index f2d05dba10..d0a4d8b6b7 100644 --- a/cl/simd.go +++ b/cl/simd.go @@ -30,8 +30,8 @@ const ( simdMaskFromBits simdMaskToBits simdScalarShift - simdVectorShift - simdSignedShift + simdUnsignedVector + simdSignedVector simdConvert ) @@ -85,12 +85,15 @@ var simdOperations = map[simdKey]simdOperation{ {"mask", "Not"}: {llssa.SIMDNot, simdUnary, 0}, {"numeric", "ShiftAllLeft"}: {llssa.SIMDShiftAllLeft, simdScalarShift, types.IsInteger}, {"numeric", "ShiftAllRight"}: {llssa.SIMDShiftAllRight, simdScalarShift, types.IsInteger}, - {"numeric", "ShiftLeft"}: {llssa.SIMDShiftLeft, simdVectorShift, types.IsInteger}, - {"numeric", "ShiftRight"}: {llssa.SIMDShiftRight, simdVectorShift, types.IsInteger}, - {"numeric", "Shift"}: {llssa.SIMDShift, simdSignedShift, types.IsInteger}, + {"numeric", "ShiftLeft"}: {llssa.SIMDShiftLeft, simdUnsignedVector, types.IsInteger}, + {"numeric", "ShiftRight"}: {llssa.SIMDShiftRight, simdUnsignedVector, types.IsInteger}, + {"numeric", "Shift"}: {llssa.SIMDShift, simdSignedVector, types.IsInteger}, {"numeric", "AddSaturated"}: {llssa.SIMDAddSaturated, simdBinary, types.IsInteger}, {"numeric", "SubSaturated"}: {llssa.SIMDSubSaturated, simdBinary, types.IsInteger}, {"numeric", "Min"}: {llssa.SIMDMin, simdBinary, 0}, + {"numeric", "LookupOrZero"}: {llssa.SIMDLookupOrZero, simdBinary, types.IsInteger}, + {"numeric", "PermuteOrZero"}: {llssa.SIMDPermuteOrZero, simdSignedVector, types.IsInteger}, + {"numeric", "Permute"}: {llssa.SIMDPermute, simdUnsignedVector, types.IsInteger}, {"numeric", "Max"}: {llssa.SIMDMax, simdBinary, 0}, } @@ -179,12 +182,15 @@ func (d simdOperation) matches(sig *types.Signature, vector types.Type) bool { if d.elements != 0 && lanes.Elem().Underlying().(*types.Basic).Info()&d.elements == 0 { return false } + if (d.op == llssa.SIMDLookupOrZero || d.op == llssa.SIMDPermuteOrZero) && lanes.Len() != 16 { + return false + } var params []types.Type result := vector switch d.signature { case simdScalarShift: params = []types.Type{types.Typ[types.Uint64]} - case simdVectorShift, simdSignedShift: + case simdUnsignedVector, simdSignedVector: if sig.Params().Len() != 1 { return false } @@ -194,7 +200,7 @@ func (d simdOperation) matches(sig *types.Signature, vector types.Type) bool { return false } info := shape.Elem().Underlying().(*types.Basic).Info() - if info&types.IsInteger == 0 || (info&types.IsUnsigned != 0) != (d.signature == simdVectorShift) { + if info&types.IsInteger == 0 || (info&types.IsUnsigned != 0) != (d.signature == simdUnsignedVector) { return false } params = []types.Type{counts} diff --git a/internal/build/simd_test.go b/internal/build/simd_test.go index c7f3a4b04f..2dec93ca92 100644 --- a/internal/build/simd_test.go +++ b/internal/build/simd_test.go @@ -5,6 +5,7 @@ package build import ( "os" "path/filepath" + "regexp" "strings" "testing" @@ -74,7 +75,12 @@ func maskTo64(x archsimd.Mask64x2) uint8 { return x.ToBits() } func simdTestDir(t *testing.T) string { t.Helper() dir := t.TempDir() - for name, text := range map[string]string{"go.mod": "module simdtest\n\ngo 1.27\n", "main.go": simd128Source, "bitmap_amd64.go": simdMaskBitmapSource} { + for name, text := range map[string]string{ + "go.mod": "module simdtest\n\ngo 1.27\n", "main.go": simd128Source, "bitmap_amd64.go": simdMaskBitmapSource, + "lookup_arm64.go": `package main; import "simd/archsimd"; func lookup(x,y archsimd.Int8x16) archsimd.Int8x16 { return x.LookupOrZero(y) }`, + "lookup_wasm.go": `package main; import "simd/archsimd"; func lookup(x,y archsimd.Int8x16) archsimd.Int8x16 { return x.LookupOrZero(y) }`, + "permute_amd64.go": `package main; import "simd/archsimd"; func permute(x archsimd.Uint8x16, y archsimd.Int8x16) archsimd.Uint8x16 { return x.PermuteOrZero(y) }`, + } { if err := os.WriteFile(filepath.Join(dir, name), []byte(text), 0600); err != nil { t.Fatal(err) } @@ -88,6 +94,9 @@ func TestSIMD128LLVM(t *testing.T) { t.Run(target.arch, func(t *testing.T) { conf := NewDefaultConf(ModeGen) conf.Goos, conf.Goarch, conf.GOEXPERIMENT = target.os, target.arch, "simd" + if target.arch == "amd64" { + conf.GOAMD64 = "v1" + } pkgs, err := Build(Invocation{Args: []string{"."}, Config: conf, Dir: dir}) if err != nil { t.Fatal(err) @@ -193,6 +202,15 @@ func TestSIMD128LLVM(t *testing.T) { if !strings.Contains(string(asm.Bytes()), want) { t.Fatalf("missing %s in assembly", want) } + if target.arch == "amd64" && regexp.MustCompile(`(?m)^\s+v[a-z][a-z0-9]*\s`).Match(asm.Bytes()) { + t.Fatal("GOAMD64=v1 emitted an AVX instruction") + } + if target.arch != "amd64" { + wantLookup := map[string]string{"arm64": "tbl", "wasm": "i8x16.swizzle"}[target.arch] + if !strings.Contains(string(asm.Bytes()), wantLookup) { + t.Fatalf("missing lookup instruction %s", wantLookup) + } + } if target.arch == "amd64" && strings.Contains(string(asm.Bytes()), "roundeven") { t.Fatal("baseline rounding requires nonportable libm roundeven") } @@ -207,6 +225,9 @@ func TestSIMDIntrinsicDefinitions(t *testing.T) { t.Run(target.os+"/"+target.arch, func(t *testing.T) { conf := NewDefaultConf(ModeGen) conf.Goos, conf.Goarch, conf.GOEXPERIMENT = target.os, target.arch, "simd" + if target.arch == "amd64" { + conf.GOAMD64 = "v1" + } pkgs, err := Build(Invocation{Args: []string{".", "simd/archsimd"}, Config: conf, Dir: dir}) if err != nil { t.Fatal(err) diff --git a/ssa/simd.go b/ssa/simd.go index 3327de63f1..88126878b7 100644 --- a/ssa/simd.go +++ b/ssa/simd.go @@ -59,6 +59,9 @@ const ( SIMDMin SIMDMax SIMDConvert + SIMDLookupOrZero + SIMDPermuteOrZero + SIMDPermute ) // SIMDNumericShape validates the official numeric aggregate representation. @@ -151,6 +154,8 @@ func (b Builder) SIMD(op SIMDOp, result Type, args ...Expr) Expr { return Expr{v, result} } switch op { + case SIMDLookupOrZero, SIMDPermuteOrZero, SIMDPermute: + return b.simdPermute(op, args[0], args[1]) case SIMDConvert: return b.simdConvert(result, args[0]) case SIMDShiftAllLeft, SIMDShiftAllRight, SIMDShiftLeft, SIMDShiftRight, SIMDShift: diff --git a/ssa/simd_permute.go b/ssa/simd_permute.go new file mode 100644 index 0000000000..14af548d77 --- /dev/null +++ b/ssa/simd_permute.go @@ -0,0 +1,31 @@ +package ssa + +import "github.com/xgo-dev/llvm" + +func (b Builder) simdPermute(op SIMDOp, x, indices Expr) Expr { + if op == SIMDLookupOrZero { + name := "llvm.aarch64.neon.tbl1" + if b.Prog.Target().GOARCH == "wasm" { + name = "llvm.wasm.swizzle" + } + value := b.impl.CreateIntrinsic(x.ll, llvm.LookupIntrinsicID(name), []llvm.Value{x.impl, indices.impl}, "") + return Expr{value, x.Type} + } + // Mask indices before extracting a lane, including indices whose sign bit + // requests zero. An out-of-bounds extract would create LLVM poison. + n := x.ll.VectorSize() + mask := llvm.ConstInt(indices.ll.ElementType(), uint64(n-1), false) + result := llvm.Undef(x.ll) + for i := 0; i < n; i++ { + lane := llvm.ConstInt(b.Prog.tyInt32(), uint64(i), false) + index := b.impl.CreateExtractElement(indices.impl, lane, "") + safe := b.impl.CreateAnd(index, mask, "") + value := b.impl.CreateExtractElement(x.impl, safe, "") + if op == SIMDPermuteOrZero { + negative := llvm.CreateICmp(b.impl, llvm.IntSLT, index, llvm.ConstNull(index.Type())) + value = b.impl.CreateSelect(negative, llvm.ConstNull(value.Type()), value, "") + } + result = b.impl.CreateInsertElement(result, value, lane, "") + } + return Expr{result, x.Type} +} diff --git a/test/simd/lookup_test.go b/test/simd/lookup_test.go new file mode 100644 index 0000000000..250f1fdc11 --- /dev/null +++ b/test/simd/lookup_test.go @@ -0,0 +1,37 @@ +//go:build goexperiment.simd && (arm64 || wasm) + +package simd_test + +import ( + "simd/archsimd" + "testing" +) + +func TestSIMDByteLookup(t *testing.T) { + var table [16]int8 + for i := range table { + table[i] = int8(i*7 - 53) + } + vector := archsimd.LoadInt8x16Array(&table) + // Exhaust all 256 index patterns, including signed-negative and >=16 values. + for base := 0; base < 256; base += 16 { + var indices, got [16]int8 + for i := range indices { + indices[i] = int8(base + i) + } + vector.LookupOrZero(archsimd.LoadInt8x16Array(&indices)).StoreArray(&got) + for i, index := range indices { + var want int8 + if index >= 0 && index < 16 { + want = table[index] + } + if got[i] != want { + t.Fatalf("index=%d got=%d want=%d", index, got[i], want) + } + } + } +} + +func lookupHexDigits(table, indices archsimd.Uint8x16) archsimd.Uint8x16 { + return table.BitsToInt8().LookupOrZero(indices.BitsToInt8()).ToBits() +} diff --git a/test/simd/permute_amd64_test.go b/test/simd/permute_amd64_test.go new file mode 100644 index 0000000000..af30e36afb --- /dev/null +++ b/test/simd/permute_amd64_test.go @@ -0,0 +1,37 @@ +//go:build goexperiment.simd && amd64 + +package simd_test + +import ( + "simd/archsimd" + "testing" +) + +func TestSIMDBytePermuteOrZero(t *testing.T) { + var table [16]uint8 + for i := range table { + table[i] = uint8(i*7 + 13) + } + vector := archsimd.LoadUint8x16Array(&table) + for base := 0; base < 256; base += 16 { + var indices [16]int8 + var got [16]uint8 + for i := range indices { + indices[i] = int8(base + i) + } + vector.PermuteOrZero(archsimd.LoadInt8x16Array(&indices)).StoreArray(&got) + for i, index := range indices { + var want uint8 + if index >= 0 { + want = table[index%16] + } + if got[i] != want { + t.Fatalf("index=%d got=%d want=%d", index, got[i], want) + } + } + } +} + +func lookupHexDigits(table, indices archsimd.Uint8x16) archsimd.Uint8x16 { + return table.PermuteOrZero(indices.BitsToInt8()) +} diff --git a/test/simd/permute_llgo_amd64_test.go b/test/simd/permute_llgo_amd64_test.go new file mode 100644 index 0000000000..129d3e839b --- /dev/null +++ b/test/simd/permute_llgo_amd64_test.go @@ -0,0 +1,40 @@ +//go:build llgo && goexperiment.simd && amd64 + +package simd_test + +import ( + "simd/archsimd" + "testing" +) + +// These official operations require AVX512. Exercise LLGo's baseline-safe +// equivalent against scalar references on machines without that extension. +func TestSIMDModuloPermute(t *testing.T) { + var table [16]uint8 + for i := range table { + table[i] = uint8(i*7 + 13) + } + vector := archsimd.LoadUint8x16Array(&table) + for base := 0; base < 256; base += 16 { + var indices, got [16]uint8 + for i := range indices { + indices[i] = uint8(base + i) + } + vector.Permute(archsimd.LoadUint8x16Array(&indices)).StoreArray(&got) + for i, index := range indices { + if got[i] != table[index%16] { + t.Fatalf("index=%d got=%d", index, got[i]) + } + } + } + table16 := [8]uint16{1, 17, 123, 999, 1024, 50000, 60000, 65535} + for _, indices := range [][8]uint16{{0, 1, 2, 3, 4, 5, 6, 7}, {8, 9, 10, 11, 12, 13, 14, 15}, {1 << 15, 65535, 256, 257, 258, 259, 260, 261}} { + var got [8]uint16 + archsimd.LoadUint16x8Array(&table16).Permute(archsimd.LoadUint16x8Array(&indices)).StoreArray(&got) + for i, index := range indices { + if got[i] != table16[index%8] { + t.Fatalf("index=%d got=%x", index, got[i]) + } + } + } +} diff --git a/test/simd/workload_test.go b/test/simd/workload_test.go new file mode 100644 index 0000000000..6b2400f96c --- /dev/null +++ b/test/simd/workload_test.go @@ -0,0 +1,57 @@ +//go:build goexperiment.simd && (amd64 || arm64 || wasm) + +package simd_test + +import ( + "bytes" + "encoding/hex" + "simd/archsimd" + "testing" +) + +func encodeHexSIMD(dst, src []byte) { + tableBytes := [16]byte{'0', '1', '2', '3', '4', '5', '6', '7', '8', '9', 'a', 'b', 'c', 'd', 'e', 'f'} + table := archsimd.LoadUint8x16Array(&tableBytes) + mask := archsimd.BroadcastUint8x16(15) + n := len(src) &^ 15 + for pos := 0; pos < n; pos += 16 { + x := archsimd.LoadUint8x16(src[pos:]) + lo := lookupHexDigits(table, x.And(mask)) + // Shifting pairs of bytes and masking each byte also works on amd64, + // whose official API does not expose a byte-lane ShiftAllRight. + hi := lookupHexDigits(table, x.ReshapeToUint16s().ShiftAllRight(4).ReshapeToUint8s().And(mask)) + var lows, highs [16]byte + lo.StoreArray(&lows) + hi.StoreArray(&highs) + for i := range lows { + dst[2*(pos+i)], dst[2*(pos+i)+1] = highs[i], lows[i] + } + } + for i := n; i < len(src); i++ { + dst[2*i], dst[2*i+1] = tableBytes[src[i]>>4], tableBytes[src[i]&15] + } +} + +func TestSIMDHexWorkload(t *testing.T) { + // Every tail length, empty input, unaligned starts, and every byte value. + for offset := 0; offset < 4; offset++ { + for n := 0; n <= 273; n++ { + backing := make([]byte, n+offset) + src := backing[offset:] + for i := range src { + src[i] = byte(i*73 + n) + } + want := make([]byte, n*2) + hex.Encode(want, src) + backingOut := bytes.Repeat([]byte{0xa5}, 2*n+4) + out := backingOut[1 : 1+2*n] + encodeHexSIMD(out, src) + if !bytes.Equal(out, want) { + t.Fatalf("offset=%d length=%d: %x want %x", offset, n, out, want) + } + if backingOut[0] != 0xa5 || backingOut[len(backingOut)-1] != 0xa5 || backingOut[len(backingOut)-2] != 0xa5 || backingOut[len(backingOut)-3] != 0xa5 { + t.Fatalf("output boundary overwritten: length=%d", n) + } + } + } +} From 76d3edc3c3d7e48c5729848dcd8a3db4922fca3b Mon Sep 17 00:00:00 2001 From: ZhouGuangyuan Date: Sat, 3 Oct 2026 01:50:01 +0800 Subject: [PATCH 11/20] simd: document qualification and verify CPU initialization with LTO --- internal/build/cpu_init_test.go | 15 ++++-- test/simd/README.md | 94 +++++++++++++++++++++++++++++---- 2 files changed, 95 insertions(+), 14 deletions(-) diff --git a/internal/build/cpu_init_test.go b/internal/build/cpu_init_test.go index c88e2cb270..3f67156d3b 100644 --- a/internal/build/cpu_init_test.go +++ b/internal/build/cpu_init_test.go @@ -82,11 +82,18 @@ func TestCPUInitializationMatchesGo(t *testing.T) { if runtime.GOARCH == "arm64" { debugOptions = debugOptions[:4] } - for _, level := range []optlevel.Level{optlevel.O0, optlevel.O2} { - t.Run(level.String(), func(t *testing.T) { + for _, mode := range []struct { + name string + level optlevel.Level + lto lto.Mode + }{ + {"O0", optlevel.O0, lto.Off}, {"O2", optlevel.O2, lto.Off}, + {"O2-thin", optlevel.O2, lto.Thin}, {"O2-full", optlevel.O2, lto.Full}, + } { + t.Run(mode.name, func(t *testing.T) { conf := NewDefaultConf(ModeBuild) - conf.GOEXPERIMENT, conf.OptLevel, conf.LTO = "simd", level, lto.Off - conf.OutFile = filepath.Join(dir, "llgo-"+level.String()) + conf.GOEXPERIMENT, conf.OptLevel, conf.LTO = "simd", mode.level, mode.lto + conf.OutFile = filepath.Join(dir, "llgo-"+mode.name) if runtime.GOOS == "windows" { conf.OutFile += ".exe" } diff --git a/test/simd/README.md b/test/simd/README.md index 930209e859..e9e1e4deb3 100644 --- a/test/simd/README.md +++ b/test/simd/README.md @@ -6,24 +6,98 @@ Shared behavior tests run under official Go and LLGo. LLGo-specific behavior uses an additional `llgo` build tag; LLVM IR/instruction assertions belong in compiler tests instead. -From the repository root: +## Implemented SIMD128 coverage + +The table-driven lowering implements the applicable official declarations on +amd64, arm64, and wasm. The architecture's API determines which type/method +combinations exist; the families below do not imply identical APIs on every +architecture. + +- Array/slice loads and stores, broadcast, and lane extraction/insertion. +- Add/subtract/multiply, floating divide, negation, absolute value, square root, + rounding, and target-specific floating min/max semantics. +- Integer bitwise operations, saturated add/subtract, min/max, uniform shifts, + and the applicable variable shifts. +- Numeric bitcasts, signedness conversions, and SIMD128 numeric conversions, + with defined results for NaN and out-of-range floating inputs. +- Four mask shapes, comparisons, mask bitmaps where provided, and the official + Go helpers for masking and conditional selection. +- Byte lookup with out-of-range zeroing on arm64/wasm, and baseline-safe amd64 + byte/word permutations and negative-index zeroing. + +Tests cover NaNs, signed zero, wrapping/saturation boundaries, large shift +counts, every byte lookup index, element-aligned addresses, short slices, +nil arrays, and values crossing interfaces, tuples, closures, package calls, +defer/recover, goroutines, and scheduling. The hex-encoding workload checks +all tail lengths, unaligned input, and output sentinels against `encoding/hex`. +It is a correctness workload, not a performance benchmark. + +Go helper bodies are compiled normally. Remaining bodyless intrinsic +implementations panic with their symbol name; `unimplemented_llgo_test.go` +checks direct, indirect, deferred, and linkname calls. SIMD reflection, +portable `simd` specialization, general FMV, and 256/512-bit vectors remain +outside this implemented stage. + +## Running + +From the repository root, using the built LLGo binary on `PATH`: ```sh GOEXPERIMENT=simd go test -count=1 ./test/simd/... GOEXPERIMENT=simd llgo test -O0 -count=1 ./test/simd/... GOEXPERIMENT=simd llgo test -O2 -count=1 ./test/simd/... +GOEXPERIMENT=simd GODEBUG=cpu.all=off llgo test -O2 -lto=full -count=1 ./test/simd/... GOEXPERIMENT=simd GOOS=wasip1 GOARCH=wasm go test -exec=wasmtime -count=1 ./test/simd/... GOEXPERIMENT=simd llgo test -O2 -target wasi -emulator -count=1 -timeout=2m ./test/simd/... +GOEXPERIMENT=simd llgo test -O2 -target emscripten -emulator -count=1 -timeout=2m ./test/simd/... ``` WASI execution requires Wasmtime and LLGo's supported Binaryen (`WASMOPT`). -CI runs the native suite on amd64/arm64 and the WASI suite in the existing wasm -test-command job. Native LLGo runs at O0 and O2; WASI runs at O2 because the unoptimized -standard testing framework exceeds Wasmtime's local-variable limit. The shared suite covers implemented SIMD128 operations, array/slice memory access, -and broadcast, including element-aligned addresses, short slices, and nil arrays. -`unimplemented_llgo_test.go` checks that remaining intrinsic declarations panic -with their symbol name, including indirect, deferred, and linkname calls. -SIMD reflection is outside this stage's scope, matching the Go 1.27 support -boundary. As operations are implemented, move their behavior cases into the -shared suite and replace the corresponding fallback assertions. +Emscripten execution requires a compatible SDK and Node.js. The local +qualification used Go 1.27.0, LLVM 22.1.8, Emscripten 6.0.8, and Node 24.19.0. +Existing CI runs native amd64/arm64 at O0/O2 and WASI at O2. + +The complete O0 Wasm test executable exceeds the engines' local-variable +limits. A small executable covers SIMD initialization, cross-package calls, +recovery, and scheduling at O0 without importing the testing framework: + +```sh +GOEXPERIMENT=simd llgo run -O0 -target wasi -emulator ./test/simd/testdata/boundary +GOEXPERIMENT=simd llgo run -O0 -target emscripten -emulator ./test/simd/testdata/boundary +GOEXPERIMENT=simd llgo run -O2 -lto=thin -target emscripten -emulator ./test/simd/testdata/boundary +GOEXPERIMENT=simd llgo run -O2 -lto=full -target emscripten -emulator ./test/simd/testdata/boundary +``` + +Emscripten's JavaScript SjLj wrappers cannot carry `v128`. Calls in functions +containing `setjmp` use a memory bridge for vector arguments/results; ordinary +Wasm vector calls retain their vector ABI. These bridges cannot be inlined, +and bodies retaining vector calls cannot be moved into recovery functions by +a later backend/LTO inliner. The earlier LLVM optimization still runs at the +requested level. + +## CPU initialization and reference boundaries + +Effective CPU flags are initialized before `archsimd` and user initialization, +using the official `GODEBUG=cpu.*` policy. The compiler tests compare startup +queries with official Go and check the platform hooks: + +```sh +go test ./cl ./ssa ./internal/build -run '^(TestSIMD|TestCPUInitialization|TestEmscriptenSIMDCallBridge)' -count=1 +``` + +Generic LLVM legalization keeps the implemented native operations valid for +the compilation baseline. The amd64 assembly check explicitly uses +`GOAMD64=v1` and rejects AVX instructions. Tests needing official AVX512 APIs +use separate LLGo-only scalar-reference cases when that hardware is not +available; this does not qualify native AVX512 execution. + +Two reference limitations remain visible rather than skipped: + +- With official Go 1.27.0 on WASI, the nil-array SIMD load/store cases in + `TestSIMDMemoryBounds` return without a panic. LLGo passes those assertions; + the other shared WASI cases passed in the local comparison. +- Running the full standard-library `internal/cpu` test package under LLGo + currently hits duplicate symbols from the original/test package archives. + This was reproduced without the CPU-initialization change. The dedicated + CPU startup executables and target-hook tests pass independently. From c00d889e6bb57895f1e66c2665e78dc4941aa3cc Mon Sep 17 00:00:00 2001 From: ZhouGuangyuan Date: Sat, 3 Oct 2026 12:02:31 +0800 Subject: [PATCH 12/20] simd: distinguish native nil checks and WASI runner limitations --- test/simd/README.md | 22 +++++++++++++++------- test/simd/memory_nil_test.go | 31 +++++++++++++++++++++++++++++++ test/simd/memory_test.go | 2 -- 3 files changed, 46 insertions(+), 9 deletions(-) create mode 100644 test/simd/memory_nil_test.go diff --git a/test/simd/README.md b/test/simd/README.md index e9e1e4deb3..7a4724649a 100644 --- a/test/simd/README.md +++ b/test/simd/README.md @@ -53,13 +53,20 @@ GOEXPERIMENT=simd llgo test -O2 -target wasi -emulator -count=1 -timeout=2m ./te GOEXPERIMENT=simd llgo test -O2 -target emscripten -emulator -count=1 -timeout=2m ./test/simd/... ``` -WASI execution requires Wasmtime and LLGo's supported Binaryen (`WASMOPT`). +LLGo WASI execution uses the threaded W32 profile and WAMR (`iwasm`), built +with `bash dev/build_iwasm.sh`. The official Go WASI comparison uses Wasmtime. Emscripten execution requires a compatible SDK and Node.js. The local qualification used Go 1.27.0, LLVM 22.1.8, Emscripten 6.0.8, and Node 24.19.0. Existing CI runs native amd64/arm64 at O0/O2 and WASI at O2. -The complete O0 Wasm test executable exceeds the engines' local-variable -limits. A small executable covers SIMD initialization, cross-package calls, +The current WAMR 2.4.5 classic-interpreter profile rejects `v128` function +types with `unknown value type`, even though its build reports SIMD enabled. +The WASI suite is therefore blocked at module loading; the same failure is +reproducible on the main-branch baseline and an import-free `v128` identity +module. Emscripten provides executable Wasm SIMD coverage independently. + +The complete O0 Emscripten test executable exceeds Node's local-variable +limit. A small executable covers SIMD initialization, cross-package calls, recovery, and scheduling at O0 without importing the testing framework: ```sh @@ -92,11 +99,12 @@ the compilation baseline. The amd64 assembly check explicitly uses use separate LLGo-only scalar-reference cases when that hardware is not available; this does not qualify native AVX512 execution. -Two reference limitations remain visible rather than skipped: +Reference boundaries: -- With official Go 1.27.0 on WASI, the nil-array SIMD load/store cases in - `TestSIMDMemoryBounds` return without a panic. LLGo passes those assertions; - the other shared WASI cases passed in the local comparison. +- Official Go 1.27.0 on WASI does not panic for nil SIMD array loads/stores. + `memory_nil_test.go` therefore checks the nil-panic contract under LLGo on + every target and under official native Go. Shared WASI tests still cover + short-slice bounds; LLGo WASI retains both nil assertions. - Running the full standard-library `internal/cpu` test package under LLGo currently hits duplicate symbols from the original/test package archives. This was reproduced without the CPU-initialization change. The dedicated diff --git a/test/simd/memory_nil_test.go b/test/simd/memory_nil_test.go new file mode 100644 index 0000000000..44093afc47 --- /dev/null +++ b/test/simd/memory_nil_test.go @@ -0,0 +1,31 @@ +//go:build goexperiment.simd && (amd64 || arm64 || (llgo && wasm)) + +package simd_test + +import ( + "simd/archsimd" + "testing" +) + +// Official Go 1.27's Wasm SIMD array intrinsics access linear-memory address +// zero without a nil check. Keep LLGo's nil-panic contract covered on all +// targets, and retain the official native comparison where it agrees. +func TestSIMDNilArrayPanics(t *testing.T) { + var v archsimd.Float32x4 + for _, tc := range []struct { + name string + f func() + }{ + {"load", func() { memoryResult = archsimd.LoadFloat32x4Array(nil) }}, + {"store", func() { v.StoreArray(nil) }}, + } { + t.Run(tc.name, func(t *testing.T) { + defer func() { + if recover() == nil { + t.Fatal("missing nil array panic") + } + }() + tc.f() + }) + } +} diff --git a/test/simd/memory_test.go b/test/simd/memory_test.go index 308495f032..01f3c3d09e 100644 --- a/test/simd/memory_test.go +++ b/test/simd/memory_test.go @@ -106,8 +106,6 @@ func TestSIMDMemoryBounds(t *testing.T) { }{ {"short load", func() { archsimd.LoadFloat32x4(make([]float32, 3)) }}, {"short store", func() { v.Store(make([]float32, 3)) }}, - {"nil array load", func() { memoryResult = archsimd.LoadFloat32x4Array(nil) }}, - {"nil array store", func() { v.StoreArray(nil) }}, } { t.Run(tc.name, func(t *testing.T) { defer func() { From 771d853c26e7be34cc95eec30180877de393464c Mon Sep 17 00:00:00 2001 From: ZhouGuangyuan Date: Sun, 4 Oct 2026 20:02:19 +0800 Subject: [PATCH 13/20] simd: address review feedback and qualify Wasmer execution --- cl/simd.go | 3 +++ cl/simd_test.go | 31 +++++++++++++++++++++++++++++++ internal/build/simd_test.go | 22 +++++++++++++++++++--- ssa/simd.go | 30 +++++++++++++++--------------- ssa/simd_convert.go | 3 ++- ssa/simd_integer.go | 6 +++++- ssa/simd_permute.go | 2 ++ ssa/simd_test.go | 4 ++-- test/simd/README.md | 18 ++++++++---------- 9 files changed, 87 insertions(+), 32 deletions(-) diff --git a/cl/simd.go b/cl/simd.go index d0a4d8b6b7..6bea82fea9 100644 --- a/cl/simd.go +++ b/cl/simd.go @@ -168,6 +168,9 @@ func lookupSIMD(fn *ssa.Function, arch string) (simdOperation, bool) { if !ok || !desc.matches(sig, vector) { return fallback() } + if desc.op == llssa.SIMDLookupOrZero && arch != "arm64" && arch != "wasm" { + return fallback() + } return desc, true } diff --git a/cl/simd_test.go b/cl/simd_test.go index 5ad8934f44..4fe1221452 100644 --- a/cl/simd_test.go +++ b/cl/simd_test.go @@ -114,3 +114,34 @@ func TestSIMDFallbackDeclarations(t *testing.T) { } } } + +func TestSIMDLookupOrZeroTargets(t *testing.T) { + const source = `package archsimd + type v128 struct { _ [0]func() } + type Int8x16 struct { tag v128; vals [16]int8 } + func (x Int8x16) LookupOrZero(y Int8x16) Int8x16 + ` + fs := token.NewFileSet() + file, err := parser.ParseFile(fs, "simd.go", source, 0) + if err != nil { + t.Fatal(err) + } + pkg, _, err := ssautil.BuildPackage(&types.Config{}, fs, types.NewPackage("simd/archsimd", "archsimd"), []*ast.File{file}, ssa.SanityCheckFunctions) + if err != nil { + t.Fatal(err) + } + vector := pkg.Pkg.Scope().Lookup("Int8x16").Type().(*types.Named) + fn := pkg.Prog.FuncValue(vector.Method(0)) + for _, arch := range []string{"arm64", "wasm", "amd64"} { + t.Run(arch, func(t *testing.T) { + want := llssa.SIMDLookupOrZero + if arch == "amd64" { + want = llssa.SIMDUnimplemented + } + op, ok := lookupSIMD(fn, arch) + if !ok || op.op != want { + t.Fatalf("lookup = (%v, %v), want %v", op.op, ok, want) + } + }) + } +} diff --git a/internal/build/simd_test.go b/internal/build/simd_test.go index 2dec93ca92..094de5fc34 100644 --- a/internal/build/simd_test.go +++ b/internal/build/simd_test.go @@ -77,9 +77,17 @@ func simdTestDir(t *testing.T) string { dir := t.TempDir() for name, text := range map[string]string{ "go.mod": "module simdtest\n\ngo 1.27\n", "main.go": simd128Source, "bitmap_amd64.go": simdMaskBitmapSource, - "lookup_arm64.go": `package main; import "simd/archsimd"; func lookup(x,y archsimd.Int8x16) archsimd.Int8x16 { return x.LookupOrZero(y) }`, - "lookup_wasm.go": `package main; import "simd/archsimd"; func lookup(x,y archsimd.Int8x16) archsimd.Int8x16 { return x.LookupOrZero(y) }`, - "permute_amd64.go": `package main; import "simd/archsimd"; func permute(x archsimd.Uint8x16, y archsimd.Int8x16) archsimd.Uint8x16 { return x.PermuteOrZero(y) }`, + "lookup_arm64.go": `package main; import "simd/archsimd"; func lookup(x,y archsimd.Int8x16) archsimd.Int8x16 { return x.LookupOrZero(y) }`, + "lookup_wasm.go": `package main; import "simd/archsimd"; func lookup(x,y archsimd.Int8x16) archsimd.Int8x16 { return x.LookupOrZero(y) }`, + "permute_amd64.go": `package main; import "simd/archsimd"; func permute(x archsimd.Uint8x16, y archsimd.Int8x16) archsimd.Uint8x16 { return x.PermuteOrZero(y) } +func permuteConstant(x archsimd.Uint8x16) archsimd.Uint8x16 { + indices := [16]uint8{31, 30, 29, 28, 27, 26, 25, 24, 23, 22, 21, 20, 19, 18, 17, 16} + return x.Permute(archsimd.LoadUint8x16Array(&indices)) +} +func permuteZeroConstant(x archsimd.Uint8x16) archsimd.Uint8x16 { + indices := [16]int8{-1, 30, -128, 28, 27, 26, 25, 24, 23, 22, 21, 20, 19, 18, 17, 16} + return x.PermuteOrZero(archsimd.LoadInt8x16Array(&indices)) +}`, } { if err := os.WriteFile(filepath.Join(dir, name), []byte(text), 0600); err != nil { t.Fatal(err) @@ -193,6 +201,14 @@ func TestSIMD128LLVM(t *testing.T) { if err := mod.RunPasses("default", prog.TargetMachine(), opts); err != nil { t.Fatal(err) } + if target.arch == "amd64" { + for _, name := range []string{"permuteConstant", "permuteZeroConstant"} { + ir := mod.NamedFunction("main." + name).String() + if !strings.Contains(ir, "shufflevector") || strings.Contains(ir, "extractelement") || strings.Contains(ir, "insertelement") { + t.Fatalf("constant permutation did not fold to a vector shuffle:\n%s", ir) + } + } + } asm, err := prog.TargetMachine().EmitToMemoryBuffer(mod, llvm.AssemblyFile) if err != nil { t.Fatal(err) diff --git a/ssa/simd.go b/ssa/simd.go index 88126878b7..71bf7cd9aa 100644 --- a/ssa/simd.go +++ b/ssa/simd.go @@ -145,7 +145,7 @@ func (b Builder) SIMD(op SIMDOp, result Type, args ...Expr) Expr { } return b.Call(b.Pkg.rtFunc("PanicSIMDUnimplemented"), args...) } - b.simdFeatures(op) + b.requireSIMDFeatures() if op == SIMDRound && b.Prog.Target().GOARCH == "amd64" { return b.simdRoundEven(args[0]) } @@ -225,22 +225,26 @@ func (b Builder) SIMD(op SIMDOp, result Type, args ...Expr) Expr { } } +var simdFloatPredicates = map[SIMDOp]llvm.FloatPredicate{ + SIMDEqual: llvm.FloatOEQ, SIMDNotEqual: llvm.FloatUNE, + SIMDLess: llvm.FloatOLT, SIMDLessEqual: llvm.FloatOLE, + SIMDGreater: llvm.FloatOGT, SIMDGreaterEqual: llvm.FloatOGE, +} + +var simdIntPredicates = map[SIMDOp]llvm.IntPredicate{ + SIMDEqual: llvm.IntEQ, SIMDNotEqual: llvm.IntNE, + SIMDLess: llvm.IntSLT, SIMDLessEqual: llvm.IntSLE, + SIMDGreater: llvm.IntSGT, SIMDGreaterEqual: llvm.IntSGE, +} + func (b Builder) simdCompare(op SIMDOp, result Type, x, y Expr) Expr { info := simdLanes(x.RawType()).Elem().Underlying().(*types.Basic).Info() var cond llvm.Value if info&types.IsFloat != 0 { - pred := map[SIMDOp]llvm.FloatPredicate{ - SIMDEqual: llvm.FloatOEQ, SIMDNotEqual: llvm.FloatUNE, - SIMDLess: llvm.FloatOLT, SIMDLessEqual: llvm.FloatOLE, - SIMDGreater: llvm.FloatOGT, SIMDGreaterEqual: llvm.FloatOGE, - }[op] + pred := simdFloatPredicates[op] cond = b.impl.CreateFCmp(pred, x.impl, y.impl, "") } else { - pred := map[SIMDOp]llvm.IntPredicate{ - SIMDEqual: llvm.IntEQ, SIMDNotEqual: llvm.IntNE, - SIMDLess: llvm.IntSLT, SIMDLessEqual: llvm.IntSLE, - SIMDGreater: llvm.IntSGT, SIMDGreaterEqual: llvm.IntSGE, - }[op] + pred := simdIntPredicates[op] if info&types.IsUnsigned != 0 { switch pred { case llvm.IntSLT: @@ -263,10 +267,6 @@ var simdFloatUnary = map[SIMDOp]string{ SIMDTrunc: "llvm.trunc", SIMDRound: "llvm.roundeven", } -func (b Builder) simdFeatures(op SIMDOp) { - b.requireSIMDFeatures() -} - func (b Builder) requireSIMDFeatures() { if b.Prog.Target().GOARCH == "wasm" { features := "+simd128" diff --git a/ssa/simd_convert.go b/ssa/simd_convert.go index 668661279e..9ae81d6404 100644 --- a/ssa/simd_convert.go +++ b/ssa/simd_convert.go @@ -59,7 +59,8 @@ func (b Builder) simdConvert(result Type, x Expr) Expr { v = b.impl.CreateBitCast(x.impl, typ, "") } if n != result.ll.VectorSize() { - // Narrowing conversions leave the unused upper lanes zero on these targets. + // Converting to a narrower element type increases the result lane count. + // Preserve the converted lanes and zero-fill the added upper lanes. mask := make([]llvm.Value, result.ll.VectorSize()) for i := range mask { index := i diff --git a/ssa/simd_integer.go b/ssa/simd_integer.go index ad6433516e..447fa0d6fa 100644 --- a/ssa/simd_integer.go +++ b/ssa/simd_integer.go @@ -56,8 +56,12 @@ func (b Builder) simdShiftValue(x Expr, count llvm.Value, right bool) llvm.Value return b.impl.CreateSelect(large, llvm.ConstNull(x.ll), shifted, "") } +var simdIntegerIntrinsics = map[SIMDOp]string{ + SIMDAddSaturated: "add.sat", SIMDSubSaturated: "sub.sat", SIMDMin: "min", SIMDMax: "max", +} + func (b Builder) simdIntegerIntrinsic(op SIMDOp, x, y Expr) Expr { - name := map[SIMDOp]string{SIMDAddSaturated: "add.sat", SIMDSubSaturated: "sub.sat", SIMDMin: "min", SIMDMax: "max"}[op] + name := simdIntegerIntrinsics[op] prefix := "llvm.s" if simdLanes(x.RawType()).Elem().Underlying().(*types.Basic).Info()&types.IsUnsigned != 0 { prefix = "llvm.u" diff --git a/ssa/simd_permute.go b/ssa/simd_permute.go index 14af548d77..6bd246a7f4 100644 --- a/ssa/simd_permute.go +++ b/ssa/simd_permute.go @@ -11,6 +11,8 @@ func (b Builder) simdPermute(op SIMDOp, x, indices Expr) Expr { value := b.impl.CreateIntrinsic(x.ll, llvm.LookupIntrinsicID(name), []llvm.Value{x.impl, indices.impl}, "") return Expr{value, x.Type} } + // Dynamic amd64 permutations use a baseline-safe lane gather; constant + // indices fold to shufflevector during LLVM optimization. // Mask indices before extracting a lane, including indices whose sign bit // requests zero. An out-of-bounds extract would create LLVM poison. n := x.ll.VectorSize() diff --git a/ssa/simd_test.go b/ssa/simd_test.go index e5121cf8a6..c11d192265 100644 --- a/ssa/simd_test.go +++ b/ssa/simd_test.go @@ -17,8 +17,8 @@ func TestSIMDFeaturesPreserveExistingRequirements(t *testing.T) { fn := pkg.NewFunc("p.f", NoArgsNoRet, InGo) fn.impl.AddFunctionAttr(prog.ctx.CreateStringAttribute("target-features", "+bulk-memory")) b := fn.MakeBody(1) - b.simdFeatures(SIMDAdd) - b.simdFeatures(SIMDSub) + b.requireSIMDFeatures() + b.requireSIMDFeatures() b.Return() b.EndBuild() for _, attr := range fn.impl.GetFunctionAttributes() { diff --git a/test/simd/README.md b/test/simd/README.md index 7a4724649a..de4504cc34 100644 --- a/test/simd/README.md +++ b/test/simd/README.md @@ -23,7 +23,8 @@ architecture. - Four mask shapes, comparisons, mask bitmaps where provided, and the official Go helpers for masking and conditional selection. - Byte lookup with out-of-range zeroing on arm64/wasm, and baseline-safe amd64 - byte/word permutations and negative-index zeroing. + byte/word permutations and negative-index zeroing. Constant permutations + fold to vector shuffles at O2; dynamic indices use a per-lane fallback. Tests cover NaNs, signed zero, wrapping/saturation boundaries, large shift counts, every byte lookup index, element-aligned addresses, short slices, @@ -49,21 +50,18 @@ GOEXPERIMENT=simd llgo test -O2 -count=1 ./test/simd/... GOEXPERIMENT=simd GODEBUG=cpu.all=off llgo test -O2 -lto=full -count=1 ./test/simd/... GOEXPERIMENT=simd GOOS=wasip1 GOARCH=wasm go test -exec=wasmtime -count=1 ./test/simd/... +GOEXPERIMENT=simd llgo test -O0 -target wasi -emulator -count=1 -timeout=2m ./test/simd/... GOEXPERIMENT=simd llgo test -O2 -target wasi -emulator -count=1 -timeout=2m ./test/simd/... GOEXPERIMENT=simd llgo test -O2 -target emscripten -emulator -count=1 -timeout=2m ./test/simd/... ``` -LLGo WASI execution uses the threaded W32 profile and WAMR (`iwasm`), built -with `bash dev/build_iwasm.sh`. The official Go WASI comparison uses Wasmtime. +LLGo WASI execution uses the threaded W32 profile and Wasmer 7.5.0, installed +with `bash dev/install_wasmer.sh`. The runner supports SIMD, shared-memory +threads, and standard Wasm exception handling; Wasmer selects an available +backend automatically. The official Go WASI comparison uses Wasmtime. Emscripten execution requires a compatible SDK and Node.js. The local qualification used Go 1.27.0, LLVM 22.1.8, Emscripten 6.0.8, and Node 24.19.0. -Existing CI runs native amd64/arm64 at O0/O2 and WASI at O2. - -The current WAMR 2.4.5 classic-interpreter profile rejects `v128` function -types with `unknown value type`, even though its build reports SIMD enabled. -The WASI suite is therefore blocked at module loading; the same failure is -reproducible on the main-branch baseline and an import-free `v128` identity -module. Emscripten provides executable Wasm SIMD coverage independently. +Existing CI runs native amd64/arm64 and WASI at O0/O2. The complete O0 Emscripten test executable exceeds Node's local-variable limit. A small executable covers SIMD initialization, cross-package calls, From 374e5eca9dd9a553875cfb2973343dcd267e1a32 Mon Sep 17 00:00:00 2001 From: ZhouGuangyuan Date: Sun, 4 Oct 2026 20:44:30 +0800 Subject: [PATCH 14/20] test: cover CRC folding and Windows CPU override policy --- internal/build/cpu_init_test.go | 48 +++++++++++++++++++++++++++++-- test/simd/README.md | 5 ++++ test/std/hash/crc32/crc32_test.go | 30 +++++++++++++++++++ 3 files changed, 80 insertions(+), 3 deletions(-) diff --git a/internal/build/cpu_init_test.go b/internal/build/cpu_init_test.go index 3f67156d3b..06536bd473 100644 --- a/internal/build/cpu_init_test.go +++ b/internal/build/cpu_init_test.go @@ -64,7 +64,7 @@ func TestCPUInitializationMatchesGo(t *testing.T) { } cmd := exec.Command("go", "build", "-o", reference, ".") cmd.Dir = dir - cmd.Env = withEnv(os.Environ(), "GOEXPERIMENT=simd") + cmd.Env = withEnv(os.Environ(), "GOEXPERIMENT=simd", "GOAMD64=v1") if out, err := cmd.CombinedOutput(); err != nil { t.Fatalf("official build: %v\n%s", err, out) } @@ -82,6 +82,39 @@ func TestCPUInitializationMatchesGo(t *testing.T) { if runtime.GOARCH == "arm64" { debugOptions = debugOptions[:4] } + // Go 1.27 initializes CPU flags before reading Windows environment + // variables. Compare default detection there, then check LLGo overrides + // against explicit expectations instead of that reference limitation. + baseline := strings.Fields(run(reference, "")) + if len(baseline) != 6 { + t.Fatalf("unexpected CPU output: %v", baseline) + } + expected := func(debug string) string { + if debug == "" { + return strings.Join(baseline, " ") + } + values := append([]string(nil), baseline...) + switch debug { + case "cpu.all=off", "cpu.all=off,cpu.aes=on": + for i := range values { + values[i] = "false" + } + if debug == "cpu.all=off,cpu.aes=on" && runtime.GOARCH == "arm64" { + values[0], values[1] = baseline[0], baseline[1] + } + case "cpu.aes=off": + if runtime.GOARCH == "arm64" { + values[0], values[1] = "false", "false" + } else { + values[5] = "false" + } + case "cpu.avx=off": + values[0], values[2], values[5] = "false", "false", "false" + default: + t.Fatalf("missing expectation for GODEBUG=%q", debug) + } + return strings.Join(values, " ") + } for _, mode := range []struct { name string level optlevel.Level @@ -93,6 +126,9 @@ func TestCPUInitializationMatchesGo(t *testing.T) { t.Run(mode.name, func(t *testing.T) { conf := NewDefaultConf(ModeBuild) conf.GOEXPERIMENT, conf.OptLevel, conf.LTO = "simd", mode.level, mode.lto + if runtime.GOARCH == "amd64" { + conf.GOAMD64 = "v1" + } conf.OutFile = filepath.Join(dir, "llgo-"+mode.name) if runtime.GOOS == "windows" { conf.OutFile += ".exe" @@ -101,8 +137,14 @@ func TestCPUInitializationMatchesGo(t *testing.T) { t.Fatal(err) } for _, debug := range debugOptions { - if got, want := run(conf.OutFile, debug), run(reference, debug); got != want { - t.Fatalf("GODEBUG=%q: LLGo %q, official Go %q", debug, got, want) + got := run(conf.OutFile, debug) + if want := expected(debug); got != want { + t.Fatalf("GODEBUG=%q: LLGo %q, expected %q", debug, got, want) + } + if runtime.GOOS != "windows" { + if want := run(reference, debug); got != want { + t.Fatalf("GODEBUG=%q: LLGo %q, official Go %q", debug, got, want) + } } } }) diff --git a/test/simd/README.md b/test/simd/README.md index de4504cc34..b8626e8831 100644 --- a/test/simd/README.md +++ b/test/simd/README.md @@ -99,6 +99,11 @@ available; this does not qualify native AVX512 execution. Reference boundaries: +- Official Go 1.27.0 initializes Windows CPU flags before reading `GODEBUG`. + The CPU startup test compares default hardware detection with Go and checks + LLGo overrides against explicit expectations; other native hosts also retain + the Go comparison for each override. Both amd64 compilers use `GOAMD64=v1`. + - Official Go 1.27.0 on WASI does not panic for nil SIMD array loads/stores. `memory_nil_test.go` therefore checks the nil-panic contract under LLGo on every target and under official native Go. Shared WASI tests still cover diff --git a/test/std/hash/crc32/crc32_test.go b/test/std/hash/crc32/crc32_test.go index 41d7fff950..600d614468 100644 --- a/test/std/hash/crc32/crc32_test.go +++ b/test/std/hash/crc32/crc32_test.go @@ -146,3 +146,33 @@ func TestHashBinaryMarshaling(t *testing.T) { t.Fatalf("post-unmarshal Sum32=%#x want %#x", h1.Sum32(), ieee) } } + +// Cover the 64-byte folding boundary and the 1024-byte AVX512 transition +// against a scalar polynomial reference, including streaming CRC state. +func TestIEEELargeBuffers(t *testing.T) { + for _, n := range []int{63, 64, 65, 1023, 1024, 1025, 2048, 4095, 4096} { + for _, offset := range []int{0, 1, 7, 15} { + data := make([]byte, n+offset)[offset:] + for i := range data { + data[i] = byte(i*31 + 7) + } + for _, seed := range []uint32{0, 0x12345678, 0xffffffff} { + want := ^seed + for _, b := range data { + want ^= uint32(b) + for range 8 { + if want&1 != 0 { + want = want>>1 ^ crc32.IEEE + } else { + want >>= 1 + } + } + } + want = ^want + if got := crc32.Update(seed, crc32.IEEETable, data); got != want { + t.Fatalf("length=%d offset=%d seed=%08x: got %08x, want %08x", n, offset, seed, got, want) + } + } + } + } +} From 09f4fe21dea0692cb0f1352dd55d23cd3e038ba9 Mon Sep 17 00:00:00 2001 From: ZhouGuangyuan Date: Mon, 5 Oct 2026 08:27:25 +0800 Subject: [PATCH 15/20] deps: update plan9asm to v0.6.2 for SIMD register aliasing --- go.mod | 2 +- go.sum | 2 ++ 2 files changed, 3 insertions(+), 1 deletion(-) diff --git a/go.mod b/go.mod index 8413dd9f1f..38b75620ad 100644 --- a/go.mod +++ b/go.mod @@ -11,7 +11,7 @@ require ( github.com/qiniu/x v1.18.3 github.com/xgo-dev/llgo/runtime v0.0.0-00010101000000-000000000000 github.com/xgo-dev/llvm v0.10.0 - github.com/xgo-dev/plan9asm v0.6.1 + github.com/xgo-dev/plan9asm v0.6.2 go.bug.st/serial v1.6.4 go.yaml.in/yaml/v3 v3.0.5 golang.org/x/mod v0.41.0 diff --git a/go.sum b/go.sum index 5e3ac8d854..1c9389a6fc 100644 --- a/go.sum +++ b/go.sum @@ -26,6 +26,8 @@ github.com/xgo-dev/llvm v0.10.0 h1:o27kypAGI7DrLFckrsUVjWChXqHUBDAFtZbwMiDFs50= github.com/xgo-dev/llvm v0.10.0/go.mod h1:42vav2/cI5BAIcL543DZSMO9do8/aCK2z7JERH+AE+M= github.com/xgo-dev/plan9asm v0.6.1 h1:WcwHdcNKBzZnEwl/m4AQPkBQ0XO96d9kgBhanmpINuA= github.com/xgo-dev/plan9asm v0.6.1/go.mod h1:UTpcZz3aQXSWYruhfzOi8aK4QZTEFPIvB61YW6jHT4M= +github.com/xgo-dev/plan9asm v0.6.2 h1:9mV7u9stDT2V3wZP+ywRP/wbWmtQwf5bb6zym5Unyww= +github.com/xgo-dev/plan9asm v0.6.2/go.mod h1:UTpcZz3aQXSWYruhfzOi8aK4QZTEFPIvB61YW6jHT4M= go.bug.st/serial v1.6.4 h1:7FmqNPgVp3pu2Jz5PoPtbZ9jJO5gnEnZIvnI1lzve8A= go.bug.st/serial v1.6.4/go.mod h1:nofMJxTeNVny/m6+KaafC6vJGj3miwQZ6vW4BZUGJPI= go.yaml.in/yaml/v3 v3.0.5 h1:N6y/pJk8buWs9NY5ERU2HSMfm+IuD/OtfdAnq6kESPw= From c6f35c79679ea1e12c4514820900a4b34b2f6b4b Mon Sep 17 00:00:00 2001 From: ZhouGuangyuan Date: Mon, 5 Oct 2026 09:03:38 +0800 Subject: [PATCH 16/20] wasm: bridge GoJS SIMD calls only for JavaScript SjLj --- internal/build/build.go | 4 +- internal/build/wasm_simd_calls.go | 61 +++++++++++++++++- internal/build/wasm_simd_calls_test.go | 88 ++++++++++++++++++++++---- internal/clang/clang.go | 6 ++ test/simd/README.md | 13 +++- 5 files changed, 154 insertions(+), 18 deletions(-) diff --git a/internal/build/build.go b/internal/build/build.go index d75461372f..552d9fee1d 100644 --- a/internal/build/build.go +++ b/internal/build/build.go @@ -3229,7 +3229,7 @@ func lowerMainCExportModule(ctx *context, pkg llssa.Package, exports []cExport) if err := optimizeLLVMModule(ctx, pkg.Path(), mod); err != nil { return true, err } - lowerEmscriptenSIMDCalls(string(ctx.crossCompile.WasmProvider), mod) + lowerEmscriptenSIMDCalls(ctx, mod) localizeWasmStackAddresses(ctx.buildConf.Goarch, mod) return true, nil } @@ -3298,7 +3298,7 @@ func compilePackageModule(ctx *context, aPkg *aPackage, externs []string, verbos if err := optimizeLLVMModule(ctx, pkgPath, ret.Module()); err != nil { return err } - lowerEmscriptenSIMDCalls(string(ctx.crossCompile.WasmProvider), ret.Module()) + lowerEmscriptenSIMDCalls(ctx, ret.Module()) localizeWasmStackAddresses(ctx.buildConf.Goarch, ret.Module()) dropUnusedWindowsTestMain(ctx, aPkg, ret.Module()) emitFuncInfoEntrySites(ctx, ret) diff --git a/internal/build/wasm_simd_calls.go b/internal/build/wasm_simd_calls.go index dcddf1d660..3f2ff23480 100644 --- a/internal/build/wasm_simd_calls.go +++ b/internal/build/wasm_simd_calls.go @@ -2,18 +2,75 @@ package build import ( "fmt" + "os" "strings" + "github.com/xgo-dev/llgo/internal/crosscompile" + "github.com/xgo-dev/llgo/xtool/safesplit" "github.com/xgo-dev/llvm" ) +func needsEmscriptenSIMDCallBridge(ctx *context) bool { + provider := ctx.crossCompile.WasmProvider + if provider != crosscompile.WasmProviderEmscripten && provider != crosscompile.WasmProviderGoJS { + return false + } + if ctx.crossCompile.WasmProfile == crosscompile.WasmProfileJ64 { + // Memory64 IR uses matching clang with explicit JS SjLj flags, rather + // than emcc. EMCC_CFLAGS does not select that compiler's lowering. + return true + } + emccFlags := ctx.commands.lookup("EMCC_CFLAGS") + if ctx.commands.environ == nil { + emccFlags = os.Getenv("EMCC_CFLAGS") + } + flags := safesplit.SplitPkgConfigFlags(emccFlags) + compiler := ctx.irCompiler() + linker := ctx.linker() + // emcc appends EMCC_CFLAGS after its command-line arguments. Require + // native SjLj at both codegen and link time, including LTO codegen. + return !emscriptenNativeSjLj(append(compiler.CompileArguments(), flags...)) || + !emscriptenNativeSjLj(append(linker.LinkArguments(), flags...)) +} + +func emscriptenNativeSjLj(args []string) bool { + wasmEH := false + wasmEHSetting := "" + longjmp := "1" + for i := 0; i < len(args); i++ { + arg := args[i] + if strings.HasPrefix(arg, "@") { + // Unknown response-file contents can override any earlier option. + return false + } + if arg == "-fwasm-exceptions" { + wasmEH = true + } + if arg == "-s" && i+1 < len(args) { + i++ + arg += args[i] + } + if value, ok := strings.CutPrefix(arg, "-sSUPPORT_LONGJMP="); ok { + longjmp = strings.Trim(value, "\"'") + } + if value, ok := strings.CutPrefix(arg, "-sWASM_EXCEPTIONS="); ok { + wasmEHSetting = value + } + } + // Emscripten applies explicit -s settings after parsing driver flags. + if wasmEHSetting != "" { + wasmEH = wasmEHSetting == "1" + } + return longjmp == "wasm" || longjmp == "1" && wasmEH +} + // lowerEmscriptenSIMDCalls keeps vectors inside Wasm when Emscripten's JS SjLj // lowering wraps a potentially throwing call. JavaScript cannot carry v128. // Only calls in setjmp functions need a memory bridge; ordinary vector calls // retain their vector ABI. Run after optimization so inlined calls are covered. // Remaining vector-call bodies cannot be inlined later into a setjmp function. -func lowerEmscriptenSIMDCalls(provider string, mod llvm.Module) int { - if provider != "emscripten" { +func lowerEmscriptenSIMDCalls(ctx *context, mod llvm.Module) int { + if !needsEmscriptenSIMDCallBridge(ctx) { return 0 } setjmp := mod.NamedFunction("setjmp") diff --git a/internal/build/wasm_simd_calls_test.go b/internal/build/wasm_simd_calls_test.go index 0813d07680..386a879dfb 100644 --- a/internal/build/wasm_simd_calls_test.go +++ b/internal/build/wasm_simd_calls_test.go @@ -4,6 +4,7 @@ import ( "strings" "testing" + "github.com/xgo-dev/llgo/internal/crosscompile" "github.com/xgo-dev/llvm" ) @@ -36,28 +37,44 @@ define <4 x float> @ordinary(<4 x float> %x, ptr %env) { ret <4 x float> %v } ` - for _, provider := range []string{"emscripten", "wasi", "gojs"} { - t.Run(provider, func(t *testing.T) { - mod := parseWasmAggregateIR(t, source) - before := mod.NamedFunction("ordinary").String() - count := lowerEmscriptenSIMDCalls(provider, mod) - want := 0 - if provider == "emscripten" { - want = 4 + t.Setenv("CCFLAGS", "") + t.Setenv("CFLAGS", "") + t.Setenv("LDFLAGS", "") + for _, tc := range []struct { + name string + provider crosscompile.WasmProvider + flags string + bridges int + }{ + {"emscripten JS", crosscompile.WasmProviderEmscripten, "", 4}, + {"gojs JS", crosscompile.WasmProviderGoJS, "", 4}, + {"wasi", crosscompile.WasmProviderWASI, "", 0}, + {"emscripten native", crosscompile.WasmProviderEmscripten, "-fwasm-exceptions -sSUPPORT_LONGJMP=wasm", 0}, + {"gojs native", crosscompile.WasmProviderGoJS, "-fwasm-exceptions -sSUPPORT_LONGJMP=wasm", 0}, + } { + t.Run(tc.name, func(t *testing.T) { + ctx := &context{ + buildConf: &Config{}, + crossCompile: crosscompile.Export{WasmProvider: tc.provider}, + commands: commandEnv{environ: []string{"EMCC_CFLAGS=" + tc.flags}}, } + mod := parseWasmAggregateIR(t, source) + before := mod.String() + count := lowerEmscriptenSIMDCalls(ctx, mod) + want := tc.bridges if count != want { t.Fatalf("bridges=%d want %d", count, want) } if err := llvm.VerifyModule(mod, llvm.ReturnStatusAction); err != nil { t.Fatalf("%v\n%s", err, mod.String()) } - if provider != "emscripten" && before != mod.NamedFunction("ordinary").String() { - t.Fatal("changed other provider") + if want == 0 && before != mod.String() { + t.Fatal("changed a module without JS SjLj") } if !strings.Contains(mod.NamedFunction("ordinary").String(), "call <4 x float> @callee") { t.Fatal("changed ordinary vector ABI") } - if provider == "emscripten" { + if want != 0 { ir := mod.NamedFunction("withjmp").String() if strings.Contains(ir, "call <4 x float> @callee") || strings.Contains(ir, "call <4 x float> %fn") { t.Fatalf("v128 crosses JS SjLj boundary:\n%s", ir) @@ -82,10 +99,57 @@ define <4 x float> @ordinary(<4 x float> %x, ptr %env) { if err := llvm.VerifyModule(mod, llvm.ReturnStatusAction); err != nil { t.Fatal(err) } - if second := lowerEmscriptenSIMDCalls(provider, mod); second != 0 { + if second := lowerEmscriptenSIMDCalls(ctx, mod); second != 0 { t.Fatalf("pass not idempotent: %d", second) } } }) } } + +func TestEmscriptenSIMDCallBridgeMode(t *testing.T) { + for _, tc := range []struct { + name, cc, c, ld, emcc string + configCC, configLD []string + cxx []string + memory64, bridge bool + }{ + {name: "default", bridge: true}, + {name: "native flag", emcc: "-fwasm-exceptions"}, + {name: "native longjmp", emcc: "-s SUPPORT_LONGJMP='wasm'"}, + {name: "explicit JS", emcc: "-sSUPPORT_LONGJMP=emscripten", bridge: true}, + {name: "last longjmp wins", emcc: "-sSUPPORT_LONGJMP=wasm -s SUPPORT_LONGJMP=emscripten", bridge: true}, + {name: "default longjmp with EH", emcc: "-fwasm-exceptions -sSUPPORT_LONGJMP=1"}, + {name: "explicit EH setting", emcc: "-sWASM_EXCEPTIONS=1"}, + {name: "EH setting overrides flag", emcc: "-sWASM_EXCEPTIONS=0 -fwasm-exceptions", bridge: true}, + {name: "shared driver flags", cc: "-fwasm-exceptions"}, + {name: "compile only", c: "-fwasm-exceptions", bridge: true}, + {name: "link only", ld: "-fwasm-exceptions", bridge: true}, + {name: "config both", configCC: []string{"-fwasm-exceptions"}, configLD: []string{"-fwasm-exceptions"}}, + {name: "CXX link driver", c: "-fwasm-exceptions", cxx: []string{"-fwasm-exceptions"}}, + {name: "emcc overrides config", configCC: []string{"-sSUPPORT_LONGJMP=emscripten"}, configLD: []string{"-sSUPPORT_LONGJMP=emscripten"}, emcc: "-sSUPPORT_LONGJMP=wasm"}, + {name: "unknown response file", emcc: "-fwasm-exceptions @options.rsp", bridge: true}, + {name: "memory64 retains JS codegen", memory64: true, emcc: "-fwasm-exceptions -sSUPPORT_LONGJMP=wasm", bridge: true}, + } { + t.Run(tc.name, func(t *testing.T) { + t.Setenv("CCFLAGS", tc.cc) + t.Setenv("CFLAGS", tc.c) + t.Setenv("LDFLAGS", tc.ld) + t.Setenv("EMCC_CFLAGS", tc.emcc) + ctx := &context{buildConf: &Config{}, crossCompile: crosscompile.Export{ + WasmProvider: crosscompile.WasmProviderGoJS, + CCFLAGS: tc.configCC, LDFLAGS: tc.configLD, + }} + if tc.cxx != nil { + ctx.crossCompile.CXX = "em++" + ctx.crossCompile.CXXArgs = tc.cxx + } + if tc.memory64 { + ctx.crossCompile.WasmProfile = crosscompile.WasmProfileJ64 + } + if got := needsEmscriptenSIMDCallBridge(ctx); got != tc.bridge { + t.Fatalf("bridge = %v, want %v", got, tc.bridge) + } + }) + } +} diff --git a/internal/clang/clang.go b/internal/clang/clang.go index 729c41cafc..f3d2619813 100644 --- a/internal/clang/clang.go +++ b/internal/clang/clang.go @@ -133,6 +133,12 @@ func (c *Cmd) Compile(args ...string) error { return c.exec(allArgs...) } +// CompileArguments returns the effective driver arguments without executing +// the compiler, including the command prefix and environment flags. +func (c *Cmd) CompileArguments(args ...string) []string { + return slices.Concat(c.prefixArgs, c.mergeCompilerFlags(), args) +} + // Link executes a linking command with merged flags. func (c *Cmd) Link(args ...string) error { return c.exec(c.linkArguments(args...)...) diff --git a/test/simd/README.md b/test/simd/README.md index b8626e8831..72b53ffad3 100644 --- a/test/simd/README.md +++ b/test/simd/README.md @@ -52,6 +52,8 @@ GOEXPERIMENT=simd GODEBUG=cpu.all=off llgo test -O2 -lto=full -count=1 ./test/si GOEXPERIMENT=simd GOOS=wasip1 GOARCH=wasm go test -exec=wasmtime -count=1 ./test/simd/... GOEXPERIMENT=simd llgo test -O0 -target wasi -emulator -count=1 -timeout=2m ./test/simd/... GOEXPERIMENT=simd llgo test -O2 -target wasi -emulator -count=1 -timeout=2m ./test/simd/... +GOEXPERIMENT=simd llgo test -O2 -lto=thin -target wasi -emulator -count=1 -timeout=2m ./test/simd/... +GOEXPERIMENT=simd llgo test -O2 -lto=full -target wasi -emulator -count=1 -timeout=2m ./test/simd/... GOEXPERIMENT=simd llgo test -O2 -target emscripten -emulator -count=1 -timeout=2m ./test/simd/... ``` @@ -70,16 +72,23 @@ recovery, and scheduling at O0 without importing the testing framework: ```sh GOEXPERIMENT=simd llgo run -O0 -target wasi -emulator ./test/simd/testdata/boundary GOEXPERIMENT=simd llgo run -O0 -target emscripten -emulator ./test/simd/testdata/boundary +GOEXPERIMENT=simd GOOS=js GOARCH=wasm llgo run -O0 -emulator ./test/simd/testdata/boundary GOEXPERIMENT=simd llgo run -O2 -lto=thin -target emscripten -emulator ./test/simd/testdata/boundary GOEXPERIMENT=simd llgo run -O2 -lto=full -target emscripten -emulator ./test/simd/testdata/boundary ``` -Emscripten's JavaScript SjLj wrappers cannot carry `v128`. Calls in functions +The default GoJS and Emscripten JavaScript SjLj wrappers cannot carry `v128`. Calls in functions containing `setjmp` use a memory bridge for vector arguments/results; ordinary Wasm vector calls retain their vector ABI. These bridges cannot be inlined, and bodies retaining vector calls cannot be moved into recovery functions by a later backend/LTO inliner. The earlier LLVM optimization still runs at the -requested level. +requested level. When compilation and linking select native Wasm SjLj (for +example, `EMCC_CFLAGS='-fwasm-exceptions -sSUPPORT_LONGJMP=wasm'`), these +bridges and late-inlining restrictions are unnecessary and omitted. Memory64 +IR still uses explicit JS SjLj codegen and retains the bridge. + +WASI SIMD execution is also qualified at O2 with Thin and Full LTO using the +commands above and Wasmer 7.5.0. ## CPU initialization and reference boundaries From 6772d71f5a84197f1b93c209ed06a9e15b6a5c94 Mon Sep 17 00:00:00 2001 From: ZhouGuangyuan Date: Mon, 5 Oct 2026 16:49:32 +0800 Subject: [PATCH 17/20] ci: temporarily bypass Codecov outage for PR 2722 --- .github/workflows/doc-link-checker.yml | 7 ++++++- .github/workflows/go.yml | 5 ++++- .github/workflows/llgo.yml | 6 ++++-- 3 files changed, 14 insertions(+), 4 deletions(-) diff --git a/.github/workflows/doc-link-checker.yml b/.github/workflows/doc-link-checker.yml index 6fcbf8b709..f956eb60d1 100644 --- a/.github/workflows/doc-link-checker.yml +++ b/.github/workflows/doc-link-checker.yml @@ -51,4 +51,9 @@ jobs: id: lychee uses: lycheeverse/lychee-action@v2 with: - args: --max-concurrency 3 --retry-wait-time 15 README.md + # The same Codecov TLS outage affects its README link. Keep all + # other links checked; restore Codecov automatically outside #2722. + args: >- + --max-concurrency 3 --retry-wait-time 15 + ${{ github.event.pull_request.number == 2722 && '--exclude "^https://(app[.])?codecov[.]io/"' || '' }} + README.md diff --git a/.github/workflows/go.yml b/.github/workflows/go.yml index dc60f2f059..0649fcf329 100644 --- a/.github/workflows/go.yml +++ b/.github/workflows/go.yml @@ -234,7 +234,8 @@ jobs: done - name: Upload coverage reports to Codecov - if: steps.test_coverage.outputs.windows_runtime_corruption != 'true' + # Temporarily bypass Codecov's TLS outage for PR #2722 only. + if: github.event.pull_request.number != 2722 && steps.test_coverage.outputs.windows_runtime_corruption != 'true' uses: codecov/codecov-action@v7 with: token: ${{secrets.CODECOV_TOKEN}} @@ -360,6 +361,8 @@ jobs: bash .github/workflows/test_demo.sh - name: Upload dev LTO GlobalDCE coverage reports to Codecov + # Temporarily bypass Codecov's TLS outage for PR #2722 only. + if: github.event.pull_request.number != 2722 continue-on-error: true uses: codecov/codecov-action@v7 with: diff --git a/.github/workflows/llgo.yml b/.github/workflows/llgo.yml index 80e285bc75..85ac869e9b 100644 --- a/.github/workflows/llgo.yml +++ b/.github/workflows/llgo.yml @@ -220,7 +220,8 @@ jobs: - name: Upload LLGo test coverage to Codecov # Preserve any profile emitted by failing tests, without uploading # unrelated profiles produced by child compiler integration tests. - if: ${{ !cancelled() && hashFiles('coverage-llgo.txt') != '' }} + # Temporarily bypass Codecov's TLS outage for PR #2722 only. + if: ${{ !cancelled() && github.event.pull_request.number != 2722 && hashFiles('coverage-llgo.txt') != '' }} uses: codecov/codecov-action@v7 with: token: ${{ secrets.CODECOV_TOKEN }} @@ -774,7 +775,8 @@ jobs: LLGO_BROWSER_COVERAGE_DIR: coverage-browser run: python3 dev/test_wasm_browser_debug.py - name: Upload browser debugger coverage to Codecov - if: ${{ !cancelled() && hashFiles('coverage-browser/*.out') != '' }} + # Temporarily bypass Codecov's TLS outage for PR #2722 only. + if: ${{ !cancelled() && github.event.pull_request.number != 2722 && hashFiles('coverage-browser/*.out') != '' }} uses: codecov/codecov-action@v7 with: token: ${{ secrets.CODECOV_TOKEN }} From 449454bf8ea2a9234c95fe84ec3fc473ec07ed6d Mon Sep 17 00:00:00 2001 From: ZhouGuangyuan Date: Mon, 5 Oct 2026 17:10:36 +0800 Subject: [PATCH 18/20] ci: pause Codecov uploads until a follow-up PR --- .github/workflows/doc-link-checker.yml | 5 ++--- .github/workflows/go.yml | 8 ++++---- .github/workflows/llgo.yml | 8 ++++---- 3 files changed, 10 insertions(+), 11 deletions(-) diff --git a/.github/workflows/doc-link-checker.yml b/.github/workflows/doc-link-checker.yml index f956eb60d1..57c85678f3 100644 --- a/.github/workflows/doc-link-checker.yml +++ b/.github/workflows/doc-link-checker.yml @@ -51,9 +51,8 @@ jobs: id: lychee uses: lycheeverse/lychee-action@v2 with: - # The same Codecov TLS outage affects its README link. Keep all - # other links checked; restore Codecov automatically outside #2722. + # Temporarily exclude Codecov during its TLS outage; restore in a follow-up PR. args: >- --max-concurrency 3 --retry-wait-time 15 - ${{ github.event.pull_request.number == 2722 && '--exclude "^https://(app[.])?codecov[.]io/"' || '' }} + --exclude "^https://(app[.])?codecov[.]io/" README.md diff --git a/.github/workflows/go.yml b/.github/workflows/go.yml index 0649fcf329..faa34cce19 100644 --- a/.github/workflows/go.yml +++ b/.github/workflows/go.yml @@ -234,8 +234,8 @@ jobs: done - name: Upload coverage reports to Codecov - # Temporarily bypass Codecov's TLS outage for PR #2722 only. - if: github.event.pull_request.number != 2722 && steps.test_coverage.outputs.windows_runtime_corruption != 'true' + # Temporarily disabled during the Codecov TLS outage; re-enable in a follow-up PR. + if: ${{ false && steps.test_coverage.outputs.windows_runtime_corruption != 'true' }} uses: codecov/codecov-action@v7 with: token: ${{secrets.CODECOV_TOKEN}} @@ -361,8 +361,8 @@ jobs: bash .github/workflows/test_demo.sh - name: Upload dev LTO GlobalDCE coverage reports to Codecov - # Temporarily bypass Codecov's TLS outage for PR #2722 only. - if: github.event.pull_request.number != 2722 + # Temporarily disabled during the Codecov TLS outage; re-enable in a follow-up PR. + if: ${{ false && success() }} continue-on-error: true uses: codecov/codecov-action@v7 with: diff --git a/.github/workflows/llgo.yml b/.github/workflows/llgo.yml index 85ac869e9b..c9c4c6dfb7 100644 --- a/.github/workflows/llgo.yml +++ b/.github/workflows/llgo.yml @@ -220,8 +220,8 @@ jobs: - name: Upload LLGo test coverage to Codecov # Preserve any profile emitted by failing tests, without uploading # unrelated profiles produced by child compiler integration tests. - # Temporarily bypass Codecov's TLS outage for PR #2722 only. - if: ${{ !cancelled() && github.event.pull_request.number != 2722 && hashFiles('coverage-llgo.txt') != '' }} + # Temporarily disabled during the Codecov TLS outage; re-enable in a follow-up PR. + if: ${{ false && !cancelled() && hashFiles('coverage-llgo.txt') != '' }} uses: codecov/codecov-action@v7 with: token: ${{ secrets.CODECOV_TOKEN }} @@ -775,8 +775,8 @@ jobs: LLGO_BROWSER_COVERAGE_DIR: coverage-browser run: python3 dev/test_wasm_browser_debug.py - name: Upload browser debugger coverage to Codecov - # Temporarily bypass Codecov's TLS outage for PR #2722 only. - if: ${{ !cancelled() && github.event.pull_request.number != 2722 && hashFiles('coverage-browser/*.out') != '' }} + # Temporarily disabled during the Codecov TLS outage; re-enable in a follow-up PR. + if: ${{ false && !cancelled() && hashFiles('coverage-browser/*.out') != '' }} uses: codecov/codecov-action@v7 with: token: ${{ secrets.CODECOV_TOKEN }} From 1f91d8ee519848f564c2805f4b77be2e4d9843e4 Mon Sep 17 00:00:00 2001 From: ZhouGuangyuan Date: Mon, 5 Oct 2026 17:59:37 +0800 Subject: [PATCH 19/20] ci: restore Codecov uploads after service recovery --- .github/workflows/doc-link-checker.yml | 6 +----- .github/workflows/go.yml | 5 +---- .github/workflows/llgo.yml | 6 ++---- 3 files changed, 4 insertions(+), 13 deletions(-) diff --git a/.github/workflows/doc-link-checker.yml b/.github/workflows/doc-link-checker.yml index 57c85678f3..6fcbf8b709 100644 --- a/.github/workflows/doc-link-checker.yml +++ b/.github/workflows/doc-link-checker.yml @@ -51,8 +51,4 @@ jobs: id: lychee uses: lycheeverse/lychee-action@v2 with: - # Temporarily exclude Codecov during its TLS outage; restore in a follow-up PR. - args: >- - --max-concurrency 3 --retry-wait-time 15 - --exclude "^https://(app[.])?codecov[.]io/" - README.md + args: --max-concurrency 3 --retry-wait-time 15 README.md diff --git a/.github/workflows/go.yml b/.github/workflows/go.yml index faa34cce19..dc60f2f059 100644 --- a/.github/workflows/go.yml +++ b/.github/workflows/go.yml @@ -234,8 +234,7 @@ jobs: done - name: Upload coverage reports to Codecov - # Temporarily disabled during the Codecov TLS outage; re-enable in a follow-up PR. - if: ${{ false && steps.test_coverage.outputs.windows_runtime_corruption != 'true' }} + if: steps.test_coverage.outputs.windows_runtime_corruption != 'true' uses: codecov/codecov-action@v7 with: token: ${{secrets.CODECOV_TOKEN}} @@ -361,8 +360,6 @@ jobs: bash .github/workflows/test_demo.sh - name: Upload dev LTO GlobalDCE coverage reports to Codecov - # Temporarily disabled during the Codecov TLS outage; re-enable in a follow-up PR. - if: ${{ false && success() }} continue-on-error: true uses: codecov/codecov-action@v7 with: diff --git a/.github/workflows/llgo.yml b/.github/workflows/llgo.yml index c9c4c6dfb7..80e285bc75 100644 --- a/.github/workflows/llgo.yml +++ b/.github/workflows/llgo.yml @@ -220,8 +220,7 @@ jobs: - name: Upload LLGo test coverage to Codecov # Preserve any profile emitted by failing tests, without uploading # unrelated profiles produced by child compiler integration tests. - # Temporarily disabled during the Codecov TLS outage; re-enable in a follow-up PR. - if: ${{ false && !cancelled() && hashFiles('coverage-llgo.txt') != '' }} + if: ${{ !cancelled() && hashFiles('coverage-llgo.txt') != '' }} uses: codecov/codecov-action@v7 with: token: ${{ secrets.CODECOV_TOKEN }} @@ -775,8 +774,7 @@ jobs: LLGO_BROWSER_COVERAGE_DIR: coverage-browser run: python3 dev/test_wasm_browser_debug.py - name: Upload browser debugger coverage to Codecov - # Temporarily disabled during the Codecov TLS outage; re-enable in a follow-up PR. - if: ${{ false && !cancelled() && hashFiles('coverage-browser/*.out') != '' }} + if: ${{ !cancelled() && hashFiles('coverage-browser/*.out') != '' }} uses: codecov/codecov-action@v7 with: token: ${{ secrets.CODECOV_TOKEN }} From 36caa4db4a8a4017a9a36830ad11105235e21f7e Mon Sep 17 00:00:00 2001 From: ZhouGuangyuan Date: Mon, 5 Oct 2026 20:20:23 +0800 Subject: [PATCH 20/20] simd: simplify bridge scanning and conversion checks --- cl/simd.go | 8 ++++++-- internal/build/wasm_simd_calls.go | 33 ++++++++++++++----------------- ssa/simd.go | 6 +----- ssa/simd_integer.go | 1 + 4 files changed, 23 insertions(+), 25 deletions(-) diff --git a/cl/simd.go b/cl/simd.go index 6bea82fea9..63690b47d6 100644 --- a/cl/simd.go +++ b/cl/simd.go @@ -243,8 +243,12 @@ func (d simdOperation) matches(sig *types.Signature, vector types.Type) bool { if !ok || d.signature == simdConvert && shape.Len() < lanes.Len() { return false } - if d.signature == simdConvert && shape.Len() != lanes.Len() && lanes.Elem().Underlying().(*types.Basic).Info()&types.IsFloat == 0 && shape.Elem().Underlying().(*types.Basic).Info()&types.IsFloat == 0 { - return false + if d.signature == simdConvert && shape.Len() != lanes.Len() { + fromFloat := lanes.Elem().Underlying().(*types.Basic).Info()&types.IsFloat != 0 + toFloat := shape.Elem().Underlying().(*types.Basic).Info()&types.IsFloat != 0 + if !fromFloat && !toFloat { + return false + } } case simdUnary: // Receiver only. diff --git a/internal/build/wasm_simd_calls.go b/internal/build/wasm_simd_calls.go index 3f2ff23480..4da6fe5aad 100644 --- a/internal/build/wasm_simd_calls.go +++ b/internal/build/wasm_simd_calls.go @@ -74,40 +74,37 @@ func lowerEmscriptenSIMDCalls(ctx *context, mod llvm.Module) int { return 0 } setjmp := mod.NamedFunction("setjmp") - functions := make(map[llvm.Value]bool) - if !setjmp.IsNil() { - for use := setjmp.FirstUse(); !use.IsNil(); use = use.NextUse() { - call := use.User().IsACallInst() - if !call.IsNil() && call.CalledValue() == setjmp { - functions[call.InstructionParent().Parent()] = true - } - } - } var calls []llvm.Value for fn := mod.FirstFunction(); !fn.IsNil(); fn = llvm.NextFunction(fn) { + var vectorCalls []llvm.Value + hasSetjmp := false for block := fn.FirstBasicBlock(); !block.IsNil(); block = llvm.NextBasicBlock(block) { for inst := block.FirstInstruction(); !inst.IsNil(); inst = llvm.NextInstruction(inst) { if inst.IsACallInst().IsNil() || strings.HasPrefix(inst.CalledValue().Name(), "llvm.") { continue } + if inst.CalledValue() == setjmp { + hasSetjmp = true + } typ := inst.CalledFunctionType() vector := typ.ReturnType().TypeKind() == llvm.VectorTypeKind for i := 0; i < inst.OperandsCount()-1; i++ { vector = vector || inst.Operand(i).Type().TypeKind() == llvm.VectorTypeKind } if vector { - if functions[fn] { - calls = append(calls, inst) - } else { - // A later backend/LTO inliner must not transplant these - // unbridged calls into a function containing setjmp. - // The body has already received the selected optimization. - fn.RemoveEnumFunctionAttribute(llvm.AttributeKindID("alwaysinline")) - fn.AddFunctionAttr(mod.Context().CreateEnumAttribute(llvm.AttributeKindID("noinline"), 0)) - } + vectorCalls = append(vectorCalls, inst) } } } + if hasSetjmp { + calls = append(calls, vectorCalls...) + } else if len(vectorCalls) != 0 { + // A later backend/LTO inliner must not transplant these + // unbridged calls into a function containing setjmp. + // The body has already received the selected optimization. + fn.RemoveEnumFunctionAttribute(llvm.AttributeKindID("alwaysinline")) + fn.AddFunctionAttr(mod.Context().CreateEnumAttribute(llvm.AttributeKindID("noinline"), 0)) + } } for i, call := range calls { bridgeEmscriptenSIMDCall(mod, call, i) diff --git a/ssa/simd.go b/ssa/simd.go index 71bf7cd9aa..23d25caa62 100644 --- a/ssa/simd.go +++ b/ssa/simd.go @@ -478,11 +478,7 @@ func (b Builder) simdRoundEven(x Expr) Expr { integer := b.Prog.ctx.IntType(width) vector := llvm.VectorType(integer, int(lanes.Len())) constant := func(value uint64) llvm.Value { - values := make([]llvm.Value, lanes.Len()) - for i := range values { - values[i] = llvm.ConstInt(integer, value, false) - } - return llvm.ConstVector(values, false) + return simdIntegerConstant(vector, value) } bits := b.impl.CreateBitCast(x.impl, vector, "") sign := b.impl.CreateAnd(bits, constant(uint64(1)<<(width-1)), "") diff --git a/ssa/simd_integer.go b/ssa/simd_integer.go index 447fa0d6fa..1f39160b4f 100644 --- a/ssa/simd_integer.go +++ b/ssa/simd_integer.go @@ -45,6 +45,7 @@ func (b Builder) simdShiftValue(x Expr, count llvm.Value, right bool) llvm.Value } unsigned := simdLanes(x.RawType()).Elem().Underlying().(*types.Basic).Info()&types.IsUnsigned != 0 if right && !unsigned { + // Clamping to width-1 already sign-extends oversized arithmetic shifts. return b.impl.CreateAShr(x.impl, safe, "") } var shifted llvm.Value