From 67e87c4607c79f9204c8e8cafd43418417b724ac Mon Sep 17 00:00:00 2001 From: ZR74 <2401889661@qq.com> Date: Thu, 6 Aug 2026 22:18:51 +0800 Subject: [PATCH 1/2] perf(evm): batch runtime stack boundary access Batch non-lifted block entry loads, depth drops, and exit stores so each boundary updates runtime stack metadata once. Keep full stack SSA unchanged and cover both SSA-on residual paths and SSA-off execution with frontend and differential tests. --- CMakeLists.txt | 4 + .../README.md | 134 ++++++++++++++++++ docs/changes/README.md | 1 + docs/modules/compiler/spec.md | 13 ++ src/CMakeLists.txt | 4 + src/action/evm_bytecode_visitor.h | 22 ++- .../evm_frontend/evm_mir_compiler.cpp | 98 +++++++++++++ src/compiler/evm_frontend/evm_mir_compiler.h | 3 + src/tests/evm_differential_tests.cpp | 61 ++++++++ src/tests/evm_jit_frontend_tests.cpp | 125 ++++++++++++++++ 10 files changed, 464 insertions(+), 1 deletion(-) create mode 100644 docs/changes/2026-07-29-evm-stack-boundary-batch/README.md diff --git a/CMakeLists.txt b/CMakeLists.txt index e514d9b9f..2382a336f 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -52,6 +52,10 @@ option(ZEN_ENABLE_EVM_GAS_REGISTER option(ZEN_ENABLE_EVM_STACK_SSA_LIFT "Enable conservative EVM stack lifting in multipass frontend" OFF ) +option( + ZEN_ENABLE_EVM_STACK_BOUNDARY_BATCH + "Enable batched runtime-stack loads/stores at EVM basic-block boundaries" OFF +) option(ZEN_ENABLE_EVM_MEM_LARGE_STATIC_WORKSPACE_LOWERING "Enable experimental EVM large static workspace memory lowering" OFF ) diff --git a/docs/changes/2026-07-29-evm-stack-boundary-batch/README.md b/docs/changes/2026-07-29-evm-stack-boundary-batch/README.md new file mode 100644 index 000000000..4d135a4ae --- /dev/null +++ b/docs/changes/2026-07-29-evm-stack-boundary-batch/README.md @@ -0,0 +1,134 @@ +# Change: Batch EVM runtime-stack boundary access + +- **Status**: Validated +- **Date**: 2026-07-29 +- **Tier**: Full + +## Overview + +Add an opt-in first stage of selective stack dematerialization for the EVM +multipass frontend. Non-lifted block entries load their required live-in values +with one batch address calculation and one depth update. Non-lifted exits store +their bottom-to-top logical stack with one top/size update. + +The change is gated by `ZEN_ENABLE_EVM_STACK_BOUNDARY_BATCH`, which is `OFF` by +default. Full operand-stack SSA remains unchanged and takes priority. + +## Contract + +`EVMMirBuilder` exposes three frontend-internal operations: + +- `peekStackBatch(count, skipTop)` returns operands in bottom-to-top order and + leaves runtime depth unchanged; +- `dropStackBatch(count)` updates runtime top and size once; +- `pushStackBatch(values)` stores bottom-to-top values and updates runtime top + and size once. + +Empty batches are strict no-ops. The operations do not change `EVMInstance`, +EVMC ABI, gas accounting, exception behavior, or dynamic-jump dispatch. + +## Validation + +Build and frontend test commands: + +```bash +cmake -S . -B build-batch \ + -DCMAKE_BUILD_TYPE=Release \ + -DZEN_ENABLE_EVM=ON \ + -DZEN_ENABLE_MULTIPASS_JIT=ON \ + -DZEN_ENABLE_SINGLEPASS_JIT=OFF \ + -DZEN_ENABLE_SPEC_TEST=ON \ + -DZEN_ENABLE_JIT_FALLBACK_TEST=ON \ + -DZEN_ENABLE_EVM_STACK_SSA_LIFT=ON \ + -DZEN_ENABLE_EVM_MEMORY_PLAN_FRAMEWORK=ON \ + -DZEN_ENABLE_EVM_STACK_BOUNDARY_BATCH=ON \ + -DLLVM_DIR=/path/to/llvm-15/lib/cmake/llvm +cmake --build build-batch --target evmJitFrontendTests evmDifferentialTests \ + evmFallbackExecutionTests evmStateTests +./build-batch/evmJitFrontendTests +./build-batch/evmDifferentialTests +./build-batch/evmFallbackExecutionTests +``` + +The final validation used two Release builds from the same source revision: + +- A: V111 with boundary batching disabled; +- B: V111 with boundary batching enabled. + +The change was also rebuilt as V011 (SSA disabled) in both A/B modes. Results: + +| Gate | Result | +| --- | --- | +| V011 frontend | A/B common suite: 123/123; B-only batch tests: 2/2 | +| V111 frontend | A/B common suite: 123/123; B-only batch tests: 2/2 | +| V011 differential | A/B: 73/73 | +| V111 differential | A/B: 73/73 | +| state tests | B: 1798/1798 | +| forced JIT-to-interpreter fallback | B: 8/8 | +| Osaka transaction-exact replay | B: 100/100, 29 code hashes | + +The exact replay used the fixture-aware state-test host, which restores complete +prestate before each transaction. Its result is +`/tmp/dtvm-pr1-b-correctness-100.json` on the measurement host, SHA-256 +`8c2e8bbcd6b68010f49800f963ee54dfef23ae6685e8604a73f1abbc40c50c3d`. + +The synthetic 16-slot, 8-boundary stress case reduced MIR instructions from +5147 to 3131 (-39.2%) and CgIR instructions from 4128 to 2614 (-36.7%). +Production compiler observation over the 29 real code hashes recorded 7103 +batch loads/drops for 22004 slots and 7383 batch stores for 26359 slots. Top +and size were each updated 14486 times, exactly once per drop or store batch. +These coverage counters were collected with the optional compiler-observation +instrumentation from PR #587; that instrumentation is intentionally not part +of this change and is not required by the optimization. + +Formal Osaka performance used CPU 24, 29 unique code hashes, 12 fresh-process +cold rounds, three warmups per variant, and 100 ms hot calibration. S0 denotes +A and S1 denotes B in the result file: + +| Metric | B relative to A | 95% bootstrap CI | +| --- | ---: | ---: | +| JIT compilation, geometric mean | -2.645% | [-3.202%, -2.098%] | +| Emitted code, geometric mean | -0.168% | [-0.216%, -0.122%] | +| Hot execution, geometric mean | -0.142% | [-0.375%, +0.082%] | +| Frequency-weighted hot execution | -0.243% | [-0.626%, +0.087%] | + +The result is `/tmp/dtvm-pr1-ab-formal-29x12.json`, SHA-256 +`c9c71d8c9ccf7b594ce9bcf28d64dacf0cc4b52f2a080055dccf80c87c6f5f6d`. +The compiler-observation result is +`/tmp/dtvm-pr1-ab-observation-29.json`, SHA-256 +`f38d5d865554ea829de3fb3a00df4d705a906d786fe859bbb0e826bed0dd834e`. +Both compile/code-size and hot-execution regression gates pass. + +Reproduce exact correctness and paired performance with the existing +fixture-aware replay tools: + +```bash +python3 tools/run_replay_correctness.py \ + --executable build-b-v111/evmStateTests \ + --fixture-manifest /path/to/fixtures-100-v2-manifest.json \ + --fixture-dir /path/to/fixtures-100-v2 \ + --output /tmp/dtvm-pr1-b-correctness-100.json \ + --raw-dir /tmp/dtvm-pr1-b-correctness-100-raw \ + --mode multipass --revision Osaka --cpu 24 + +python3 tools/run_ssa_replay_exact_performance.py \ + --s0 build-a-v111/evmStateTests \ + --s1 build-b-v111/evmStateTests \ + --performance-set /path/to/performance-set/manifest.json \ + --fixture-dir /path/to/29-fixtures \ + --output /tmp/dtvm-pr1-ab-formal-29x12.json \ + --raw-output-dir /tmp/dtvm-pr1-ab-formal-29x12-raw \ + --revision Osaka --cpu 24 --rounds 12 \ + --warmup-iterations 3 --target-hot-ms 100 \ + --bootstrap-samples 10000 --seed 20260729 +``` + +## Risks + +- A wrong base offset can reverse U256 or stack-slot order. Batch tests cover + empty, 1-, 16-, and 17-slot shapes; differential and exact replay remain + mandatory. +- Batching reduces address and top/size maintenance but not the four limb + loads/stores per U256 value. +- Observation counters are compile-time structural evidence and are not a + substitute for hot-execution measurement. diff --git a/docs/changes/README.md b/docs/changes/README.md index e040f04cb..9032bfe3a 100644 --- a/docs/changes/README.md +++ b/docs/changes/README.md @@ -54,6 +54,7 @@ Typical triggers: | 2026-05-13 | [evm-ngram-macro-ops](2026-05-13-evm-ngram-macro-ops/README.md) | Implemented | Full | Initial EVM n-gram macro-op lowering and specialized keccak helpers for multipass JIT | | 2026-07-21 | [evm-memory-alias-and-expansion](2026-07-21-evm-memory-alias-and-expansion/README.md) | Implemented | Full | Stronger memory alias proofs, wider precheck/expansion coverage, DSE, load forwarding, grouping, and MCOPY roadmap | | 2026-07-28 | [ssa-shared-dynamic-dispatch](2026-07-28-ssa-shared-dynamic-dispatch/README.md) | Implemented | Light | Share unfiltered full-table dynamic dispatch in stack-SSA builds | +| 2026-07-29 | [evm-stack-boundary-batch](2026-07-29-evm-stack-boundary-batch/README.md) | Validated | Full | Batch runtime-stack loads, drops, and stores at non-lifted EVM block boundaries | Each active proposal lives in its own subdirectory. Browse `docs/changes/*/README.md` to see all current proposals, or use: diff --git a/docs/modules/compiler/spec.md b/docs/modules/compiler/spec.md index 9d52f9dcf..6682b6eae 100644 --- a/docs/modules/compiler/spec.md +++ b/docs/modules/compiler/spec.md @@ -197,6 +197,19 @@ and destination ends for expansion elision while retaining the original - Provide bytecode, gas chunk end/cost arrays for chunk-based metering - Use register to hold gas when `ZEN_ENABLE_EVM_GAS_REGISTER` is enabled +### EVM Stack Boundary Batching + +`ZEN_ENABLE_EVM_STACK_BOUNDARY_BATCH` is an opt-in, default-off lowering for +non-lifted EVM block boundaries. Entry loads use `peekStackBatch` followed by +one `dropStackBatch`; exit materialization uses one `pushStackBatch`. Batch +vectors are always bottom-to-top. Empty batches emit no MIR and do not update +runtime stack state. Full stack-SSA blocks retain the existing entry-state and +phi protocol. + +The runtime stack remains the complete authoritative representation in this +stage. Every non-lifted exit still writes all logical values, and dynamic +dispatch, fallback, gas, and runtime ABI behavior are unchanged. + ### EVM Stack SSA Lift Safety - `ZEN_ENABLE_EVM_STACK_SSA_LIFT` permits compatible EVM operand-stack values diff --git a/src/CMakeLists.txt b/src/CMakeLists.txt index 9ade28423..7b5f886dc 100644 --- a/src/CMakeLists.txt +++ b/src/CMakeLists.txt @@ -77,6 +77,10 @@ if(ZEN_ENABLE_EVM_STACK_SSA_LIFT) add_definitions(-DZEN_ENABLE_EVM_STACK_SSA_LIFT) endif() +if(ZEN_ENABLE_EVM_STACK_BOUNDARY_BATCH) + add_definitions(-DZEN_ENABLE_EVM_STACK_BOUNDARY_BATCH) +endif() + if(ZEN_ENABLE_EVM AND ZEN_ENABLE_EVM_MEM_LARGE_STATIC_WORKSPACE_LOWERING) add_definitions(-DZEN_ENABLE_EVM_MEM_LARGE_STATIC_WORKSPACE_LOWERING) endif() diff --git a/src/action/evm_bytecode_visitor.h b/src/action/evm_bytecode_visitor.h index 2617b9548..ec2f36828 100644 --- a/src/action/evm_bytecode_visitor.h +++ b/src/action/evm_bytecode_visitor.h @@ -1037,9 +1037,13 @@ template class EVMByteCodeVisitor { // missed underflow traps in later blocks. spillTrackedStackPreservingPrefix(Values, /*PrefixDepth=*/0); } else { +#ifdef ZEN_ENABLE_EVM_STACK_BOUNDARY_BATCH + Builder.pushStackBatch(Values); +#else for (const Operand &Opnd : Values) { Builder.stackPush(Opnd); } +#endif } } InDeadCode = true; @@ -1290,13 +1294,28 @@ template class EVMByteCodeVisitor { CurrentBlockLifted = false; int32_t TotalPopSize = -BlockInfo.MinPopHeight; - EvalStack ReverseStack; // Refine each popped Operand's ValueRange from analyzer-computed entry // ranges so u64-narrow fast paths fire on values flowing through CFG // joins (see EVMRangeAnalyzer / // docs/changes/2026-05-07-value-range-cfg-join). EntryStackRanges[0] is the // bottom of entry stack; pop order is top-first. const auto &EntryRanges = BlockInfo.EntryStackRanges; +#ifdef ZEN_ENABLE_EVM_STACK_BOUNDARY_BATCH + std::vector EntryValues = + Builder.peekStackBatch(static_cast(TotalPopSize)); + Builder.dropStackBatch(static_cast(TotalPopSize)); + const int32_t FirstEntrySlot = + static_cast(EntryRanges.size()) - TotalPopSize; + for (int32_t Index = 0; Index < TotalPopSize; ++Index) { + Operand &Opnd = EntryValues[static_cast(Index)]; + const int32_t SlotIdx = FirstEntrySlot + Index; + if (SlotIdx >= 0 && SlotIdx < static_cast(EntryRanges.size())) { + Opnd.setRange(EntryRanges[static_cast(SlotIdx)]); + } + Stack.push(Opnd); + } +#else + EvalStack ReverseStack; const int32_t EntryTopIdx = static_cast(EntryRanges.size()) - 1; int32_t PopIter = 0; while (TotalPopSize > 0) { @@ -1313,6 +1332,7 @@ template class EVMByteCodeVisitor { Operand Opnd = ReverseStack.pop(); Stack.push(Opnd); } +#endif } void diff --git a/src/compiler/evm_frontend/evm_mir_compiler.cpp b/src/compiler/evm_frontend/evm_mir_compiler.cpp index 8188f2c2d..06c860dfd 100644 --- a/src/compiler/evm_frontend/evm_mir_compiler.cpp +++ b/src/compiler/evm_frontend/evm_mir_compiler.cpp @@ -1108,6 +1108,104 @@ typename EVMMirBuilder::Operand EVMMirBuilder::stackPop() { return Operand(PopComponents, EVMType::UINT256); } +std::vector +EVMMirBuilder::peekStackBatch(uint32_t Count, uint32_t SkipTop) { + std::vector Values; + if (Count == 0) { + return Values; + } + + const uint64_t AddressedSlots = + static_cast(Count) + static_cast(SkipTop); + ZEN_ASSERT(AddressedSlots <= 1024 && + "runtime stack batch peek exceeds the EVM stack"); + + MType *I64Type = EVMFrontendContext::getMIRTypeFromEVMType(EVMType::UINT64); + MPointerType *U64PtrType = MPointerType::create(Ctx, Ctx.I64Type); + MInstruction *StackTopInt = getInstanceStackTopInt(); + MInstruction *BaseOffset = + createIntConstInstruction(I64Type, AddressedSlots * 32ULL); + MInstruction *BatchBase = createInstruction( + false, OP_sub, I64Type, StackTopInt, BaseOffset); + MInstruction *BatchPtr = createInstruction( + false, OP_inttoptr, U64PtrType, BatchBase); + + Values.reserve(Count); + for (uint32_t Slot = 0; Slot < Count; ++Slot) { + U256Inst Components = {}; + for (size_t Limb = 0; Limb < EVM_ELEMENTS_COUNT; ++Limb) { + const int32_t Offset = + static_cast(static_cast(Slot) * 32ULL + + static_cast(Limb) * 8ULL); + MInstruction *LoadInstr = createInstruction( + false, I64Type, BatchPtr, 1, nullptr, Offset); + Variable *ValVar = storeInstructionInTemp(LoadInstr, I64Type); + Components[Limb] = loadVariable(ValVar); + } + Values.emplace_back(Components, EVMType::UINT256); + } + + return Values; +} + +void EVMMirBuilder::dropStackBatch(uint32_t Count) { + if (Count == 0) { + return; + } + ZEN_ASSERT(Count <= 1024 && "runtime stack batch drop exceeds the EVM stack"); + + MType *I64Type = EVMFrontendContext::getMIRTypeFromEVMType(EVMType::UINT64); + MInstruction *StackBytes = + createIntConstInstruction(I64Type, static_cast(Count) * 32ULL); + MInstruction *StackTopInt = getInstanceStackTopInt(); + MInstruction *StackSize = loadVariable(StackSizeVar); + MInstruction *NewTop = createInstruction( + false, OP_sub, I64Type, StackTopInt, StackBytes); + MInstruction *NewSize = createInstruction( + false, OP_sub, I64Type, StackSize, StackBytes); + createInstruction(true, &(Ctx.VoidType), NewTop, + StackTopVar->getVarIdx()); + createInstruction(true, &(Ctx.VoidType), NewSize, + StackSizeVar->getVarIdx()); + +} + +void EVMMirBuilder::pushStackBatch(const std::vector &Values) { + if (Values.empty()) { + return; + } + ZEN_ASSERT(Values.size() <= 1024 && + "runtime stack batch push exceeds the EVM stack"); + + MType *I64Type = EVMFrontendContext::getMIRTypeFromEVMType(EVMType::UINT64); + MPointerType *U64PtrType = MPointerType::create(Ctx, Ctx.I64Type); + MInstruction *StackTopInt = getInstanceStackTopInt(); + MInstruction *StackTopPtr = createInstruction( + false, OP_inttoptr, U64PtrType, StackTopInt); + + for (size_t Slot = 0; Slot < Values.size(); ++Slot) { + U256Inst Components = extractU256Operand(Values[Slot]); + for (size_t Limb = 0; Limb < EVM_ELEMENTS_COUNT; ++Limb) { + const int32_t Offset = static_cast(Slot * 32ULL + Limb * 8ULL); + createInstruction(true, &Ctx.VoidType, Components[Limb], + StackTopPtr, Offset); + } + } + + MInstruction *StackSize = loadVariable(StackSizeVar); + MInstruction *StackBytes = createIntConstInstruction( + I64Type, static_cast(Values.size()) * 32ULL); + MInstruction *NewTop = createInstruction( + false, OP_add, I64Type, StackTopInt, StackBytes); + MInstruction *NewSize = createInstruction( + false, OP_add, I64Type, StackSize, StackBytes); + createInstruction(true, &(Ctx.VoidType), NewTop, + StackTopVar->getVarIdx()); + createInstruction(true, &(Ctx.VoidType), NewSize, + StackSizeVar->getVarIdx()); + +} + void EVMMirBuilder::stackSet(int32_t IndexFromTop, Operand SetValue) { // This set element to stack with index from top U256Inst SetComponents = extractU256Operand(SetValue); diff --git a/src/compiler/evm_frontend/evm_mir_compiler.h b/src/compiler/evm_frontend/evm_mir_compiler.h index c6b51f0f7..21f5f5274 100644 --- a/src/compiler/evm_frontend/evm_mir_compiler.h +++ b/src/compiler/evm_frontend/evm_mir_compiler.h @@ -351,6 +351,9 @@ class EVMMirBuilder final { void stackPush(Operand PushValue); Operand stackPop(); + std::vector peekStackBatch(uint32_t Count, uint32_t SkipTop = 0); + void dropStackBatch(uint32_t Count); + void pushStackBatch(const std::vector &Values); void stackSet(int32_t IndexFromTop, Operand SetValue); Operand stackGet(int32_t IndexFromTop); diff --git a/src/tests/evm_differential_tests.cpp b/src/tests/evm_differential_tests.cpp index 4d7156b6f..3c35c398b 100644 --- a/src/tests/evm_differential_tests.cpp +++ b/src/tests/evm_differential_tests.cpp @@ -822,6 +822,67 @@ TEST(EVMRangeDifferential, DeadUnderResolvedFallthroughMatchesInterpreter) { Bytecode, {})); } +TEST(EVMStackBoundaryDifferential, + SixteenSlotForwardBoundaryStressMatchesInterpreter) { + std::vector Bytecode; + auto AppendPush2 = [&](uint16_t Value) { + Bytecode.push_back(0x61); + Bytecode.push_back(static_cast(Value >> 8)); + Bytecode.push_back(static_cast(Value)); + }; + + constexpr uint32_t Slots = 16; + constexpr uint32_t Boundaries = 8; + for (uint32_t Slot = 0; Slot < Slots; ++Slot) { + AppendPush2(static_cast(Slot + 1)); + } + for (uint32_t Boundary = 0; Boundary < Boundaries; ++Boundary) { + const uint16_t TargetPC = static_cast(Bytecode.size() + 4); + AppendPush2(TargetPC); + Bytecode.push_back(0x56); // JUMP + Bytecode.push_back(0x5b); // JUMPDEST + Bytecode.insert(Bytecode.end(), Slots, 0x50); // POP x16 + for (uint32_t Slot = 0; Slot < Slots; ++Slot) { + AppendPush2(static_cast((Boundary + 1) * Slots + Slot + 1)); + } + } + Bytecode.push_back(0x00); // STOP + + EXPECT_TRUE(expectInterpMatchesMultipass( + "sixteen_slot_forward_boundary_stress", Bytecode, {})); +} + +TEST(EVMStackBoundaryDifferential, + PopDup16AndSwap16AcrossBoundaryMatchInterpreter) { + std::vector Bytecode; + for (uint8_t Value = 1; Value <= 17; ++Value) { + Bytecode.push_back(0x60); // PUSH1 + Bytecode.push_back(Value); + } + Bytecode.push_back(0x60); // PUSH1 target + Bytecode.push_back(static_cast(Bytecode.size() + 2)); + Bytecode.push_back(0x56); // JUMP + Bytecode.push_back(0x5b); // JUMPDEST + Bytecode.push_back(0x8f); // DUP16 + Bytecode.push_back(0x9f); // SWAP16 + Bytecode.push_back(0x50); // POP + Bytecode.insert(Bytecode.end(), + {0x5f, 0x52, 0x60, 0x20, 0x5f, 0xf3}); // return top word + + EXPECT_TRUE(expectInterpMatchesMultipass("pop_dup16_swap16_across_boundary", + Bytecode, {})); +} + +TEST(EVMStackBoundaryDifferential, UnderflowAndOverflowMatchInterpreter) { + EXPECT_TRUE(expectInterpStatusMatchesMultipass( + "batch_boundary_underflow", {0x50, 0x00}, EVMC_STACK_UNDERFLOW)); + + std::vector Overflow(1025, 0x5f); // PUSH0 x1025 + Overflow.push_back(0x00); + EXPECT_TRUE(expectInterpStatusMatchesMultipass( + "batch_boundary_overflow", Overflow, EVMC_STACK_OVERFLOW)); +} + TEST(EVMLiftedStackMerge, JumpiFallthroughSharedJumpDestMatchesInterpreterOnBothEdges) { const std::vector Bytecode = { diff --git a/src/tests/evm_jit_frontend_tests.cpp b/src/tests/evm_jit_frontend_tests.cpp index bf0b81aa0..b00c35b1b 100644 --- a/src/tests/evm_jit_frontend_tests.cpp +++ b/src/tests/evm_jit_frontend_tests.cpp @@ -2332,6 +2332,17 @@ struct MockStackAccessStats { uint32_t StackSetCount = 0; }; +struct MockBatchStackAccessStats { + uint32_t PeekCalls = 0; + uint32_t PeekSlots = 0; + uint32_t DropCalls = 0; + uint32_t DropSlots = 0; + uint32_t PushCalls = 0; + uint32_t PushSlots = 0; + uint32_t StackTopUpdates = 0; + uint32_t StackSizeUpdates = 0; +}; + struct MockMeterOpcodeRangeRecord { uint64_t StartPC = 0; uint64_t EndPCExclusive = 0; @@ -2443,6 +2454,44 @@ class MockEVMBuilder { return Top; } + std::vector peekStackBatch(uint32_t Count, uint32_t SkipTop = 0) { + if (Count == 0) { + return {}; + } + ZEN_ASSERT(static_cast(Count) + SkipTop <= RuntimeStack.size() && + "mock runtime stack batch peek underflow"); + BatchStats.PeekCalls++; + BatchStats.PeekSlots += Count; + const size_t Begin = + RuntimeStack.size() - static_cast(SkipTop) - Count; + return std::vector(RuntimeStack.begin() + Begin, + RuntimeStack.begin() + Begin + Count); + } + + void dropStackBatch(uint32_t Count) { + if (Count == 0) { + return; + } + ZEN_ASSERT(Count <= RuntimeStack.size() && + "mock runtime stack batch drop underflow"); + RuntimeStack.resize(RuntimeStack.size() - Count); + BatchStats.DropCalls++; + BatchStats.DropSlots += Count; + BatchStats.StackTopUpdates++; + BatchStats.StackSizeUpdates++; + } + + void pushStackBatch(const std::vector &Values) { + if (Values.empty()) { + return; + } + RuntimeStack.insert(RuntimeStack.end(), Values.begin(), Values.end()); + BatchStats.PushCalls++; + BatchStats.PushSlots += Values.size(); + BatchStats.StackTopUpdates++; + BatchStats.StackSizeUpdates++; + } + void stackSet(int32_t IndexFromTop, Operand SetValue) { Stats[CurrentOpcode].StackSetCount++; size_t Index = RuntimeStack.size() - static_cast(IndexFromTop) - 1; @@ -2696,6 +2745,10 @@ class MockEVMBuilder { return Stats[static_cast(Opcode)]; } + const MockBatchStackAccessStats &batchAccessStats() const { + return BatchStats; + } + uint32_t meteredOpcodeCount(evmc_opcode Opcode) const { return MeteredOpcodeCounts[static_cast(Opcode)]; } @@ -2760,6 +2813,12 @@ class MockEVMBuilder { size_t runtimeStackDepth() const { return RuntimeStack.size(); } + MockOperand::U256Value runtimeStackValueFromBottom(size_t Index) const { + ZEN_ASSERT(Index < RuntimeStack.size() && + "mock runtime stack index is out of range"); + return RuntimeStack[Index].resolvedValue(); + } + MockOperand::U256Value topStackValue() const { ZEN_ASSERT(!RuntimeStack.empty() && "mock runtime stack is empty"); return RuntimeStack.back().resolvedValue(); @@ -2816,6 +2875,7 @@ class MockEVMBuilder { bool EnableRuntimeStackChecks = false; uint8_t CurrentOpcode = 0xff; std::array Stats = {}; + MockBatchStackAccessStats BatchStats = {}; std::array MeteredOpcodeCounts = {}; std::vector MeteredRanges; std::vector HelperOpcodes; @@ -2983,6 +3043,71 @@ bool compileWithMockBuilder(const std::vector &Bytecode, return Visitor.compile(); } +#ifdef ZEN_ENABLE_EVM_STACK_BOUNDARY_BATCH +TEST(EVMMirBuilderStackBoundaryBatchTest, + EmptyOneSixteenAndSeventeenSlotBatchesBuild) { + MirBuilderConstFoldHarness Harness; + using Operand = EVMMirBuilder::Operand; + + const size_t InitialStatements = + Harness.Func.getBasicBlock(0)->getNumStatements(); + EXPECT_TRUE(Harness.Builder.peekStackBatch(0).empty()); + Harness.Builder.dropStackBatch(0); + Harness.Builder.pushStackBatch({}); + EXPECT_EQ(Harness.Func.getBasicBlock(0)->getNumStatements(), + InitialStatements); + + for (uint32_t Count : {1u, 16u, 17u}) { + std::vector Values; + Values.reserve(Count); + for (uint32_t Index = 0; Index < Count; ++Index) { + Values.emplace_back( + EVMMirBuilder::U256Value{Index, Index + 1, Index + 2, Index + 3}); + } + Harness.Builder.pushStackBatch(Values); + EXPECT_EQ(Harness.Builder.peekStackBatch(Count).size(), Count); + Harness.Builder.dropStackBatch(Count); + } +} + +TEST(EVMJITFrontendVisitorTest, + BatchProtocolPreservesBottomToTopOrderAndSkipTop) { + MockEVMBuilder Builder; + for (uint64_t Value = 1; Value <= 17; ++Value) { + Builder.stackPush(MockOperand(Value)); + } + + std::vector Bottom = + Builder.peekStackBatch(/*Count=*/1, /*SkipTop=*/16); + ASSERT_EQ(Bottom.size(), 1u); + EXPECT_EQ(Bottom[0].resolvedValue()[0], 1u); + + std::vector All = Builder.peekStackBatch(17); + ASSERT_EQ(All.size(), 17u); + for (uint64_t Index = 0; Index < All.size(); ++Index) { + EXPECT_EQ(All[Index].resolvedValue()[0], Index + 1); + } + Builder.dropStackBatch(17); + EXPECT_EQ(Builder.runtimeStackDepth(), 0u); + + Builder.pushStackBatch(All); + ASSERT_EQ(Builder.runtimeStackDepth(), 17u); + for (uint64_t Index = 0; Index < All.size(); ++Index) { + EXPECT_EQ(Builder.runtimeStackValueFromBottom(Index)[0], Index + 1); + } + + const MockBatchStackAccessStats &Stats = Builder.batchAccessStats(); + EXPECT_EQ(Stats.PeekCalls, 2u); + EXPECT_EQ(Stats.PeekSlots, 18u); + EXPECT_EQ(Stats.DropCalls, 1u); + EXPECT_EQ(Stats.DropSlots, 17u); + EXPECT_EQ(Stats.PushCalls, 1u); + EXPECT_EQ(Stats.PushSlots, 17u); + EXPECT_EQ(Stats.StackTopUpdates, 2u); + EXPECT_EQ(Stats.StackSizeUpdates, 2u); +} +#endif + TEST(EVMJITFrontendVisitorTest, TerminatingMemoryHelpersRetainExactOpcodePC) { const std::vector ReturnBytecode = {OP_PUSH0, OP_PUSH0, OP_RETURN}; const std::vector RevertBytecode = {OP_PUSH0, OP_PUSH0, OP_REVERT}; From 7a032dd03149266b7013a6e81a23d2ba82244a19 Mon Sep 17 00:00:00 2001 From: ZR74 <2401889661@qq.com> Date: Thu, 6 Aug 2026 14:58:07 +0000 Subject: [PATCH 2/2] refactor(evm): make stack boundary batching unconditional Use batched runtime-stack access for every non-lifted boundary and remove the experimental build gate and scalar fallback. Full stack SSA continues to take priority. --- CMakeLists.txt | 4 --- .../README.md | 15 +++++------ docs/modules/compiler/spec.md | 11 ++++---- src/CMakeLists.txt | 4 --- src/action/evm_bytecode_visitor.h | 26 ------------------- .../evm_frontend/evm_mir_compiler.cpp | 2 -- src/tests/evm_jit_frontend_tests.cpp | 2 -- 7 files changed, 12 insertions(+), 52 deletions(-) diff --git a/CMakeLists.txt b/CMakeLists.txt index 2382a336f..e514d9b9f 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -52,10 +52,6 @@ option(ZEN_ENABLE_EVM_GAS_REGISTER option(ZEN_ENABLE_EVM_STACK_SSA_LIFT "Enable conservative EVM stack lifting in multipass frontend" OFF ) -option( - ZEN_ENABLE_EVM_STACK_BOUNDARY_BATCH - "Enable batched runtime-stack loads/stores at EVM basic-block boundaries" OFF -) option(ZEN_ENABLE_EVM_MEM_LARGE_STATIC_WORKSPACE_LOWERING "Enable experimental EVM large static workspace memory lowering" OFF ) diff --git a/docs/changes/2026-07-29-evm-stack-boundary-batch/README.md b/docs/changes/2026-07-29-evm-stack-boundary-batch/README.md index 4d135a4ae..33b368d7c 100644 --- a/docs/changes/2026-07-29-evm-stack-boundary-batch/README.md +++ b/docs/changes/2026-07-29-evm-stack-boundary-batch/README.md @@ -6,13 +6,14 @@ ## Overview -Add an opt-in first stage of selective stack dematerialization for the EVM +Add a first stage of selective stack dematerialization for the EVM multipass frontend. Non-lifted block entries load their required live-in values with one batch address calculation and one depth update. Non-lifted exits store their bottom-to-top logical stack with one top/size update. -The change is gated by `ZEN_ENABLE_EVM_STACK_BOUNDARY_BATCH`, which is `OFF` by -default. Full operand-stack SSA remains unchanged and takes priority. +Batching is the standard runtime-stack boundary lowering. Full operand-stack +SSA remains unchanged and takes priority, so batching only handles non-lifted +blocks in SSA builds and all block boundaries when SSA is disabled. ## Contract @@ -38,19 +39,17 @@ cmake -S . -B build-batch \ -DZEN_ENABLE_MULTIPASS_JIT=ON \ -DZEN_ENABLE_SINGLEPASS_JIT=OFF \ -DZEN_ENABLE_SPEC_TEST=ON \ - -DZEN_ENABLE_JIT_FALLBACK_TEST=ON \ -DZEN_ENABLE_EVM_STACK_SSA_LIFT=ON \ -DZEN_ENABLE_EVM_MEMORY_PLAN_FRAMEWORK=ON \ - -DZEN_ENABLE_EVM_STACK_BOUNDARY_BATCH=ON \ -DLLVM_DIR=/path/to/llvm-15/lib/cmake/llvm cmake --build build-batch --target evmJitFrontendTests evmDifferentialTests \ - evmFallbackExecutionTests evmStateTests + evmStateTests ./build-batch/evmJitFrontendTests ./build-batch/evmDifferentialTests -./build-batch/evmFallbackExecutionTests ``` -The final validation used two Release builds from the same source revision: +Before the experimental gate was removed, validation used two Release builds +from the same source revision: - A: V111 with boundary batching disabled; - B: V111 with boundary batching enabled. diff --git a/docs/modules/compiler/spec.md b/docs/modules/compiler/spec.md index 6682b6eae..6e09f7dfe 100644 --- a/docs/modules/compiler/spec.md +++ b/docs/modules/compiler/spec.md @@ -199,12 +199,11 @@ and destination ends for expansion elision while retaining the original ### EVM Stack Boundary Batching -`ZEN_ENABLE_EVM_STACK_BOUNDARY_BATCH` is an opt-in, default-off lowering for -non-lifted EVM block boundaries. Entry loads use `peekStackBatch` followed by -one `dropStackBatch`; exit materialization uses one `pushStackBatch`. Batch -vectors are always bottom-to-top. Empty batches emit no MIR and do not update -runtime stack state. Full stack-SSA blocks retain the existing entry-state and -phi protocol. +Non-lifted EVM block boundaries use batched runtime-stack access. Entry loads +use `peekStackBatch` followed by one `dropStackBatch`; exit materialization uses +one `pushStackBatch`. Batch vectors are always bottom-to-top. Empty batches emit +no MIR and do not update runtime stack state. Full stack-SSA blocks retain the +existing entry-state and phi protocol. The runtime stack remains the complete authoritative representation in this stage. Every non-lifted exit still writes all logical values, and dynamic diff --git a/src/CMakeLists.txt b/src/CMakeLists.txt index 7b5f886dc..9ade28423 100644 --- a/src/CMakeLists.txt +++ b/src/CMakeLists.txt @@ -77,10 +77,6 @@ if(ZEN_ENABLE_EVM_STACK_SSA_LIFT) add_definitions(-DZEN_ENABLE_EVM_STACK_SSA_LIFT) endif() -if(ZEN_ENABLE_EVM_STACK_BOUNDARY_BATCH) - add_definitions(-DZEN_ENABLE_EVM_STACK_BOUNDARY_BATCH) -endif() - if(ZEN_ENABLE_EVM AND ZEN_ENABLE_EVM_MEM_LARGE_STATIC_WORKSPACE_LOWERING) add_definitions(-DZEN_ENABLE_EVM_MEM_LARGE_STATIC_WORKSPACE_LOWERING) endif() diff --git a/src/action/evm_bytecode_visitor.h b/src/action/evm_bytecode_visitor.h index ec2f36828..585eec43f 100644 --- a/src/action/evm_bytecode_visitor.h +++ b/src/action/evm_bytecode_visitor.h @@ -1037,13 +1037,7 @@ template class EVMByteCodeVisitor { // missed underflow traps in later blocks. spillTrackedStackPreservingPrefix(Values, /*PrefixDepth=*/0); } else { -#ifdef ZEN_ENABLE_EVM_STACK_BOUNDARY_BATCH Builder.pushStackBatch(Values); -#else - for (const Operand &Opnd : Values) { - Builder.stackPush(Opnd); - } -#endif } } InDeadCode = true; @@ -1300,7 +1294,6 @@ template class EVMByteCodeVisitor { // docs/changes/2026-05-07-value-range-cfg-join). EntryStackRanges[0] is the // bottom of entry stack; pop order is top-first. const auto &EntryRanges = BlockInfo.EntryStackRanges; -#ifdef ZEN_ENABLE_EVM_STACK_BOUNDARY_BATCH std::vector EntryValues = Builder.peekStackBatch(static_cast(TotalPopSize)); Builder.dropStackBatch(static_cast(TotalPopSize)); @@ -1314,25 +1307,6 @@ template class EVMByteCodeVisitor { } Stack.push(Opnd); } -#else - EvalStack ReverseStack; - const int32_t EntryTopIdx = static_cast(EntryRanges.size()) - 1; - int32_t PopIter = 0; - while (TotalPopSize > 0) { - Operand Opnd = Builder.stackPop(); - const int32_t SlotIdx = EntryTopIdx - PopIter; - if (SlotIdx >= 0 && SlotIdx < static_cast(EntryRanges.size())) { - Opnd.setRange(EntryRanges[SlotIdx]); - } - ReverseStack.push(Opnd); - ++PopIter; - --TotalPopSize; - } - while (!ReverseStack.empty()) { - Operand Opnd = ReverseStack.pop(); - Stack.push(Opnd); - } -#endif } void diff --git a/src/compiler/evm_frontend/evm_mir_compiler.cpp b/src/compiler/evm_frontend/evm_mir_compiler.cpp index 06c860dfd..b41adf570 100644 --- a/src/compiler/evm_frontend/evm_mir_compiler.cpp +++ b/src/compiler/evm_frontend/evm_mir_compiler.cpp @@ -1167,7 +1167,6 @@ void EVMMirBuilder::dropStackBatch(uint32_t Count) { StackTopVar->getVarIdx()); createInstruction(true, &(Ctx.VoidType), NewSize, StackSizeVar->getVarIdx()); - } void EVMMirBuilder::pushStackBatch(const std::vector &Values) { @@ -1203,7 +1202,6 @@ void EVMMirBuilder::pushStackBatch(const std::vector &Values) { StackTopVar->getVarIdx()); createInstruction(true, &(Ctx.VoidType), NewSize, StackSizeVar->getVarIdx()); - } void EVMMirBuilder::stackSet(int32_t IndexFromTop, Operand SetValue) { diff --git a/src/tests/evm_jit_frontend_tests.cpp b/src/tests/evm_jit_frontend_tests.cpp index b00c35b1b..58827bfcb 100644 --- a/src/tests/evm_jit_frontend_tests.cpp +++ b/src/tests/evm_jit_frontend_tests.cpp @@ -3043,7 +3043,6 @@ bool compileWithMockBuilder(const std::vector &Bytecode, return Visitor.compile(); } -#ifdef ZEN_ENABLE_EVM_STACK_BOUNDARY_BATCH TEST(EVMMirBuilderStackBoundaryBatchTest, EmptyOneSixteenAndSeventeenSlotBatchesBuild) { MirBuilderConstFoldHarness Harness; @@ -3106,7 +3105,6 @@ TEST(EVMJITFrontendVisitorTest, EXPECT_EQ(Stats.StackTopUpdates, 2u); EXPECT_EQ(Stats.StackSizeUpdates, 2u); } -#endif TEST(EVMJITFrontendVisitorTest, TerminatingMemoryHelpersRetainExactOpcodePC) { const std::vector ReturnBytecode = {OP_PUSH0, OP_PUSH0, OP_RETURN};