From 7d0d277f81f962ee9e2672a9a1dfa3b77c951c9b Mon Sep 17 00:00:00 2001 From: Cheng-Hsin Weng Date: Tue, 18 Aug 2026 15:46:32 +0800 Subject: [PATCH 1/3] Qualcomm AI Engine Direct - Add HTP context graph splitting option Exposes QNN_HTP_CONTEXT_CONFIG_OPTION_GRAPH_SPLITTING_CONFIGS through generate_htp_compiler_spec(use_graph_splitting=...) for the offline-prepare (host) export path. Gated behind QNN HTP API >= 5.49 since this option does not exist in older version. Co-authored-by with assistance from Claude Code (Anthropic). --- .../runtime/backends/QnnBackendFactory.cpp | 3 + .../htp/host/HtpContextCustomConfig.cpp | 21 ++++++ .../serialization/qc_compiler_spec.fbs | 4 ++ backends/qualcomm/serialization/qc_schema.py | 1 + .../tests/rework/htp/feature/v68/test.py | 10 +++ backends/qualcomm/tests/rework/src/feature.py | 69 +++++++++++++++++++ backends/qualcomm/tests/test_qnn_delegate.py | 55 +++++++++++++++ backends/qualcomm/utils/utils.py | 4 ++ 8 files changed, 167 insertions(+) diff --git a/backends/qualcomm/runtime/backends/QnnBackendFactory.cpp b/backends/qualcomm/runtime/backends/QnnBackendFactory.cpp index 1b2a89eceb0..6adf7b9e9fb 100644 --- a/backends/qualcomm/runtime/backends/QnnBackendFactory.cpp +++ b/backends/qualcomm/runtime/backends/QnnBackendFactory.cpp @@ -60,6 +60,9 @@ std::unique_ptr QnnBackendFactory::Create( QNN_EXECUTORCH_LOG_INFO( "use_weight_sharing in htp_options: %d", htp_options->use_weight_sharing()); + QNN_EXECUTORCH_LOG_INFO( + "use_graph_splitting in htp_options: %d", + htp_options->use_graph_splitting()); } backend_params->qnn_backend_cache_ptr_ = std::make_unique( diff --git a/backends/qualcomm/runtime/backends/htp/host/HtpContextCustomConfig.cpp b/backends/qualcomm/runtime/backends/htp/host/HtpContextCustomConfig.cpp index 037998132a8..6b59b3f8430 100644 --- a/backends/qualcomm/runtime/backends/htp/host/HtpContextCustomConfig.cpp +++ b/backends/qualcomm/runtime/backends/htp/host/HtpContextCustomConfig.cpp @@ -26,6 +26,27 @@ HtpContextCustomConfig::CreateContextCustomConfig() { ret.push_back(static_cast(p_custom_config)); } +#if (QNN_HTP_API_VERSION_MAJOR >= 5 && QNN_HTP_API_VERSION_MINOR >= 49) + if (htp_options_->use_graph_splitting()) { + p_custom_config = AllocContextCustomConfig(); + p_custom_config->option = + QNN_HTP_CONTEXT_CONFIG_OPTION_GRAPH_SPLITTING_CONFIGS; + QnnHtpContext_GraphSplit_t graph_split_info; + graph_split_info.graphSplittingEnabled = true; + p_custom_config->graphSplittingConfigs = graph_split_info; + ret.push_back(static_cast(p_custom_config)); + } +#else + if (htp_options_->use_graph_splitting()) { + QNN_EXECUTORCH_LOG_WARN( + "use_graph_splitting is enabled but the QNN SDK used for this build " + "(API %d.%d) does not support graph splitting, which requires HTP API " + "5.49 or newer. The option will be ignored.", + QNN_API_VERSION_MAJOR, + QNN_API_VERSION_MINOR); + } +#endif + return ret; } diff --git a/backends/qualcomm/serialization/qc_compiler_spec.fbs b/backends/qualcomm/serialization/qc_compiler_spec.fbs index ad8a2184e3a..2c0ff6444da 100644 --- a/backends/qualcomm/serialization/qc_compiler_spec.fbs +++ b/backends/qualcomm/serialization/qc_compiler_spec.fbs @@ -206,6 +206,10 @@ table QnnExecuTorchHtpBackendOptions { /// It will help the by reducing overall bandwith on the use case. /// The feature is only supported by specific SOCs. use_slc_allocator:bool; + + /// When enabled, the compiled graph is split based on its structure and + /// each part is compiled as an independent subgraph. + use_graph_splitting:bool; } /// Real-time: Indicates that the model is intended for real-time use cases, where a specific performance threshold must be met. diff --git a/backends/qualcomm/serialization/qc_schema.py b/backends/qualcomm/serialization/qc_schema.py index 09407830dcc..55887d3b89b 100644 --- a/backends/qualcomm/serialization/qc_schema.py +++ b/backends/qualcomm/serialization/qc_schema.py @@ -203,6 +203,7 @@ class QnnExecuTorchHtpBackendOptions: use_multi_contexts: bool = False use_weight_sharing: bool = False use_slc_allocator: bool = False + use_graph_splitting: bool = False @unique diff --git a/backends/qualcomm/tests/rework/htp/feature/v68/test.py b/backends/qualcomm/tests/rework/htp/feature/v68/test.py index 85269adaa6a..17f31ca711e 100644 --- a/backends/qualcomm/tests/rework/htp/feature/v68/test.py +++ b/backends/qualcomm/tests/rework/htp/feature/v68/test.py @@ -12,6 +12,16 @@ from executorch.backends.qualcomm.tests.rework.src.feature import * # noqa: F403 +@pytest.mark.parametrize( + "kwargs", + [ + pytest.param({"expected": Tolerance()}, id="e2e"), + ], +) +def test_graph_splitting(request, kwargs): + GraphSplit.test_graph_splitting(request, kwargs) # noqa: F405 + + @pytest.mark.parametrize( "kwargs", [ diff --git a/backends/qualcomm/tests/rework/src/feature.py b/backends/qualcomm/tests/rework/src/feature.py index 49499fc04e6..f1b635d611c 100644 --- a/backends/qualcomm/tests/rework/src/feature.py +++ b/backends/qualcomm/tests/rework/src/feature.py @@ -71,6 +71,75 @@ def get_quantizer(qnn_config: QnnConfig): ) +class GraphSplit: + class Model(torch.nn.Module): + def __init__(self): + super().__init__() + + def example_inputs(self): + return (torch.randn(1, 2, 3, 4),) + + def forward(self, x): + return torch.nn.ReLU()(x) + + @staticmethod + def _test(qnn_config, compile_specs, expected): + def callback(adb: SimpleADB, pattern): + def verify(log): + msg = log.stdout + assert pattern in msg, f"{pattern} in log" + + # QnnExecuTorchLogLevel.kLogLevelVerbose + adb.extra_cmds += " --log_level 4" + adb.execute(output_callback=verify) + + with expected: + # model declaration + model = __class__.Model() + inputs = model.example_inputs() + # perform ptq + with calibrate( + model, [inputs], make_quantizer(soc_model=qnn_config.soc_model) + ) as model: + executorch_prog_mgr = to_edge_transform_and_lower_to_qnn( + module=model, + inputs=inputs, + compiler_specs=compile_specs, + ).to_executorch() + # file for subgraph 0 + assert os.path.isfile("forward_schematic.bin_sg_0.py") + os.remove("forward_schematic.bin_sg_0.py") + # remote testing + invoke_remote( + qnn_config=qnn_config, + executorch_prog=executorch_prog_mgr, + callback=partial(callback, pattern="Found blob with gpe enabled"), + ) + + @staticmethod + @unpack_fixtures + def test_graph_splitting(qnn_config, compile_specs, expected): + # extend this for other backends + backend_compile_specs = { + QnnExecuTorchBackendType.kHtpBackend: compile_specs( + tuple( + { + "soc_model": getattr(QcomChipset, qnn_config.soc_model), + "use_graph_splitting": True, + "use_fp16": False, + "profile_level": 3, + }.items() + ) + ), + } + + __class__._test( + qnn_config=qnn_config, + compile_specs=backend_compile_specs[qnn_config.backend], + expected=expected, + ) + + class Logging: class Model(torch.nn.Module): def __init__(self): diff --git a/backends/qualcomm/tests/test_qnn_delegate.py b/backends/qualcomm/tests/test_qnn_delegate.py index 78f0e0f4350..7a485aedf7a 100644 --- a/backends/qualcomm/tests/test_qnn_delegate.py +++ b/backends/qualcomm/tests/test_qnn_delegate.py @@ -7248,6 +7248,32 @@ def test_qnn_backend_multi_contexts_composite(self): exec_prog = edge_prog.to_executorch() self.verify_output(module.get_reference_module(), sample_input, exec_prog) + @unittest.skipIf( + is_qnn_sdk_version_less_than("2.49"), + "feature is enable after 2.49.", + ) + def test_qnn_backend_graph_splitting(self): + backend_options = generate_htp_compiler_spec( + use_fp16=True, + use_graph_splitting=True, + ) + compiler_spec = generate_qnn_executorch_compiler_spec( + soc_model=self.chipset_table[TestQNN.soc_model], + backend_options=backend_options, + profile_level=3, + ) + sample_input = (torch.randn([2, 5, 1, 3]),) + module = Relu() # noqa: F405 + edge_prog_mgr = to_edge_transform_and_lower_to_qnn( + module, sample_input, compiler_spec + ).to_executorch() + # file for subgraph 0 + self.assertTrue(os.path.isfile("forward_schematic.bin_sg_0.py")) + os.remove("forward_schematic.bin_sg_0.py") + self.verify_output( + module=module, sample_inputs=sample_input, executorch_prog=edge_prog_mgr + ) + def test_qnn_backend_multi_graphs(self): if self.enable_x86_64: self.skipTest("weight sharing is not supported on host machine") @@ -8326,6 +8352,35 @@ def test_qnn_backend_multi_contexts_composite(self): exec_prog = edge_prog.to_executorch() self.verify_output(module.get_reference_module(), sample_input, exec_prog) + @unittest.skipIf( + is_qnn_sdk_version_less_than("2.49"), + "feature is enable after 2.49.", + ) + def test_qnn_backend_graph_splitting(self): + backend_options = generate_htp_compiler_spec( + use_fp16=False, + use_graph_splitting=True, + ) + compiler_spec = generate_qnn_executorch_compiler_spec( + soc_model=self.chipset_table[TestQNN.soc_model], + backend_options=backend_options, + profile_level=3, + ) + sample_input = (torch.randn([2, 5, 1, 3]),) + module = Relu() # noqa: F405 + module = self.get_qdq_module( + module, sample_input, quant_dtype=QuantDtype.use_8a8w + ) + edge_prog_mgr = to_edge_transform_and_lower_to_qnn( + module, sample_input, compiler_spec + ).to_executorch() + # file for subgraph 0 + self.assertTrue(os.path.isfile("forward_schematic.bin_sg_0.py")) + os.remove("forward_schematic.bin_sg_0.py") + self.verify_output( + module=module, sample_inputs=sample_input, executorch_prog=edge_prog_mgr + ) + def test_qnn_backend_multi_graphs(self): if self.enable_x86_64: self.skipTest("weight sharing is not supported on host machine") diff --git a/backends/qualcomm/utils/utils.py b/backends/qualcomm/utils/utils.py index db215d91675..2f831a15b55 100644 --- a/backends/qualcomm/utils/utils.py +++ b/backends/qualcomm/utils/utils.py @@ -1094,6 +1094,7 @@ def generate_htp_compiler_spec( use_multi_contexts: bool = False, use_weight_sharing: bool = False, use_slc_allocator: bool = False, + use_graph_splitting: bool = False, htp_performance_mode: QnnExecuTorchHtpPerformanceMode = QnnExecuTorchHtpPerformanceMode.kHtpBurst, ) -> QnnExecuTorchBackendOptions: """ @@ -1113,6 +1114,8 @@ def generate_htp_compiler_spec( use_slc_allocator: Allows user to enable the usage of the System Level Cache Allocator for a given graph. It will help the by reducing overall bandwith on the use case. The feature is only supported by specific SOCs. + use_graph_splitting: When enabled, the compiled graph is split based on + its structure and each part is compiled as an independent subgraph. Returns: QnnExecuTorchHtpBackendOptions: backend options for QNN HTP. @@ -1131,6 +1134,7 @@ def generate_htp_compiler_spec( htp_options.use_weight_sharing = use_weight_sharing htp_options.use_dlbc = use_dlbc htp_options.use_slc_allocator = use_slc_allocator + htp_options.use_graph_splitting = use_graph_splitting return QnnExecuTorchBackendOptions( backend_type=QnnExecuTorchBackendType.kHtpBackend, htp_options=htp_options, From 68481285a5fb1b8edcc17ef8d81e843956eca30c Mon Sep 17 00:00:00 2001 From: Cheng-Hsin Weng Date: Thu, 17 Sep 2026 09:54:10 +0800 Subject: [PATCH 2/3] fix comment --- .../runtime/backends/htp/host/HtpContextCustomConfig.cpp | 8 ++++---- backends/qualcomm/tests/rework/src/feature.py | 8 ++++++-- backends/qualcomm/tests/test_qnn_delegate.py | 8 ++++++-- 3 files changed, 16 insertions(+), 8 deletions(-) diff --git a/backends/qualcomm/runtime/backends/htp/host/HtpContextCustomConfig.cpp b/backends/qualcomm/runtime/backends/htp/host/HtpContextCustomConfig.cpp index 6b59b3f8430..a488ba11c85 100644 --- a/backends/qualcomm/runtime/backends/htp/host/HtpContextCustomConfig.cpp +++ b/backends/qualcomm/runtime/backends/htp/host/HtpContextCustomConfig.cpp @@ -26,12 +26,12 @@ HtpContextCustomConfig::CreateContextCustomConfig() { ret.push_back(static_cast(p_custom_config)); } -#if (QNN_HTP_API_VERSION_MAJOR >= 5 && QNN_HTP_API_VERSION_MINOR >= 49) +#if (QNN_HTP_API_VERSION_MAJOR > 5 || (QNN_HTP_API_VERSION_MAJOR == 5 && QNN_HTP_API_VERSION_MINOR >= 49)) if (htp_options_->use_graph_splitting()) { p_custom_config = AllocContextCustomConfig(); p_custom_config->option = QNN_HTP_CONTEXT_CONFIG_OPTION_GRAPH_SPLITTING_CONFIGS; - QnnHtpContext_GraphSplit_t graph_split_info; + QnnHtpContext_GraphSplit_t graph_split_info{}; graph_split_info.graphSplittingEnabled = true; p_custom_config->graphSplittingConfigs = graph_split_info; ret.push_back(static_cast(p_custom_config)); @@ -42,8 +42,8 @@ HtpContextCustomConfig::CreateContextCustomConfig() { "use_graph_splitting is enabled but the QNN SDK used for this build " "(API %d.%d) does not support graph splitting, which requires HTP API " "5.49 or newer. The option will be ignored.", - QNN_API_VERSION_MAJOR, - QNN_API_VERSION_MINOR); + QNN_HTP_API_VERSION_MAJOR, + QNN_HTP_API_VERSION_MINOR); } #endif diff --git a/backends/qualcomm/tests/rework/src/feature.py b/backends/qualcomm/tests/rework/src/feature.py index f1b635d611c..848c370760b 100644 --- a/backends/qualcomm/tests/rework/src/feature.py +++ b/backends/qualcomm/tests/rework/src/feature.py @@ -107,8 +107,12 @@ def verify(log): compiler_specs=compile_specs, ).to_executorch() # file for subgraph 0 - assert os.path.isfile("forward_schematic.bin_sg_0.py") - os.remove("forward_schematic.bin_sg_0.py") + # delete artifact before assertion to avoid leak + file_name = "forward_schematic.bin_sg_0.py" + file_exist = os.path.isfile(file_name) + if file_exist: + os.remove(file_name) + assert os.path.isfile(file_exist) # remote testing invoke_remote( qnn_config=qnn_config, diff --git a/backends/qualcomm/tests/test_qnn_delegate.py b/backends/qualcomm/tests/test_qnn_delegate.py index 7a485aedf7a..9d047fe040a 100644 --- a/backends/qualcomm/tests/test_qnn_delegate.py +++ b/backends/qualcomm/tests/test_qnn_delegate.py @@ -7268,8 +7268,12 @@ def test_qnn_backend_graph_splitting(self): module, sample_input, compiler_spec ).to_executorch() # file for subgraph 0 - self.assertTrue(os.path.isfile("forward_schematic.bin_sg_0.py")) - os.remove("forward_schematic.bin_sg_0.py") + # delete artifact before assertion to avoid leak + file_name = "forward_schematic.bin_sg_0.py" + file_exist = os.path.isfile(file_name) + if file_exist: + os.remove(file_name) + self.assertTrue(file_exist) self.verify_output( module=module, sample_inputs=sample_input, executorch_prog=edge_prog_mgr ) From ccb9f0226b2166468bcf59b2af6690710b3524fc Mon Sep 17 00:00:00 2001 From: Cheng-Hsin Weng Date: Thu, 17 Sep 2026 10:58:16 +0800 Subject: [PATCH 3/3] fix lint --- .../runtime/backends/htp/host/HtpContextCustomConfig.cpp | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/backends/qualcomm/runtime/backends/htp/host/HtpContextCustomConfig.cpp b/backends/qualcomm/runtime/backends/htp/host/HtpContextCustomConfig.cpp index a488ba11c85..874458c9eb2 100644 --- a/backends/qualcomm/runtime/backends/htp/host/HtpContextCustomConfig.cpp +++ b/backends/qualcomm/runtime/backends/htp/host/HtpContextCustomConfig.cpp @@ -26,7 +26,9 @@ HtpContextCustomConfig::CreateContextCustomConfig() { ret.push_back(static_cast(p_custom_config)); } -#if (QNN_HTP_API_VERSION_MAJOR > 5 || (QNN_HTP_API_VERSION_MAJOR == 5 && QNN_HTP_API_VERSION_MINOR >= 49)) +#if ( \ + QNN_HTP_API_VERSION_MAJOR > 5 || \ + (QNN_HTP_API_VERSION_MAJOR == 5 && QNN_HTP_API_VERSION_MINOR >= 49)) if (htp_options_->use_graph_splitting()) { p_custom_config = AllocContextCustomConfig(); p_custom_config->option =