Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions backends/qualcomm/runtime/backends/QnnBackendFactory.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -60,6 +60,9 @@ std::unique_ptr<BackendConfigParameters> QnnBackendFactory::Create(
QNN_EXECUTORCH_LOG_INFO(
"use_weight_sharing in htp_options: %d",
htp_options->use_weight_sharing());
QNN_EXECUTORCH_LOG_INFO(
"use_graph_splitting in htp_options: %d",
htp_options->use_graph_splitting());
}
backend_params->qnn_backend_cache_ptr_ =
std::make_unique<HtpBackendCache>(
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -26,6 +26,29 @@ HtpContextCustomConfig::CreateContextCustomConfig() {
ret.push_back(static_cast<QnnContext_CustomConfig_t>(p_custom_config));
}

#if ( \
QNN_HTP_API_VERSION_MAJOR > 5 || \
(QNN_HTP_API_VERSION_MAJOR == 5 && QNN_HTP_API_VERSION_MINOR >= 49))
if (htp_options_->use_graph_splitting()) {
p_custom_config = AllocContextCustomConfig();
p_custom_config->option =
QNN_HTP_CONTEXT_CONFIG_OPTION_GRAPH_SPLITTING_CONFIGS;
QnnHtpContext_GraphSplit_t graph_split_info{};
graph_split_info.graphSplittingEnabled = true;
p_custom_config->graphSplittingConfigs = graph_split_info;
ret.push_back(static_cast<QnnContext_CustomConfig_t>(p_custom_config));
}
#else
if (htp_options_->use_graph_splitting()) {
QNN_EXECUTORCH_LOG_WARN(
"use_graph_splitting is enabled but the QNN SDK used for this build "
"(API %d.%d) does not support graph splitting, which requires HTP API "
"5.49 or newer. The option will be ignored.",
QNN_HTP_API_VERSION_MAJOR,
QNN_HTP_API_VERSION_MINOR);
}
#endif

return ret;
}

Expand Down
4 changes: 4 additions & 0 deletions backends/qualcomm/serialization/qc_compiler_spec.fbs
Original file line number Diff line number Diff line change
Expand Up @@ -206,6 +206,10 @@ table QnnExecuTorchHtpBackendOptions {
/// It will help the by reducing overall bandwith on the use case.
/// The feature is only supported by specific SOCs.
use_slc_allocator:bool;

/// When enabled, the compiled graph is split based on its structure and
/// each part is compiled as an independent subgraph.
use_graph_splitting:bool;
}

/// Real-time: Indicates that the model is intended for real-time use cases, where a specific performance threshold must be met.
Expand Down
1 change: 1 addition & 0 deletions backends/qualcomm/serialization/qc_schema.py
Original file line number Diff line number Diff line change
Expand Up @@ -203,6 +203,7 @@ class QnnExecuTorchHtpBackendOptions:
use_multi_contexts: bool = False
use_weight_sharing: bool = False
use_slc_allocator: bool = False
use_graph_splitting: bool = False


@unique
Expand Down
10 changes: 10 additions & 0 deletions backends/qualcomm/tests/rework/htp/feature/v68/test.py
Original file line number Diff line number Diff line change
Expand Up @@ -12,6 +12,16 @@
from executorch.backends.qualcomm.tests.rework.src.feature import * # noqa: F403


@pytest.mark.parametrize(
"kwargs",
[
pytest.param({"expected": Tolerance()}, id="e2e"),
],
)
def test_graph_splitting(request, kwargs):
GraphSplit.test_graph_splitting(request, kwargs) # noqa: F405


@pytest.mark.parametrize(
"kwargs",
[
Expand Down
73 changes: 73 additions & 0 deletions backends/qualcomm/tests/rework/src/feature.py
Original file line number Diff line number Diff line change
Expand Up @@ -71,6 +71,79 @@ def get_quantizer(qnn_config: QnnConfig):
)


class GraphSplit:
class Model(torch.nn.Module):
def __init__(self):
super().__init__()

def example_inputs(self):
return (torch.randn(1, 2, 3, 4),)

def forward(self, x):
return torch.nn.ReLU()(x)

@staticmethod
def _test(qnn_config, compile_specs, expected):
def callback(adb: SimpleADB, pattern):
def verify(log):
msg = log.stdout
assert pattern in msg, f"{pattern} in log"

# QnnExecuTorchLogLevel.kLogLevelVerbose
adb.extra_cmds += " --log_level 4"
adb.execute(output_callback=verify)

with expected:
# model declaration
model = __class__.Model()
inputs = model.example_inputs()
# perform ptq
with calibrate(
model, [inputs], make_quantizer(soc_model=qnn_config.soc_model)
) as model:
executorch_prog_mgr = to_edge_transform_and_lower_to_qnn(
module=model,
inputs=inputs,
compiler_specs=compile_specs,
).to_executorch()
# file for subgraph 0
# delete artifact before assertion to avoid leak
file_name = "forward_schematic.bin_sg_0.py"
file_exist = os.path.isfile(file_name)
if file_exist:
os.remove(file_name)
assert os.path.isfile(file_exist)
# remote testing
invoke_remote(
qnn_config=qnn_config,
executorch_prog=executorch_prog_mgr,
callback=partial(callback, pattern="Found blob with gpe enabled"),
)

@staticmethod
@unpack_fixtures
def test_graph_splitting(qnn_config, compile_specs, expected):
# extend this for other backends
backend_compile_specs = {
QnnExecuTorchBackendType.kHtpBackend: compile_specs(
tuple(
{
"soc_model": getattr(QcomChipset, qnn_config.soc_model),
"use_graph_splitting": True,
"use_fp16": False,
"profile_level": 3,
}.items()
)
),
}

__class__._test(
qnn_config=qnn_config,
compile_specs=backend_compile_specs[qnn_config.backend],
expected=expected,
)


class Logging:
class Model(torch.nn.Module):
def __init__(self):
Expand Down
59 changes: 59 additions & 0 deletions backends/qualcomm/tests/test_qnn_delegate.py
Original file line number Diff line number Diff line change
Expand Up @@ -7248,6 +7248,36 @@ def test_qnn_backend_multi_contexts_composite(self):
exec_prog = edge_prog.to_executorch()
self.verify_output(module.get_reference_module(), sample_input, exec_prog)

@unittest.skipIf(
is_qnn_sdk_version_less_than("2.49"),
"feature is enable after 2.49.",
)
def test_qnn_backend_graph_splitting(self):
backend_options = generate_htp_compiler_spec(
use_fp16=True,
use_graph_splitting=True,
)
compiler_spec = generate_qnn_executorch_compiler_spec(
soc_model=self.chipset_table[TestQNN.soc_model],
backend_options=backend_options,
profile_level=3,
)
sample_input = (torch.randn([2, 5, 1, 3]),)
module = Relu() # noqa: F405
edge_prog_mgr = to_edge_transform_and_lower_to_qnn(
module, sample_input, compiler_spec
).to_executorch()
# file for subgraph 0
# delete artifact before assertion to avoid leak
file_name = "forward_schematic.bin_sg_0.py"
file_exist = os.path.isfile(file_name)

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

os.path.isfile accepts a file descriptor, so this stats fd 1 or 0 - stdout/stdin not the artifact, did I miss anything ?

Copy link
Copy Markdown
Contributor Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

it accepts a file path.

if file_exist:
os.remove(file_name)
self.assertTrue(file_exist)
self.verify_output(
module=module, sample_inputs=sample_input, executorch_prog=edge_prog_mgr
)

def test_qnn_backend_multi_graphs(self):
if self.enable_x86_64:
self.skipTest("weight sharing is not supported on host machine")
Expand Down Expand Up @@ -8326,6 +8356,35 @@ def test_qnn_backend_multi_contexts_composite(self):
exec_prog = edge_prog.to_executorch()
self.verify_output(module.get_reference_module(), sample_input, exec_prog)

@unittest.skipIf(
is_qnn_sdk_version_less_than("2.49"),
"feature is enable after 2.49.",
)
def test_qnn_backend_graph_splitting(self):
backend_options = generate_htp_compiler_spec(
use_fp16=False,
use_graph_splitting=True,
)
compiler_spec = generate_qnn_executorch_compiler_spec(
soc_model=self.chipset_table[TestQNN.soc_model],
backend_options=backend_options,
profile_level=3,
)
sample_input = (torch.randn([2, 5, 1, 3]),)
module = Relu() # noqa: F405
module = self.get_qdq_module(
module, sample_input, quant_dtype=QuantDtype.use_8a8w
)
edge_prog_mgr = to_edge_transform_and_lower_to_qnn(
module, sample_input, compiler_spec
).to_executorch()
# file for subgraph 0
self.assertTrue(os.path.isfile("forward_schematic.bin_sg_0.py"))
os.remove("forward_schematic.bin_sg_0.py")
self.verify_output(
module=module, sample_inputs=sample_input, executorch_prog=edge_prog_mgr
)

def test_qnn_backend_multi_graphs(self):
if self.enable_x86_64:
self.skipTest("weight sharing is not supported on host machine")
Expand Down
4 changes: 4 additions & 0 deletions backends/qualcomm/utils/utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -1094,6 +1094,7 @@ def generate_htp_compiler_spec(
use_multi_contexts: bool = False,
use_weight_sharing: bool = False,
use_slc_allocator: bool = False,
use_graph_splitting: bool = False,
htp_performance_mode: QnnExecuTorchHtpPerformanceMode = QnnExecuTorchHtpPerformanceMode.kHtpBurst,
) -> QnnExecuTorchBackendOptions:
"""
Expand All @@ -1113,6 +1114,8 @@ def generate_htp_compiler_spec(
use_slc_allocator: Allows user to enable the usage of the System Level Cache Allocator for a given graph.
It will help the by reducing overall bandwith on the use case.
The feature is only supported by specific SOCs.
use_graph_splitting: When enabled, the compiled graph is split based on
its structure and each part is compiled as an independent subgraph.

Returns:
QnnExecuTorchHtpBackendOptions: backend options for QNN HTP.
Expand All @@ -1131,6 +1134,7 @@ def generate_htp_compiler_spec(
htp_options.use_weight_sharing = use_weight_sharing
htp_options.use_dlbc = use_dlbc
htp_options.use_slc_allocator = use_slc_allocator
htp_options.use_graph_splitting = use_graph_splitting
return QnnExecuTorchBackendOptions(
backend_type=QnnExecuTorchBackendType.kHtpBackend,
htp_options=htp_options,
Expand Down
Loading