Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions docs/execution_providers/QNN-ExecutionProvider.md
Original file line number Diff line number Diff line change
Expand Up @@ -1728,6 +1728,9 @@ To enable new operator support in EP, areas to visit:

A **User-Defined Operation (UDO)** allows developers to extend the Qualcomm® Neural Network (QNN) runtimes with custom operators. UDO enables execution of operations that are not natively supported in the default QNN op set, while maintaining compatibility with model conversion, compilation, and runtime execution.

For an end-to-end MyAdd UDO reference, including CPU, HTP, and on-device
commands, see the [QNN UDO sample](../../qcom/samples/qnn_udo_myadd/README.md).

### Overview

A UDO lets you define and register custom operations—describing their inputs, outputs, parameters, data types, and backend behavior—so they can run on:
Expand Down
9 changes: 9 additions & 0 deletions qcom/samples/qnn_udo_myadd/.gitignore
Original file line number Diff line number Diff line change
@@ -0,0 +1,9 @@
# Artifacts generated while building or running the sample.
/artifacts/
/build/
/run_udo_sample
/test_data_set_0/
/__pycache__/
QNNExecutionProvider_*.json
/*.onnx
/*.bin
351 changes: 351 additions & 0 deletions qcom/samples/qnn_udo_myadd/README.md

Large diffs are not rendered by default.

105 changes: 105 additions & 0 deletions qcom/samples/qnn_udo_myadd/build_op_package.sh
Original file line number Diff line number Diff line change
@@ -0,0 +1,105 @@
#!/usr/bin/env bash
# Copyright (c) Qualcomm Technologies, Inc. and/or its subsidiaries.
# SPDX-License-Identifier: MIT
#
# build_op_package.sh -- Build MyAdd QNN op package(s).
#
# Usage:
# ./build_op_package.sh cpu # build CPU x86 op package
# ./build_op_package.sh htp # build HTP x86 op package
# ./build_op_package.sh all # build both
#
# Required environment variables:
# QNN_SDK_ROOT – path to QAIRT SDK root (e.g. .../qairt/<version>)
# LLVM_TOOL_DIR – path to LLVM bin dir (e.g. .../LLVM-21.1.8-Linux-X64)
# HEXAGON_SDK_ROOT – (HTP only) path to Hexagon SDK version dir (e.g. .../6.5.0.0)
#
# Outputs (under this sample's artifacts/ directory):
# artifacts/libMyAddOpPackage_cpu.so (CPU target)
# artifacts/libMyAddOpPackage_htp.so (HTP target)

set -euo pipefail

SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
REPO_ROOT="$(cd "${SCRIPT_DIR}/../../.." && pwd)"
OP_PACKAGE_DIR="${REPO_ROOT}/onnxruntime/test/providers/qnn/udo"
ARTIFACT_DIR="${SCRIPT_DIR}/artifacts"
BUILD_DIR="${ARTIFACT_DIR}/build"

build_cpu() {
QNN_SDK_ROOT="${QNN_SDK_ROOT:?QNN_SDK_ROOT must be set}"
LLVM_TOOL_DIR="${LLVM_TOOL_DIR:?LLVM_TOOL_DIR must be set}"
echo ">>> Building CPU x86 op package..."
mkdir -p "${ARTIFACT_DIR}"
local cpu_build="${BUILD_DIR}/cpu"
rm -rf "${cpu_build}"

# Step 1: generate skeleton
PYTHONPATH="${QNN_SDK_ROOT}/lib/python" \
python3 "${QNN_SDK_ROOT}/bin/x86_64-linux-clang/qnn-op-package-generator" \
-p "${OP_PACKAGE_DIR}/MyAddOpPackageCpu.xml" \
-o "${cpu_build}"

# Step 2: copy pre-implemented kernel
/bin/cp "${OP_PACKAGE_DIR}/MyAddCPU.cpp" \
"${cpu_build}/MyAddOpPackage/src/ops/MyAdd.cpp"

# Step 3: build
QNN_SDK_ROOT="${QNN_SDK_ROOT}" \
PATH="${LLVM_TOOL_DIR}/bin:${PATH}" \
make -C "${cpu_build}/MyAddOpPackage" \
"CXX=${LLVM_TOOL_DIR}/bin/clang++ -stdlib=libc++ -static-libstdc++ -Wl,--exclude-libs,ALL" \
all_x86

# Step 4: copy output
/bin/cp "${cpu_build}/MyAddOpPackage/libs/x86_64-linux-clang/libMyAddOpPackage.so" \
"${ARTIFACT_DIR}/libMyAddOpPackage_cpu.so"
echo ">>> CPU package: ${ARTIFACT_DIR}/libMyAddOpPackage_cpu.so"
}

build_htp() {
QNN_SDK_ROOT="${QNN_SDK_ROOT:?QNN_SDK_ROOT must be set}"
LLVM_TOOL_DIR="${LLVM_TOOL_DIR:?LLVM_TOOL_DIR must be set}"
HEXAGON_SDK_ROOT="${HEXAGON_SDK_ROOT:?HEXAGON_SDK_ROOT must be set for HTP build}"
mkdir -p "${ARTIFACT_DIR}"
local htp_build="${BUILD_DIR}/htp"
rm -rf "${htp_build}"

# Step 1: generate skeleton
PYTHONPATH="${QNN_SDK_ROOT}/lib/python" \
python3 "${QNN_SDK_ROOT}/bin/x86_64-linux-clang/qnn-op-package-generator" \
-p "${OP_PACKAGE_DIR}/MyAddOpPackageHtp.xml" \
-o "${htp_build}"

# Step 2: copy pre-implemented kernel + custom HTP Makefile
/bin/cp "${OP_PACKAGE_DIR}/MyAddHTP.cpp" \
"${htp_build}/MyAddOpPackage/src/ops/MyAdd.cpp"
/bin/cp "${OP_PACKAGE_DIR}/HTP_Makefile" \
"${htp_build}/MyAddOpPackage/Makefile"

# Step 3: build
QNN_SDK_ROOT="${QNN_SDK_ROOT}" \
HEXAGON_SDK_ROOT="${HEXAGON_SDK_ROOT}" \
PATH="${LLVM_TOOL_DIR}/bin:${PATH}" \
make -C "${htp_build}/MyAddOpPackage" \
"X86_CXX=${LLVM_TOOL_DIR}/bin/clang++ -stdlib=libc++" \
htp_x86

# Step 4: copy output
/bin/cp "${htp_build}/MyAddOpPackage/build/x86_64-linux-clang/libQnnMyAddOpPackage.so" \
"${ARTIFACT_DIR}/libMyAddOpPackage_htp.so"
echo ">>> HTP package: ${ARTIFACT_DIR}/libMyAddOpPackage_htp.so"
}

TARGET="${1:-all}"
case "${TARGET}" in
cpu) build_cpu ;;
htp) build_htp ;;
all) build_cpu; build_htp ;;
*)
echo "Usage: $0 [cpu|htp|all]"
exit 1
;;
esac

echo ">>> Done."
135 changes: 135 additions & 0 deletions qcom/samples/qnn_udo_myadd/gen_myadd_model.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,135 @@
# Copyright (c) Qualcomm Technologies, Inc. and/or its subsidiaries.
# SPDX-License-Identifier: MIT
"""
Generate two ONNX models containing a single MyAdd UDO node:
- myadd_fp32.onnx : float32 model (for QNN CPU backend)
- myadd_qdq.onnx : uint8 QDQ model (DQ -> MyAdd -> Q, for QNN HTP backend)

MyAdd computes: output = input + constant
input : shape [1, 32], float32
output : shape [1, 32], float32
constant attr : float, default 2.0

Usage:
python gen_myadd_model.py [--constant 2.0] [--outdir .]
"""

import argparse
from pathlib import Path

import numpy as np
import onnx
from onnx import TensorProto, helper, numpy_helper

DOMAIN = "example"
OP_TYPE = "MyAdd"
INPUT_SHAPE = [1, 32]


def make_fp32_model(constant: float) -> onnx.ModelProto:
"""Float32 model: input -> MyAdd -> output."""
input_vi = helper.make_tensor_value_info("input", TensorProto.FLOAT, INPUT_SHAPE)
output_vi = helper.make_tensor_value_info("output", TensorProto.FLOAT, INPUT_SHAPE)

constant_attr = helper.make_attribute("constant", constant)
node = helper.make_node(OP_TYPE, inputs=["input"], outputs=["output"], domain=DOMAIN)
node.attribute.append(constant_attr)

graph = helper.make_graph([node], "myadd_fp32", [input_vi], [output_vi])
opset = helper.make_opsetid(DOMAIN, 1)
model = helper.make_model(graph, opset_imports=[opset])
model.ir_version = 8
onnx.checker.check_model(model, full_check=False)
return model


def make_qdq_model(constant: float) -> onnx.ModelProto:
"""
QDQ model for HTP backend:
input (f32) -> QuantizeLinear -> input_q (u8) -> DequantizeLinear -> input_dq (f32)
-> MyAdd -> output_dq (f32) -> QuantizeLinear -> output_q (u8) -> DequantizeLinear -> output (f32)

Scale/zero-point are computed from [-1, 1] range matching udo_op_test.cc test input.
"""
# Input quantization: [-1, 1] range -> uint8 (scale=2/255, zp=128 so 0.0 maps to 128)
scale_in = np.float32(2.0 / 255.0)
zp_in = np.uint8(128)

# Output quantization: [0, 4] range (conservatively covers input[-1,1]+constant[2.0])
# zp=0 (asymmetric, all values positive)
scale_out = np.float32(4.0 / 255.0)
zp_out = np.uint8(0)

def quant_init(name: str, val) -> onnx.TensorProto:
return numpy_helper.from_array(np.array(val), name=name)

# scale/zero_point initializers
inits = [
quant_init("scale_in", scale_in),
quant_init("zp_in", zp_in),
quant_init("scale_out", scale_out),
quant_init("zp_out", zp_out),
]

# Value infos
def make_f32_value_info(name: str) -> onnx.ValueInfoProto:
return helper.make_tensor_value_info(name, TensorProto.FLOAT, INPUT_SHAPE)

input_vi = make_f32_value_info("input")
output_vi = make_f32_value_info("output")

# Declared type/shape for the MyAdd output (intermediate tensor feeding the
# output QuantizeLinear). The QNN EP auto-registration path registers a
# placeholder op with no shape/type inference, so ORT resolves the custom-op
# output type from this value_info at model-load time. Without it, load fails
# with "type inference failed" for the custom-domain node.
output_dq_vi = make_f32_value_info("output_dq")

# Nodes: Q -> DQ -> MyAdd -> Q -> DQ
q_in = helper.make_node("QuantizeLinear", ["input", "scale_in", "zp_in"], ["input_q"], axis=None)
dq_in = helper.make_node("DequantizeLinear", ["input_q", "scale_in", "zp_in"], ["input_dq"], axis=None)

constant_attr = helper.make_attribute("constant", constant)
myadd = helper.make_node(OP_TYPE, ["input_dq"], ["output_dq"], domain=DOMAIN)
myadd.attribute.append(constant_attr)

q_out = helper.make_node("QuantizeLinear", ["output_dq", "scale_out", "zp_out"], ["output_q"], axis=None)
dq_out = helper.make_node("DequantizeLinear", ["output_q", "scale_out", "zp_out"], ["output"], axis=None)

graph = helper.make_graph(
[q_in, dq_in, myadd, q_out, dq_out],
"myadd_qdq",
[input_vi],
[output_vi],
initializer=inits,
value_info=[output_dq_vi],
)
onnx_opset = helper.make_opsetid("", 21)
custom_opset = helper.make_opsetid(DOMAIN, 1)
model = helper.make_model(graph, opset_imports=[onnx_opset, custom_opset])
model.ir_version = 8
onnx.checker.check_model(model, full_check=False)
return model


def main():
parser = argparse.ArgumentParser(description="Generate MyAdd UDO ONNX models")
parser.add_argument("--constant", type=float, default=2.0, help="Value added to each input element (default: 2.0)")
parser.add_argument("--outdir", default=".", help="Output directory")
args = parser.parse_args()

outdir = Path(args.outdir)
outdir.mkdir(parents=True, exist_ok=True)

fp32_path = outdir / "myadd_fp32.onnx"
qdq_path = outdir / "myadd_qdq.onnx"

onnx.save(make_fp32_model(args.constant), fp32_path)
print(f"Saved {fp32_path}")

onnx.save(make_qdq_model(args.constant), qdq_path)
print(f"Saved {qdq_path}")


if __name__ == "__main__":
main()
37 changes: 37 additions & 0 deletions qcom/samples/qnn_udo_myadd/gen_myadd_test_data.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,37 @@
# Copyright (c) Qualcomm Technologies, Inc. and/or its subsidiaries.
# SPDX-License-Identifier: MIT
"""Generate protobuf test data for onnxruntime_plugin_ep_onnx_test.

The input matches the standalone samples: 32 float32 values evenly spaced in
[-1, 1]. The reference is the unquantized mathematical result (input +
constant); the on-device runner applies the documented QDQ tolerance.
"""

import argparse
from pathlib import Path

import numpy as np
from onnx import numpy_helper


def main():
parser = argparse.ArgumentParser(description="Generate MyAdd ONNX test data")
parser.add_argument("--constant", type=float, default=2.0, help="Value added by MyAdd (default: 2.0)")
parser.add_argument("--outdir", required=True, help="Test-case directory that will contain test_data_set_0")
args = parser.parse_args()

data_dir = Path(args.outdir) / "test_data_set_0"
data_dir.mkdir(parents=True, exist_ok=True)
input_data = np.linspace(-1.0, 1.0, 32, dtype=np.float32).reshape(1, 32)
output_data = input_data + np.float32(args.constant)

with (data_dir / "input_0.pb").open("wb") as f:
f.write(numpy_helper.from_array(input_data, name="input").SerializeToString())
with (data_dir / "output_0.pb").open("wb") as f:
f.write(numpy_helper.from_array(output_data, name="output").SerializeToString())

print(f"Saved {data_dir}/input_0.pb and output_0.pb")


if __name__ == "__main__":
main()
Loading