Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion backends/nxp/edge_passes/neutron_edge_pass_manager.py
Original file line number Diff line number Diff line change
Expand Up @@ -17,7 +17,7 @@
from executorch.backends.nxp.edge_passes.remove_as_strided_copy_nodes import (
RemoveUselessAsStridedCopyNodes,
)
from torch.fx.passes.infra.pass_manager import PassManager
from executorch.exir.pass_manager import PassManager


class NeutronEdgePassManager(PassManager):
Expand Down
374 changes: 374 additions & 0 deletions backends/nxp/recipes/nxp_recipe_provider.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,374 @@
# Copyright 2026 NXP
#
# This source code is licensed under the BSD-style license found in the
# LICENSE file in the root directory of this source tree.

import logging
from dataclasses import dataclass
from functools import partial
from typing import Any, Callable, cast, Iterable, Optional, Sequence

import torch

from executorch.backends.nxp.aten_passes.fuse_batch_norm_with_linear_pass import (
FuseBatchNormWithLinearPass,
)
from executorch.backends.nxp.aten_passes.simulated_linear_bn_fusion_passes import (
AddSimulatedLinearBatchNormFusionQATPass,
RemoveSimulatedLinearBatchNormFusionQATPass,
)
from executorch.backends.nxp.backend.custom_delegation_options import (
CustomDelegationOptions,
)
from executorch.backends.nxp.backend.neutron_target_spec import NeutronTargetSpec
from executorch.backends.nxp.edge_passes.neutron_edge_pass import NeutronEdgePass
from executorch.backends.nxp.edge_passes.neutron_edge_pass_manager import (
NeutronEdgePassManager,
)
from executorch.backends.nxp.edge_passes.remove_additional_quantize_dequantize_nodes_pass import (
RemoveAdditionalQDQClustersPass,
)
from executorch.backends.nxp.edge_passes.remove_io_quant_ops_pass import (
RemoveIOQuantOpsPass,
)
from executorch.backends.nxp.neutron_partitioner import NeutronPartitioner
from executorch.backends.nxp.nxp_backend import (
core_aten_ops_exception_list,
generate_neutron_compile_spec,
)
from executorch.backends.nxp.recipes.nxp_recipe_types import NXP_BACKEND, NXPRecipeType
from executorch.backends.nxp.tests.executorch_pipeline import (
_get_default_quantizer,

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

nit: From my coding experience in Python, methods with _ prefix are meant to be "private" and used only in the class/module they are defined or implemented in. Thus I think it would be better to rename the method in executorch_pipeline to get_default_quantizer without the prefix.

get_random_calibration_inputs,
GetCalibrationInputsFn,
handle_kernel_selection,
ModelInputSpec,
to_model_input_spec,
)
from executorch.backends.transforms.quantize_fused_convbn_bias_pass import (
QuantizeFusedConvBnBiasAtenPass,
)
from executorch.exir import EdgeCompileConfig, EdgeProgramManager, ExportedProgram

from executorch.exir.backend.compile_spec_schema import CompileSpec
from executorch.exir.backend.partitioner import Partitioner
from executorch.export import (
BackendRecipeProvider,
ExportRecipe,
LoweringRecipe,
QuantizationRecipe,
RecipeType,
)
from torchao.quantization.pt2e.quantizer import Quantizer


class NeutronEdgePassManagerWrapper:
def __init__(self, passes: list[NeutronEdgePass] | None = None):
self.neutron_edge_pass_manager = NeutronEdgePassManager(passes)

def __call__(self, s: str, epm: EdgeProgramManager) -> NeutronEdgePassManager:
return self.neutron_edge_pass_manager
Comment on lines +69 to +70

@novak-vaclav novak-vaclav Aug 4, 2026

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I think this is applicable (the Copilot's comment)



NEUTRON_RECIPE_CONFIG_KEY = "neutron_recipe_config"


@dataclass
class NeutronRecipeConfig:
"""Configuration shared by all NXP recipe types.

Parameters that vary the *type* of export (PTQ vs QAT, delegate vs no-delegate)
are expressed by choosing a different NXPRecipeType rather than by flags here.

Attributes:
input_spec: Model input description. Accepts a single shape tuple, a list of
shape tuples (one per input), or a list of ModelInputSpec objects.
target: Neutron hardware target string. Default: "imxrt700".
operators_not_to_delegate: Optional list of op names excluded from NPU delegation.

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

nit: I would add an example of entry in the operators_not_to_delegate list, like "aten::add".

intermediates_dir: Optional directory to dump intermediate compilation artefacts.
get_quantizer_fn: Optional factory that returns a custom Quantizer. When None,
the default NeutronQuantizer is used.
get_calibration_inputs_fn: Optional function that, given the input_spec, returns
calibration input samples. When None, random inputs are
used.
train_fn: QAT training callback. Required when using INT8_QAT_NEUTRON.
custom_delegation_options: Optional fine-grained control over which ops are
delegated. Default: CustomDelegationOptions().
remove_quant_io_ops: If True, remove quantize/dequantize ops at the IO boundary
(useful for integer-IO deployments).
use_quant_state_dict: If False, the post-quantization parameter values are not
passed to NeutronPartitioner.
use_neutron_for_format_conversion: Whether Neutron handles data-format conversion.
fetch_constants_to_sram: Place constant tensors in SRAM on the target.
dump_kernel_selection_code: Emit kernel-selection files after compilation.
use_profiling: Enable execution profiling / ETRecord generation.
"""

input_spec: Iterable[ModelInputSpec] | tuple[int, ...] | list[tuple[int, ...]]
target: str = "imxrt700"
operators_not_to_delegate: list[str] = None

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

nit: the type of operators_not_to_delegate should be list[str] | None

intermediates_dir: str | None = None
Comment on lines +107 to +110

@novak-vaclav novak-vaclav Aug 4, 2026

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Applicable (the Copilot's comment).

get_quantizer_fn: Callable[[], Quantizer] | None = None
get_calibration_inputs_fn: GetCalibrationInputsFn | None = None
train_fn: Callable[[torch.fx.GraphModule], None] | None = None
custom_delegation_options: CustomDelegationOptions | None = None
remove_quant_io_ops: bool = False
use_quant_state_dict: bool = True
use_neutron_for_format_conversion: bool = True
fetch_constants_to_sram: bool = False
dump_kernel_selection_code: bool = False
use_profiling: bool = False


class NXPRecipeProvider(BackendRecipeProvider):

@property
def backend_name(self) -> str:
return NXP_BACKEND

def get_supported_recipes(self) -> Sequence[RecipeType]:
return list(NXPRecipeType)

def create_recipe(
self, recipe_type: RecipeType, **kwargs: Any
) -> Optional[ExportRecipe]:
if recipe_type not in self.get_supported_recipes():
logging.warning(f"NXP backend: Recipe `{recipe_type}` is not valid.")
return None

rc = cast(NeutronRecipeConfig, kwargs.get(NEUTRON_RECIPE_CONFIG_KEY))
if rc is None:
raise KeyError(
f"NXP backend: create_recipe() requires `{NEUTRON_RECIPE_CONFIG_KEY}=<NeutronRecipeConfig>`."
)

if rc.custom_delegation_options is None:

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I'm not a big fan of modifying the config that is passed to the recipe creator, I view it as an anti-pattern. Additionally, it might have unwanted consequences if someone decides to reuse the config somewhere else. Same goes for line 177, where the issue is even more severe.
If you decide to keep it anyway, please add a doc string to this method with a mention about this behavior.

rc.custom_delegation_options = CustomDelegationOptions()

rc.input_spec = to_model_input_spec(rc.input_spec)

match recipe_type:
case NXPRecipeType.INT8_PTQ_NEUTRON:
return self._build_recipe(recipe_type, rc, is_qat=False, delegate=True)
case NXPRecipeType.INT8_QAT_NEUTRON:
if rc.train_fn is None:

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Is the train_fn really needed to be required? In calibrate_and_quantize() we don't require it. In _build_quantization_recipe() you reference it "or after prepare when no train_fn".

raise ValueError(
"NXP backend: INT8_QAT_NEUTRON requires train_fn in NeutronRecipeConfig."
)
return self._build_recipe(recipe_type, rc, is_qat=True, delegate=True)
case NXPRecipeType.INT8_PTQ_NO_DELEGATE:
return self._build_recipe(recipe_type, rc, is_qat=False, delegate=False)
case _:
raise NotImplementedError(
f"NXP backend: Recipe `{recipe_type}` is not supported."
)

def _build_recipe(
self,
recipe_type: NXPRecipeType,
rc: NeutronRecipeConfig,
*,
is_qat: bool,
delegate: bool,
) -> ExportRecipe:
neutron_target_spec = NeutronTargetSpec(rc.target)

if rc.get_quantizer_fn is None:
rc.get_quantizer_fn = partial(
_get_default_quantizer, neutron_target_spec, is_qat
)

quantization_recipe = _build_quantization_recipe(rc, is_qat)
compile_spec = generate_neutron_compile_spec(
rc.target,
intermediates_dir=rc.intermediates_dir,
operators_not_to_delegate=rc.operators_not_to_delegate,
use_neutron_for_format_conversion=rc.use_neutron_for_format_conversion,
fetch_constants_to_sram=rc.fetch_constants_to_sram,
dump_kernel_selection_code=rc.dump_kernel_selection_code,
use_profiling=rc.use_profiling,
)
lowering_recipe = _build_lowering_recipe(
compile_spec, neutron_target_spec, rc, delegate=delegate
)

return ExportRecipe(
name=recipe_type.value,
quantization_recipe=quantization_recipe,
lowering_recipe=lowering_recipe,
)


# ---------------------------------------------------------------------------
# Module-level builder helpers
# ---------------------------------------------------------------------------


def _build_quantization_recipe(
rc: NeutronRecipeConfig, is_qat: bool
) -> QuantizationRecipe:
"""Build the QuantizationRecipe for PTQ or QAT.

PTQ uses the standard QuantizeStage default flow (prepare_pt2e -> calibrate ->
convert_pt2e) with calibration_inputs_fn supplying the calibration samples.

QAT uses the same QuantizeStage flow with is_qat=True (prepare_qat_pt2e) and
the NXP-specific BN-fusion passes distributed across the four pass-list hooks:
- post_prepare_passes: AddSimulatedLinearBatchNormFusionQATPass
- post_train_passes: RemoveSimulatedLinearBatchNormFusionQATPass,
FuseBatchNormWithLinearPass
- post_calibration_passes: RemoveSimulatedLinearBatchNormFusionQATPass,
FuseBatchNormWithLinearPass (applied again after
observer-only calibration when train_fn is None)
- post_convert_passes: QuantizeFusedConvBnBiasAtenPass
"""
_input_spec = rc.input_spec
_calibration_fn = rc.get_calibration_inputs_fn or get_random_calibration_inputs
_quantizer = rc.get_quantizer_fn()

def _calibration_inputs_fn() -> Iterable:
return _calibration_fn(_input_spec)

if not is_qat:
return QuantizationRecipe(
quantizers=[_quantizer],
calibration_inputs_fn=_calibration_inputs_fn,
)

# QAT pass hooks -- keep NXP-specific pass classes out of shared modules
def _post_prepare(m: torch.fx.GraphModule) -> torch.fx.GraphModule:
return AddSimulatedLinearBatchNormFusionQATPass()(m).graph_module

def _remove_simulated_bn_and_fuse(m: torch.fx.GraphModule) -> torch.fx.GraphModule:
m = RemoveSimulatedLinearBatchNormFusionQATPass()(m).graph_module
return FuseBatchNormWithLinearPass()(m).graph_module

def _post_convert(m: torch.fx.GraphModule) -> torch.fx.GraphModule:
return QuantizeFusedConvBnBiasAtenPass(
default_zero_bias=False, symmetric_quant=True
)(m).graph_module

return QuantizationRecipe(
quantizers=[_quantizer],
calibration_inputs_fn=_calibration_inputs_fn,
is_qat=True,
train_fn=rc.train_fn,
post_prepare_passes=[_post_prepare],
# Applied after training (or after prepare when no train_fn).
post_train_passes=[_remove_simulated_bn_and_fuse],
# Applied after calibration (only reached when train_fn is None).

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Are you sure post_calibration_passes are applied only when train_fn is None?
in stages.py on line 512, I see post_calibration_passes are always run.

post_calibration_passes=[_remove_simulated_bn_and_fuse],
post_convert_passes=[_post_convert],
)


def _build_lowering_recipe(
compile_spec: list[CompileSpec],
neutron_target_spec: NeutronTargetSpec,
rc: NeutronRecipeConfig,
*,
delegate: bool,
) -> LoweringRecipe:
"""Build the LoweringRecipe, optionally including NPU delegation."""
partitioners = _build_partitioners(compile_spec, neutron_target_spec, rc, delegate)
pre_partitioning_callback = _build_pre_partitioning_callback(rc)
post_partitioning_transforms = _build_post_partitioning_transforms(rc)

# The edge pass manager must be wrapped: EdgeTransformAndLowerStage calls
# edge_transform_passes with (method_name, ep) and expects a PassManager back.
return LoweringRecipe(
partitioners=partitioners,
edge_transform_passes=[NeutronEdgePassManagerWrapper()],
edge_compile_config=EdgeCompileConfig(
_check_ir_validity=False,
_core_aten_ops_exception_list=core_aten_ops_exception_list,
),
pre_partitioning_callback=pre_partitioning_callback,
post_partitioning_transforms=post_partitioning_transforms,
)


def _build_partitioners(
compile_spec: list[CompileSpec],
neutron_target_spec: NeutronTargetSpec,
rc: NeutronRecipeConfig,
delegate: bool,
) -> list:
"""Create the NeutronPartitioner list. Empty when delegate=False."""
if not delegate:
return []
return [
NeutronPartitioner(
compile_spec,
neutron_target_spec,
rc.custom_delegation_options,
preserve_ops=[torch.ops.aten.prelu.default],

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

This comment might not be directly related to this PR, but I think it's time to refactor the preserve_ops list we now use in multiple different calls (afaik it's here and in executorch_pipeline).
I'm saying this because the logic of preserving ops became more complicated and conditional in the aten.pad PR, and also Roman added aten.hardswish in his PR.
I suggest modifying NeutronPartitioner to set preserve_ops to [prelu, hardswish, pad] as default or extracting [prelu, hardswish, pad] to some global variable.
There is also a core_aten_ops_exception_list variable in our backend, which seems to do something similar to preserve_ops.
If you don't want to solve it in this PR, just add aten.pad and aten.harswish here and we will tackle the refactoring in another issue.

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

It should be unified as I commented too. We have it also in aot_neutron_example.py.

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

We should place the preserve_ops list in one place, I suggest a global variable in executor_pipeline.py.

)
]


def _build_pre_partitioning_callback(rc: NeutronRecipeConfig):
"""Return a callback that assigns the post-quantization state_dict to NeutronPartitioner.

NeutronPartitioner requires static parameter data. Since the partitioner is instantiated
during recipe creation (before model data is available), assignment is deferred to a
callback invoked just before partitioning.
"""
_use_quant_state_dict = rc.use_quant_state_dict

def _callback(
_partitioners: list[Partitioner] | None,
programs: dict[str, ExportedProgram],
) -> None:
if _partitioners is None:
return

if _use_quant_state_dict:
post_quant_state_dict: dict | None = {}
for _, program in programs.items():
post_quant_state_dict.update(program.state_dict)
else:
post_quant_state_dict = None

for _partitioner in _partitioners:
if isinstance(_partitioner, NeutronPartitioner):
_partitioner.post_quantization_state_dict = post_quant_state_dict

return _callback


def _build_post_partitioning_transforms(rc: NeutronRecipeConfig) -> list:
"""Build the list of post-partitioning EdgeProgramManager transforms.

These mirror what the imperative pipeline did after to_edge_transform_and_lower:
- RemoveIOQuantOpsPass (optional, when remove_quant_io_ops=True)
- RemoveAdditionalQDQClustersPass (always applied)
- handle_kernel_selection side-effect (optional, when dump_kernel_selection_code=True)
"""
transforms = []

if rc.remove_quant_io_ops:

def _remove_io_quant_ops(epm: EdgeProgramManager) -> EdgeProgramManager:
return epm.transform([RemoveIOQuantOpsPass(edge_program_manager=epm)])

transforms.append(_remove_io_quant_ops)

def _remove_additional_qdq_clusters(epm: EdgeProgramManager) -> EdgeProgramManager:
return epm.transform(
NeutronEdgePassManager([RemoveAdditionalQDQClustersPass()])
)

transforms.append(_remove_additional_qdq_clusters)

if rc.dump_kernel_selection_code:

def _handle_kernel_selection_transform(
epm: EdgeProgramManager,
) -> EdgeProgramManager:
handle_kernel_selection()
return epm

transforms.append(_handle_kernel_selection_transform)

return transforms
Loading
Loading