diff --git a/backends/samsung/builders/__init__.py b/backends/samsung/builders/__init__.py index cccaf5e4409..780ed787231 100644 --- a/backends/samsung/builders/__init__.py +++ b/backends/samsung/builders/__init__.py @@ -41,6 +41,7 @@ op_pad, op_permute, op_pixel_shuffle, + op_pixel_unshuffle, op_placeholder, op_pow, op_prelu, @@ -106,6 +107,7 @@ "op_mul", "op_permute", "op_pixel_shuffle", + "op_pixel_unshuffle", "op_placeholder", "op_pow", "op_prelu", diff --git a/backends/samsung/builders/op_pixel_unshuffle.py b/backends/samsung/builders/op_pixel_unshuffle.py new file mode 100644 index 00000000000..a16026e95e8 --- /dev/null +++ b/backends/samsung/builders/op_pixel_unshuffle.py @@ -0,0 +1,44 @@ +# Copyright (c) 2025 Samsung Electronics Co. LTD +# All rights reserved +# +# This source code is licensed under the BSD-style license found in the +# LICENSE file in the root directory of this source tree. +from typing import cast, Dict + +import torch +from executorch.backends.samsung.builders.node_visitor import ( + NodeVisitor, + register_node_visitor, +) +from executorch.backends.samsung.serialization.enn_graph_schema import EnnGraph +from executorch.backends.transforms import get_shape + + +@register_node_visitor +class PixelUnshuffleVisitor(NodeVisitor): + target = "aten.pixel_unshuffle.default" + + def __init__(self, *args) -> None: + super().__init__(*args) + + def define_node( + self, + node: torch.fx.Node, + enn_graph: EnnGraph, + vals_to_ids: Dict[torch.Tensor, int], + ) -> bool: + if len(get_shape(node.args[0])) != 4: + return False + + input_id = self.define_tensor(node.args[0], enn_graph, vals_to_ids) + + downscale_factor = cast(int, node.args[1]) + params = {"block_size": downscale_factor, "mode": "CRD"} + + output_id = self.define_tensor(node, enn_graph, vals_to_ids) + + enn_graph.define_op( + node.name, "SPACE_TO_DEPTH", [input_id], [output_id], params + ) + + return True diff --git a/backends/samsung/builders/op_rms_norm.py b/backends/samsung/builders/op_rms_norm.py index 0ff01701d1b..b36d7b35802 100644 --- a/backends/samsung/builders/op_rms_norm.py +++ b/backends/samsung/builders/op_rms_norm.py @@ -4,6 +4,7 @@ # This source code is licensed under the BSD-style license found in the # LICENSE file in the root directory of this source tree. +import logging from typing import cast, Dict, List import torch @@ -13,6 +14,7 @@ ) from executorch.backends.samsung.builders.utils import get_tensor from executorch.backends.samsung.serialization.enn_graph_schema import EnnGraph +from executorch.backends.transforms import get_shape @register_node_visitor @@ -31,6 +33,12 @@ def define_node( # input2 normalized_shape = cast(List[int], node.args[1]) + input_shape = get_shape(input) + if len(normalized_shape) != 1 or normalized_shape[0] != input_shape[-1]: + logging.warning( + "Currently, Enn backend only supports rms norm with last input dimension." + ) + return False gamma_node = node.args[2] gamma_id = self.define_tensor(gamma_node, enn_graph, vals_to_ids) @@ -44,6 +52,7 @@ def define_node( params["normalize_shape"] = normalized_shape params["param_num"] = 2 params["epsilon"] = epsilon + params["axis"] = [len(input_shape) - 1] output_id = self.define_tensor(node, enn_graph, vals_to_ids) diff --git a/backends/samsung/partition/enn_partitioner.py b/backends/samsung/partition/enn_partitioner.py index 3ca656db68e..a61187ae2ea 100644 --- a/backends/samsung/partition/enn_partitioner.py +++ b/backends/samsung/partition/enn_partitioner.py @@ -197,6 +197,7 @@ def ops_to_not_decompose( torch.ops.aten.prelu.default, torch.ops.aten.layer_norm.default, torch.ops.aten.pixel_shuffle.default, + torch.ops.aten.pixel_unshuffle.default, torch.ops.aten.hardsigmoid.default, torch.ops.aten.silu.default, torch.ops.aten.pad.default, diff --git a/backends/samsung/test/models/test_mobilebert_qat.py b/backends/samsung/test/models/test_mobilebert_qat.py index d871301ad78..5dc2ddbc538 100644 --- a/backends/samsung/test/models/test_mobilebert_qat.py +++ b/backends/samsung/test/models/test_mobilebert_qat.py @@ -96,10 +96,10 @@ def test_mobilebert_qat_a8w8(self): example_inputs[1].size(2), example_inputs[1].size(3), ) - vector_input_ids = torch.randint(0, 256, size_input_ids).to(device) + vector_input_ids = torch.randint(0, 256, size_input_ids, device=device) vector_attention_mask = torch.zeros( - size_attention_mask, dtype=torch.float32 - ).to(device) + size_attention_mask, dtype=torch.float32, device=device + ) export_inputs = ( vector_input_ids, vector_attention_mask, diff --git a/backends/samsung/test/ops/test_hardsigmoid.py b/backends/samsung/test/ops/test_hardsigmoid.py new file mode 100644 index 00000000000..2370b3d258a --- /dev/null +++ b/backends/samsung/test/ops/test_hardsigmoid.py @@ -0,0 +1,47 @@ +# Copyright (c) Samsung Electronics Co. LTD +# All rights reserved +# +# Licensed under the BSD License (the "License"); you may not use this file +# except in compliance with the License. See the license file in the root +# directory of this source tree for more details. + +import unittest + +import torch + +from executorch.backends.samsung.serialization.compile_options import ( + gen_samsung_backend_compile_spec, +) +from executorch.backends.samsung.test.tester import SamsungTester +from executorch.backends.samsung.test.utils.utils import TestConfig + + +class HardSigmoid(torch.nn.Module): + def __init__(self) -> None: + super().__init__() + self.module = torch.nn.Hardsigmoid() + + def forward(self, x: torch.Tensor) -> torch.Tensor: + return self.module(x) + + +class TestHardSigmoid(unittest.TestCase): + def _test(self, module: torch.nn.Module, inputs): + tester = SamsungTester( + module, + inputs, + [gen_samsung_backend_compile_spec(TestConfig.chipset)], + ) + ( + tester.export() + .check_count({"torch.ops.aten.hardsigmoid.default": 1}) + .to_edge_transform_and_lower() + .check_not(["executorch_exir_dialects_edge__ops_aten_hardsigmoid_default"]) + .check_count({"torch.ops.higher_order.executorch_call_delegate": 1}) + .to_executorch() + .run_method_and_compare_outputs(inputs=inputs) + ) + + def test_fp32_hard_sigmoid(self): + inputs = (torch.randn(1, 16, 32, 32),) + self._test(HardSigmoid(), inputs) diff --git a/backends/samsung/test/ops/test_hardswish.py b/backends/samsung/test/ops/test_hardswish.py new file mode 100644 index 00000000000..17afdd148bc --- /dev/null +++ b/backends/samsung/test/ops/test_hardswish.py @@ -0,0 +1,47 @@ +# Copyright (c) Samsung Electronics Co. LTD +# All rights reserved +# +# Licensed under the BSD License (the "License"); you may not use this file +# except in compliance with the License. See the license file in the root +# directory of this source tree for more details. + +import unittest + +import torch + +from executorch.backends.samsung.serialization.compile_options import ( + gen_samsung_backend_compile_spec, +) +from executorch.backends.samsung.test.tester import SamsungTester +from executorch.backends.samsung.test.utils.utils import TestConfig + + +class HardSwish(torch.nn.Module): + def __init__(self) -> None: + super().__init__() + self.module = torch.nn.Hardswish() + + def forward(self, x: torch.Tensor) -> torch.Tensor: + return self.module(x) + + +class TestHardSwish(unittest.TestCase): + def _test(self, module: torch.nn.Module, inputs): + tester = SamsungTester( + module, + inputs, + [gen_samsung_backend_compile_spec(TestConfig.chipset)], + ) + ( + tester.export() + .check_count({"torch.ops.aten.hardswish.default": 1}) + .to_edge_transform_and_lower() + .check_not(["executorch_exir_dialects_edge__ops_aten_hardswish_default"]) + .check_count({"torch.ops.higher_order.executorch_call_delegate": 1}) + .to_executorch() + .run_method_and_compare_outputs(inputs=inputs, atol=0.002) + ) + + def test_fp32_hard_swish(self): + inputs = (torch.randn(1, 16, 32, 32),) + self._test(HardSwish(), inputs) diff --git a/backends/samsung/test/ops/test_hardtanh.py b/backends/samsung/test/ops/test_hardtanh.py new file mode 100644 index 00000000000..44df1740af1 --- /dev/null +++ b/backends/samsung/test/ops/test_hardtanh.py @@ -0,0 +1,47 @@ +# Copyright (c) Samsung Electronics Co. LTD +# All rights reserved +# +# Licensed under the BSD License (the "License"); you may not use this file +# except in compliance with the License. See the license file in the root +# directory of this source tree for more details. + +import unittest + +import torch + +from executorch.backends.samsung.serialization.compile_options import ( + gen_samsung_backend_compile_spec, +) +from executorch.backends.samsung.test.tester import SamsungTester +from executorch.backends.samsung.test.utils.utils import TestConfig + + +class Hardtanh(torch.nn.Module): + def __init__(self, min_val, max_val) -> None: + super().__init__() + self.module = torch.nn.Hardtanh(min_val=min_val, max_val=max_val) + + def forward(self, x: torch.Tensor) -> torch.Tensor: + return self.module(x) + + +class TestHardtanh(unittest.TestCase): + def _test(self, module: torch.nn.Module, inputs): + tester = SamsungTester( + module, + inputs, + [gen_samsung_backend_compile_spec(TestConfig.chipset)], + ) + ( + tester.export() + .check_count({"torch.ops.aten.hardtanh.default": 1}) + .to_edge_transform_and_lower() + .check_not(["executorch_exir_dialects_edge__ops_aten_hardtanh_default"]) + .check_count({"torch.ops.higher_order.executorch_call_delegate": 1}) + .to_executorch() + .run_method_and_compare_outputs(inputs=inputs) + ) + + def test_fp32_hardtanh(self): + inputs = (torch.randn(1, 3, 16, 16),) + self._test(Hardtanh(min_val=0, max_val=1), inputs) diff --git a/backends/samsung/test/ops/test_layer_norm.py b/backends/samsung/test/ops/test_layer_norm.py new file mode 100644 index 00000000000..211902f2b9b --- /dev/null +++ b/backends/samsung/test/ops/test_layer_norm.py @@ -0,0 +1,47 @@ +# Copyright (c) Samsung Electronics Co. LTD +# All rights reserved +# +# Licensed under the BSD License (the "License"); you may not use this file +# except in compliance with the License. See the license file in the root +# directory of this source tree for more details. + +import unittest + +import torch + +from executorch.backends.samsung.serialization.compile_options import ( + gen_samsung_backend_compile_spec, +) +from executorch.backends.samsung.test.tester import SamsungTester +from executorch.backends.samsung.test.utils.utils import TestConfig + + +class LayerNorm(torch.nn.Module): + def __init__(self) -> None: + super().__init__() + self.module = torch.nn.LayerNorm(10) + + def forward(self, x: torch.Tensor) -> torch.Tensor: + return self.module(x) + + +class TestLayerNorm(unittest.TestCase): + def _test(self, module: torch.nn.Module, inputs): + tester = SamsungTester( + module, + inputs, + [gen_samsung_backend_compile_spec(TestConfig.chipset)], + ) + ( + tester.export() + .check_count({"torch.ops.aten.layer_norm.default": 1}) + .to_edge_transform_and_lower() + .check_not(["executorch_exir_dialects_edge__ops_aten_layer_norm_default"]) + .check_count({"torch.ops.higher_order.executorch_call_delegate": 1}) + .to_executorch() + .run_method_and_compare_outputs(inputs=inputs) + ) + + def test_fp32_layer_norm(self): + inputs = (torch.randn(1, 32, 10),) + self._test(LayerNorm(), inputs) diff --git a/backends/samsung/test/ops/test_maximum.py b/backends/samsung/test/ops/test_maximum.py new file mode 100644 index 00000000000..6eafebe358f --- /dev/null +++ b/backends/samsung/test/ops/test_maximum.py @@ -0,0 +1,46 @@ +# Copyright (c) Samsung Electronics Co. LTD +# All rights reserved +# +# Licensed under the BSD License (the "License"); you may not use this file +# except in compliance with the License. See the license file in the root +# directory of this source tree for more details. + +import unittest + +import torch + +from executorch.backends.samsung.serialization.compile_options import ( + gen_samsung_backend_compile_spec, +) +from executorch.backends.samsung.test.tester import SamsungTester +from executorch.backends.samsung.test.utils.utils import TestConfig + + +class Maximum(torch.nn.Module): + def __init__(self) -> None: + super().__init__() + + def forward(self, x: torch.Tensor, y: torch.Tensor) -> torch.Tensor: + return torch.maximum(x, y) + + +class TestMaximum(unittest.TestCase): + def _test(self, module: torch.nn.Module, inputs): + tester = SamsungTester( + module, + inputs, + [gen_samsung_backend_compile_spec(TestConfig.chipset)], + ) + ( + tester.export() + .check_count({"torch.ops.aten.maximum.default": 1}) + .to_edge_transform_and_lower() + .check_not(["executorch_exir_dialects_edge__ops_aten_maximum_default"]) + .check_count({"torch.ops.higher_order.executorch_call_delegate": 1}) + .to_executorch() + .run_method_and_compare_outputs(inputs=inputs) + ) + + def test_fp32_maximum(self): + inputs = (torch.randn(1, 3, 56, 56), torch.randn(1, 3, 56, 56)) + self._test(Maximum(), inputs) diff --git a/backends/samsung/test/ops/test_pixel_unshuffle.py b/backends/samsung/test/ops/test_pixel_unshuffle.py new file mode 100644 index 00000000000..b010b38d8d5 --- /dev/null +++ b/backends/samsung/test/ops/test_pixel_unshuffle.py @@ -0,0 +1,50 @@ +# Copyright (c) Samsung Electronics Co. LTD +# All rights reserved +# +# Licensed under the BSD License (the "License"); you may not use this file +# except in compliance with the License. See the license file in the root +# directory of this source tree for more details. + + +import unittest + +import torch + +from executorch.backends.samsung.serialization.compile_options import ( + gen_samsung_backend_compile_spec, +) +from executorch.backends.samsung.test.tester import SamsungTester +from executorch.backends.samsung.test.utils.utils import TestConfig + + +class PixelUnshuffle(torch.nn.Module): + def __init__(self) -> None: + super().__init__() + self.module = torch.nn.PixelUnshuffle(downscale_factor=2) + + def forward(self, x: torch.Tensor) -> torch.Tensor: + return self.module(x) + + +class TestPixelUnshuffle(unittest.TestCase): + def _test(self, module: torch.nn.Module, inputs): + tester = SamsungTester( + module, + inputs, + [gen_samsung_backend_compile_spec(TestConfig.chipset)], + ) + ( + tester.export() + .check_count({"torch.ops.aten.pixel_unshuffle.default": 1}) + .to_edge_transform_and_lower() + .check_not( + ["executorch_exir_dialects_edge__ops_aten_pixel_unshuffle_default"] + ) + .check_count({"torch.ops.higher_order.executorch_call_delegate": 1}) + .to_executorch() + .run_method_and_compare_outputs(inputs=inputs) + ) + + def test_fp32_pixel_unshuffle(self): + inputs = (torch.randn(1, 4, 8, 8),) + self._test(PixelUnshuffle(), inputs) diff --git a/backends/samsung/test/ops/test_prelu.py b/backends/samsung/test/ops/test_prelu.py new file mode 100644 index 00000000000..ace781a6333 --- /dev/null +++ b/backends/samsung/test/ops/test_prelu.py @@ -0,0 +1,51 @@ +# Copyright (c) Samsung Electronics Co. LTD +# All rights reserved +# +# Licensed under the BSD License (the "License"); you may not use this file +# except in compliance with the License. See the license file in the root +# directory of this source tree for more details. + +import unittest + +import torch + +from executorch.backends.samsung.serialization.compile_options import ( + gen_samsung_backend_compile_spec, +) +from executorch.backends.samsung.test.tester import SamsungTester +from executorch.backends.samsung.test.utils.utils import TestConfig + + +class PReLU(torch.nn.Module): + def __init__(self, input_channel=3, per_channel=False) -> None: + super().__init__() + self.module = torch.nn.PReLU(input_channel if per_channel else 1) + + def forward(self, x: torch.Tensor) -> torch.Tensor: + return self.module(x) + + +class TestPReLU(unittest.TestCase): + def _test(self, module: torch.nn.Module, inputs): + tester = SamsungTester( + module, + inputs, + [gen_samsung_backend_compile_spec(TestConfig.chipset)], + ) + ( + tester.export() + .check_count({"torch.ops.aten.prelu.default": 1}) + .to_edge_transform_and_lower() + .check_not(["executorch_exir_dialects_edge__ops_aten_prelu_default"]) + .check_count({"torch.ops.higher_order.executorch_call_delegate": 1}) + .to_executorch() + .run_method_and_compare_outputs(inputs=inputs) + ) + + def test_fp32_prelu(self): + inputs = (torch.randn(1, 3, 56, 56),) + self._test(PReLU(per_channel=False), inputs) + + def test_fp32_prelu_per_channel(self): + inputs = (torch.randn(1, 3, 56, 56),) + self._test(PReLU(input_channel=3, per_channel=True), inputs) diff --git a/backends/samsung/utils/export_utils.py b/backends/samsung/utils/export_utils.py index 86512b75dec..55dfcfd7d11 100644 --- a/backends/samsung/utils/export_utils.py +++ b/backends/samsung/utils/export_utils.py @@ -33,6 +33,7 @@ def get_edge_compile_config(): exir_ops.edge.aten.hardswish.default, exir_ops.edge.aten.prelu.default, exir_ops.edge.aten.pixel_shuffle.default, + exir_ops.edge.aten.pixel_unshuffle.default, exir_ops.edge.aten._safe_softmax.default, exir_ops.edge.aten.layer_norm.default, exir_ops.edge.aten.matmul.default, diff --git a/examples/samsung/scripts/mobilebert_finetune_QAT.py b/examples/samsung/scripts/mobilebert_finetune_QAT.py index 584388b2107..40954900feb 100644 --- a/examples/samsung/scripts/mobilebert_finetune_QAT.py +++ b/examples/samsung/scripts/mobilebert_finetune_QAT.py @@ -426,9 +426,9 @@ def build_aten_to_qat_mobilebert( inputs[1].size(2), inputs[1].size(3), ) - vector_input_ids = torch.randint(0, 256, size_input_ids).to(device) - vector_attention_mask = torch.zeros(size_attention_mask, dtype=torch.float32).to( - device + vector_input_ids = torch.randint(0, 256, size_input_ids, device=device) + vector_attention_mask = torch.zeros( + size_attention_mask, dtype=torch.float32, device=device ) example_inputs = ( vector_input_ids, @@ -483,9 +483,9 @@ def build_aten_to_qat_mobilebert( inputs[1].size(2), inputs[1].size(3), ) - vector_input_ids = torch.randint(0, 256, size_input_ids).to(device) - vector_attention_mask = torch.zeros(size_attention_mask, dtype=torch.float32).to( - device + vector_input_ids = torch.randint(0, 256, size_input_ids, device=device) + vector_attention_mask = torch.zeros( + size_attention_mask, dtype=torch.float32, device=device ) example_inputs = ( vector_input_ids, @@ -500,10 +500,10 @@ def build_aten_to_qat_mobilebert( device_cpu = torch.device(type="cpu") quantized_model = quantized_model.to(device_cpu) quantized_model = removing_gpu_node_in_graph(quantized_model) - cpu_vector_input_ids = torch.randint(0, 256, size_input_ids).to(device_cpu) + cpu_vector_input_ids = torch.randint(0, 256, size_input_ids, device=device_cpu) cpu_vector_attention_mask = torch.zeros( - size_attention_mask, dtype=torch.float32 - ).to(device_cpu) + size_attention_mask, dtype=torch.float32, device=device_cpu + ) example_inputs_cpu = ( cpu_vector_input_ids, cpu_vector_attention_mask,