diff --git a/.ci/docker/ci_commit_pins/pytorch.txt b/.ci/docker/ci_commit_pins/pytorch.txt index 401a0594d98..7a1fe71d429 100644 --- a/.ci/docker/ci_commit_pins/pytorch.txt +++ b/.ci/docker/ci_commit_pins/pytorch.txt @@ -1 +1 @@ -release/2.13 +69985f4c2d95401063ec5739934ba09d9aae3792 diff --git a/backends/vulkan/test/op_tests/CMakeLists.txt b/backends/vulkan/test/op_tests/CMakeLists.txt index 0f8456accf5..59a38b1ce50 100644 --- a/backends/vulkan/test/op_tests/CMakeLists.txt +++ b/backends/vulkan/test/op_tests/CMakeLists.txt @@ -72,6 +72,7 @@ function(vulkan_op_test test_name test_src) set(extra_deps ${ARGN}) add_executable(${test_name} ${test_src}) + target_compile_features(${test_name} PRIVATE cxx_std_20) target_include_directories(${test_name} PRIVATE ${COMMON_INCLUDES}) target_link_libraries( ${test_name} @@ -89,6 +90,7 @@ endfunction() if(TARGET vulkan_backend AND LIB_TORCH) add_library(test_utils ${CMAKE_CURRENT_SOURCE_DIR}/test_utils.cpp) + target_compile_features(test_utils PRIVATE cxx_std_20) target_include_directories(test_utils PRIVATE ${COMMON_INCLUDES}) target_link_libraries( test_utils PRIVATE vulkan_backend ${LIB_TORCH} ${LIB_TORCH_CPU} diff --git a/backends/xnnpack/operators/quant_params.py b/backends/xnnpack/operators/quant_params.py index 6a3acd0df39..224d6b92431 100644 --- a/backends/xnnpack/operators/quant_params.py +++ b/backends/xnnpack/operators/quant_params.py @@ -163,6 +163,26 @@ def __str__(self) -> str: f")" ) + def matches(self, other: QuantParams) -> bool: + def values_match(lhs, rhs) -> bool: + if isinstance(lhs, torch.Tensor) and isinstance(rhs, torch.Tensor): + return torch.equal(lhs, rhs) + return bool(lhs == rhs) + + return ( + self.per_channel == other.per_channel + and self.per_channel_group == other.per_channel_group + and values_match(self.scale, other.scale) + and values_match(self.zp, other.zp) + and self.axis == other.axis + and self.dtype == other.dtype + and self.qmin == other.qmin + and self.qmax == other.qmax + and self.is_dynamic == other.is_dynamic + and self.num_nonbatch_dims == other.num_nonbatch_dims + and self.group_size == other.group_size + ) + def quantize_tensor(self, tensor: torch.Tensor) -> torch.Tensor: # Do nothing if already quantized by the Quantizer if tensor.dtype == self.dtype: diff --git a/backends/xnnpack/partition/config/generic_node_configs.py b/backends/xnnpack/partition/config/generic_node_configs.py index c7a3be5f65f..623d1a298de 100644 --- a/backends/xnnpack/partition/config/generic_node_configs.py +++ b/backends/xnnpack/partition/config/generic_node_configs.py @@ -12,6 +12,7 @@ import numpy as np import torch +from executorch.backends.xnnpack.operators.quant_params import QuantParams from executorch.backends.xnnpack.partition.config.xnnpack_config import ( ConfigPrecisionType, XNNPartitionerConfig, @@ -418,6 +419,21 @@ def check_constraints(self, node: torch.fx.Node, ep: ExportedProgram) -> bool: if not self.check_common_constraints(node, ep): return False + input_node = get_input_node(node, 0) + output_node = next(iter(node.users), None) + if is_dequant(input_node) and output_node is not None and is_quant(output_node): + input_quant_params = QuantParams.from_q_dq_node(input_node, ep) + output_quant_params = QuantParams.from_q_dq_node(output_node, ep) + if not input_quant_params.matches(output_quant_params): + why( + node, + reason=( + "XNNPACK static reshape requires matching input and " + "output quantization parameters." + ), + ) + return False + new_shape = node.args[1] # Check for symbolic dims. They aren't lowerable to XNNPACK currently. diff --git a/backends/xnnpack/test/test_xnnpack_partitioner.py b/backends/xnnpack/test/test_xnnpack_partitioner.py index 894fab4098f..36f317014f7 100644 --- a/backends/xnnpack/test/test_xnnpack_partitioner.py +++ b/backends/xnnpack/test/test_xnnpack_partitioner.py @@ -12,11 +12,17 @@ import torch.nn.functional as F from executorch.backends.xnnpack.partition.xnnpack_partitioner import XnnpackPartitioner -from executorch.exir import to_edge, to_edge_transform_and_lower +from executorch.backends.xnnpack.quantizer.xnnpack_quantizer import ( + get_symmetric_quantization_config, + XNNPACKQuantizer, +) +from executorch.backends.xnnpack.utils.configs import get_transform_passes +from executorch.exir import EdgeCompileConfig, to_edge, to_edge_transform_and_lower from executorch.extension.pybindings.portable_lib import ( _load_for_executorch_from_buffer, ) from torch.export import export +from torchao.quantization.pt2e.quantize_pt2e import convert_pt2e, prepare_pt2e class TestXnnpackPartitioner(unittest.TestCase): @@ -88,6 +94,62 @@ def test_no_warning_for_to_edge_transform_and_lower_workflow(self): log_contents = log_capture_string.getvalue() self.assertNotIn("DEPRECATION WARNING", log_contents) + def test_quantized_view_with_mismatched_qparams_stays_portable(self): + class ConvPoolFlattenLinear(torch.nn.Module): + def __init__(self): + super().__init__() + self.conv = torch.nn.Conv2d(3, 4, 3) + self.linear = torch.nn.Linear(4, 2) + + def forward(self, x): + x = self.conv(x) + x = F.adaptive_avg_pool2d(x, (1, 1)) + return self.linear(torch.flatten(x, 1)) + + inputs = (torch.randn(1, 3, 8, 8),) + exported = export(ConvPoolFlattenLinear().eval(), inputs, strict=False) + quantizer = XNNPACKQuantizer().set_global( + get_symmetric_quantization_config(is_per_channel=False) + ) + prepared = prepare_pt2e(exported.module(), quantizer) + prepared(*inputs) + converted = convert_pt2e(prepared) + + output_quant = next( + node + for node in converted.graph.nodes + if "quantize_per_tensor" in str(node.target) + and any("flatten" in str(inp.target) for inp in node.all_input_nodes) + ) + output_scale = output_quant.args[1] * 2 + output_quant.args = ( + output_quant.args[0], + output_scale, + *output_quant.args[2:], + ) + output_dequant = next(iter(output_quant.users)) + output_dequant.args = ( + output_dequant.args[0], + output_scale, + *output_dequant.args[2:], + ) + converted.recompile() + + lowered = to_edge_transform_and_lower( + export(converted, inputs, strict=False), + partitioner=[XnnpackPartitioner()], + transform_passes=get_transform_passes(), + compile_config=EdgeCompileConfig( + _check_ir_validity=False, _skip_dim_order=True + ), + ) + self.assertTrue( + any( + "view_copy" in str(node.target) + for node in lowered.exported_program().graph.nodes + ) + ) + def test_multi_method_partitioning_with_shared_weights(self): """ Test that multi-method models with shared weights are correctly partitioned. diff --git a/runtime/core/portable_type/c10/c10/util/complex.h b/runtime/core/portable_type/c10/c10/util/complex.h index 4e699684bc3..be99e270aa1 100644 --- a/runtime/core/portable_type/c10/c10/util/complex.h +++ b/runtime/core/portable_type/c10/c10/util/complex.h @@ -31,19 +31,11 @@ C10_HOST_DEVICE T abs(const c10::complex& z) { #endif } -#if defined(USE_ROCM) -#define ROCm_Bug(x) -#else -#define ROCm_Bug(x) x -#endif - template C10_HOST_DEVICE T arg(const c10::complex& z) { - return ROCm_Bug(std)::atan2(std::imag(z), std::real(z)); + return std::atan2(std::imag(z), std::real(z)); } -#undef ROCm_Bug - template constexpr T norm(const c10::complex& z) { return z.real() * z.real() + z.imag() * z.imag(); diff --git a/runtime/core/portable_type/c10/c10/util/llvmMathExtras.h b/runtime/core/portable_type/c10/c10/util/llvmMathExtras.h index da297241449..8ae5cde4f02 100644 --- a/runtime/core/portable_type/c10/c10/util/llvmMathExtras.h +++ b/runtime/core/portable_type/c10/c10/util/llvmMathExtras.h @@ -400,6 +400,7 @@ constexpr inline bool isShiftedUInt(uint64_t x) { N + S <= 64, "isShiftedUInt with N + S > 64 is too wide."); // Per the two static_asserts above, S must be strictly less than 64. So // 1 << S is not undefined behavior. + // NOLINTNEXTLINE(bugprone-chained-comparison) return isUInt(x) && (x % (UINT64_C(1) << S) == 0); } diff --git a/runtime/core/portable_type/c10/c10/util/overflows.h b/runtime/core/portable_type/c10/c10/util/overflows.h index 183a2f62a32..348ee5ffa40 100644 --- a/runtime/core/portable_type/c10/c10/util/overflows.h +++ b/runtime/core/portable_type/c10/c10/util/overflows.h @@ -61,13 +61,28 @@ template std::enable_if_t, bool> overflows( From f, bool strict_unsigned [[maybe_unused]] = false) { - using limit = std::numeric_limits::type>; + using ToScalar = typename scalar_value_type::type; + using limit = std::numeric_limits; if (limit::has_infinity && std::isinf(static_cast(f))) { return false; } if (!limit::has_quiet_NaN && (f != f)) { return true; } + if constexpr (std::is_integral_v) { + // limit::max() for wide integer types is NOT exactly representable in + // floating point (e.g. int64 max = 2^63-1 rounds up to 2^63), so `f > + // limit::max()` lets a just-out-of-range value like 2^63 slip through and + // then become INT64_MIN via static_cast. Compare against the + // exactly-representable upper bound max()+1 == 2^digits instead. lowest() + // is 0 or a negated power of two, so it stays exact. (digits-1 keeps the + // shift < 64 for the uint64 case; the *2 recovers 2^digits without a 1<<64 + // overflow.) + constexpr int digits = limit::digits; + constexpr From upper = + static_cast(uint64_t{1} << (digits - 1)) * From{2}; + return f < static_cast(limit::lowest()) || f >= upper; + } return f < limit::lowest() || f > limit::max(); } diff --git a/runtime/core/portable_type/c10/torch/headeronly/macros/Macros.h b/runtime/core/portable_type/c10/torch/headeronly/macros/Macros.h index cef99df3f56..08c4e9f1f84 100644 --- a/runtime/core/portable_type/c10/torch/headeronly/macros/Macros.h +++ b/runtime/core/portable_type/c10/torch/headeronly/macros/Macros.h @@ -123,6 +123,15 @@ #define C10_HAS_CPP_ATTRIBUTE(x) (0) #endif +/// Bind a returned reference/pointer's lifetime to a parameter (or *this) so +/// Clang can warn when it would dangle. Expands to nothing on compilers that +/// lack the attribute (e.g. non-clang, older nvcc). +#if C10_HAS_CPP_ATTRIBUTE(clang::lifetimebound) +#define C10_LIFETIMEBOUND [[clang::lifetimebound]] +#else +#define C10_LIFETIMEBOUND +#endif + #ifndef FBCODE_CAFFE2 /// DEPRECATED: Warn if a type or return value is discarded. #define C10_NODISCARD [[nodiscard]] diff --git a/runtime/core/portable_type/c10/torch/headeronly/util/Half.h b/runtime/core/portable_type/c10/torch/headeronly/util/Half.h index e5aa622656c..401472357ec 100644 --- a/runtime/core/portable_type/c10/torch/headeronly/util/Half.h +++ b/runtime/core/portable_type/c10/torch/headeronly/util/Half.h @@ -213,7 +213,7 @@ C10_HOST_DEVICE inline float fp16_ieee_to_fp32_value(uint16_t h) { * Now, remember that denormalized half-precision numbers are represented as: * FP16 = mantissa * 2**(-24). * The trick is to construct a normalized single-precision number with the - * same mantissa and thehalf-precision input and with an exponent which would + * same mantissa and the half-precision input and with an exponent which would * scale the corresponding mantissa bits to 2**(-24). A normalized * single-precision floating-point number is represented as: FP32 = (1 + * mantissa * 2**(-23)) * 2**(exponent - 127) Therefore, when the biased diff --git a/runtime/core/portable_type/c10/torch/headeronly/util/TypeSafeSignMath.h b/runtime/core/portable_type/c10/torch/headeronly/util/TypeSafeSignMath.h index c33a286bc5b..8e897957fee 100644 --- a/runtime/core/portable_type/c10/torch/headeronly/util/TypeSafeSignMath.h +++ b/runtime/core/portable_type/c10/torch/headeronly/util/TypeSafeSignMath.h @@ -14,20 +14,6 @@ C10_CLANG_DIAGNOSTIC_IGNORE("-Wimplicit-int-float-conversion") namespace c10 { -/// Returns false since we cannot have x < 0 if x is unsigned. -template -inline constexpr bool is_negative( - const T& /*x*/, - std::true_type /*is_unsigned*/) { - return false; -} - -/// Returns true if a signed variable x < 0 -template -inline constexpr bool is_negative(const T& x, std::false_type /*is_unsigned*/) { - return x < T(0); -} - /// Returns true if x < 0 /// NOTE: Will fail on an unsigned custom type /// For the most part it's possible to fix this if @@ -35,19 +21,12 @@ inline constexpr bool is_negative(const T& x, std::false_type /*is_unsigned*/) { /// However, notably, c10::Half does not :-( template inline constexpr bool is_negative(const T& x) { - return is_negative(x, std::is_unsigned()); -} - -/// Returns the sign of an unsigned variable x as 0, 1 -template -inline constexpr int signum(const T& x, std::true_type /*is_unsigned*/) { - return T(0) < x; -} - -/// Returns the sign of a signed variable x as -1, 0, 1 -template -inline constexpr int signum(const T& x, std::false_type /*is_unsigned*/) { - return (T(0) < x) - (x < T(0)); + if constexpr (std::is_unsigned_v) { + // An unsigned value can never be less than zero. + return false; + } else { + return x < T(0); + } } /// Returns the sign of x as -1, 0, 1 @@ -57,7 +36,11 @@ inline constexpr int signum(const T& x, std::false_type /*is_unsigned*/) { /// However, notably, c10::Half does not :-( template inline constexpr int signum(const T& x) { - return signum(x, std::is_unsigned()); + if constexpr (std::is_unsigned_v) { + return T(0) < x; + } else { + return (T(0) < x) - (x < T(0)); + } } /// Returns true if a and b are not both negative @@ -86,53 +69,22 @@ inline constexpr bool greater_than_max(const T& x) { #pragma GCC diagnostic pop #endif -/// Returns true if x < lowest(Limit). Standard comparison -template -inline constexpr bool less_than_lowest( - const T& x, - std::false_type /*limit_is_unsigned*/, - std::false_type /*x_is_unsigned*/) { - return x < std::numeric_limits::lowest(); -} - -/// Returns false since all the limit is signed and therefore includes -/// negative values but x cannot be negative because it is unsigned -template -inline constexpr bool less_than_lowest( - const T& /*x*/, - std::false_type /*limit_is_unsigned*/, - std::true_type /*x_is_unsigned*/) { - return false; -} - -/// Returns true if x < 0, where 0 is constructed from T. -/// Limit is not signed, so its lower value is zero -template -inline constexpr bool less_than_lowest( - const T& x, - std::true_type /*limit_is_unsigned*/, - std::false_type /*x_is_unsigned*/) { - return x < T(0); -} - -/// Returns false sign both types are unsigned -template -inline constexpr bool less_than_lowest( - const T& /*x*/, - std::true_type /*limit_is_unsigned*/, - std::true_type /*x_is_unsigned*/) { - return false; -} - -/// Returns true if x is less than the lowest value of type T +/// Returns true if x is less than the lowest value of type Limit /// NOTE: Will fail on an unsigned custom type /// For the most part it's possible to fix this if /// the custom type has a constexpr constructor. /// However, notably, c10::Half does not : template inline constexpr bool less_than_lowest(const T& x) { - return less_than_lowest( - x, std::is_unsigned(), std::is_unsigned()); + if constexpr (std::is_unsigned_v) { + // x is unsigned, so it can never be below the lowest value of any type. + return false; + } else if constexpr (std::is_unsigned_v) { + // Limit is unsigned, so its lowest value is zero. + return x < T(0); + } else { + return x < std::numeric_limits::lowest(); + } } } // namespace c10 diff --git a/runtime/core/portable_type/c10/torch/headeronly/util/complex.h b/runtime/core/portable_type/c10/torch/headeronly/util/complex.h index 733a22d5dbb..41430766952 100644 --- a/runtime/core/portable_type/c10/torch/headeronly/util/complex.h +++ b/runtime/core/portable_type/c10/torch/headeronly/util/complex.h @@ -3,6 +3,7 @@ #include #include +#include #include #if defined(__CUDACC__) || defined(__HIPCC__) @@ -588,6 +589,60 @@ struct alignas(4) complex { } }; +template <> +struct alignas(4) complex { + BFloat16 real_; + BFloat16 imag_; + + // Constructors + complex() = default; + // BFloat16 constructor is not constexpr so the following constructor can't + // be constexpr + C10_HOST_DEVICE explicit inline complex( + const BFloat16& real, + const BFloat16& imag) + : real_(real), imag_(imag) {} + C10_HOST_DEVICE inline complex(const c10::complex& value) + : real_(value.real()), imag_(value.imag()) {} + + // Conversion operator + inline C10_HOST_DEVICE operator c10::complex() const { + return {real_, imag_}; + } + + constexpr C10_HOST_DEVICE BFloat16 real() const { + return real_; + } + constexpr C10_HOST_DEVICE BFloat16 imag() const { + return imag_; + } + + C10_HOST_DEVICE complex& operator+=( + const complex& other) { + real_ = static_cast(real_) + static_cast(other.real_); + imag_ = static_cast(imag_) + static_cast(other.imag_); + return *this; + } + + C10_HOST_DEVICE complex& operator-=( + const complex& other) { + real_ = static_cast(real_) - static_cast(other.real_); + imag_ = static_cast(imag_) - static_cast(other.imag_); + return *this; + } + + C10_HOST_DEVICE complex& operator*=( + const complex& other) { + auto a = static_cast(real_); + auto b = static_cast(imag_); + auto c = static_cast(other.real()); + auto d = static_cast(other.imag()); + real_ = a * c - b * d; + imag_ = a * d + b * c; + return *this; + } +}; + } // namespace c10 HIDDEN_NAMESPACE_BEGIN(torch, headeronly) diff --git a/tools/cmake/Utils.cmake b/tools/cmake/Utils.cmake index 958b425c47c..7219bf30ab4 100644 --- a/tools/cmake/Utils.cmake +++ b/tools/cmake/Utils.cmake @@ -142,6 +142,10 @@ macro(find_package_torch) add_torch_to_cmake_prefix_path() find_package(Torch CONFIG REQUIRED) endif() + # PyTorch sets CXX_STANDARD on its imported target, which does not propagate + # to consumers. Advertise the requirement as an interface compile feature so + # only targets that link Torch are compiled as C++20. + target_compile_features(torch INTERFACE cxx_std_20) endmacro() # Modify ${targetName}'s INTERFACE_INCLUDE_DIRECTORIES by wrapping each entry in diff --git a/torch_pin.py b/torch_pin.py index ca593b1ef05..45f0b9a20f2 100644 --- a/torch_pin.py +++ b/torch_pin.py @@ -1,2 +1,2 @@ TORCH_VERSION = "2.13.0" -# NIGHTLY_VERSION = "dev20260318" Temporarily pinning to stable release candidate. Revert https://github.com/pytorch/executorch/pull/18287 +NIGHTLY_VERSION = "dev20260809"