summaryrefslogtreecommitdiff
path: root/sci-ml
diff options
context:
space:
mode:
authorroot <root@alpha.trunkmasters.com>2026-08-05 03:04:47 -0500
committerroot <root@alpha.trunkmasters.com>2026-08-05 03:04:47 -0500
commitff03324f0b5c30484659977ecb3a56e73f5856a5 (patch)
treed3d9a7636e37b0d2eec93ddf61a47caeda399de0 /sci-ml
parenteed03ed6da4435154826115469c3956ad88ee8a5 (diff)
downloadbaldeagleos-repo-ff03324f0b5c30484659977ecb3a56e73f5856a5.tar.gz
baldeagleos-repo-ff03324f0b5c30484659977ecb3a56e73f5856a5.tar.xz
baldeagleos-repo-ff03324f0b5c30484659977ecb3a56e73f5856a5.zip
Adding metadata
Diffstat (limited to 'sci-ml')
-rw-r--r--sci-ml/pytorch/Manifest3
-rw-r--r--sci-ml/pytorch/files/composable-kernel-7fe50dc-expand-isa.patch141
-rw-r--r--sci-ml/pytorch/files/pytorch-2.13.0-aotriton-fixes.patch46
-rw-r--r--sci-ml/pytorch/files/pytorch-2.13.0-glog.patch42
-rw-r--r--sci-ml/pytorch/files/pytorch-2.13.0-removekineto-pr178960.patch293
-rw-r--r--sci-ml/pytorch/files/pytorch-2.13.0-unbundle_fbgemm.patch19
-rw-r--r--sci-ml/pytorch/files/pytorch-2.13.0-unbundle_flatbuffers.patch23
-rw-r--r--sci-ml/pytorch/files/pytorch-2.13.0-unbundle_fmt.patch39
-rw-r--r--sci-ml/pytorch/files/pytorch-2.13.0-unbundle_kineto.patch33
-rw-r--r--sci-ml/pytorch/files/pytorch-2.13.0-unbundle_mkldnn.patch19
-rw-r--r--sci-ml/pytorch/files/pytorch-2.13.0-unbundle_nnpack.patch11
-rw-r--r--sci-ml/pytorch/files/pytorch-2.13.0-unbundle_pocketfft.patch18
-rw-r--r--sci-ml/pytorch/files/pytorch-2.13.0-xnnpack.patch20
-rw-r--r--sci-ml/pytorch/metadata.xml24
-rw-r--r--sci-ml/pytorch/pytorch-2.13.0-r1.ebuild340
15 files changed, 1071 insertions, 0 deletions
diff --git a/sci-ml/pytorch/Manifest b/sci-ml/pytorch/Manifest
index e11be7b5ef39..1811ce5b902d 100644
--- a/sci-ml/pytorch/Manifest
+++ b/sci-ml/pytorch/Manifest
@@ -1,2 +1,5 @@
+DIST composable_kernel-7fe50dc3.tar.gz 5380728 BLAKE2B c89c346d8e2d7a93a9cf26409e477fcdd25c43bc3f99d904c3bfe1bc282c6844ef2f2c80aceabe3bf4494db3457285384d5de5a22281aa426ba7479af82b0caf SHA512 a62f92e2dd7da944bd34bab6cf3bf624f630dc316d29c755e9fd523343c3f7648b0b7e0c9a0c8f5e9654477599ae8be9dac687d4054b0390f064ac2e40fc1cd3
+DIST flash-attention-2.7.4.gh.tar.gz 5841323 BLAKE2B 432999d763f2b3d732580ddfea5d3e01370351db0656546259a5e500a07516dd03c98828bfb55855dabe4adc651033b5d97ea4725ca46158b9970f0fbc662710 SHA512 05a4afb09e666f7404d6a3f8b5256e7bed6eba60a6f1bde2b7dbb96d318975f0b458c2521c7a38d88e97b6e4c27f29077cf787849daf82586e33f43a3d9a84b3
DIST pytorch-2.11.0.tar.gz 63504636 BLAKE2B d7227bd2d5475acf696bb48a3ca8cd8b7b6cd75fb99e64ebc71434e23817f94dea5164f48a176dcad1acd12186c33b02452582c1b41d6265b89cf54b29630670 SHA512 a5b27efdbce0506b49808e9dc5fbf1a56a22b1651b60236570823e1fe2c7395a0de9d3d7a01a568a201d593c52e7aa402e5d7645a54cb5932d5866e374ebf945
DIST pytorch-2.12.0.tar.gz 64740318 BLAKE2B e5e46a93422454da420563d0a8057c87d1ed4b27e0bd9b3b30e46b5e40cfbb489072144ac00267af2f085432d55e12adf750daf15ea165a1252449b00dca1ff9 SHA512 f5ab0f6933d88271f772b416f8c9b3b0d3e1ffaf8d00838b455206266b40d7c805e34003d84b01cddc7f1ad917dd72d6f79a05063b0962cbf223d9042cff3206
+DIST pytorch-2.13.0.tar.gz 66369404 BLAKE2B 373df49b4f2d504aafedd948e34562696335d316ebcd497dbd86a54a3a3c39310d1595050cc2bcae858d0085e6a0c59ed5299962d6a2706441ab28c1699c4821 SHA512 0f7465835ae9206746d507e1b927afb94f8860248473f2d1c5d71ccd40defbd23d0e3e27a1b09deb907e753a7c9008352830170b53441701ade0916d68a42562
diff --git a/sci-ml/pytorch/files/composable-kernel-7fe50dc-expand-isa.patch b/sci-ml/pytorch/files/composable-kernel-7fe50dc-expand-isa.patch
new file mode 100644
index 000000000000..53dcaf71fb44
--- /dev/null
+++ b/sci-ml/pytorch/files/composable-kernel-7fe50dc-expand-isa.patch
@@ -0,0 +1,141 @@
+Fix for "undeclared identifier 'CK_BUFFER_RESOURCE_3RD_DWORD'" for AMDGPU_TARGETS="gfx1012".
+Combines of 3 patches from https://github.com/ROCm/composable_kernel/issues/775#issuecomment-2726315348
+
+Bug: https://bugs.gentoo.org/947583
+Bug: https://bugs.gentoo.org/show_bug.cgi?id=959808
+--- a/include/ck/ck.hpp
++++ b/include/ck/ck.hpp
+@@ -78,7 +78,7 @@
+ #define CK_BUFFER_RESOURCE_3RD_DWORD -1
+ #elif defined(__gfx803__) || defined(__gfx900__) || defined(__gfx906__) || defined(__gfx9__)
+ #define CK_BUFFER_RESOURCE_3RD_DWORD 0x00020000
+-#elif defined(__gfx103__)
++#elif defined(__gfx101__) || defined(__gfx103__)
+ #define CK_BUFFER_RESOURCE_3RD_DWORD 0x31014000
+ #elif defined(__gfx11__) || defined(__gfx12__)
+ #define CK_BUFFER_RESOURCE_3RD_DWORD 0x31004000
+@@ -86,12 +86,12 @@
+
+ // FMA instruction
+ #ifndef __HIP_DEVICE_COMPILE__ // for host code, define nothing
+-#elif defined(__gfx803__) || defined(__gfx900__) // for GPU code
+-#define CK_USE_AMD_V_MAC_F32
+-#elif defined(__gfx906__) || defined(__gfx9__) || defined(__gfx103__) // for GPU code
++#elif defined(__gfx906__) || defined(__gfx9__) || defined(__gfx103__) || defined(__gfx1011__) || defined(__gfx1012__) // for GPU code
+ #define CK_USE_AMD_V_FMAC_F32
+ #define CK_USE_AMD_V_DOT2_F32_F16
+ #define CK_USE_AMD_V_DOT4_I32_I8
++#elif defined(__gfx803__) || defined(__gfx900__) || defined(__gfx101__) // for GPU code
++#define CK_USE_AMD_V_MAC_F32
+ #elif defined(__gfx11__) || defined(__gfx12__)
+ #define CK_USE_AMD_V_FMAC_F32
+ #define CK_USE_AMD_V_DOT2_F32_F16
+--- a/include/ck/tensor_operation/gpu/device/impl/device_batched_gemm_multiple_d_dl.hpp
++++ b/include/ck/tensor_operation/gpu/device/impl/device_batched_gemm_multiple_d_dl.hpp
+@@ -71,7 +71,7 @@ __launch_bounds__(CK_MAX_THREAD_PER_BLOCK, CK_MIN_BLOCK_PER_CU)
+ const Block2CTileMap block_2_ctile_map)
+ {
+ #if(defined(__gfx906__) || defined(__gfx908__) || defined(__gfx90a__) || defined(__gfx94__) || \
+- defined(__gfx103__) || defined(__gfx11__) || defined(__gfx12__))
++ defined(__gfx101__) || defined(__gfx103__) || defined(__gfx11__) || defined(__gfx12__))
+
+ const index_t num_blocks_per_batch =
+ __builtin_amdgcn_readfirstlane(get_grid_size() / batch_count);
+--- a/include/ck/tensor_operation/gpu/device/impl/device_gemm_multiple_d_dl.hpp
++++ b/include/ck/tensor_operation/gpu/device/impl/device_gemm_multiple_d_dl.hpp
+@@ -50,7 +50,7 @@ __launch_bounds__(CK_MAX_THREAD_PER_BLOCK, CK_MIN_BLOCK_PER_CU)
+ const CGridDesc_M0_M10_M11_N0_N10_N11 e_grid_desc_m0_m10_m11_n0_n10_n11,
+ const Block2CTileMap block_2_ctile_map)
+ {
+-#if(defined(__gfx906__) || defined(__gfx9__) || defined(__gfx103__) || defined(__gfx11__) || \
++#if(defined(__gfx906__) || defined(__gfx9__) || defined(__gfx101__) || defined(__gfx103__) || defined(__gfx11__) || \
+ defined(__gfx12__))
+
+ constexpr index_t shared_block_size =
+--- a/include/ck/tensor_operation/gpu/device/impl/device_grouped_conv_bwd_weight_dl.hpp
++++ b/include/ck/tensor_operation/gpu/device/impl/device_grouped_conv_bwd_weight_dl.hpp
+@@ -48,7 +48,7 @@ __launch_bounds__(CK_MAX_THREAD_PER_BLOCK, CK_MIN_BLOCK_PER_CU)
+ const Block2CTileMap block_2_ctile_map,
+ const ComputePtrOffsetOfBatch compute_ptr_offset_of_batch)
+ {
+-#if(defined(__gfx906__) || defined(__gfx103__) || defined(__gfx90a__) || defined(__gfx908__) || \
++#if(defined(__gfx906__) || defined(__gfx101__) || defined(__gfx103__) || defined(__gfx90a__) || defined(__gfx908__) || \
+ defined(__gfx94__) || defined(__gfx11__) || defined(__gfx12__))
+ const index_t num_blocks_per_batch =
+ __builtin_amdgcn_readfirstlane(get_grid_size() / batch_count);
+--- a/include/ck/tensor_operation/gpu/device/impl/device_grouped_conv_fwd_dl_multiple_d_nhwc_kyxc_nhwk.hpp
++++ b/include/ck/tensor_operation/gpu/device/impl/device_grouped_conv_fwd_dl_multiple_d_nhwc_kyxc_nhwk.hpp
+@@ -90,7 +90,7 @@ __launch_bounds__(CK_MAX_THREAD_PER_BLOCK, CK_MIN_BLOCK_PER_CU)
+ const Block2CTileMap block_2_ctile_map,
+ const ComputePtrOffsetOfBatch compute_ptr_offset_of_batch)
+ {
+-#if(defined(__gfx906__) || defined(__gfx103__) || defined(__gfx90a__) || defined(__gfx908__) || \
++#if(defined(__gfx906__) || defined(__gfx101__) || defined(__gfx101__) || defined(__gfx103__) || defined(__gfx90a__) || defined(__gfx908__) || \
+ defined(__gfx94__) || defined(__gfx11__) || defined(__gfx12__))
+ // offset base pointer for each work-group
+ const index_t num_blocks_per_batch =
+--- a/include/ck/tensor_operation/gpu/device/impl/device_grouped_conv_fwd_dl_nhwc_kyxc_nhwk.hpp
++++ b/include/ck/tensor_operation/gpu/device/impl/device_grouped_conv_fwd_dl_nhwc_kyxc_nhwk.hpp
+@@ -106,7 +106,7 @@ __launch_bounds__(CK_MAX_THREAD_PER_BLOCK, CK_MIN_BLOCK_PER_CU)
+ const Block2CTileMap block_2_ctile_map,
+ const ComputePtrOffsetOfBatch compute_ptr_offset_of_batch)
+ {
+-#if(defined(__gfx906__) || defined(__gfx103__) || defined(__gfx11__) || defined(__gfx12__))
++#if(defined(__gfx906__) || defined(__gfx101__) || defined(__gfx103__) || defined(__gfx11__) || defined(__gfx12__))
+ // offset base pointer for each work-group
+ const index_t num_blocks_per_batch =
+ __builtin_amdgcn_readfirstlane(get_grid_size() / batch_count);
+--- a/include/ck/tensor_operation/gpu/device/impl/device_grouped_gemm_multiple_d_dl.hpp
++++ b/include/ck/tensor_operation/gpu/device/impl/device_grouped_gemm_multiple_d_dl.hpp
+@@ -40,7 +40,7 @@ __launch_bounds__(CK_MAX_THREAD_PER_BLOCK, CK_MIN_BLOCK_PER_CU)
+ const BElementwiseOperation b_element_op,
+ const CDEElementwiseOperation cde_element_op)
+ {
+-#if(defined(__gfx906__) || defined(__gfx908__) || defined(__gfx90a__) || defined(__gfx103__) || \
++#if(defined(__gfx906__) || defined(__gfx908__) || defined(__gfx90a__) || defined(__gfx101__) || defined(__gfx101__) || defined(__gfx103__) || \
+ defined(__gfx11__) || defined(__gfx94__) || defined(__gfx12__))
+ __shared__ char p_shared[GridwiseGemm::GetSharedMemoryNumberOfByte()];
+
+--- a/include/ck/tensor_operation/gpu/grid/gridwise_gemm_dpp.hpp
++++ b/include/ck/tensor_operation/gpu/grid/gridwise_gemm_dpp.hpp
+@@ -28,7 +28,7 @@ __launch_bounds__(CK_MAX_THREAD_PER_BLOCK, CK_MIN_BLOCK_PER_CU)
+ #endif
+ kernel_gemm_dpp(const typename GridwiseGemm::Argument karg)
+ {
+-#if(defined(__gfx103__) || defined(__gfx11__))
++#if(defined(__gfx101__) || defined(__gfx103__) || defined(__gfx11__))
+ __shared__ char p_shared[GridwiseGemm::GetSharedMemoryNumberOfByte()];
+
+ const auto a_grid_desc_ak0_m_ak1 = amd_wave_read_first_lane(
+--- a/include/ck/tensor_operation/gpu/grid/gridwise_tensor_rearrange.hpp
++++ b/include/ck/tensor_operation/gpu/grid/gridwise_tensor_rearrange.hpp
+@@ -36,7 +36,7 @@ __launch_bounds__(CK_MAX_THREAD_PER_BLOCK, CK_MIN_BLOCK_PER_CU)
+ const ComputePtrOffsetOfStridedBatch compute_ptr_offset_of_batch)
+ {
+ #if(defined(__gfx906__) || defined(__gfx908__) || defined(__gfx90a__) || defined(__gfx94__) || \
+- defined(__gfx103__) || defined(__gfx11__) || defined(__gfx12__))
++ defined(__gfx101__) || defined(__gfx103__) || defined(__gfx11__) || defined(__gfx12__))
+ GridwiseTensorRearrangeKernel::Run(in_grid_desc,
+ p_in_global,
+ out_grid_desc,
+--- a/include/ck_tile/core/config.hpp
++++ b/include/ck_tile/core/config.hpp
+@@ -9,6 +9,9 @@
+ #if defined(__gfx942__) || defined(__gfx950__)
+ #define __gfx94__
+ #endif
++#if defined(__gfx1010__) || defined(__gfx1011__) || defined(__gfx1012__)
++#define __gfx101__
++#endif
+ #if defined(__gfx1030__) || defined(__gfx1031__) || defined(__gfx1032__) || \
+ defined(__gfx1034__) || defined(__gfx1035__) || defined(__gfx1036__) || \
+ defined(__gfx10_3_generic__)
+@@ -200,7 +203,7 @@
+ #elif defined(__gfx803__) || defined(__gfx900__) || defined(__gfx906__) || \
+ defined(__gfx9__) // for GPU code
+ #define CK_TILE_BUFFER_RESOURCE_3RD_DWORD 0x00020000
+-#elif defined(__gfx103__) // for GPU code
++#elif defined(__gfx101__) || defined(__gfx103__) // for GPU code
+ #define CK_TILE_BUFFER_RESOURCE_3RD_DWORD 0x31014000
+ #elif defined(__gfx11__) || defined(__gfx12__) // for GPU code
+ #define CK_TILE_BUFFER_RESOURCE_3RD_DWORD 0x31004000
diff --git a/sci-ml/pytorch/files/pytorch-2.13.0-aotriton-fixes.patch b/sci-ml/pytorch/files/pytorch-2.13.0-aotriton-fixes.patch
new file mode 100644
index 000000000000..cb2b7ffd106b
--- /dev/null
+++ b/sci-ml/pytorch/files/pytorch-2.13.0-aotriton-fixes.patch
@@ -0,0 +1,46 @@
+Fix installation with aotriton
+
+Upstream bug: https://github.com/pytorch/pytorch/issues/158109
+--- a/cmake/External/aotriton.cmake
++++ b/cmake/External/aotriton.cmake
+@@ -56,7 +56,8 @@ if(NOT __AOTRITON_INCLUDED)
+ set(__AOTRITON_Z "gz")
+ # Set the default __AOTRITON_LIB path
+ if(NOT WIN32)
+- set(__AOTRITON_LIB "lib/libaotriton_v2.so")
++ include(GNUInstallDirs)
++ set(__AOTRITON_LIB "${CMAKE_INSTALL_LIBDIR}/libaotriton_v2.so")
+ else()
+ set(__AOTRITON_LIB "lib/aotriton_v2.lib")
+ endif()
+@@ -240,10 +241,6 @@ if(NOT __AOTRITON_INCLUDED)
+
+ # Note it is INSTALL"ED"
+ if(DEFINED ENV{AOTRITON_INSTALLED_PREFIX})
+- install(DIRECTORY
+- $ENV{AOTRITON_INSTALLED_PREFIX}/lib
+- $ENV{AOTRITON_INSTALLED_PREFIX}/include
+- DESTINATION ${__AOTRITON_INSTALL_DIR})
+ set(__AOTRITON_INSTALL_DIR "$ENV{AOTRITON_INSTALLED_PREFIX}")
+ message(STATUS "Using Preinstalled AOTriton at ${__AOTRITON_INSTALL_DIR}")
+ elseif(DEFINED ENV{AOTRITON_INSTALL_FROM_SOURCE})
+@@ -280,19 +277,4 @@ if(NOT __AOTRITON_INCLUDED)
+ target_link_libraries(__caffe2_aotriton INTERFACE "${__AOTRITON_INSTALL_DIR}/${__AOTRITON_LIB}")
+ target_include_directories(__caffe2_aotriton INTERFACE ${__AOTRITON_INSTALL_DIR}/include)
+ set(AOTRITON_FOUND TRUE)
+- # Install libaotriton_v2.so into the cmake install tree so it ends up in
+- # site-packages/torch/lib/ when building with scikit-build-core.
+- # aotriton's ExternalProject puts the library directly in the source tree
+- # (${PROJECT_SOURCE_DIR}/torch/lib/) without a cmake install() rule, so it
+- # is absent from the installed wheel and causes link failures in downstream
+- # cmake builds (e.g., custom op builds) that link against installed torch.
+- install(DIRECTORY "${__AOTRITON_INSTALL_DIR}/lib/"
+- DESTINATION "lib"
+- FILES_MATCHING PATTERN "libaotriton_v2*.so*"
+- )
+- # Install aotriton GPU kernel images (compressed ISA blobs) into the wheel.
+- install(DIRECTORY "${__AOTRITON_INSTALL_DIR}/lib/aotriton.images"
+- DESTINATION "lib"
+- OPTIONAL
+- )
+ endif() # __AOTRITON_INCLUDED
diff --git a/sci-ml/pytorch/files/pytorch-2.13.0-glog.patch b/sci-ml/pytorch/files/pytorch-2.13.0-glog.patch
new file mode 100644
index 000000000000..935a002f02fd
--- /dev/null
+++ b/sci-ml/pytorch/files/pytorch-2.13.0-glog.patch
@@ -0,0 +1,42 @@
+--- a/c10/util/Exception.cpp 2026-07-20 08:01:49.241949729 +0200
++++ b/c10/util/Exception.cpp 2026-07-20 08:02:53.751321052 +0200
+@@ -15,9 +15,7 @@
+ // (e.g. Torch, common/init). See also Logging.cpp in this directory.
+ #ifdef C10_USE_GLOG
+ namespace google {
+-namespace glog_internal_namespace_ {
+ bool IsGoogleLoggingInitialized();
+-} // namespace glog_internal_namespace_
+ } // namespace google
+ #endif
+
+@@ -265,7 +263,7 @@
+ // During static initialization (before InitGoogleLogging), glog's global
+ // flags may not be constructed yet. Accessing them causes SIOF crashes
+ // (T253115013, D96553733). Fall back to stderr in that case.
+- if (!::google::glog_internal_namespace_::IsGoogleLoggingInitialized()) {
++ if (!::google::IsGoogleLoggingInitialized()) {
+ std::cerr << warning.source_location().file << ':'
+ << warning.source_location().line
+ << ": Warning: " << warning.msg() << " (function "
+--- a/c10/util/Logging.cpp 2026-08-04 17:56:50.870727752 +0200
++++ b/c10/util/Logging.cpp 2026-08-04 18:04:50.369891538 +0200
+@@ -390,9 +390,7 @@
+ // declaring it here. This is a hack but has been used by a bunch of others too
+ // (e.g. Torch).
+ namespace google {
+-namespace glog_internal_namespace_ {
+ bool IsGoogleLoggingInitialized();
+-} // namespace glog_internal_namespace_
+ } // namespace google
+
+ namespace c10 {
+@@ -401,7 +399,7 @@
+ void initGoogleLogging(char const* name) {
+ #if !defined(_MSC_VER)
+ // This trick can only be used on UNIX platforms
+- if (!::google::glog_internal_namespace_::IsGoogleLoggingInitialized())
++ if (!::google::IsGoogleLoggingInitialized())
+ #endif
+ {
+ ::google::InitGoogleLogging(name);
diff --git a/sci-ml/pytorch/files/pytorch-2.13.0-removekineto-pr178960.patch b/sci-ml/pytorch/files/pytorch-2.13.0-removekineto-pr178960.patch
new file mode 100644
index 000000000000..6be8ebedfb85
--- /dev/null
+++ b/sci-ml/pytorch/files/pytorch-2.13.0-removekineto-pr178960.patch
@@ -0,0 +1,293 @@
+From 37e33881b9e995f8328da0a149d71f4314bd7aa1 Mon Sep 17 00:00:00 2001
+From: Samaresh Kumar Singh <ssam3003@gmail.com>
+Date: Tue, 31 Mar 2026 21:39:08 -0500
+Subject: [PATCH 1/3] profiler: fix USE_KINETO=OFF build failure due to
+ unconditional ActivityType.h include
+
+kineto_shim.h was unconditionally including <ActivityType.h>, which is a Kineto header, causing a fatal compile error when building with USE_KINETO=OFF and Kineto is not installed (e.g. on Gentoo with system libraries).
+
+Guard the include with #ifdef USE_KINETO and provide a minimal stub (enum class ActivityType : uint8_t { NONE = 0 }) so the data structures and function signatures that reference libkineto::ActivityType still compile. Guard the corresponding function bodies in kineto_shim.cpp (addCPUActivity,
+deviceTypeFromActivity) and collection.cpp (scopeToType, kinetoType) that use concrete enum values, adding no-op stubs for the non-Kineto paths.Fixes the bug #178939
+---
+ torch/csrc/profiler/collection.cpp | 8 ++++++++
+ torch/csrc/profiler/kineto_shim.cpp | 16 ++++++++++++++--
+ torch/csrc/profiler/kineto_shim.h | 7 +++++++
+ 3 files changed, 29 insertions(+), 2 deletions(-)
+
+diff --git a/torch/csrc/profiler/collection.cpp b/torch/csrc/profiler/collection.cpp
+index 1be2d80310910..e8054d4f44df7 100644
+--- a/torch/csrc/profiler/collection.cpp
++++ b/torch/csrc/profiler/collection.cpp
+@@ -573,11 +573,13 @@ std::string toString(const ExtraFields<EventType::PyCall>& e) {
+ e.callsite_.funcname_.str());
+ }
+
++#ifdef USE_KINETO
+ auto scopeToType(at::RecordScope scope) {
+ return scope == at::RecordScope::USER_SCOPE
+ ? libkineto::ActivityType::USER_ANNOTATION
+ : libkineto::ActivityType::CPU_OP;
+ }
++#endif
+
+ int64_t torchOpEndNS(
+ const ExtraFields<EventType::TorchOp>& e,
+@@ -626,6 +628,7 @@ std::string Result::overload_name() const {
+ [](const auto& e) -> std::string { return ""; }));
+ }
+
++#ifdef USE_KINETO
+ libkineto::ActivityType Result::kinetoType() const {
+ return visit(c10::overloaded(
+ ATTRIBUTE(TorchOp, scopeToType(e.scope_)),
+@@ -638,6 +641,11 @@ libkineto::ActivityType Result::kinetoType() const {
+ ATTRIBUTE(PythonGC, libkineto::ActivityType::PYTHON_FUNCTION),
+ ATTRIBUTE(Kineto, e.activity_type_)));
+ }
++#else
++libkineto::ActivityType Result::kinetoType() const {
++ return libkineto::ActivityType::NONE;
++}
++#endif
+
+ uint64_t Result::correlationID() const {
+ return visit(c10::overloaded(
+diff --git a/torch/csrc/profiler/kineto_shim.cpp b/torch/csrc/profiler/kineto_shim.cpp
+index fa232e1a01016..0ddf62997a07e 100644
+--- a/torch/csrc/profiler/kineto_shim.cpp
++++ b/torch/csrc/profiler/kineto_shim.cpp
+@@ -138,6 +138,7 @@ TraceWrapper::TraceWrapper(const int64_t start_time, const std::string& name)
+ }
+ #endif // USE_KINETO
+
++#ifdef USE_KINETO
+ activity_t* TraceWrapper::addCPUActivity(
+ const std::string& name,
+ const libkineto::ActivityType type,
+@@ -145,7 +146,6 @@ activity_t* TraceWrapper::addCPUActivity(
+ const uint64_t correlation_id,
+ const int64_t start_time,
+ const int64_t end_time) {
+-#ifdef USE_KINETO
+ TORCH_CHECK((bool)(*this), "Cannot add event to non-existent trace.");
+ cpu_trace_->emplace_activity(cpu_trace_->span, type, name);
+ auto& act = libkineto::CpuTraceBuffer::toRef(cpu_trace_->activities.back());
+@@ -157,10 +157,18 @@ activity_t* TraceWrapper::addCPUActivity(
+ act.endTime = end_time;
+ }
+ return cpu_trace_->activities.back().get();
++}
+ #else
++activity_t* TraceWrapper::addCPUActivity(
++ const std::string& name,
++ const libkineto::ActivityType type,
++ const DeviceAndResource device_and_resource,
++ const uint64_t correlation_id,
++ const int64_t start_time,
++ const int64_t end_time) {
+ return nullptr;
+-#endif // USE_KINETO
+ }
++#endif // USE_KINETO
+
+ void TraceWrapper::transferCpuTrace(int64_t end_time) {
+ #ifdef USE_KINETO
+@@ -473,6 +481,7 @@ void logInvariantViolation(
+
+ namespace autograd::profiler {
+ c10::DeviceType deviceTypeFromActivity(libkineto::ActivityType activity_type) {
++#ifdef USE_KINETO
+ // PrivateUse1 kineto backend reuse some ActivityTypes,
+ // If PrivateUse1 backend is enabled, this should return
+ // c10::DeviceType::PrivateUse1.
+@@ -524,6 +533,9 @@ c10::DeviceType deviceTypeFromActivity(libkineto::ActivityType activity_type) {
+ return c10::DeviceType::CPU;
+ }
+ }
++#else
++ return c10::DeviceType::CPU;
++#endif // USE_KINETO
+ }
+
+ void addMetadataJson(const std::string& key, const std::string& value) {
+diff --git a/torch/csrc/profiler/kineto_shim.h b/torch/csrc/profiler/kineto_shim.h
+index 4f9bdc6770507..44bcb0e18a6e0 100644
+--- a/torch/csrc/profiler/kineto_shim.h
++++ b/torch/csrc/profiler/kineto_shim.h
+@@ -12,7 +12,14 @@
+ #undef USE_KINETO
+ #endif
+
++#ifdef USE_KINETO
+ #include <ActivityType.h>
++#else
++// Minimal stub so non-Kineto builds can compile types that hold ActivityType.
++namespace libkineto {
++enum class ActivityType : uint8_t { NONE = 0 };
++} // namespace libkineto
++#endif
+
+ #include <torch/csrc/Export.h>
+ #include <torch/csrc/profiler/api.h>
+
+From f5e3a9d9aa12c6e2006a9243c97ef7703fcbbe2f Mon Sep 17 00:00:00 2001
+From: Samaresh Kumar Singh <ssam3003@gmail.com>
+Date: Wed, 1 Apr 2026 10:34:24 -0500
+Subject: [PATCH 2/3] Fixed the USER_ANNOTATION/GPU_USER_ANNOTATION build
+ errors with external kineto
+
+When building against a system-installed kineto (e.g. Gentoo's sci-ml/kineto) that lacks USER_ANNOTATION and GPU_USER_ANNOTATION in its ActivityType enum,
+init.cpp failed to compile because it referenced those enum member names directly.
+---
+ torch/csrc/autograd/init.cpp | 7 +------
+ torch/csrc/autograd/profiler_kineto.cpp | 7 +++++++
+ torch/csrc/autograd/profiler_kineto.h | 1 +
+ torch/csrc/profiler/kineto_shim.h | 11 +++++++++--
+ 4 files changed, 18 insertions(+), 8 deletions(-)
+
+diff --git a/torch/csrc/autograd/init.cpp b/torch/csrc/autograd/init.cpp
+index 1e2442a5d99db..f8e8795c5805e 100644
+--- a/torch/csrc/autograd/init.cpp
++++ b/torch/csrc/autograd/init.cpp
+@@ -305,12 +305,7 @@ PyObject* THPAutograd_initExtension(PyObject* _unused, PyObject* unused) {
+ .def("privateuse1_elapsed_us", &KinetoEvent::privateuse1ElapsedUs)
+ .def(
+ "is_user_annotation",
+- [](const KinetoEvent& e) {
+- return e.activityType() ==
+- (uint8_t)libkineto::ActivityType::USER_ANNOTATION ||
+- e.activityType() ==
+- (uint8_t)libkineto::ActivityType::GPU_USER_ANNOTATION;
+- })
++ [](const KinetoEvent& e) { return e.isUserAnnotation(); })
+ .def(
+ "is_python_function",
+ [](const KinetoEvent& e) { return e.isPythonFunction(); })
+diff --git a/torch/csrc/autograd/profiler_kineto.cpp b/torch/csrc/autograd/profiler_kineto.cpp
+index 9a076f58d7143..6b5165f0872fb 100644
+--- a/torch/csrc/autograd/profiler_kineto.cpp
++++ b/torch/csrc/autograd/profiler_kineto.cpp
+@@ -1081,6 +1081,13 @@ int64_t KinetoEvent::privateuse1ElapsedUs() const {
+ return -1;
+ }
+
++bool KinetoEvent::isUserAnnotation() const {
++ constexpr uint8_t kUserAnnotation = 1;
++ constexpr uint8_t kGpuUserAnnotation = 2;
++ const auto type = activityType();
++ return type == kUserAnnotation || type == kGpuUserAnnotation;
++}
++
+ void KinetoEvent::getPerfEventCounters(std::vector<uint64_t>& in) const {
+ return result_->visit(c10::overloaded(
+ [&in](const ExtraFields<EventType::TorchOp>& e) -> void {
+diff --git a/torch/csrc/autograd/profiler_kineto.h b/torch/csrc/autograd/profiler_kineto.h
+index 777b8a5851ed5..c3135c61245d4 100644
+--- a/torch/csrc/autograd/profiler_kineto.h
++++ b/torch/csrc/autograd/profiler_kineto.h
+@@ -67,6 +67,7 @@ struct TORCH_API KinetoEvent {
+ bool isPythonFunction() const;
+ int64_t cudaElapsedUs() const;
+ int64_t privateuse1ElapsedUs() const;
++ bool isUserAnnotation() const;
+ void getPerfEventCounters(torch::profiler::perf_counters_t& /*in*/) const;
+ extra_meta_t extraMeta() const;
+ std::string metadataJson() const;
+diff --git a/torch/csrc/profiler/kineto_shim.h b/torch/csrc/profiler/kineto_shim.h
+index 44bcb0e18a6e0..ddc00c03fa9cc 100644
+--- a/torch/csrc/profiler/kineto_shim.h
++++ b/torch/csrc/profiler/kineto_shim.h
+@@ -15,9 +15,16 @@
+ #ifdef USE_KINETO
+ #include <ActivityType.h>
+ #else
+-// Minimal stub so non-Kineto builds can compile types that hold ActivityType.
+ namespace libkineto {
+-enum class ActivityType : uint8_t { NONE = 0 };
++enum class ActivityType : uint8_t {
++ CPU_OP = 0,
++ USER_ANNOTATION,
++ GPU_USER_ANNOTATION,
++ NONE = CPU_OP,
++};
++inline const char* toString(ActivityType) {
++ return "CPU_OP";
++}
+ } // namespace libkineto
+ #endif
+
+
+From 717526b7ce90d242cb917db90f4cb13c85c93661 Mon Sep 17 00:00:00 2001
+From: Samaresh Kumar Singh <ssam3003@gmail.com>
+Date: Fri, 3 Apr 2026 13:01:01 -0500
+Subject: [PATCH 3/3] profiler: guard GPU ActivityType checks with KINETO GPU
+ backend macros
+
+The externalId() function referenced GPU-specific libkineto::ActivityType
+values (GPU_MEMCPY, GPU_MEMSET, CONCURRENT_KERNEL, CUDA_RUNTIME,
+CUDA_DRIVER, PRIVATEUSE1_RUNTIME, PRIVATEUSE1_DRIVER) that are only
+present in kineto builds with GPU backend support.
+
+System-installed kineto packages built without CUDA/ROCm support
+(LIBKINETO_NOCUPTI + LIBKINETO_NOROCTRACER) omit these enum values,
+causing compile errors. Guard the check with the existing GPU backend
+macros. Also restructures the condition to an early-return to avoid
+duplicating the correlation ID lookup.
+---
+ torch/csrc/autograd/profiler_kineto.cpp | 38 ++++++++++++++-----------
+ 1 file changed, 22 insertions(+), 16 deletions(-)
+
+diff --git a/torch/csrc/autograd/profiler_kineto.cpp b/torch/csrc/autograd/profiler_kineto.cpp
+index 6b5165f0872fb..466e8d66ab518 100644
+--- a/torch/csrc/autograd/profiler_kineto.cpp
++++ b/torch/csrc/autograd/profiler_kineto.cpp
+@@ -1125,27 +1125,33 @@ int64_t KinetoEvent::externalId() const {
+ return static_cast<int64_t>(linked);
+ }
+
++#if defined(USE_KINETO) && \
++ (!defined(LIBKINETO_NOCUPTI) || !defined(LIBKINETO_NOROCTRACER))
+ // Orphaned GPU activities (no linked CPU op) in these types should not get
+ // an External id, to avoid incorrect cross-linking in trace viewers.
++ // These GPU-specific ActivityType values are only present when kineto is
++ // built with GPU backend support (CUPTI or ROCtracer). CPU-only kineto
++ // builds (e.g. system packages without GPU support) omit them.
+ auto type = static_cast<libkineto::ActivityType>(activityType());
+- if (type != libkineto::ActivityType::GPU_MEMCPY &&
+- type != libkineto::ActivityType::GPU_MEMSET &&
+- type != libkineto::ActivityType::CONCURRENT_KERNEL &&
+- type != libkineto::ActivityType::CUDA_RUNTIME &&
+- type != libkineto::ActivityType::CUDA_DRIVER &&
+- type != libkineto::ActivityType::PRIVATEUSE1_RUNTIME &&
+- type != libkineto::ActivityType::PRIVATEUSE1_DRIVER) {
+- return static_cast<int64_t>(result_->visit(c10::overloaded(
+- [](const ExtraFields<EventType::TorchOp>& e) -> uint64_t {
+- return e.correlation_id_;
+- },
+- [](const ExtraFields<EventType::Kineto>& e) -> uint64_t {
+- return e.correlation_id_;
+- },
+- [](const auto&) -> uint64_t { return 0; })));
++ if (type == libkineto::ActivityType::GPU_MEMCPY ||
++ type == libkineto::ActivityType::GPU_MEMSET ||
++ type == libkineto::ActivityType::CONCURRENT_KERNEL ||
++ type == libkineto::ActivityType::CUDA_RUNTIME ||
++ type == libkineto::ActivityType::CUDA_DRIVER ||
++ type == libkineto::ActivityType::PRIVATEUSE1_RUNTIME ||
++ type == libkineto::ActivityType::PRIVATEUSE1_DRIVER) {
++ return 0;
+ }
++#endif
+
+- return 0;
++ return static_cast<int64_t>(result_->visit(c10::overloaded(
++ [](const ExtraFields<EventType::TorchOp>& e) -> uint64_t {
++ return e.correlation_id_;
++ },
++ [](const ExtraFields<EventType::Kineto>& e) -> uint64_t {
++ return e.correlation_id_;
++ },
++ [](const auto&) -> uint64_t { return 0; })));
+ }
+
+ #define FORWARD_FROM_RESULT(method_name, result_expr) \
diff --git a/sci-ml/pytorch/files/pytorch-2.13.0-unbundle_fbgemm.patch b/sci-ml/pytorch/files/pytorch-2.13.0-unbundle_fbgemm.patch
new file mode 100644
index 000000000000..a8cd385e7737
--- /dev/null
+++ b/sci-ml/pytorch/files/pytorch-2.13.0-unbundle_fbgemm.patch
@@ -0,0 +1,19 @@
+--- a/cmake/Dependencies.cmake
++++ b/cmake/Dependencies.cmake
+@@ -674,7 +672,7 @@ if(BUILD_TEST OR BUILD_MOBILE_BENCHMARK OR BUILD_MOBILE_TEST)
+ endif()
+
+ # ---[ FBGEMM
+-if(USE_FBGEMM)
++if(FALSE)
+ set(CAFFE2_THIRD_PARTY_ROOT "${PROJECT_SOURCE_DIR}/third_party")
+ if(NOT DEFINED FBGEMM_SOURCE_DIR)
+ set(FBGEMM_SOURCE_DIR "${CAFFE2_THIRD_PARTY_ROOT}/fbgemm" CACHE STRING "FBGEMM source directory")
+@@ -698,6 +696,7 @@ if(USE_FBGEMM)
+ endif()
+
+ if(USE_FBGEMM)
++ list(APPEND Caffe2_DEPENDENCY_LIBS fbgemm)
+ caffe2_update_option(USE_FBGEMM ON)
+ else()
+ caffe2_update_option(USE_FBGEMM OFF)
diff --git a/sci-ml/pytorch/files/pytorch-2.13.0-unbundle_flatbuffers.patch b/sci-ml/pytorch/files/pytorch-2.13.0-unbundle_flatbuffers.patch
new file mode 100644
index 000000000000..e410a124da92
--- /dev/null
+++ b/sci-ml/pytorch/files/pytorch-2.13.0-unbundle_flatbuffers.patch
@@ -0,0 +1,23 @@
+--- a/torch/CMakeLists.txt
++++ b/torch/CMakeLists.txt
+@@ -59,7 +59,6 @@
+
+ ${TORCH_ROOT}/third_party/gloo
+ ${TORCH_ROOT}/third_party/onnx
+- ${TORCH_ROOT}/third_party/flatbuffers/include
+ ${TORCH_ROOT}/third_party/kineto/libkineto/include
+ ${TORCH_ROOT}/third_party/cpp-httplib
+ ${TORCH_ROOT}/third_party/nlohmann/include
+--- a/cmake/FlatBuffers.cmake
++++ b/cmake/FlatBuffers.cmake
+@@ -1,10 +1 @@
+-set(FlatBuffers_Include ${PROJECT_SOURCE_DIR}/third_party/flatbuffers/include)
+-file(GLOB FlatBuffers_Library_SRCS
+- ${FlatBuffers_Include}/flatbuffers/*.h
+-)
+ add_library(flatbuffers INTERFACE)
+-target_sources(
+- flatbuffers
+- INTERFACE ${FlatBuffers_Library_SRCS}
+-)
+-target_include_directories(flatbuffers INTERFACE ${FlatBuffers_Include})
diff --git a/sci-ml/pytorch/files/pytorch-2.13.0-unbundle_fmt.patch b/sci-ml/pytorch/files/pytorch-2.13.0-unbundle_fmt.patch
new file mode 100644
index 000000000000..6be91dbbdcf6
--- /dev/null
+++ b/sci-ml/pytorch/files/pytorch-2.13.0-unbundle_fmt.patch
@@ -0,0 +1,39 @@
+--- a/cmake/Dependencies.cmake
++++ b/cmake/Dependencies.cmake
+@@ -1522,7 +1522,6 @@
+ # CMAKE_CXX_FLAGS in ways that break feature checks. Since we already know
+ # `fmt` is compatible with a superset of the compilers that PyTorch is, it
+ # shouldn't be too bad to just disable the checks.
+-set_target_properties(fmt-header-only PROPERTIES INTERFACE_COMPILE_FEATURES "")
+
+ list(APPEND Caffe2_DEPENDENCY_LIBS fmt)
+ set(BUILD_SHARED_LIBS ${TEMP_BUILD_SHARED_LIBS} CACHE BOOL "Build shared libs" FORCE)
+@@ -1641,8 +1641,6 @@
+ else()
+ set(_fmt_no_unique_address "[[no_unique_address]]")
+ endif()
+-target_compile_definitions(fmt PUBLIC "FMT_NO_UNIQUE_ADDRESS=${_fmt_no_unique_address}")
+-target_compile_definitions(fmt-header-only INTERFACE "FMT_NO_UNIQUE_ADDRESS=${_fmt_no_unique_address}")
+ unset(_fmt_no_unique_address)
+
+ list(APPEND Caffe2_DEPENDENCY_LIBS fmt)
+--- a/aten/src/ATen/CMakeLists.txt 2025-02-27 14:23:02.402742165 +0100
++++ b/aten/src/ATen/CMakeLists.txt 2025-02-27 14:23:40.445850718 +0100
+@@ -448,8 +448,6 @@
+ if(USE_CUDA)
+ list(APPEND ATen_CUDA_INCLUDE ${CMAKE_CURRENT_SOURCE_DIR}/cuda)
+ # Next two lines are needed because TunableOp uses third-party/fmt
+- list(APPEND ATen_CUDA_INCLUDE $<TARGET_PROPERTY:fmt::fmt-header-only,INTERFACE_INCLUDE_DIRECTORIES>)
+- list(APPEND ATen_CUDA_DEPENDENCY_LIBS fmt::fmt-header-only)
+ list(APPEND ATen_CUDA_CU_SRCS
+ ${cuda_cu}
+ ${native_cuda_cu}
+@@ -521,8 +519,6 @@
+ endif()
+
+ # Next two lines are needed because TunableOp uses third-party/fmt
+- list(APPEND ATen_HIP_INCLUDE $<TARGET_PROPERTY:fmt::fmt-header-only,INTERFACE_INCLUDE_DIRECTORIES>)
+- list(APPEND ATen_HIP_DEPENDENCY_LIBS fmt::fmt-header-only)
+ if(USE_FLASH_ATTENTION AND USE_ROCM_CK_SDPA)
+ list(APPEND ATen_HIP_INCLUDE ${CMAKE_CURRENT_SOURCE_DIR}/native/transformers/hip/flash_attn/ck)
+ endif()
diff --git a/sci-ml/pytorch/files/pytorch-2.13.0-unbundle_kineto.patch b/sci-ml/pytorch/files/pytorch-2.13.0-unbundle_kineto.patch
new file mode 100644
index 000000000000..928073410480
--- /dev/null
+++ b/sci-ml/pytorch/files/pytorch-2.13.0-unbundle_kineto.patch
@@ -0,0 +1,33 @@
+--- a/torch/CMakeLists.txt 2024-11-09 20:50:21.806784527 +0100
++++ b/torch/CMakeLists.txt 2024-11-09 20:52:56.303892418 +0100
+@@ -67,7 +67,7 @@
+
+ ${TORCH_ROOT}/third_party/gloo
+ ${TORCH_ROOT}/third_party/onnx
+- ${TORCH_ROOT}/third_party/kineto/libkineto/include
++ "/usr/include/kineto"
+ ${TORCH_ROOT}/third_party/cpp-httplib
+ ${TORCH_ROOT}/third_party/nlohmann/include
+
+--- a/caffe2/CMakeLists.txt 2024-11-09 20:58:37.698085144 +0100
++++ b/caffe2/CMakeLists.txt 2024-11-09 20:59:15.771391643 +0100
+@@ -1185,7 +1185,7 @@
+ ${TORCH_ROOT}/third_party/miniz-2.1.0)
+
+ target_include_directories(torch_cpu PRIVATE
+- ${TORCH_ROOT}/third_party/kineto/libkineto/include)
++ "/usr/include/kineto")
+
+ if(USE_KINETO)
+ target_include_directories(torch_cpu PRIVATE
+--- a/cmake/Dependencies.cmake 2025-05-04 15:30:00.268862558 +0200
++++ b/cmake/Dependencies.cmake 2025-05-04 15:30:13.275934233 +0200
+@@ -1711,7 +1711,7 @@
+ endif()
+ endif()
+
+- if(NOT TARGET kineto)
++ if(FALSE)
+ add_subdirectory("${KINETO_SOURCE_DIR}")
+ set_property(TARGET kineto PROPERTY POSITION_INDEPENDENT_CODE ON)
+ endif()
diff --git a/sci-ml/pytorch/files/pytorch-2.13.0-unbundle_mkldnn.patch b/sci-ml/pytorch/files/pytorch-2.13.0-unbundle_mkldnn.patch
new file mode 100644
index 000000000000..384e1ce9a4ab
--- /dev/null
+++ b/sci-ml/pytorch/files/pytorch-2.13.0-unbundle_mkldnn.patch
@@ -0,0 +1,19 @@
+--- a/cmake/Modules/FindMKLDNN.cmake 2026-08-02 19:43:26.710474163 +0200
++++ b/cmake/Modules/FindMKLDNN.cmake 2026-08-02 19:45:24.546256910 +0200
+@@ -12,7 +12,7 @@
+ # MKLDNN_USE_NATIVE_ARCH : Whether native CPU instructions should be used in MKLDNN. This should be turned off for
+ # general packaging to avoid incompatible CPU instructions. Default: OFF.
+
+-IF(NOT MKLDNN_FOUND)
++IF(FALSE)
+ SET(MKLDNN_LIBRARIES)
+ SET(MKLDNN_INCLUDE_DIR)
+
+@@ -193,4 +193,6 @@
+ SET(MKLDNN_FOUND TRUE)
+ MESSAGE(STATUS "Found MKL-DNN: TRUE")
+
+-ENDIF(NOT MKLDNN_FOUND)
++ENDIF(FALSE)
++SET(MKLDNN_FOUND TRUE)
++SET(MKLDNN_LIBRARIES dnnl)
diff --git a/sci-ml/pytorch/files/pytorch-2.13.0-unbundle_nnpack.patch b/sci-ml/pytorch/files/pytorch-2.13.0-unbundle_nnpack.patch
new file mode 100644
index 000000000000..fe2e5743e0bc
--- /dev/null
+++ b/sci-ml/pytorch/files/pytorch-2.13.0-unbundle_nnpack.patch
@@ -0,0 +1,11 @@
+--- a/cmake/External/nnpack.cmake
++++ b/cmake/External/nnpack.cmake
+@@ -56,7 +56,7 @@
+ set(PTHREADPOOL_SOURCE_DIR "${CAFFE2_THIRD_PARTY_ROOT}/pthreadpool" CACHE STRING "pthreadpool source directory")
+ set(GOOGLETEST_SOURCE_DIR "${CAFFE2_THIRD_PARTY_ROOT}/googletest" CACHE STRING "Google Test source directory")
+
+- if(NOT TARGET nnpack)
++ if(FALSE)
+ set(NNPACK_BUILD_TESTS OFF CACHE BOOL "")
+ set(NNPACK_BUILD_BENCHMARKS OFF CACHE BOOL "")
+ set(NNPACK_LIBRARY_TYPE "static" CACHE STRING "")
diff --git a/sci-ml/pytorch/files/pytorch-2.13.0-unbundle_pocketfft.patch b/sci-ml/pytorch/files/pytorch-2.13.0-unbundle_pocketfft.patch
new file mode 100644
index 000000000000..3ffe9c775b28
--- /dev/null
+++ b/sci-ml/pytorch/files/pytorch-2.13.0-unbundle_pocketfft.patch
@@ -0,0 +1,18 @@
+--- a/cmake/Dependencies.cmake
++++ b/cmake/Dependencies.cmake
+@@ -276,15 +276,8 @@
+ # --- [ PocketFFT
+ set(AT_POCKETFFT_ENABLED 0)
+ if(NOT AT_MKL_ENABLED)
+- set(POCKETFFT_INCLUDE_DIR "${Torch_SOURCE_DIR}/third_party/pocketfft/")
+- if(NOT EXISTS "${POCKETFFT_INCLUDE_DIR}")
+- message(FATAL_ERROR "pocketfft directory not found, expected ${POCKETFFT_INCLUDE_DIR}")
+- elseif(NOT EXISTS "${POCKETFFT_INCLUDE_DIR}/pocketfft_hdronly.h")
+- message(FATAL_ERROR "pocketfft headers not found in ${POCKETFFT_INCLUDE_DIR}")
+- endif()
+
+ set(AT_POCKETFFT_ENABLED 1)
+- message(STATUS "Using pocketfft in directory: ${POCKETFFT_INCLUDE_DIR}")
+ endif()
+
+ # ---[ Dependencies
diff --git a/sci-ml/pytorch/files/pytorch-2.13.0-xnnpack.patch b/sci-ml/pytorch/files/pytorch-2.13.0-xnnpack.patch
new file mode 100644
index 000000000000..41621e846fba
--- /dev/null
+++ b/sci-ml/pytorch/files/pytorch-2.13.0-xnnpack.patch
@@ -0,0 +1,20 @@
+--- a/cmake/Dependencies.cmake 2026-08-04 10:54:55.396437127 +0200
++++ b/cmake/Dependencies.cmake 2026-08-04 11:02:44.251900961 +0200
+@@ -541,7 +541,8 @@
+ endif()
+
+ # ---[ XNNPACK
+-if(USE_XNNPACK AND NOT USE_SYSTEM_XNNPACK)
++if(USE_XNNPACK)
++if(NOT USE_SYSTEM_XNNPACK)
+ if(NOT DEFINED XNNPACK_SOURCE_DIR)
+ set(XNNPACK_SOURCE_DIR "${CAFFE2_THIRD_PARTY_ROOT}/XNNPACK" CACHE STRING "XNNPACK source directory")
+ endif()
+@@ -615,6 +616,7 @@
+ message("-- Found XNNPACK: ${XNNPACK_LIBRARY}")
+ list(APPEND Caffe2_DEPENDENCY_LIBS XNNPACK microkernels-prod)
+ endif()
++endif()
+
+ # ---[ Vulkan deps
+ if(USE_VULKAN)
diff --git a/sci-ml/pytorch/metadata.xml b/sci-ml/pytorch/metadata.xml
index 4de637d0a010..ed0c68a8eac3 100644
--- a/sci-ml/pytorch/metadata.xml
+++ b/sci-ml/pytorch/metadata.xml
@@ -5,6 +5,30 @@
<email>tupone@gentoo.org</email>
<name>Tupone Alfredo</name>
</maintainer>
+ <use>
+ <flag name="cusparselt">Use the CUDA/HIP Sparse Matrix Multiplication</flag>
+ <flag name="distributed">Support distributed applications</flag>
+ <flag name="fbgemm">Use <pkg>sci-ml/FBGEMM</pkg>
+ </flag>
+ <flag name="flash">Enable flash attention</flag>
+ <flag name="gloo">Use <pkg>sci-ml/gloo</pkg>
+ </flag>
+ <flag name="kineto">Use<pkg>sci-ml/kineto</pkg>profiling library</flag>
+ <flag name="memefficient">Enable mem efficient attention</flag>
+ <flag name="mimalloc">Use <pkg>dev-libs/mimalloc</pkg> as replacement for system malloc</flag>
+ <flag name="mkl">Use <pkg>sci-libs/mkl</pkg> for blas, lapack and sparse blas routines</flag>
+ <flag name="nccl">Use <pkg>dev-libs/rccl</pkg> (NCCL compatible) backend for distributed operations</flag>
+ <flag name="nnpack">Use <pkg>sci-ml/NNPACK</pkg>
+ </flag>
+ <flag name="numpy">Add support for math operations through numpy</flag>
+ <flag name="onednn">Use <pkg>sci-ml/oneDNN</pkg>
+ </flag>
+ <flag name="openblas">Use <pkg>sci-libs/openblas</pkg> for blas routines</flag>
+ <flag name="qnnpack">Use QNNPACK</flag>
+ <flag name="rocm">Enable ROCm gpu computing support</flag>
+ <flag name="xnnpack">Use <pkg>sci-ml/XNNPACK</pkg>
+ </flag>
+ </use>
<upstream>
<bugs-to>https://github.com/pytorch/pytorch/issues</bugs-to>
<changelog>https://github.com/pytorch/pytorch/releases</changelog>
diff --git a/sci-ml/pytorch/pytorch-2.13.0-r1.ebuild b/sci-ml/pytorch/pytorch-2.13.0-r1.ebuild
new file mode 100644
index 000000000000..93ada8738fde
--- /dev/null
+++ b/sci-ml/pytorch/pytorch-2.13.0-r1.ebuild
@@ -0,0 +1,340 @@
+# Copyright 2022-2026 Gentoo Authors
+# Distributed under the terms of the GNU General Public License v2
+
+EAPI=8
+
+DISTUTILS_USE_PEP517=setuptools
+PYTHON_COMPAT=( python3_{13..14} )
+DISTUTILS_SINGLE_IMPL=1
+DISTUTILS_EXT=1
+
+ROCM_VERSION=6.1
+inherit distutils-r1 prefix cuda flag-o-matic rocm multiprocessing
+
+# pytorch-2.13.0 depends on future version of composable kernel
+# TODO: replace it with DEPEND in the future
+CK_COMMIT=7fe50dc3da2069d6645d9deb8c017a876472a977
+CK_P=composable_kernel-${CK_COMMIT:0:8}
+
+# Starting from 2.7.0 pytorch moved flash attention out-of-tree,
+# but hardcoded it as third_party subproject
+# TODO: unbundle
+FLASH_PV=2.7.4
+FLASH_PN=flash-attention
+FLASH_P=${FLASH_PN}-${FLASH_PV}
+FLASH_ATT_URI="https://github.com/Dao-AILab/${FLASH_PN}/archive/refs/tags/v${FLASH_PV}.tar.gz -> ${FLASH_P}.gh.tar.gz"
+
+DESCRIPTION="Tensors and Dynamic neural networks in Python with strong GPU acceleration"
+HOMEPAGE="https://pytorch.org/"
+SRC_URI="https://github.com/pytorch/${PN}/archive/refs/tags/v${PV}.tar.gz
+ -> ${P}.tar.gz
+ rocm? (
+ https://github.com/ROCm/composable_kernel/archive/${CK_COMMIT}.tar.gz
+ -> ${CK_P}.tar.gz
+ )
+ cuda? (
+ flash? ( ${FLASH_ATT_URI} )
+ memefficient? ( ${FLASH_ATT_URI} )
+ )
+"
+
+LICENSE="BSD"
+SLOT="0"
+KEYWORDS="~amd64"
+IUSE="cuda cusparselt distributed fbgemm flash gloo kineto memefficient
+ mimalloc mkl mpi nccl nnpack +numpy onednn openblas opencl openmp qnnpack
+ rocm xnnpack"
+RESTRICT="test"
+
+REQUIRED_USE="
+ ${PYTHON_REQUIRED_USE}
+ mpi? ( distributed )
+ gloo? ( distributed )
+ ?? ( cuda rocm )
+ rocm? (
+ || ( ${ROCM_REQUIRED_USE} )
+ memefficient? ( flash )
+ )
+ cusparselt? ( || ( cuda rocm ) )
+ flash? ( || ( cuda rocm ) )
+ memefficient? ( || ( cuda rocm ) )
+ nccl? ( rocm )
+"
+
+RDEPEND="
+ ${PYTHON_DEPS}
+ !sci-ml/caffe2
+ dev-cpp/abseil-cpp:=
+ dev-cpp/gflags:=
+ >=dev-cpp/glog-0.6.0:=
+ >=dev-libs/cpuinfo-2025.11.14
+ dev-libs/libfmt:=
+ dev-libs/protobuf:=
+ dev-libs/sleef
+ sci-ml/onnx
+ $(python_gen_cond_dep '
+ dev-python/sympy[${PYTHON_USEDEP}]
+ dev-python/typing-extensions[${PYTHON_USEDEP}]
+ ')
+ cuda? (
+ dev-libs/cudnn
+ >=sci-ml/cudnn-frontend-1.12.0:=
+ >=dev-util/nvidia-cuda-toolkit-12.9:=[profiler]
+ cusparselt? ( dev-libs/cusparselt )
+ )
+ distributed? (
+ !rocm? ( sci-ml/tensorpipe[cuda?] )
+ dev-cpp/cpp-httplib:=
+ )
+ fbgemm? ( >=sci-ml/FBGEMM-1.4 )
+ gloo? ( >=sci-ml/gloo-2025.06.04[cuda?,rocm?] )
+ kineto? ( ~sci-ml/kineto-0.4.0_p20260603 )
+ mimalloc? ( dev-libs/mimalloc )
+ mpi? ( virtual/mpi )
+ nnpack? (
+ sci-ml/NNPACK
+ dev-libs/pthreadpool
+ )
+ numpy? ( $(python_gen_cond_dep '
+ dev-python/numpy[${PYTHON_USEDEP}]
+ ') )
+ onednn? ( sci-ml/oneDNN )
+ opencl? ( virtual/opencl )
+ qnnpack? (
+ !sci-libs/QNNPACK
+ sci-ml/gemmlowp
+ dev-libs/pthreadpool
+ )
+ rocm? (
+ nccl? ( >=dev-libs/rccl-6.3:= <dev-libs/rccl-7.3:= )
+ >=dev-util/hip-6.3:= <dev-util/hip-7.3:=
+ >=dev-util/roctracer-6.3:= <dev-util/roctracer-7.3:=
+ >=sci-libs/hipBLAS-6.3:= <sci-libs/hipBLAS-7.3:=[rocsolver(+)]
+ >=sci-libs/hipBLASLt-6.3:= <sci-libs/hipBLASLt-7.3:=
+ >=sci-libs/hipFFT-6.3:= <sci-libs/hipFFT-7.3:=
+ >=sci-libs/hipRAND-6.3:= <sci-libs/hipRAND-7.3:=
+ >=sci-libs/hipSOLVER-6.3:= <sci-libs/hipSOLVER-7.3:=
+ >=sci-libs/hipSPARSE-6.3:= <sci-libs/hipSPARSE-7.3:=
+ >=sci-libs/miopen-6.3:= <sci-libs/miopen-7.3:=
+ >=sci-libs/rocBLAS-6.3:= <sci-libs/rocBLAS-7.3:=
+ >=sci-libs/rocRAND-6.3:= <sci-libs/rocRAND-7.3:=
+ >=sci-libs/rocSOLVER-6.3:= <sci-libs/rocSOLVER-7.3:=
+ memefficient? ( =sci-libs/aotriton-bin-0.13*:= )
+ distributed? (
+ >=dev-util/rocm-smi-6.3:= <dev-util/rocm-smi-7.3:=
+ >=dev-util/amdsmi-6.3:= <dev-util/amdsmi-7.3:=
+ )
+ cusparselt? ( >=sci-libs/hipsparselt-6.3:= <sci-libs/hipsparselt-7.3:= )
+ )
+ xnnpack? (
+ >=sci-ml/XNNPACK-2024.11
+ dev-libs/pthreadpool
+ )
+ mkl? ( sci-libs/mkl )
+ openblas? ( sci-libs/openblas )
+"
+
+DEPEND="
+ ${RDEPEND}
+ dev-cpp/nlohmann_json
+ dev-libs/flatbuffers
+ dev-libs/FXdiv
+ dev-libs/pocketfft
+ dev-libs/psimd
+ sci-ml/FP16
+ $(python_gen_cond_dep '
+ <dev-python/pybind11-3.0.5[${PYTHON_USEDEP}]
+ dev-python/pyyaml[${PYTHON_USEDEP}]
+ dev-python/typing-extensions[${PYTHON_USEDEP}]
+ ')
+ cuda? ( ~dev-libs/cutlass-4.4.2[tools(+)] )
+ onednn? ( sci-ml/ideep )
+ rocm? (
+ >=sci-libs/hipCUB-6.3:= <sci-libs/hipCUB-7.3:=
+ >=sci-libs/rocPRIM-6.3:= <sci-libs/rocPRIM-7.3:=
+ >=sci-libs/rocThrust-6.3:= <sci-libs/rocThrust-7.3:=
+ )
+ qnnpack? ( dev-libs/clog )
+"
+
+BDEPEND="dev-build/cmake"
+
+PATCHES=(
+ "${FILESDIR}"/${P}-aotriton-fixes.patch
+ "${FILESDIR}"/${P}-glog.patch
+ "${FILESDIR}"/${P}-removekineto-pr178960.patch
+ "${FILESDIR}"/${P}-unbundle_fbgemm.patch
+ "${FILESDIR}"/${P}-unbundle_flatbuffers.patch
+ "${FILESDIR}"/${P}-unbundle_fmt.patch
+ "${FILESDIR}"/${P}-unbundle_kineto.patch
+ "${FILESDIR}"/${P}-unbundle_mkldnn.patch
+ "${FILESDIR}"/${P}-unbundle_nnpack.patch
+ "${FILESDIR}"/${P}-unbundle_pocketfft.patch
+ "${FILESDIR}"/${P}-xnnpack.patch
+)
+
+src_prepare() {
+ if use cuda && ( use flash || use memefficient ); then
+ mv "${WORKDIR}"/${FLASH_P}/* third_party/${FLASH_PN}/ || die
+ fi
+ filter-lto #bug 862672
+
+ eapply "${FILESDIR}/${PN}-2.10.0-cpp-extension-multilib.patch"
+ # Replace placeholders added by cpp-extension.patch
+ sed -e "s|%LIB_DIR%|$(get_libdir)|g" \
+ -i torch/utils/cpp_extension.py || die
+
+ # Unbundle fmt
+ sed -i \
+ -e 's|::fmt-header-only||' \
+ c10/CMakeLists.txt \
+ cmake/Dependencies.cmake \
+ torch/CMakeLists.txt \
+ || die
+
+ # tensorpipe is in system, not a build target of pytorch
+ sed -e '/target_compile_options_if_supported(tensorpipe/d' -i cmake/Dependencies.cmake || die
+
+ # Drop third_party from CMake tree
+ sed -i \
+ -e '/add_subdirectory.*third_party/d' \
+ CMakeLists.txt \
+ cmake/Dependencies.cmake \
+ cmake/ProtoBuf.cmake \
+ aten/src/ATen/CMakeLists.txt \
+ || die
+
+ # Add needed file for cutlass as symbolic link
+ ln -sf /usr/share/cutlass/examples third_party/cutlass/examples || die
+
+ distutils-r1_src_prepare
+
+ # Noisy warnings from Logging.h
+ sed -i 's/-Wextra-semi//' cmake/public/utils.cmake || die
+
+ pushd torch/csrc/jit/serialization > /dev/null || die
+ flatc --cpp --gen-mutable --scoped-enums mobile_bytecode.fbs || die
+ popd > /dev/null || die
+
+ # prefixify the hardcoded paths, after all patches are applied
+ hprefixify \
+ aten/CMakeLists.txt \
+ caffe2/CMakeLists.txt \
+ cmake/Metal.cmake \
+ cmake/Modules/*.cmake \
+ cmake/Modules_CUDA_fix/FindCUDNN.cmake \
+ cmake/Modules_CUDA_fix/upstream/FindCUDA/make2cmake.cmake \
+ cmake/Modules_CUDA_fix/upstream/FindPackageHandleStandardArgs.cmake \
+ cmake/public/LoadHIP.cmake \
+ cmake/public/cuda.cmake \
+ cmake/Dependencies.cmake \
+ tools/setup_helpers/env.py \
+ torch/CMakeLists.txt \
+ CMakeLists.txt
+
+ if use rocm; then
+ # TODO: delete, when caffe2 depends on systemwide composable_kernel
+ sed -e "s:third_party/composable_kernel:../composable_kernel-${CK_COMMIT}:g" \
+ -i aten/src/ATen/CMakeLists.txt || die
+
+ # Bug 959808: fix for gfx101x targets
+ pushd "${WORKDIR}/composable_kernel-${CK_COMMIT}" > /dev/null || die
+ eapply "${FILESDIR}"/composable-kernel-7fe50dc-expand-isa.patch
+ popd > /dev/null || die
+
+ # Workaround for libc++ issue https://github.com/llvm/llvm-project/issues/100802
+ sed -e 's/std::memcpy/memcpy/g' -i torch/headeronly/util/Half.h || die
+
+ ebegin "HIPifying cuda sources"
+ FBCODE_BUILD_TOOL="buck" ${EPYTHON} tools/amd_build/build_amd.py || die
+ eend $?
+ fi
+}
+
+src_configure() {
+ if use cuda && [[ -z ${TORCH_CUDA_ARCH_LIST} ]]; then
+ ewarn "WARNING: pytorch is being built with its default CUDA compute capabilities: 3.5 and 7.0."
+ ewarn "These may not be optimal for your GPU."
+ ewarn ""
+ ewarn "To configure pytorch with the CUDA compute capability that is optimal for your GPU,"
+ ewarn "set TORCH_CUDA_ARCH_LIST in your make.conf, and re-emerge pytorch."
+ ewarn "For example, to use CUDA capability 7.5 & 3.5, add: TORCH_CUDA_ARCH_LIST=7.5 3.5"
+ ewarn "For a Maxwell model GPU, an example value would be: TORCH_CUDA_ARCH_LIST=Maxwell"
+ ewarn ""
+ ewarn "You can look up your GPU's CUDA compute capability at https://developer.nvidia.com/cuda-gpus"
+ ewarn "or by running /opt/cuda/extras/demo_suite/deviceQuery | grep 'CUDA Capability'"
+ fi
+}
+
+python_compile() {
+ local -x BUILD_TEST=OFF
+ local -x CMAKE_BUILD_DIR="${BUILD_DIR}"
+ local -x MAX_JOBS=$(makeopts_jobs)
+ local -x PYTORCH_BUILD_VERSION=${PV}
+ local -x PYTORCH_BUILD_NUMBER=0
+ local -x USE_CCACHE=OFF
+ local -x USE_CUDA=$(usex cuda)
+ local -x USE_DISTRIBUTED=$(usex distributed)
+ local -x USE_FBGEMM=$(usex fbgemm)
+ local -x USE_FLASH_ATTENTION=$(usex flash)
+ local -x USE_GFLAGS=ON
+ local -x USE_GLOG=ON
+ local -x USE_GLOO=$(usex gloo)
+ local -x USE_ITT=OFF
+ local -x USE_KINETO=$(usex kineto)
+ local -x USE_KLEIDIAI=OFF # TODO
+ local -x USE_MAGMA=OFF # TODO: In GURU as sci-libs/magma
+ local -x USE_MEM_EFF_ATTENTION=$(usex memefficient)
+ local -x USE_MIMALLOC=$(usex mimalloc)
+ local -x USE_MKLDNN=$(usex onednn)
+ local -x USE_MPI=$(usex mpi)
+ local -x USE_NNPACK=$(usex nnpack)
+ local -x USE_NUMA=OFF
+ local -x USE_NUMPY=$(usex numpy)
+ local -x USE_OPENCL=$(usex opencl)
+ local -x USE_OPENMP=$(usex openmp)
+ local -x USE_PYTORCH_QNNPACK=$(usex qnnpack)
+ local -x USE_PYTORCH_METAL=OFF
+ local -x USE_ROCM=$(usex rocm)
+ local -x USE_SYSTEM_LIBS=ON
+ local -x USE_SYSTEM_XNNPACK=$(usex xnnpack)
+ local -x USE_TENSORPIPE=$(usex distributed $(usex !rocm))
+ local -x DUSE_UCC=OFF
+ local -x USE_VALGRIND=OFF
+ local -x USE_XNNPACK=$(usex xnnpack)
+ local -x USE_XPU=OFF
+
+ if use cuda; then
+ # bug 867706 926116
+ cuda_add_sandbox
+ addpredict "/dev/char/"
+
+ local -x CMAKE_CUDA_FLAGS="$(cuda_gccdir -f | tr -d \")"
+ local -x TORCH_CUDA_ARCH_LIST="${TORCH_CUDA_ARCH_LIST:7.0}"
+ local -x USE_CUDNN=ON
+ local -x USE_FLASH_ATTENTION=OFF
+ local -x USE_MEM_EFF_ATTENTION=OFF
+ local -x USE_NCCL=OFF # TODO: NVIDIA Collective Communication Library
+
+ elif use rocm; then
+ export PYTORCH_ROCM_ARCH="$(get_amdgpu_flags)"
+
+ if use memefficient; then
+ export AOTRITON_INSTALLED_PREFIX="${ESYSROOT}/usr"
+ fi
+
+ local -x CMAKE_REQUIRE_FIND_PACKAGE_HIP=ON
+ local -x USE_NCCL=$(usex nccl)
+ local -x CMAKE_DISABLE_FIND_PACKAGE_hipsparselt=$(usex !cusparselt) # disable automagic
+ local -x USE_ROCM_CK_SDPA=OFF # requires flash + aiter, works only on gfx90a/gfx942/gfx950
+ local -x ROCM_PATH=/usr
+ local -x HIP_CLANG_PATH=$(hipconfig --hipclangpath)
+
+ # ROCm libraries produce too much warnings
+ append-cxxflags -Wno-deprecated-declarations -Wno-unused-result -Wno-unused-value
+ else
+ local -x USE_NCCL=OFF
+ fi
+
+ distutils-r1_python_compile develop sdist
+}