ROCm · pragupta · Nov 3, 2025 · Nov 4, 2025 · Nov 4, 2025 · Nov 4, 2025
@@ -7,13 +7,13 @@ ENV LC_ALL en_US.UTF-8
 ENV LANG en_US.UTF-8
 ENV LANGUAGE en_US.UTF-8
 
-ARG DEVTOOLSET_VERSION=11
+ARG DEVTOOLSET_VERSION=13
 
 RUN yum -y update
 RUN yum -y install epel-release
 # install glibc-langpack-en make sure en_US.UTF-8 locale is available
 RUN yum -y install glibc-langpack-en
-RUN yum install -y sudo wget curl perl util-linux xz bzip2 git patch which perl zlib-devel openssl-devel yum-utils autoconf automake make gcc-toolset-${DEVTOOLSET_VERSION}-toolchain
+RUN yum install -y sudo wget curl perl util-linux xz bzip2 git patch which perl zlib-devel openssl-devel yum-utils autoconf automake make gcc-toolset-${DEVTOOLSET_VERSION}-gcc gcc-toolset-${DEVTOOLSET_VERSION}-gcc-c++ gcc-toolset-${DEVTOOLSET_VERSION}-gcc-gfortran gcc-toolset-${DEVTOOLSET_VERSION}-gdb
 # Just add everything as a safe.directory for git since these will be used in multiple places with git
 RUN git config --global --add safe.directory '*'
 ENV PATH=/opt/rh/gcc-toolset-${DEVTOOLSET_VERSION}/root/usr/bin:$PATH
@@ -41,6 +41,7 @@ RUN bash ./install_conda.sh && rm install_conda.sh
 # Install CUDA
 FROM base as cuda
 ARG CUDA_VERSION=12.6
+ARG DEVTOOLSET_VERSION=13
 RUN rm -rf /usr/local/cuda-*
 ADD ./common/install_cuda.sh install_cuda.sh
 COPY ./common/install_nccl.sh install_nccl.sh
@@ -50,7 +51,8 @@ ENV CUDA_HOME=/usr/local/cuda-${CUDA_VERSION}
 # Preserve CUDA_VERSION for the builds
 ENV CUDA_VERSION=${CUDA_VERSION}
 # Make things in our path by default
-ENV PATH=/usr/local/cuda-${CUDA_VERSION}/bin:$PATH
+ENV PATH=/usr/local/cuda-${CUDA_VERSION}/bin:/opt/rh/gcc-toolset-${DEVTOOLSET_VERSION}/root/usr/bin:$PATH
+
 
 FROM cuda as cuda12.6
 RUN bash ./install_cuda.sh 12.6
@@ -68,8 +70,22 @@ FROM cuda as cuda13.0
 RUN bash ./install_cuda.sh 13.0
 ENV DESIRED_CUDA=13.0
 
-FROM ${ROCM_IMAGE} as rocm
+FROM ${ROCM_IMAGE} as rocm_base
+ARG DEVTOOLSET_VERSION=13
+ENV LC_ALL en_US.UTF-8
+ENV LANG en_US.UTF-8
+ENV LANGUAGE en_US.UTF-8
+# Install devtoolset on ROCm base image
+RUN yum -y update && \
+    yum -y install epel-release && \
+    yum -y install glibc-langpack-en && \
+    yum install -y sudo wget curl perl util-linux xz bzip2 git patch which perl zlib-devel openssl-devel yum-utils autoconf automake make gcc-toolset-${DEVTOOLSET_VERSION}-gcc gcc-toolset-${DEVTOOLSET_VERSION}-gcc-c++ gcc-toolset-${DEVTOOLSET_VERSION}-gcc-gfortran gcc-toolset-${DEVTOOLSET_VERSION}-gdb
+RUN git config --global --add safe.directory '*'
+ENV PATH=/opt/rh/gcc-toolset-${DEVTOOLSET_VERSION}/root/usr/bin:$PATH
+
+FROM rocm_base as rocm
 ARG PYTORCH_ROCM_ARCH
+ARG DEVTOOLSET_VERSION=13
 ENV PYTORCH_ROCM_ARCH ${PYTORCH_ROCM_ARCH}
 ADD ./common/install_mkl.sh install_mkl.sh
 RUN bash ./install_mkl.sh && rm install_mkl.sh
@@ -88,6 +104,7 @@ COPY --from=cuda13.0  /usr/local/cuda-13.0 /usr/local/cuda-13.0
 
 # Final step
 FROM ${BASE_TARGET} as final
+ARG DEVTOOLSET_VERSION=13
 COPY --from=openssl            /opt/openssl           /opt/openssl
 COPY --from=patchelf           /patchelf              /usr/local/bin/patchelf
 COPY --from=conda              /opt/conda             /opt/conda

@@ -63,7 +63,7 @@ docker build \
   --target final \
   --progress plain \
   --build-arg "BASE_TARGET=${BASE_TARGET}" \
-  --build-arg "DEVTOOLSET_VERSION=11" \
+  --build-arg "DEVTOOLSET_VERSION=13" \
   ${EXTRA_BUILD_ARGS} \
   -t ${tmp_tag} \
   $@ \

@@ -261,19 +261,29 @@ case "$tag" in
     PYTHON_VERSION=3.10
     CUDA_VERSION=12.8.1
     ;;
-  pytorch-linux-jammy-aarch64-py3.10-gcc11)
+  pytorch-linux-jammy-aarch64-py3.10-gcc13)
     ANACONDA_PYTHON_VERSION=3.10
-    GCC_VERSION=11
+    GCC_VERSION=13
     ACL=yes
     VISION=yes
     OPENBLAS=yes
     # snadampal: skipping llvm src build install because the current version
     # from pytorch/llvm:9.0.1 is x86 specific
     SKIP_LLVM_SRC_BUILD_INSTALL=yes
     ;;
-  pytorch-linux-jammy-aarch64-py3.10-gcc11-inductor-benchmarks)
+  pytorch-linux-jammy-aarch64-py3.10-clang21)
     ANACONDA_PYTHON_VERSION=3.10
-    GCC_VERSION=11
+    CLANG_VERSION=21
+    ACL=yes
+    VISION=yes
+    OPENBLAS=yes
+    # snadampal: skipping llvm src build install because the current version
+    # from pytorch/llvm:9.0.1 is x86 specific
+    SKIP_LLVM_SRC_BUILD_INSTALL=yes
+    ;;
+  pytorch-linux-jammy-aarch64-py3.10-gcc13-inductor-benchmarks)
+    ANACONDA_PYTHON_VERSION=3.10
+    GCC_VERSION=13
     ACL=yes
     VISION=yes
     OPENBLAS=yes

@@ -1 +1,5 @@
+<<<<<<< HEAD
 ac80c4190aa0321f761a08af97e1e1eee41f01d9
+=======
+bfeb066872bc1e8b2d2bc0a3b295b99dd77206e7
+>>>>>>> upstream/main
@@ -8,8 +8,8 @@ if [ -n "$CLANG_VERSION" ]; then
     # work around ubuntu apt-get conflicts
     sudo apt-get -y -f install
     wget --no-check-certificate -O - https://apt.llvm.org/llvm-snapshot.gpg.key | sudo apt-key add -
-    if [[ $CLANG_VERSION == 18 ]]; then
-      apt-add-repository "deb http://apt.llvm.org/jammy/ llvm-toolchain-jammy-18 main"
+    if [[ $CLANG_VERSION -ge 18 ]]; then
+      apt-add-repository "deb http://apt.llvm.org/jammy/ llvm-toolchain-jammy-${CLANG_VERSION} main"
     fi
   fi
 

@@ -7,11 +7,11 @@ if [ -n "$GCC_VERSION" ]; then
   # Need the official toolchain repo to get alternate packages
   add-apt-repository ppa:ubuntu-toolchain-r/test
   apt-get update
-  apt-get install -y g++-$GCC_VERSION
+  apt-get install -y g++-$GCC_VERSION gfortran-$GCC_VERSION
   update-alternatives --install /usr/bin/gcc gcc /usr/bin/gcc-"$GCC_VERSION" 50
   update-alternatives --install /usr/bin/g++ g++ /usr/bin/g++-"$GCC_VERSION" 50
   update-alternatives --install /usr/bin/gcov gcov /usr/bin/gcov-"$GCC_VERSION" 50
-
+  update-alternatives --install /usr/bin/gfortran gfortran /usr/bin/gfortran-"$GCC_VERSION" 50
 
   # Cleanup package manager
   apt-get autoclean && apt-get clean

@@ -10,6 +10,7 @@ git clone https://github.com/OpenMathLib/OpenBLAS.git -b "${OPENBLAS_VERSION}" -
 
 OPENBLAS_CHECKOUT_DIR="OpenBLAS"
 OPENBLAS_BUILD_FLAGS="
+CC=gcc
 NUM_THREADS=128
 USE_OPENMP=1
 NO_SHARED=0

@@ -1 +1 @@
-3.5.0
+3.5.1
diff --git a/.ci/magma-rocm/build_magma.sh b/.ci/magma-rocm/build_magma.sh
@@ -6,8 +6,8 @@ set -eou pipefail
 # The script expects DESIRED_CUDA and PACKAGE_NAME to be set
 ROOT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
 
-# post merge of https://github.com/icl-utk-edu/magma/pull/65
-MAGMA_VERSION=c0792ae825fb36872784892ea643dd6f3456bc5f
+# https://github.com/icl-utk-edu/magma/pull/65
+MAGMA_VERSION=d6e4117bc88e73f06d26c6c2e14f064e8fc3d1ec
 
 # Folders for the build
 PACKAGE_FILES=${ROOT_DIR}/magma-rocm/package_files # metadata
@@ -20,7 +20,7 @@ mkdir -p ${PACKAGE_DIR} ${PACKAGE_OUTPUT}/linux-64 ${PACKAGE_BUILD} ${PACKAGE_RE
 
 # Fetch magma sources and verify checksum
 pushd ${PACKAGE_DIR}
-git clone https://github.com/icl-utk-edu/magma
+git clone https://github.com/jeffdaily/magma
 pushd magma
 git checkout ${MAGMA_VERSION}
 popd

diff --git a/.ci/pytorch/test.sh b/.ci/pytorch/test.sh
@@ -337,7 +337,7 @@ test_python() {
 
 test_python_smoke() {
   # Smoke tests for H100/B200
-  time python test/run_test.py --include test_matmul_cuda test_scaled_matmul_cuda inductor/test_fp8 inductor/test_max_autotune inductor/test_cutedsl_grouped_mm $PYTHON_TEST_EXTRA_OPTION --upload-artifacts-while-running
+  time python test/run_test.py --include test_matmul_cuda test_scaled_matmul_cuda inductor/test_fp8 inductor/test_max_autotune $PYTHON_TEST_EXTRA_OPTION --upload-artifacts-while-running
   assert_git_not_dirty
 }
 

diff --git a/.github/ISSUE_TEMPLATE/release-feature-request.yml b/.github/ISSUE_TEMPLATE/release-feature-request.yml
@@ -1,11 +1,11 @@
-name: 🚀 Release highlight for proposed Feature
+name: 🚀 New Feature for Release
 description: Submit a Release highlight for proposed Feature
 labels: ["release-feature-request"]
 
 body:
 - type: textarea
   attributes:
-    label: Release highlight for proposed Feature
+    label: New Feature for Release
     description: >
       Example: “A torch.special module, analogous to SciPy's special module.”
 - type: input

diff --git a/.github/actions/pytest-cache-download/action.yml b/.github/actions/pytest-cache-download/action.yml
@@ -38,9 +38,9 @@ runs:
       run: |
         python3 .github/scripts/pytest_cache.py \
           --download \
-          --cache_dir $GITHUB_WORKSPACE/$CACHE_DIR \
-          --pr_identifier $GITHUB_REF \
-          --job_identifier $JOB_IDENTIFIER \
-          --temp_dir $RUNNER_TEMP \
-          --repo $REPO \
-          --bucket $BUCKET \
+          --cache_dir "$GITHUB_WORKSPACE/$CACHE_DIR" \
+          --pr_identifier "$GITHUB_REF" \
+          --job_identifier "$JOB_IDENTIFIER" \
+          --temp_dir "$RUNNER_TEMP" \
+          --repo "$REPO" \
+          --bucket "$BUCKET" \
diff --git a/.github/actions/pytest-cache-upload/action.yml b/.github/actions/pytest-cache-upload/action.yml
@@ -47,11 +47,11 @@ runs:
       run: |
         python3 .github/scripts/pytest_cache.py \
           --upload \
-          --cache_dir $GITHUB_WORKSPACE/$CACHE_DIR \
-          --pr_identifier $GITHUB_REF \
-          --job_identifier $JOB_IDENTIFIER \
-          --sha $SHA \
-          --test_config $TEST_CONFIG \
-          --shard $SHARD \
-          --repo $REPO \
-          --temp_dir $RUNNER_TEMP \
+          --cache_dir "$GITHUB_WORKSPACE/$CACHE_DIR" \
+          --pr_identifier "$GITHUB_REF" \
+          --job_identifier "$JOB_IDENTIFIER" \
+          --sha "$SHA" \
+          --test_config "$TEST_CONFIG" \
+          --shard "$SHARD" \
+          --repo "$REPO" \
+          --temp_dir "$RUNNER_TEMP" \
diff --git a/.github/ci_commit_pins/audio.txt b/.github/ci_commit_pins/audio.txt
@@ -1 +1 @@
-3b0e7a6f192ca2715e7e6cbe5db007aea7165fe2
+ad5816f0eee1c873df1b7d371c69f1f811a89387
diff --git a/.github/copilot-instructions.md b/.github/copilot-instructions.md
@@ -0,0 +1,125 @@
+# PyTorch Copilot Instructions
+
+This is the PyTorch machine learning framework codebase. These instructions help AI agents navigate and contribute effectively.
+
+## Architecture Overview
+
+### Core Components
+
+- **c10/** - Core library (C++-10 compatible) for essential, binary-size-conscious functionality
+- **aten/** - ATen tensor library (C++), PyTorch's foundation without autograd
+  - `aten/src/ATen/native/` - Modern operator implementations (CPU/CUDA/MPS/sparse)
+  - `aten/src/ATen/native/native_functions.yaml` - **Critical**: Declarative operator registry
+- **torch/** - Python bindings and public API
+  - `torch/csrc/` - C++ Python bindings (hand-written and generated)
+  - `torch/csrc/autograd/` - Reverse-mode automatic differentiation
+  - `torch/csrc/jit/` - TorchScript JIT compiler
+- **torchgen/** - Code generation tooling that reads `native_functions.yaml`
+- **tools/** - Build scripts, autograd derivatives, code generation
+
+### The Code Generation Workflow
+
+**Most operator changes require editing `native_functions.yaml`**, not direct C++ files. This YAML file:
+1. Declares operator signatures, variants (function/method), and dispatch behavior
+2. Gets processed by `torchgen/` to generate C++/Python bindings
+3. Produces headers in `build/aten/src/ATen/` during compilation
+
+Example entry structure:
+```yaml
+- func: my_op(Tensor self, Scalar alpha=1) -> Tensor
+  variants: function, method
+  dispatch:
+    CPU: my_op_cpu
+    CUDA: my_op_cuda
+```
+
+After editing `native_functions.yaml`, implement kernels in `aten/src/ATen/native/` (see `aten/src/ATen/native/README.md`).
+
+## Development Workflows
+
+### Building from Source
+
+**Never run `setup.py` directly** - use pip with editable install:
+```bash
+python -m pip install --no-build-isolation -v -e .
+```
+
+Speed up builds:
+- `DEBUG=1` - Debug symbols with `-g -O0`
+- `USE_CUDA=0` - Skip CUDA compilation
+- `BUILD_TEST=0` - Skip C++ test binaries
+- Install `ninja` (`pip install ninja`) for faster builds
+- Use `ccache` for incremental compilation caching
+
+Rebuild specific targets: `(cd build && ninja <target>)`
+
+### Testing
+
+**Critical**: DO NOT run entire test suites. Run specific tests only:
+```bash
+python test/test_torch.py TestTorch.test_specific_case
+```
+
+**Test structure**: All tests use `torch.testing._internal.common_utils`:
+```python
+from torch.testing._internal.common_utils import run_tests, TestCase
+
+class TestFeature(TestCase):
+    def test_something(self):
+        # Use self.assertEqual for tensor comparisons
+        pass
+
+if __name__ == "__main__":
+    run_tests()
+```
+
+**For bug fixes**: Create a standalone reproduction script first, verify it fails, then fix and add to appropriate test file.
+
+### Linting
+
+Run linter (not pre-commit): `lintrunner -a` (auto-applies fixes)
+
+## Project-Specific Conventions
+
+### Memory and Storage
+- **Storage is never nullptr** (but `StorageImpl.data` may be nullptr for unallocated outputs)
+- CUDA device info lives in storage objects
+
+### Python-C++ Integration (`torch/csrc/`)
+- Always include `Python.h` **first** to avoid `_XOPEN_SOURCE` redefinition errors
+- Use `pybind11::gil_scoped_acquire` before calling Python API or using `THPObjectPtr`
+- Wrap entry points with `HANDLE_TH_ERRORS` / `END_HANDLE_TH_ERRORS` for exception conversion
+
+### Dispatch System
+- PyTorch uses operator dispatch to route calls to backend-specific kernels
+- Prefer `CompositeExplicitAutograd` dispatch when writing device-agnostic compound ops
+- See `aten/src/ATen/native/README.md` for dispatch keyword guidance
+
+## Git Workflow (AI Agent Specific)
+
+When preparing PRs from this environment:
+```bash
+git stash -u
+git reset --hard $(cat /tmp/orig_work.txt)  # Reset to LOCAL branch
+git stash pop
+# Resolve conflicts if necessary
+```
+
+## Common Gotchas
+
+1. **Editing generated files** - If it's in `build/`, don't edit it. Edit the source template or `native_functions.yaml`
+2. **NVCC template compilation** - NVCC is stricter about C++ than gcc/clang; code working on Linux may fail Windows CI
+3. **Windows symbol visibility** - Use `TORCH_API` macros for exported symbols (required on Windows, optional on Linux)
+4. **No internet access** - DO NOT attempt to install dependencies during development
+
+## Key Files Reference
+
+- `AGENTS.md` - Instructions specific to AI coding agents
+- `CONTRIBUTING.md` - Comprehensive human contributor guide
+- `GLOSSARY.md` - Terminology (ATen, kernels, operations, JIT, TorchScript)
+- `aten/src/ATen/native/README.md` - Operator implementation guide
+- `tools/autograd/derivatives.yaml` - Gradient definitions for autograd
+
+## Performance Debugging
+
+Use `TORCH_SHOW_CPP_STACKTRACES=1` for C++ traces in Python errors. For profiling, prefer `py-spy` over manual instrumentation.
diff --git a/.github/workflows/_rocm-test.yml b/.github/workflows/_rocm-test.yml
@@ -97,8 +97,8 @@ jobs:
         shell: bash
         run: |
           ngpu=$(rocminfo | grep -c -E 'Name:.*\sgfx')
-          if [[ $ngpu -lt 4 ]]; then
-            echo "Error: only $ngpu GPU(s) detected, at least 4 GPUs are needed for distributed jobs"
+          if [[ $ngpu -lt 2 ]]; then #We are temporarily reducing this down to 2 from 4 so that we can run tests on nodes with less gpus.
+            echo "Error: only $ngpu GPU(s) detected, at least 2 GPUs are needed for distributed jobs"
             exit 1
           fi
 

diff --git a/.github/workflows/_xpu-test.yml b/.github/workflows/_xpu-test.yml
@@ -344,5 +344,21 @@ jobs:
           if-no-files-found: ignore
           path: ./**/core.[1-9]*
 
+      - name: Authenticate with AWS
+        uses: aws-actions/configure-aws-credentials@ececac1a45f3b08a01d2dd070d28d111c5fe6722 # v4.1.0
+        with:
+          role-to-assume: arn:aws:iam::308535385114:role/gha_workflow_upload-benchmark-results
+          # The max duration enforced by the server side
+          role-duration-seconds: 18000
+          aws-region: us-east-1
+
+      - name: Upload the benchmark results
+        uses: pytorch/test-infra/.github/actions/upload-benchmark-results@main
+        with:
+          benchmark-results-dir: test/test-reports
+          dry-run: false
+          schema-version: v3
+          github-token: ${{ secrets.GITHUB_TOKEN }}
+
       - name: Teardown XPU
         uses: ./.github/actions/teardown-xpu
Original file line number	Diff line number	Diff line change
		@@ -1 +1 @@
		3b0e7a6f192ca2715e7e6cbe5db007aea7165fe2
		ad5816f0eee1c873df1b7d371c69f1f811a89387