torch-xpu-ops
No description available.
파일 탐색기
최종 버전 다운로드 (.zip)- arc-runner-cleanup.sh
- arc-runner-entrypoint.sh
- arc-xpu-ubuntu-24.04-values.yaml
- Dockerfile.arc-runner-ubuntu-24.04
- github-token.secret.example.yaml
- README.md
- setup-arc-runner.sh
- huggingface_models_list.txt
- run_ut.sh
- skip.yaml
- timm_models_list.txt
- torchbench_models_list.txt
- huggingface_models_list.txt
- skip.yaml
- timm_models_list.txt
- torchbench_models_list.txt
- huggingface_models_list.txt
- skip.yaml
- timm_models_list.txt
- torchbench_models_list.txt
- huggingface.yaml
- huggingface_models_list.txt
- timm_models.yaml
- timm_models_list.txt
- torchbench.yaml
- torchbench_models_list.txt
- install_xpu.sh
- Dockerfile
- README.md
- validate-pr-review.md
- l0_igpu_check.py
- SKILL.md
- SKILL.md
- SKILL.md
- SKILL.md
- SKILL.md
- SKILL.md
- SKILL.md
- SKILL.md
- SKILL.md
- SKILL.md
- SKILL.md
- SKILL.md
- SKILL.md
- SKILL.md
- environment-setup.md
- execution-modes.md
- SKILL.md
- SKILL.md
- SKILL.md
- hardware_specs.yaml
- fleet-summary.md
- graph-consistency.md
- inputs.md
- insights.md
- methodology.md
- per-model-report.md
- troubleshooting.md
- SKILL.md
- bc-guidelines.md
- pr-submission-guidelines.md
- review-checklist.md
- torch-xpu-ops-review-notes.md
- SKILL.md
- SKILL.md
- SKILL.md
- SKILL.md
- xpu-alignment-buckets-and-routing.md
- xpu-alignment-environment-setup.md
- xpu-alignment-report-and-issue-format.md
- xpu-alignment-repro-precheck.md
- xpu-alignment-review.md
- SKILL.md
- reference.md
- SKILL.md
- collect_failures.py
- SKILL.md
- SKILL.md
- reference.md
- SKILL.md
- SKILL.md
- SKILL.md
- action.yml
- action.yml
- action.yml
- action.yml
- action.yml
- action.yml
- torchbench.txt
- triton.txt
- inductor_huggingface_inference.csv
- inductor_huggingface_training.csv
- inductor_timm_models_inference.csv
- inductor_timm_models_training.csv
- inductor_torchbench_inference.csv
- inductor_torchbench_training.csv
- inductor_huggingface_inference.csv
- inductor_huggingface_training.csv
- inductor_timm_models_inference.csv
- inductor_timm_models_training.csv
- inductor_torchbench_inference.csv
- inductor_torchbench_training.csv
- check_expected.py
- xpu-kernels.instructions.md
- xpu-tests.instructions.md
- xpu-yaml.instructions.md
- agent-issue-body-nonbug.yml
- agent-issue-body.yml
- bug-report.yml
- ci-failure-tracking.yml
- documentation.yml
- dynamic-skip.yml
- feature-request.yml
- apply_torch_pr.py
- bisect_search.sh
- bot_cherry_pick.py
- bot_revert.py
- bot_ut_check.py
- build.sh
- build_windows.py
- calculate_best_perf.py
- check-topology.py
- check-transformers.py
- check-ut.py
- check_aot.py
- check_llama_baseline.sh
- check_nightly_status.py
- compare-e2e.py
- compare-ut.py
- conftest.py
- detect-devices.sh
- e2e_summary.sh
- env.sh
- fetch_issues.sh
- freq-fixed.sh
- get_issue.py
- inductor_summary.py
- inductor_xpu_test.sh
- install_xpu.bat
- lintrunner.sh
- llama_baseline.csv
- llama_summary.py
- microbench_summary.py
- op_calculate_best_perf.py
- op_perf_comparison.py
- parse-junitxml.py
- perf_comparison.py
- prepare_pytorch.py
- spec.py
- ut_result_check.sh
- _arc_runner_smoke.yml
- _linux_accelerate.yml
- _linux_build.yml
- _linux_e2e.yml
- _linux_e2e_summary.yml
- _linux_op_benchmark.yml
- _linux_reproducer.yml
- _linux_test_image.yml
- _linux_transformers.yml
- _linux_ut.yml
- _performance_comparison.yml
- _windows_build.yml
- _windows_ut.yml
- acceptance_ondemand.yml
- auto_label.yml
- auto_merge_by_comment.yml
- bisect_search.yml
- bot.yml
- ci_break_monitor.yml
- issue_operator.yml
- nightly_ondemand.yml
- pull.yml
- pull_dpclang.yml
- copilot-instructions.md
- run_sycl.cmake
- FindONEMKL.cmake
- FindSYCL.cmake
- FindSYCLToolkit.cmake
- FindXCCL.cmake
- BuildFlags.cmake
- ONEMKL.cmake
- SYCL.cmake
- SYCLTLA.cmake
- XCCL.cmake
- torch_xpu_ops.jpg
- NestedTensorBinaryOpsKernels.cpp
- NestedTensorBinaryOpsKernels.h
- NestedTensorTransformerFunctionKernels.cpp
- NestedTensorTransformerFunctionKernels.h
- NestedTensorBinaryOps.cpp
- NestedTensorMatmul.cpp
- NestedTensorTransformerFunctions.cpp
- NestedTensorTransformerFunctions.cpp
- AffineQuantizerKernels.cpp
- AffineQuantizerKernels.h
- FakeQuantizeCoreKernels.cpp
- FakeQuantizeCoreKernels.h
- FusedObsFakeQuantKernels.cpp
- FusedObsFakeQuantKernels.h
- MakePerTensorQuantizedTensorKernels.cpp
- MakePerTensorQuantizedTensorKernels.h
- QuantizedMaxPool2d.cpp
- QuantizedMaxPool2d.h
- AffineQuantizer.cpp
- FakeQuantizeCore.cpp
- FusedObsFakeQuant.cpp
- MakePerTensorQuantizedTensor.cpp
- QuantizedMaxPool2d.cpp
- SparseBinaryOpIntersectionKernels.cpp
- SparseBinaryOpIntersectionKernels.h
- SparseCsrTensorAddKernels.cpp
- SparseCsrTensorAddKernels.h
- SparseCsrTensorMathKernels.cpp
- SparseCsrTensorMathKernels.h
- SparseSoftmaxKernels.cpp
- SparseSoftmaxKernels.h
- SparseTensorKernels.cpp
- SparseTensorKernels.h
- SparseTensorMathKernels.cpp
- SparseTensorMathKernels.h
- SparseBinaryOpIntersection.cpp
- SparseBlas.cpp
- SparseCsrTensorMath.cpp
- SparseSoftmax.cpp
- SparseTensor.cpp
- SparseTensorMath.cpp
- AttentionKernels.cpp
- AttentionKernels.h
- xe_fmha_fwd_epilogue.h
- xe_fmha_fwd_mainloop.h
- xe_fmha_fwd_kernel.h
- xe_tile_scheduler.h
- dropout.h
- flash_api.h
- generate_kernels.py
- mha_bwd.cpp
- mha_bwd.h
- mha_bwd_hdim128.cpp
- mha_bwd_hdim192.cpp
- mha_bwd_hdim256.cpp
- mha_bwd_hdim32.cpp
- mha_bwd_hdim64.cpp
- mha_bwd_hdim96.cpp
- mha_bwd_launch.h
- mha_common.h
- mha_fwd.cpp
- mha_fwd.h
- mha_fwd_hdim128.cpp
- mha_fwd_hdim192.cpp
- mha_fwd_hdim256.cpp
- mha_fwd_hdim32.cpp
- mha_fwd_hdim64.cpp
- mha_fwd_hdim96.cpp
- mha_fwd_launch.h
- philox.h
- flash_api.cpp
- flash_api.h
- utils.h
- Attention.cpp
- SDPUtils.cpp
- SDPUtils.h
- BatchLinearAlgebra.cpp
- BatchLinearAlgebra.h
- BlasImpl.cpp
- BlasImpl.h
- SpectralOps.cpp
- SpectralOps.h
- TorchToMklType.h
- PSTLFunctions.h
- AbsKernel.cpp
- AbsKernel.h
- ActivationEluKernels.cpp
- ActivationEluKernels.h
- ActivationGeluKernel.cpp
- ActivationGeluKernel.h
- ActivationGluKernels.cpp
- ActivationGluKernels.h
- ActivationHardshrinkKernels.cpp
- ActivationHardshrinkKernels.h
- ActivationHardsigmoidKernels.cpp
- ActivationHardsigmoidKernels.h
- ActivationHardswishKernels.cpp
- ActivationHardswishKernels.h
- ActivationHardtanhKernels.cpp
- ActivationHardtanhKernels.h
- ActivationLeakyReluKernels.cpp
- ActivationLeakyReluKernels.h
- ActivationLogSigmoidKernels.cpp
- ActivationLogSigmoidKernels.h
- ActivationMishKernels.cpp
- ActivationMishKernels.h
- ActivationPreluKernels.cpp
- ActivationPreluKernels.h
- ActivationSiluKernels.cpp
- ActivationSiluKernels.h
- ActivationSoftplusKernels.cpp
- ActivationSoftplusKernels.h
- ActivationSoftshrinkKernels.cpp
- ActivationSoftshrinkKernels.h
- ActivationThresholdKernel.cpp
- ActivationThresholdKernel.h
- AdaptiveAveragePooling2dKernels.cpp
- AdaptiveAveragePooling2dKernels.h
- AdaptiveAveragePooling3dKernels.cpp
- AdaptiveAveragePooling3dKernels.h
- AdaptiveMaxPooling2dKernels.cpp
- AdaptiveMaxPooling2dKernels.h
- AdaptiveMaxPooling3dKernels.cpp
- AdaptiveMaxPooling3dKernels.h
- AiryAiKernel.cpp
- AiryAiKernel.h
- AmpKernels.cpp
- AmpKernels.h
- Atomics.h
- AveragePool2dKernels.cpp
- AveragePool2dKernels.h
- AveragePool3dKernels.cpp
- AveragePool3dKernels.h
- BatchKernel.h
- BatchNormKernels.cpp
- BatchNormKernels.h
- BesselJ0Kernel.cpp
- BesselJ0Kernel.h
- BesselJ1Kernel.cpp
- BesselJ1Kernel.h
- BesselY0Kernel.cpp
- BesselY0Kernel.h
- BesselY1Kernel.cpp
- BesselY1Kernel.h
- BinaryBitwiseOpsKernels.cpp
- BinaryBitwiseOpsKernels.h
- BinaryDivFloorKernel.cpp
- BinaryDivTrueKernel.cpp
- BinaryDivTruncKernel.cpp
- BinaryGeometricKernels.cpp
- BinaryGeometricKernels.h
- BinaryInternal.h
- BinaryKernels.cpp
- BinaryKernels.h
- BinaryLogicalOpsKernels.cpp
- BinaryLogicalOpsKernels.h
- BinaryMiscBackwardOpsKernels.cpp
- BinaryMiscBackwardOpsKernels.h
- BinaryMiscOpsKernels.cpp
- BinaryMiscOpsKernels.h
- BinaryRemainderKernel.cpp
- BinaryRemainderKernel.h
- BinaryShiftOpsKernels.cpp
- BinaryShiftOpsKernels.h
- BucketizationKernels.cpp
- BucketizationKernels.h
- ChebyshevPolynomialKernels.h
- ChebyshevPolynomialTKernel.cpp
- ChebyshevPolynomialUKernel.cpp
- ChebyshevPolynomialVKernel.cpp
- ChebyshevPolynomialWKernel.cpp
- Col2ImKernel.cpp
- Col2ImKernel.h
- CompareKernels.cpp
- CompareKernels.h
- ComplexKernels.cpp
- ComplexKernels.h
- CopyKernel.cpp
- CopyKernel.h
- CopysignKernel.cpp
- CopysignKernel.h
- CrossKernel.cpp
- CrossKernel.h
- CumminmaxKernel.cpp
- CumprodKernel.cpp
- CumsumKernel.cpp
- DeformConv2dKernels.cpp
- DeformConv2dKernels.h
- DepthwiseConv2dKernels.cpp
- DepthwiseConv2dKernels.h
- DepthwiseConv3dKernels.cpp
- DepthwiseConv3dKernels.h
- Dequant_int4.cpp
- Dequant_int4.h
- DeviceAddCmulCdiv.h
- DilatedMaxPool2d.cpp
- DilatedMaxPool2d.h
- DilatedMaxPool3d.cpp
- DilatedMaxPool3d.h
- DistanceKernels.cpp
- DistanceKernels.h
- DistributionBernoulli.cpp
- DistributionCauchyKernel.cpp
- DistributionExponentialKernel.cpp
- DistributionGeometricKernel.cpp
- DistributionKernels.h
- DistributionLogNormalKernel.cpp
- DistributionNormal.cpp
- DistributionRandomKernel.cpp
- Distributions.cpp
- Distributions.h
- DistributionTemplates.h
- DistributionUniform.cpp
- Dropout.cpp
- DropoutKernels.h
- ElementwiseInvoke.h
- Embedding.cpp
- EmbeddingBackwardKernel.h
- EmbeddingBag.cpp
- EmbeddingBag.h
- EmbeddingBagKernels.h
- EmbeddingKernels.h
- FFTKernelFunctor.cpp
- FFTKernelFunctor.h
- FillKernel.cpp
- FillKernel.h
- ForeachBinaryOpListKernels.cpp
- ForeachBinaryOpListKernels.h
- ForeachBinaryOpScalarKernels.cpp
- ForeachBinaryOpScalarKernels.h
- ForeachBinaryOpScalarListKernels.cpp
- ForeachBinaryOpScalarListKernels.h
- ForeachBinaryOpScalarTensorKernels.cpp
- ForeachBinaryOpScalarTensorKernels.h
- ForeachCopyKernels.cpp
- ForeachCopyKernels.h
- ForeachFunctors.h
- ForeachPointwiseKernels.cpp
- ForeachPointwiseOpListKernels.h
- ForeachPointwiseOpScalarKernels.h
- ForeachPointwiseOpScalarListKernels.h
- ForeachReduceKernels.cpp
- ForeachReduceKernels.h
- ForeachTernaryKernels.cpp
- ForeachTernaryOpListKernels.h
- ForeachTernaryOpScalarKernels.h
- ForeachTernaryOpScalarListKernels.h
- ForeachUnaryKernels.cpp
- ForeachUnaryKernels.h
- FractionalMaxPool2dKernels.cpp
- FractionalMaxPool2dKernels.h
- FractionalMaxPool3dKernels.cpp
- FractionalMaxPool3dKernels.h
- FunctionOfAMatrixUtilsKernels.cpp
- FunctionOfAMatrixUtilsKernels.h
- FusedAdagradKernels.cpp
- FusedAdagradKernels.h
- FusedAdamAmsgradKernels.cpp
- FusedAdamKernels.cpp
- FusedAdamKernels.h
- FusedAdamUtils.h
- FusedAdamWAmsgradKernels.cpp
- FusedAdamWKernels.cpp
- FusedAdamWKernels.h
- FusedSgdKernels.cpp
- FusedSgdKernels.h
- GcdLcmKernels.cpp
- GcdLcmKernels.h
- GridSampler.cpp
- GridSampler.h
- GridSamplerKernels.h
- GroupNormKernels.cpp
- GroupNormKernels.h
- GroupReduceUtils.h
- HermitePolynomialHeKernel.cpp
- HermitePolynomialHeKernel.h
- HermitePolynomialHKernel.cpp
- HermitePolynomialHKernel.h
- HistogramddKernels.cpp
- HistogramKernels.h
- IGammaKernel.cpp
- IGammaKernel.h
- Im2ColKernel.cpp
- Im2ColKernel.h
- Indexing.cpp
- Indexing.h
- IndexingKernels.h
- IndexingUtils.h
- IndexKernelUtils.h
- IndexUtils.h
- IntegerDivider.h
- KernelUtils.h
- LaguerrePolynomialLKernel.cpp
- LaguerrePolynomialLKernel.h
- LaunchUtils.h
- LayerNormKernels.cpp
- LayerNormKernels.h
- LegendrePolynomialPKernel.cpp
- LegendrePolynomialPKernel.h
- LerpKernels.cpp
- LerpKernels.h
- LinearAlgebraKernels.cpp
- LinearAlgebraKernels.h
- LinearInt4.cpp
- LinearInt4.h
- LogAddExpKernels.cpp
- LogAddExpKernels.h
- LogcumsumexpKernel.cpp
- Loops.h
- LossCTCKernels.cpp
- LossCTCKernels.h
- LossKernels.cpp
- LossKernels.h
- LossNLL2dKernels.cpp
- LossNLL2dKernels.h
- LossNLLKernel.cpp
- LossNLLKernel.h
- MathExtensions.h
- MaxMinElementwiseKernels.cpp
- MaxMinElementwiseKernels.h
- MaxUnpoolingKernels.cpp
- MaxUnpoolingKernels.h
- MemoryAccess.h
- MemoryAccessUtils.h
- ModifiedBesselI0Kernel.cpp
- ModifiedBesselI0Kernel.h
- ModifiedBesselI1Kernel.cpp
- ModifiedBesselI1Kernel.h
- ModifiedBesselK0Kernel.cpp
- ModifiedBesselK0Kernel.h
- ModifiedBesselK1Kernel.cpp
- ModifiedBesselK1Kernel.h
- MultiLabelMarginLossKernels.cpp
- MultiLabelMarginLossKernels.h
- MultiMarginLossKernels.cpp
- MultiMarginLossKernels.h
- MultinomialKernel.cpp
- MultinomialKernel.h
- MultiTensorApply.h
- NMSKernel.cpp
- NMSKernel.h
- NonzeroKernel.cpp
- NonzeroKernel.h
- Norm.h
- NumericLimits.h
- OffsetCalculator.h
- Philox4x32.h
- PhiloxKeySplitKernels.cpp
- PhiloxKeySplitKernels.h
- PointwiseOpsKernels.cpp
- PointwiseOpsKernels.h
- Pow.h
- PowKernels.cpp
- PowKernels.h
- PsRoiAlignKernels.cpp
- PsRoiAlignKernels.h
- PsRoiPoolKernels.cpp
- PsRoiPoolKernels.h
- RandpermKernel.cpp
- RandpermKernel.h
- RangeFactoriesKernel.cpp
- RangeFactoriesKernel.h
- Reduce.h
- ReduceAMinMaxKernel.cpp
- ReduceArgMaxKernel.cpp
- ReduceArgMinKernel.cpp
- ReduceLogicKernels.cpp
- ReduceMaxValuesKernels.cpp
- ReduceMaxValuesKernels.h
- ReduceMinValuesKernels.cpp
- ReduceMinValuesKernels.h
- ReduceMomentKernels.cpp
- ReduceNormKernel.cpp
- ReduceNormKernel.h
- ReduceOps.h
- ReduceOpsKernels.h
- ReduceSumProdKernels.cpp
- ReflectionPadKernels.cpp
- ReflectionPadKernels.h
- RenormKernel.cpp
- RenormKernel.h
- RepeatKernel.cpp
- RepeatKernel.h
- ReplicationPaddingKernels.cpp
- ReplicationPaddingKernels.h
- ResizeKernel.cpp
- ResizeKernel.h
- RNNKernels.cpp
- RNNKernels.h
- RoiAlignKernels.cpp
- RoiAlignKernels.h
- RoiPoolKernels.cpp
- RoiPoolKernels.h
- RreluWithNoiseKernels.cpp
- RreluWithNoiseKernels.h
- ScaledModifiedBesselK0Kernel.cpp
- ScaledModifiedBesselK0Kernel.h
- ScaledModifiedBesselK1Kernel.cpp
- ScaledModifiedBesselK1Kernel.h
- ScanUtils.h
- ScatterGatherKernels.cpp
- ScatterGatherKernels.h
- SegmentReduceKernels.cpp
- SegmentReduceKernels.h
- Shape.cpp
- ShapeKernels.h
- SharedReduceOps.h
- ShiftedChebyshevPolynomialKernels.h
- ShiftedChebyshevPolynomialTKernel.cpp
- ShiftedChebyshevPolynomialUKernel.cpp
- ShiftedChebyshevPolynomialVKernel.cpp
- ShiftedChebyshevPolynomialWKernel.cpp
- SoftMaxKernels.cpp
- SoftMaxKernels.h
- Sorting.cpp
- Sorting.h
- SortingCommon.h
- SortingKernels.h
- SortingRadixSelect.h
- SortingRadixSort.h
- SphericalBesselJ0Kernel.cpp
- SphericalBesselJ0Kernel.h
- StepKernels.cpp
- StepKernels.h
- SummaryOpsKernels.cpp
- SummaryOpsKernels.h
- SYCLGroupAlgorithm.h
- TensorApplyUtils.h
- TensorCompareKernels.cpp
- TensorCompareKernels.h
- TensorFactoriesKernels.cpp
- TensorFactoriesKernels.h
- TensorModeKernel.cpp
- TensorModeKernel.h
- TensorShapeKernels.cpp
- TensorShapeKernels.h
- TensorTopKKernel.cpp
- TensorTopKKernel.h
- TensorTopKSbtopkKernel.cpp
- TensorTopKSbtopkKernel.h
- TensorTopKSbtopkKernel_k1.cpp
- TensorTopKSbtopkKernel_k2.cpp
- TensorTopKSbtopkKernel_k4.cpp
- TensorTopKSbtopkKernel_k8.cpp
- TensorTopKSbtopkKernelImpl.h
- TensorTopKSingleWgKernel.cpp
- TensorTopKSingleWgKernel.h
- TensorTransformationsKernels.cpp
- TensorTransformationsKernels.h
- TransposeKernel.cpp
- TransposeKernel.h
- TriangularOpsKernels.cpp
- TriangularOpsKernels.h
- UnaryComplexKernels.cpp
- UnaryComplexKernels.h
- UnaryFractionKernels.cpp
- UnaryFractionKernels.h
- UnaryGammaKernels.cpp
- UnaryGammaKernels.h
- UnaryGeometricAcoshKernel.cpp
- UnaryGeometricAcoshKernel.h
- UnaryGeometricAcosKernel.cpp
- UnaryGeometricAcosKernel.h
- UnaryGeometricAsinhKernel.cpp
- UnaryGeometricAsinhKernel.h
- UnaryGeometricAsinKernel.cpp
- UnaryGeometricAsinKernel.h
- UnaryGeometricAtanhKernel.cpp
- UnaryGeometricAtanhKernel.h
- UnaryGeometricAtanKernel.cpp
- UnaryGeometricAtanKernel.h
- UnaryGeometricCoshKernel.cpp
- UnaryGeometricCoshKernel.h
- UnaryGeometricCosKernel.cpp
- UnaryGeometricCosKernel.h
- UnaryGeometricSinhKernel.cpp
- UnaryGeometricSinhKernel.h
- UnaryGeometricSinKernel.cpp
- UnaryGeometricSinKernel.h
- UnaryGeometricTanhKernel.cpp
- UnaryGeometricTanhKernel.h
- UnaryGeometricTanKernel.cpp
- UnaryGeometricTanKernel.h
- UnaryKernels.cpp
- UnaryKernels.h
- UnaryLogKernels.cpp
- UnaryLogKernels.h
- UnarySignKernels.cpp
- UnarySignKernels.h
- UnarySpecialOpsKernels.cpp
- UnarySpecialOpsKernels.h
- UnfoldBackwardKernels.cpp
- UnfoldBackwardKernels.h
- UniqueKernels.cpp
- UniqueKernels.h
- UpSampleBicubic2dKernels.cpp
- UpSampleBicubic2dKernels.h
- UpSampleBilinear2dKernels.cpp
- UpSampleBilinear2dKernels.h
- UpSampleLinear1dKernels.cpp
- UpSampleLinear1dKernels.h
- UpSampleNearest1dKernels.cpp
- UpSampleNearest1dKernels.h
- UpSampleNearest2dKernels.cpp
- UpSampleNearest2dKernels.h
- UpSampleNearest3dKernels.cpp
- UpSampleNearest3dKernels.h
- UpSampleTrilinear3dKernels.cpp
- UpSampleTrilinear3dKernels.h
- WeightInt4PackKernel.cpp
- WeightInt4PackKernel.h
- WeightNormKernels.cpp
- WeightNormKernels.h
- WelfordNorm.h
- ZetaKernel.cpp
- ZetaKernel.h
- Activation.cpp
- AdaptiveAveragePooling2d.cpp
- AdaptiveAveragePooling3d.cpp
- AdaptiveMaxPooling2d.cpp
- AdaptiveMaxPooling3d.cpp
- AiryAi.cpp
- AmpKernels.cpp
- AveragePool2d.cpp
- AveragePool3d.cpp
- BatchLinearAlgebra.cpp
- BatchNorm.cpp
- Bessel.cpp
- BinaryOps.cpp
- Blas.cpp
- Blas.h
- Bucketization.cpp
- Col2Im.cpp
- CompareOps.cpp
- Copy.cpp
- Cross.cpp
- DeformConv2d.cpp
- DepthwiseConv2d.cpp
- DepthwiseConv3d.cpp
- DilatedMaxPool2d.cpp
- DilatedMaxPool3d.cpp
- Distance.cpp
- Distributions.cpp
- Dropout.cpp
- Embedding.cpp
- EmbeddingBag.cpp
- Equal.cpp
- Fill.cpp
- ForeachOpList.cpp
- ForeachOpScalar.cpp
- ForeachOpScalarList.cpp
- ForeachOpScalarTensor.cpp
- ForeachReduceOp.cpp
- ForeachUnaryOp.cpp
- FractionalMaxPool2d.cpp
- FractionalMaxPool3d.cpp
- FunctionOfAMatrixUtils.cpp
- FusedAdagrad.cpp
- FusedAdam.cpp
- FusedAdamW.cpp
- FusedSgd.cpp
- GatedLinearUnit.cpp
- GridSampler.cpp
- GroupNorm.cpp
- Histogram.cpp
- Im2Col.cpp
- Indexing.cpp
- LayerNorm.cpp
- Lerp.cpp
- LinearAlgebra.cpp
- LinearInt4.cpp
- Loss.cpp
- LossCTC.cpp
- LossMultiLabelMargin.cpp
- LossMultiMargin.cpp
- LossNLL.cpp
- LossNLL2d.cpp
- MaxUnpooling.cpp
- NMS.cpp
- Nonzero.cpp
- Normalization.cpp
- PhiloxKeySplit.cpp
- PointwiseOps.cpp
- Pow.cpp
- PsRoiAlign.cpp
- PsRoiPool.cpp
- RangeFactories.cpp
- RecordStream.cpp
- ReduceAllOps.cpp
- ReduceOps.cpp
- ReflectionPad.cpp
- Repeat.cpp
- ReplicationPadding.cpp
- Resize.cpp
- RNN.cpp
- RoiAlign.cpp
- RoiPool.cpp
- RreluWithNoise.cpp
- ScanKernels.cpp
- ScanKernels.h
- SegmentReduce.cpp
- SoftMax.cpp
- Sorting.cpp
- SpectralOps.cpp
- SummaryOps.cpp
- TensorAdvancedIndexing.cpp
- TensorCompare.cpp
- TensorFactories.cpp
- TensorProperties.cpp
- TensorShape.cpp
- TensorTopK.cpp
- TensorTransformations.cpp
- TriangluarOps.cpp
- UnaryOps.cpp
- UnfoldBackward.cpp
- Unique.cpp
- UpSample.h
- UpSampleBicubic2d.cpp
- UpSampleBilinear2d.cpp
- UpSampleLinear1d.cpp
- UpSampleNearest1d.cpp
- UpSampleNearest2d.cpp
- UpSampleNearest3d.cpp
- UpSampleTrilinear3d.cpp
- WeightInt4Pack.cpp
- WeightNorm.cpp
- XPUFallback.cpp
- XPUScalar.cpp
- Sleep.cpp
- Sleep.h
- CMakeLists.txt
- DeviceProperties.h
- Macros.h
- Memory.h
- MemoryFormat.h
- Runtime.h
- SYCLContext.h
- SYCLHelpers.h
- TensorInfo.h
- TensorOptions.h
- xpu_aten.h
- XPUMathCompat.h
- XPUPair.h
- CMakeLists.txt
- FlightRecorderXCCL.cpp
- NanCheck_XPU.cpp
- NanCheck_XPU.hpp
- ProcessGroupXCCL.cpp
- ProcessGroupXCCL.hpp
- ProcessGroupXCCLMonitor.cpp
- ProcessGroupXCCLMonitor.hpp
- reducer_xpu.cpp
- Register.cpp
- Signal.cpp
- Signal.hpp
- xccl.cpp
- xccl.h
- XPUEventCache.cpp
- XPUEventCache.hpp
- XPUSymmetricMemory.cpp
- XPUSymmetricMemory.hpp
- XPUSymmetricMemoryTypes.hpp
- XPUSymmetricMemoryUtils.cpp
- XPUSymmetricMemoryUtils.hpp
- BuildOnLinux.cmake
- BuildOnWindows.cmake
- CMakeLists.txt
- subgroup_reduction_antipattern_cases.cpp
- vectorization_pattern_cases.cpp
- adaptive_avg_pool2d.py
- avg_pool2d.py
- avg_pool3d.py
- batch_norm_1d.py
- batch_norm_2d.py
- batch_norm_3d.py
- col2im.py
- distance.cdist.py
- distance.pdist.py
- distribution.bernoulli.py
- distribution.cauchy.py
- distribution.exponential.py
- distribution.geometric.py
- distribution.log_normal.py
- distribution.multinomial.py
- distribution.normal.py
- distribution.random.py
- distribution.uniform.py
- dropout.py
- eltwise.add.py
- embedding.py
- embedding_bag.py
- flip.py
- grid_sampler.grid_sampler_2d.py
- grid_sampler.grid_sampler_3d.py
- group_norm.py
- im2col.py
- indexing.diag.py
- indexing.index.py
- indexing.index_add.py
- indexing.index_copy.py
- indexing.index_fill.py
- indexing.index_put.py
- indexing.index_select.py
- indexing.masked_fill.py
- indexing.put.py
- indexing.take.py
- layer_norm.py
- loss.binary_cross_entropy.py
- loss.ctc_loss.py
- loss.l1_loss.py
- loss.mse_loss.py
- loss.multilabel_margin_loss.py
- loss.nll_loss.py
- loss.smooth_l1_loss.py
- matmul.py
- pad_sequence.py
- pooling.adaptive_max_pool2d.py
- pooling.fractional_max_pool2d.py
- pooling.fractional_max_pool3d.py
- pooling.max_pool2d.py
- pooling.max_pool3d.py
- pooling.max_unpool2d.py
- pooling.max_unpool3d.py
- reduce.max.py
- reduce.sum.py
- remainder.py
- repeat_interleave.py
- roll.py
- scan.cumsum.py
- scan.masked_select.py
- scan.nonzero.py
- scan.topk.py
- scan.unique.py
- scatter_gather.gather.py
- scatter_gather.scatter.py
- scatter_gather.scatter_add.py
- softmax.py
- sort.py
- sort.randperm.py
- upsample_bicubic2d.py
- upsample_bilinear2d.py
- upsample_nearest2d.py
- upsample_nearest3d.py
- upsample_nearest_exact2d.py
- correlation_id_mixed.py
- llama.py
- profile_partial_runtime_ops.py
- reproducer.missing.gpu.kernel.time.py
- rn50.py
- test_for_overlapping_kernels.py
- test_profiler_correctness.py
- time_precision_in_profile.py
- triton_xpu_ops_time.py
- optests_failures_dict.json
- test_batch_norm.py
- test_bcomplex32.py
- test_binary.py
- test_cat.py
- test_clamp_promotion.py
- test_compare.py
- test_conv_transpose_complex32.py
- test_conversion.py
- test_copy.py
- test_copy_downcast_fp8.py
- test_deform_conv.py
- test_div_mode.py
- test_fft_c2c_sycl_kernel.py
- test_fft_r2c_bf16.py
- test_foreach_list.py
- test_foreach_scalar.py
- test_foreach_scalarlist.py
- test_grid_sample.py
- test_group_norm.py
- test_index_and_index_put.py
- test_int4pack.py
- test_layer_norm.py
- test_linalg.py
- test_loops.py
- test_max_pool2d_bwd.py
- test_max_pool3d_fwd_int64.py
- test_nms.py
- test_operation_on_device_1.py
- test_override_warning.py
- test_pin_memory.py
- test_rand.py
- test_record_stream.py
- test_reduce.py
- test_resize.py
- test_rms_norm.py
- test_roi_align.py
- test_safe_softmax.py
- test_scatter_gather.py
- test_softmax.py
- test_sort.py
- test_tensor_factory.py
- test_torchvision_roi_ops.py
- test_triangular_solve_sparse.py
- test_tril.py
- test_unary.py
- test_upsample_bilinear_bwd.py
- test_upsample_nearest.py
- test_where.py
- test_xpu_ops_header.py
- .gitkeep
- test_addcmul_cpu_scalar.py
- test_fused_obs_fake_quant_ch_axis_bounds.py
- test_layer_norm_group_norm_align.py
- CMakeLists.txt
- main.cpp
- simple_kernel.cpp
- simple_kernel.hpp
- test_complex_tensor_xpu.py
- bench_ccl_allgather_latency.cpp
- bench_ccl_allreduce_latency.cpp
- bench_ccl_alltoall_latency.cpp
- README.md
- __init__.py
- test_c10d_ops_xccl.py
- test_c10d_xccl.py
- test_aot_autograd_cache_xpu.py
- test_compiler_bisector_xpu.py
- test_ctx_manager_xpu.py
- test_cudagraphs_xpu.py
- test_deviceguard_xpu.py
- test_misc_xpu.py
- test_recompiles_xpu.py
- test_regional_inductor_xpu.py
- test_streams_xpu.py
- test_wrap_inductor_compiled_regions_xpu.py
- TestSparseCompressedCPU.test_print_SparseBSC_cpu.expect
- TestSparseCompressedCPU.test_print_SparseBSR_cpu.expect
- TestSparseCompressedCPU.test_print_SparseCSC_cpu.expect
- TestSparseCompressedCPU.test_print_SparseCSR_cpu.expect
- TestSparseCompressedXPU.test_print_SparseBSC_xpu.expect
- TestSparseCompressedXPU.test_print_SparseBSR_xpu.expect
- TestSparseCompressedXPU.test_print_SparseCSC_xpu.expect
- TestSparseCompressedXPU.test_print_SparseCSR_xpu.expect
- TestSparseXPU.test_print_coalesced_xpu_float64.expect
- TestSparseXPU.test_print_uncoalesced_xpu_float64.expect
- test_converter_xpu.py
- test_cpp_serdes_xpu.py
- test_draft_export_xpu.py
- test_experimental_xpu.py
- test_export_opinfo_xpu.py
- test_export_strict_xpu.py
- test_export_training_ir_to_run_decomp_xpu.py
- test_export_xpu.py
- test_hop_xpu.py
- test_passes_xpu.py
- test_retraceability_xpu.py
- test_serdes_xpu.py
- test_serialize_xpu.py
- test_strict_export_v2_xpu.py
- test_torchbind_xpu.py
- testing_xpu.py
- __init__.py
- run_test_with_skip.py
- run_test_with_skip_arc.py
- run_test_with_skip_bmg.py
- run_test_with_skip_lnl.py
- run_test_with_skip_mtl.py
- skip_list_arc.py
- skip_list_common.py
- skip_list_win.py
- skip_list_win_arc.py
- skip_list_win_bmg.py
- skip_list_win_lnl.py
- skip_list_win_mtl.py
- test_ops_xpu.py
- test_tensor_creation_ops_xpu.py
- __init__.py
- test_ac_xpu.py
- test_aot_joint_with_descriptors_xpu.py
- test_aotdispatch_xpu.py
- test_control_flow_xpu.py
- test_eager_transforms_xpu.py
- test_memory_efficient_fusion_xpu.py
- test_ops_xpu.py
- test_vmap_xpu.py
- test_invoke_subgraph_xpu.py
- test_with_effects_xpu.py
- __init__.py
- test_convolution_xpu.py
- test_dropout_xpu.py
- test_init_xpu.py
- test_lazy_modules_xpu.py
- test_multihead_attention_xpu.py
- test_packed_sequence_xpu.py
- test_parametrization_xpu.py
- test_pooling_xpu.py
- test_cpp_thread_xpu.py
- test_execution_trace_xpu.py
- test_memory_profiler.py
- test_profiler_xpu.py
- __init__.py
- test_quantized_op_xpu.py
- test_quantized_tensor_xpu.py
- test_workflow_module_xpu.py
- test_workflow_ops_xpu.py
- __init__.py
- run_distributed.py
- run_test_win_with_skip_mtl.py
- run_test_with_skip.py
- run_test_with_skip_arc.py
- run_test_with_skip_bmg.py
- run_test_with_skip_lnl.py
- run_test_with_skip_mtl.py
- run_test_with_windows_nighltly.py
- skip_list_arc.py
- skip_list_common.py
- skip_list_dist.py
- skip_list_mtl.py
- skip_list_win.py
- skip_list_win_arc.py
- skip_list_win_bmg.py
- skip_list_win_lnl.py
- skip_list_win_mtl.py
- test_autocast_xpu.py
- test_autograd_fallback_xpu.py
- test_autograd_xpu.py
- test_basic_torch_np_xpu.py
- test_binary_ufuncs_xpu.py
- test_comparison_utils_xpu.py
- test_compile_benchmark_util_xpu.py
- test_complex_xpu.py
- test_content_store_xpu.py
- test_cpp_api_parity_xpu.py
- test_cpp_extensions_aot_xpu.py
- test_cuda_multigpu_xpu.py
- test_cuda_nvml_based_avail_xpu.py
- test_cuda_primary_ctx_xpu.py
- test_cuda_sanitizer_xpu.py
- test_cuda_trace_xpu.py
- test_custom_ops_xpu.py
- test_dataloader_xpu.py
- test_decomp_xpu.py
- test_distributions_xpu.py
- test_dynamic_shapes_xpu.py
- test_expanded_weights_xpu.py
- test_fake_tensor_xpu.py
- test_flop_counter_xpu.py
- test_foreach_xpu.py
- test_fx_experimental_xpu.py
- test_fx_xpu.py
- test_indexing_xpu.py
- test_legacy_vmap_xpu.py
- test_linalg_xpu.py
- test_masked_xpu.py
- test_maskedtensor_xpu.py
- test_matmul_cuda_xpu.py
- test_meta_xpu.py
- test_mkldnn_fusion_xpu.py
- test_modules_xpu.py
- test_multiprocessing_xpu.py
- test_native_mha_xpu.py
- test_nestedtensor_xpu.py
- test_nn_xpu.py
- test_numba_integration_xpu.py
- test_numpy_interop_xpu.py
- test_ops_fwd_gradients_xpu.py
- test_ops_gradients_xpu.py
- test_ops_xpu.py
- test_optim_xpu.py
- test_out_dtype_op_xpu.py
- test_prims_xpu.py
- test_proxy_tensor_xpu.py
- test_python_dispatch_xpu.py
- test_reductions_xpu.py
- test_scaled_matmul_cuda_xpu.py
- test_scatter_gather_ops_xpu.py
- test_schema_check.py
- test_segment_reductions_xpu.py
- test_serialization_xpu.py
- test_shape_ops_xpu.py
- test_sort_and_select_xpu.py
- test_sparse_csr_xpu.py
- test_sparse_xpu.py
- test_spectral_ops_xpu.py
- test_tensor_creation_ops_xpu.py
- test_testing_xpu.py
- test_torch_xpu.py
- test_transformers_xpu.py
- test_type_promotion_xpu.py
- test_unary_ufuncs_xpu.py
- test_utils_xpu.py
- test_view_ops_xpu.py
- windows_skip_dict.py
- xpu_test_utils.py
- cutlass.yaml
- default.yaml
- fixheaders.py
- pytorch.yaml
- pytorchexamples.yaml
- README.md
- run.sh
- third_party.yaml
- torchvision.yaml
- _linter.py
- actionlint_linter.py
- clangformat_linter.py
- clangtidy_linter.py
- cmake_linter.py
- docstring_linter.py
- exec_linter.py
- flake8_linter.py
- gha_linter.py
- grep_linter.py
- import_linter.py
- lintrunner_version_linter.py
- nativefunctions_linter.py
- newlines_linter.py
- no_merge_conflict_csv_linter.py
- no_workflows_on_fork.py
- pyfmt_linter.py
- README.md
- ruff_linter.py
- s3_init.py
- s3_init_config.json
- set_linter.py
- shellcheck_linter.py
- test_has_main_linter.py
- testowners_linter.py
- update_s3.py
- workflow_consistency_linter.py
- __init__.py
- generate_build_files.py
- __init__.py
- check_ops.py
- .clang-format
- .clang-tidy
- .cmakelintrc
- .flake8
- .gitignore
- .lintrunner.toml
- AGENTS.md
- CLAUDE.md
- CMakeLists.txt
- CODE_OF_CONDUCT.md
- CONTRIBUTING.md
- LICENSE
- pyproject.toml
- README.md
- SECURITY.md
// repository documentation
Was this content helpful?
(0 ratings)
