Files
rustytorch/crates/core/rtx-tensor/Cargo.toml
T
osobhandClaude Opus 4.8 e8a2036db4
CI / Build (macos-latest) (push) Failing after 26s
CI / Format Check (push) Failing after 10s
CI / Clippy Check (push) Failing after 19s
Performance Benchmarks / Run Benchmarks (push) Successful in 28s
CI / Build (ubuntu-latest) (push) Failing after 15s
CI / Test (macos-latest) (push) Has been skipped
CI / Test (ubuntu-latest) (push) Has been skipped
CI / Python Bindings (maturin) (macos-latest) (push) Has been skipped
CI / Python Bindings (maturin) (ubuntu-latest) (push) Has been skipped
CI / WASM Build + Size Check (push) Has been skipped
CI / Distributed Training Tests (push) Has been skipped
GPU Tests / Check GPU Availability (push) Successful in 1s
GPU Tests / CUDA Tests (11.8) (push) Has been skipped
GPU Tests / CUDA Tests (12.1) (push) Has been skipped
GPU Tests / Metal Tests (push) Has been skipped
Documentation / Build User Guide (push) Successful in 6s
Documentation / Build API Documentation (push) Failing after 50s
CI / Build CPU-Only (Explicit) (push) Failing after 1m0s
CI / CI Success (push) Failing after 0s
fix(ci,rtx-tensor): resolve clippy --all-features intel-mkl conflict; gate MKL to x86_64-linux
clippy --all-features enabled both rtx-tensor/mkl (intel-mkl-src mkl-static-lp64-seq)
and rtx-csm/candle mkl (mkl-static-lp64-iomp) -> two conflicting intel-mkl-src link
configs -> E0428 'MKL_CONFIG defined multiple times'.

- clippy: drop --all-features (lint default features; --all-features is unsound for a
  multi-platform, mutually-exclusive-backend workspace).
- rtx-tensor: gate intel-mkl-src to cfg(all(target_os=linux, target_arch=x86_64)) so
  mkl is never pulled on macOS/arm.

Co-Authored-By: Claude Opus 4.8 <[email protected]>
2026-06-27 08:12:27 -07:00

140 lines
4.5 KiB
TOML

[package]
name = "rtx-tensor"
version = "1.0.0"
edition = "2024"
rust-version = "1.92"
authors = ["RustyTorch++ Team"]
license = "Apache-2.0"
description = "GPU-native tensor library with PyTorch-compatible API for RustyTorch++"
[dependencies]
# Backend abstraction
rtx-backend = { path = "../rtx-backend" }
# Core utilities
thiserror = "2"
tracing = "0.1"
once_cell = "1.19"
parking_lot = "0.12"
# Serialization for tensor persistence
serde = { version = "1.0", features = ["derive"] }
bincode = "1.3"
serde_json = "1.0"
# Compression algorithms for sparse tensor serialization
lz4_flex = "0.11"
zstd = "0.13"
# Checksums for data integrity
crc32fast = "1.3"
sha2 = "0.10"
# Numeric computing
ndarray = "0.15"
nalgebra = "0.32"
half = "2.3"
# Parallel processing
rayon = "1.10"
# BLAS for optimized matrix operations (optional - manual impl used by default)
# NOTE: OpenBLAS sgemm returns zeros on ARM64 macOS, so we use manual implementation
blas = { version = "0.22", optional = true }
cblas = { version = "0.5.0", optional = true }
blas-src = { version = "0.10", default-features = false, optional = true }
# OpenBLAS backend - good baseline performance
# Uses system-installed OpenBLAS (via Homebrew on macOS) instead of building from source
# NOTE: Requires gfortran to be installed. Currently disabled by default due to ARM64 issues.
openblas-src = { version = "0.10", features = ["cblas", "system"], optional = true }
# Intel MKL backend (optional, x86_64-Linux only - see target section below).
# MKL has no Apple Silicon build and collides with candle's mkl variant, so it is
# gated to x86_64 Linux. Enable with: cargo build --features mkl (on x86_64 Linux).
# Real CUDA GPU acceleration with cudarc 0.18.1 for production RTX 5090 support (matching workspace version)
cudarc = { workspace = true, optional = true }
# Alternative: Direct CUDA bindings
# cuda-runtime-sys = "0.3"
# cublas-sys = "0.8"
# curand-sys = "0.10"
rand = "0.8"
rand_distr = "0.4"
# Concurrency testing framework
loom = { workspace = true, optional = true }
# Removed rtx-runtime dependency - rtx-tensor is the foundation layer
[dev-dependencies]
# Testing framework
proptest = "1.4"
criterion = "0.5"
tokio = "1.35"
# Additional testing utilities
approx = "0.5"
rand = "0.8"
ctor = "0.2"
# Backend testing (for generic tensor tests)
rtx-backend = { path = "../rtx-backend" }
rtx-backend-cpu = { path = "../rtx-backend-cpu" }
[build-dependencies]
# CUDA kernel compilation support
cc = "1.0"
# Metal GPU support for Apple Silicon (macOS only)
[target.'cfg(target_os = "macos")'.dependencies]
objc2 = { version = "0.6", optional = true }
objc2-foundation = { version = "0.3", optional = true }
objc2-metal = { version = "0.3", optional = true }
objc2-metal-performance-shaders = { version = "0.3", optional = true, features = [
"MPSMatrix",
"MPSCore",
"MPSKernel"
]}
# Intel MKL is x86_64-Linux only (no Apple Silicon build); gate it so the `mkl`
# feature can never pull it on macOS/arm and collide with candle's mkl variant.
[target.'cfg(all(target_os = "linux", target_arch = "x86_64"))'.dependencies]
intel-mkl-src = { version = "0.8", features = ["mkl-static-lp64-seq"], optional = true }
[features]
default = ["cpu"] # No BLAS by default - manual impl works on all platforms
cuda = ["cudarc"]
rocm = []
metal = ["objc2", "objc2-foundation", "objc2-metal", "objc2-metal-performance-shaders"]
cpu = [] # CPU fallback (always available)
test = ["cpu"] # Enable CPU for tests
disabled_tests = [] # Temporarily disable CUDA tests during workspace builds
# Backend trait integration (new architecture)
# Note: rtx-backend is now always included for GenericTensor support
backend = []
# BLAS backend selection (mutually exclusive - pick one)
# NOTE: Disabled by default due to ARM64 macOS issues with sgemm returning zeros
openblas = ["blas", "cblas", "blas-src", "openblas-src"] # Requires gfortran
mkl = ["blas", "cblas", "blas-src", "intel-mkl-src"] # Best CPU performance for large matrices (2-5x faster)
# SIMD vectorization (requires nightly Rust for portable_simd)
portable_simd = [] # Enable std::simd for 3-4x faster element-wise ops
# cuSPARSELt library linking (requires libcusparseLt to be installed)
# Enable with: cargo build --features cuda,cusparselt_link
# Falls back to CPU implementation when not enabled
cusparselt_link = ["cuda"] # Links against libcusparseLt for 2:4 sparsity acceleration
# Concurrency testing with loom
loom = ["dep:loom"]
[lib]
name = "rtx_tensor"
path = "src/lib.rs"
[lints]
workspace = true