Files
rustytorch/crates/production/rtx-inference/Cargo.toml
T
osobhandClaude Fable 5 733b02cd8b
GPU Tests / Check GPU Availability (push) Successful in 1s
CI / Format Check (push) Failing after 6s
CI / Build (ubuntu-latest) (push) Failing after 6s
GPU Tests / CUDA Tests (11.8) (push) Has been skipped
GPU Tests / CUDA Tests (12.1) (push) Has been skipped
CI / Build CPU-Only (Explicit) (push) Failing after 6s
CI / Build (macos-latest) (push) Failing after 11s
GPU Tests / Metal Tests (push) Has been skipped
Documentation / Build User Guide (push) Successful in 7s
CI / Test (macos-latest) (push) Has been skipped
CI / Test (ubuntu-latest) (push) Has been skipped
CI / Python Bindings (maturin) (macos-latest) (push) Has been skipped
CI / Python Bindings (maturin) (ubuntu-latest) (push) Has been skipped
CI / WASM Build + Size Check (push) Has been skipped
CI / Distributed Training Tests (push) Has been skipped
CI / Clippy Check (push) Failing after 21s
CI / CI Success (push) Failing after 0s
Performance Benchmarks / Run Benchmarks (push) Successful in 28s
Documentation / Build API Documentation (push) Failing after 25s
feat(inference): concrete EAGLE draft model + real tokenizer at the serving boundary
EAGLE (rtx-inference/src/eagle.rs, ~610 lines, mirrors medusa.rs
conventions): EagleDraftHead autoregressive FFN with Concat/Add/
Attention feature fusion, EagleHeads draft model with draft/
draft_steps (per-step top-k for candidate trees) and teacher-forced
training_loss; implements the speculative::EagleDraftModel trait so it
plugs into the orchestration layer. 38 unit tests.

Tokenizer (rtx-inference/src/tokenizer.rs): ServingTokenizer enum —
Vocab (HuggingFace tokenizers, loadable from tokenizer.json) or
ByteLevel fallback preserving previous behavior. rtx-serving-api's
AppState and rtx-streaming's token generator now encode/decode through
it (with_engine_and_tokenizer / set_tokenizer added; existing
signatures unchanged). Also fixes two pre-existing compile errors in
rtx-streaming (missing import, stray .await) that blocked its lib
tests entirely.

Tests: rtx-inference 328 pass, rtx-serving-api 193 pass, rtx-streaming
53 pass (2 pre-existing mock-server connection failures unrelated to
these changes).

Co-Authored-By: Claude Fable 5 <[email protected]>
2026-07-09 21:54:05 -07:00

119 lines
3.1 KiB
TOML

[package]
name = "rtx-inference"
version = "1.0.0"
edition.workspace = true
rust-version = "1.92"
authors.workspace = true
license.workspace = true
repository.workspace = true
[dependencies]
# Workspace dependencies
rtx-runtime = { path = "../../core/rtx-runtime" }
rtx-tensor = { path = "../../core/rtx-tensor", features = ["cpu"] }
rtx-synthesis = { path = "../../specialized/rtx-synthesis" }
# ONNX Runtime integration (optional)
rtx-onnx = { path = "../rtx-onnx", optional = true }
# Burn ML framework integration (optional)
rtx-burn = { path = "../../integration/rtx-burn", optional = true }
# Candle ML framework integration (optional)
rtx-candle = { path = "../../integration/rtx-candle", optional = true }
# Error handling
anyhow.workspace = true
thiserror.workspace = true
# Async runtime for request handling
tokio = { workspace = true, features = ["full", "time", "rt-multi-thread"] }
async-trait.workspace = true
# Serialization for request/response
serde = { workspace = true, features = ["derive"] }
serde_json.workspace = true
# Concurrent data structures for request queues and KV cache
parking_lot.workspace = true
crossbeam.workspace = true
dashmap.workspace = true
# UUID for request tracking
uuid = { version = "1.0", features = ["v4", "serde"] }
# For build-time metadata
once_cell = "1.19"
# Configuration management
config.workspace = true
# Logging and telemetry
tracing.workspace = true
tracing-subscriber.workspace = true
# Performance monitoring
criterion.workspace = true
# Quantization support (commented out for testing)
# cudarc.workspace = true
# Mathematical operations for cache statistics (commented out for testing)
# nalgebra.workspace = true
# Threading for batch processing
rayon.workspace = true
# Real vocabulary tokenizer for the serving boundary (byte-level remains a fallback)
tokenizers = { workspace = true, features = ["onig"] }
# Temporary files for model caching (commented out for testing)
# tempfile.workspace = true
# Hashing for cache keys (commented out for testing)
# sha2 = "0.10"
# Pattern matching for kernel selection (commented out for testing)
# regex = "1.0"
# Random number generation for testing
fastrand = "2.0"
# Streaming support
tokio-stream = "0.1"
futures = "0.3"
# Zip archive support for PyTorch file parsing
zip = "2.2"
# Byte order handling for binary formats
byteorder = "1.5"
[dev-dependencies]
proptest.workspace = true
tokio-test = "0.4"
[build-dependencies]
chrono = { version = "0.4", features = ["serde"] }
[features]
default = []
metrics = ["tracing-subscriber/env-filter"]
# ONNX Runtime support for high-performance inference
onnx-runtime = ["rtx-onnx"]
# ONNX Runtime with CUDA support
onnx-cuda = ["onnx-runtime", "rtx-onnx/cuda"]
# ONNX Runtime with CoreML support (macOS)
onnx-coreml = ["onnx-runtime", "rtx-onnx/coreml"]
# ONNX Runtime with TensorRT support
onnx-tensorrt = ["onnx-runtime", "rtx-onnx/tensorrt"]
# Burn ML framework support
burn = ["rtx-burn"]
burn-wgpu = ["burn", "rtx-burn/wgpu"]
# Candle ML framework support
candle = ["rtx-candle"]
candle-cuda = ["candle", "rtx-candle/cuda"]
candle-metal = ["candle", "rtx-candle/metal"]
[lints]
workspace = true