GPU Tests / Check GPU Availability (push) Successful in 1s
CI / Format Check (push) Failing after 6s
CI / Build (ubuntu-latest) (push) Failing after 6s
GPU Tests / CUDA Tests (11.8) (push) Has been skipped
GPU Tests / CUDA Tests (12.1) (push) Has been skipped
CI / Build CPU-Only (Explicit) (push) Failing after 6s
CI / Build (macos-latest) (push) Failing after 11s
GPU Tests / Metal Tests (push) Has been skipped
Documentation / Build User Guide (push) Successful in 7s
CI / Test (macos-latest) (push) Has been skipped
CI / Test (ubuntu-latest) (push) Has been skipped
CI / Python Bindings (maturin) (macos-latest) (push) Has been skipped
CI / Python Bindings (maturin) (ubuntu-latest) (push) Has been skipped
CI / WASM Build + Size Check (push) Has been skipped
CI / Distributed Training Tests (push) Has been skipped
CI / Clippy Check (push) Failing after 21s
CI / CI Success (push) Failing after 0s
Performance Benchmarks / Run Benchmarks (push) Successful in 28s
Documentation / Build API Documentation (push) Failing after 25s
EAGLE (rtx-inference/src/eagle.rs, ~610 lines, mirrors medusa.rs conventions): EagleDraftHead autoregressive FFN with Concat/Add/ Attention feature fusion, EagleHeads draft model with draft/ draft_steps (per-step top-k for candidate trees) and teacher-forced training_loss; implements the speculative::EagleDraftModel trait so it plugs into the orchestration layer. 38 unit tests. Tokenizer (rtx-inference/src/tokenizer.rs): ServingTokenizer enum — Vocab (HuggingFace tokenizers, loadable from tokenizer.json) or ByteLevel fallback preserving previous behavior. rtx-serving-api's AppState and rtx-streaming's token generator now encode/decode through it (with_engine_and_tokenizer / set_tokenizer added; existing signatures unchanged). Also fixes two pre-existing compile errors in rtx-streaming (missing import, stray .await) that blocked its lib tests entirely. Tests: rtx-inference 328 pass, rtx-serving-api 193 pass, rtx-streaming 53 pass (2 pre-existing mock-server connection failures unrelated to these changes). Co-Authored-By: Claude Fable 5 <[email protected]>
119 lines
3.1 KiB
TOML
119 lines
3.1 KiB
TOML
[package]
|
|
name = "rtx-inference"
|
|
version = "1.0.0"
|
|
edition.workspace = true
|
|
rust-version = "1.92"
|
|
authors.workspace = true
|
|
license.workspace = true
|
|
repository.workspace = true
|
|
|
|
[dependencies]
|
|
# Workspace dependencies
|
|
rtx-runtime = { path = "../../core/rtx-runtime" }
|
|
rtx-tensor = { path = "../../core/rtx-tensor", features = ["cpu"] }
|
|
rtx-synthesis = { path = "../../specialized/rtx-synthesis" }
|
|
|
|
# ONNX Runtime integration (optional)
|
|
rtx-onnx = { path = "../rtx-onnx", optional = true }
|
|
|
|
# Burn ML framework integration (optional)
|
|
rtx-burn = { path = "../../integration/rtx-burn", optional = true }
|
|
|
|
# Candle ML framework integration (optional)
|
|
rtx-candle = { path = "../../integration/rtx-candle", optional = true }
|
|
|
|
# Error handling
|
|
anyhow.workspace = true
|
|
thiserror.workspace = true
|
|
|
|
# Async runtime for request handling
|
|
tokio = { workspace = true, features = ["full", "time", "rt-multi-thread"] }
|
|
async-trait.workspace = true
|
|
|
|
# Serialization for request/response
|
|
serde = { workspace = true, features = ["derive"] }
|
|
serde_json.workspace = true
|
|
|
|
# Concurrent data structures for request queues and KV cache
|
|
parking_lot.workspace = true
|
|
crossbeam.workspace = true
|
|
dashmap.workspace = true
|
|
|
|
# UUID for request tracking
|
|
uuid = { version = "1.0", features = ["v4", "serde"] }
|
|
|
|
# For build-time metadata
|
|
once_cell = "1.19"
|
|
|
|
# Configuration management
|
|
config.workspace = true
|
|
|
|
# Logging and telemetry
|
|
tracing.workspace = true
|
|
tracing-subscriber.workspace = true
|
|
|
|
# Performance monitoring
|
|
criterion.workspace = true
|
|
|
|
# Quantization support (commented out for testing)
|
|
# cudarc.workspace = true
|
|
|
|
# Mathematical operations for cache statistics (commented out for testing)
|
|
# nalgebra.workspace = true
|
|
|
|
# Threading for batch processing
|
|
rayon.workspace = true
|
|
|
|
# Real vocabulary tokenizer for the serving boundary (byte-level remains a fallback)
|
|
tokenizers = { workspace = true, features = ["onig"] }
|
|
|
|
# Temporary files for model caching (commented out for testing)
|
|
# tempfile.workspace = true
|
|
|
|
# Hashing for cache keys (commented out for testing)
|
|
# sha2 = "0.10"
|
|
|
|
# Pattern matching for kernel selection (commented out for testing)
|
|
# regex = "1.0"
|
|
|
|
# Random number generation for testing
|
|
fastrand = "2.0"
|
|
|
|
# Streaming support
|
|
tokio-stream = "0.1"
|
|
futures = "0.3"
|
|
|
|
# Zip archive support for PyTorch file parsing
|
|
zip = "2.2"
|
|
|
|
# Byte order handling for binary formats
|
|
byteorder = "1.5"
|
|
|
|
[dev-dependencies]
|
|
proptest.workspace = true
|
|
tokio-test = "0.4"
|
|
|
|
[build-dependencies]
|
|
chrono = { version = "0.4", features = ["serde"] }
|
|
|
|
[features]
|
|
default = []
|
|
metrics = ["tracing-subscriber/env-filter"]
|
|
# ONNX Runtime support for high-performance inference
|
|
onnx-runtime = ["rtx-onnx"]
|
|
# ONNX Runtime with CUDA support
|
|
onnx-cuda = ["onnx-runtime", "rtx-onnx/cuda"]
|
|
# ONNX Runtime with CoreML support (macOS)
|
|
onnx-coreml = ["onnx-runtime", "rtx-onnx/coreml"]
|
|
# ONNX Runtime with TensorRT support
|
|
onnx-tensorrt = ["onnx-runtime", "rtx-onnx/tensorrt"]
|
|
# Burn ML framework support
|
|
burn = ["rtx-burn"]
|
|
burn-wgpu = ["burn", "rtx-burn/wgpu"]
|
|
# Candle ML framework support
|
|
candle = ["rtx-candle"]
|
|
candle-cuda = ["candle", "rtx-candle/cuda"]
|
|
candle-metal = ["candle", "rtx-candle/metal"]
|
|
[lints]
|
|
workspace = true
|