Files
rustytorch/crates/production/rtx-serving-api/tests/integration_tests.rs
T
2026-03-04 00:08:42 +00:00

144 lines
4.5 KiB
Rust

//! Integration tests for the rtx-serving-api crate
//!
//! These tests demonstrate the complete TDD workflow and verify that all
//! API endpoints work together correctly.
use axum::{
Router,
routing::{get, post},
};
use axum_test::TestServer;
use rtx_serving_api::inference::{InferenceRequest, InferenceResponse};
use rtx_serving_api::models::ModelInfo;
use rtx_serving_api::{HealthStatus, health, inference, models};
/// Helper function to create a complete test server
fn create_test_server() -> TestServer {
let app = Router::new()
// Health endpoints
.route("/health", get(health::health_check))
.route("/health/ready", get(health::health_check))
.route("/health/live", get(health::health_check))
// Model endpoints
.route("/v1/models", get(models::list_models))
// Inference endpoints
.route("/v1/completions", post(inference::inference))
.route("/v1/chat/completions", post(inference::inference));
TestServer::new(app).unwrap()
}
#[tokio::test]
async fn test_complete_api_workflow() {
let server = create_test_server();
// 1. Check service health
let health_response = server.get("/health").await;
health_response.assert_status_ok();
let health: HealthStatus = health_response.json();
assert_eq!(health.status, "healthy");
assert!(!health.version.is_empty());
// 2. List available models
let models_response = server.get("/v1/models").await;
models_response.assert_status_ok();
let models: Vec<ModelInfo> = models_response.json();
assert!(!models.is_empty());
// 3. Use the first model for inference
let model_id = &models[0].id;
let inference_request = InferenceRequest {
model: model_id.clone(),
prompt: "Complete this sentence: The weather today is".to_string(),
max_tokens: Some(50),
temperature: Some(0.7),
stream: Some(false),
};
let inference_response = server
.post("/v1/completions")
.json(&inference_request)
.await;
inference_response.assert_status_ok();
let inference: InferenceResponse = inference_response.json();
assert!(!inference.text.is_empty());
assert_eq!(inference.finish_reason, "stop");
assert!(inference.usage.total_tokens > 0);
}
#[tokio::test]
async fn test_all_health_endpoints() {
let server = create_test_server();
// All health endpoints should work the same way
let endpoints = ["/health", "/health/ready", "/health/live"];
for endpoint in &endpoints {
let response = server.get(endpoint).await;
response.assert_status_ok();
let health: HealthStatus = response.json();
assert_eq!(health.status, "healthy");
assert!(!health.version.is_empty());
assert!(health.uptime_seconds >= 0);
assert_eq!(health.details.components.inference_runtime, "operational");
assert_eq!(health.details.components.model_registry, "operational");
assert_eq!(health.details.components.cache_system, "operational");
}
}
#[tokio::test]
async fn test_inference_with_different_models() {
let server = create_test_server();
// Get available models
let models_response = server.get("/v1/models").await;
let models: Vec<ModelInfo> = models_response.json();
// Test inference with each available model
for model in &models {
let request = InferenceRequest {
model: model.id.clone(),
prompt: "Test prompt".to_string(),
max_tokens: Some(10),
temperature: Some(0.5),
stream: Some(false),
};
let response = server.post("/v1/completions").json(&request).await;
response.assert_status_ok();
let inference: InferenceResponse = response.json();
assert!(!inference.text.is_empty());
assert!(inference.text.contains("Test prompt"));
assert!(inference.usage.prompt_tokens > 0);
assert!(inference.usage.completion_tokens > 0);
}
}
#[tokio::test]
async fn test_chat_completions_endpoint() {
let server = create_test_server();
let request = InferenceRequest {
model: "default-model".to_string(),
prompt: "Hello! How are you?".to_string(),
max_tokens: Some(20),
temperature: Some(0.8),
stream: Some(false),
};
let response = server.post("/v1/chat/completions").json(&request).await;
response.assert_status_ok();
let inference: InferenceResponse = response.json();
assert!(!inference.text.is_empty());
assert_eq!(inference.finish_reason, "stop");
}