feat(metrics): prometheus /metrics + generation timing in usage

- /metrics endpoint: request/token/latency counters, tok/s gauge, build info (std-only, no deps)
- usage.duration_ms + usage.tokens_per_second in non-streaming and streaming responses
- /health now reports uptime_s, n_ctx, version
- MAX_TOKENS env config (hard cap, default 2048; 0 = unlimited)
- metrics unit tests (counters, prometheus shape, uptime monotonic)
This commit is contained in:
asepharyana
2026-08-03 11:44:21 +07:00
parent 7cec411cba
commit e8f90fc9b1
8 changed files with 891 additions and 102 deletions
+26 -1
View File
@@ -1,14 +1,39 @@
//! Health check endpoint.
use std::sync::atomic::AtomicU64;
use std::sync::LazyLock;
use std::time::Instant;
use axum::Json;
use crate::config::MODEL_ID;
use crate::config::{CONFIG, MODEL_ID};
use crate::domain::entity::HealthResponse;
/// Process start instant — used to compute uptime for /health and /metrics.
pub static START_INSTANT: LazyLock<Instant> = LazyLock::new(Instant::now);
/// Process start timestamp (unix seconds) — exported as a Prometheus gauge.
pub static START_TIMESTAMP: LazyLock<AtomicU64> = LazyLock::new(|| {
AtomicU64::new(
std::time::SystemTime::now()
.duration_since(std::time::UNIX_EPOCH)
.map(|d| d.as_secs())
.unwrap_or(0),
)
});
/// Seconds since process start.
pub fn uptime_secs() -> u64 {
START_INSTANT.elapsed().as_secs()
}
pub async fn health_check() -> Json<HealthResponse> {
Json(HealthResponse {
status: "ok".into(),
// Report the exact model id served by /v1/models (no extra suffix).
model: MODEL_ID.into(),
uptime_s: Some(uptime_secs()),
n_ctx: Some(CONFIG.n_ctx),
version: Some(env!("CARGO_PKG_VERSION").into()),
})
}