From 8d26907c8961e91314efa14f0fbc09a6400cae2e Mon Sep 17 00:00:00 2001 From: LucasX Ubuntu Date: Thu, 9 Apr 2026 20:51:20 +0200 Subject: [PATCH] feat: add models list endpoint --- readme.md | 392 +++++++++++++++++++++++++++++++++++++++- src/main.rs | 55 ++---- src/providers/mod.rs | 1 + src/providers/ollama.rs | 25 +++ src/routes/mod.rs | 1 + src/routes/v1/mod.rs | 12 ++ src/routes/v1/models.rs | 16 ++ src/state/app_state.rs | 7 + src/state/mod.rs | 1 + 9 files changed, 467 insertions(+), 43 deletions(-) create mode 100644 src/providers/mod.rs create mode 100644 src/providers/ollama.rs create mode 100644 src/routes/mod.rs create mode 100644 src/routes/v1/mod.rs create mode 100644 src/routes/v1/models.rs create mode 100644 src/state/app_state.rs create mode 100644 src/state/mod.rs diff --git a/readme.md b/readme.md index cb4818d..98e7632 100644 --- a/readme.md +++ b/readme.md @@ -1,6 +1,396 @@ # TODO - Race condition on jwks token refresh +- Rate Limiting -git tag -d v1.0.0; git push origin :refs/tags/v1.0.0; git tag -a v1.0.0 -m "Release v1.0.0"; git push origin v1.0.0 \ No newline at end of file +git tag -d v1.0.0; git push origin :refs/tags/v1.0.0; git tag -a v1.0.0 -m "Release v1.0.0"; git push origin v1.0.0 + + +# ๐Ÿฆ™ Ollama Rust API Wrapper + +A high-performance Rust API wrapper around Ollama, providing an OpenAI-compatible interface, model lifecycle management, and advanced runtime features. + +--- + +# ๐Ÿš€ Features + +* โœ… OpenAI-compatible API (`/v1/...`) +* โšก Streaming (Server-Sent Events) +* ๐Ÿง  Model lifecycle management (load/unload) +* ๐Ÿ” API key authentication (optional) +* ๐Ÿ“Š Usage tracking & observability +* ๐Ÿ”€ Model routing & abstraction +* ๐Ÿงฉ Extensible architecture (multi-provider ready) + +--- + +# ๐Ÿ“ก API Endpoints + +## 1. Core LLM API (OpenAI-compatible) + +### Chat Completions + +``` +POST /v1/chat/completions +``` + +### Text Completions + +``` +POST /v1/completions +``` + +### Embeddings + +``` +POST /v1/embeddings +``` + +### List Models + +``` +GET /v1/models +``` + +--- + +## 2. Model Lifecycle Management + +### Load (Warmup) + +``` +POST /v1/models/{model}/load +``` + +```json +{ + "keep_alive": "10m" +} +``` + +--- + +### Unload (Free Memory) + +``` +POST /v1/models/{model}/unload +``` + +Internally uses: + +```json +{ + "model": "...", + "keep_alive": 0 +} +``` + +--- + +### Reload (Optional) + +``` +POST /v1/models/{model}/reload +``` + +--- + +## 3. Model Management + +### Pull Model + +``` +POST /v1/models/pull +``` + +### Delete Model + +``` +DELETE /v1/models/{model} +``` + +--- + +## 4. Runtime & Observability + +### Model Status + +``` +GET /v1/models/{model}/status +``` + +### List Loaded Models + +``` +GET /v1/runtime/models +``` + +--- + +## 5. Streaming + +Enable streaming with: + +```json +{ + "stream": true +} +``` + +Response format (SSE): + +``` +data: {"choices":[{"delta":{"content":"Hello"}}]} + +data: {"choices":[{"delta":{"content":" world"}}]} + +data: [DONE] +``` + +--- + +## 6. Health Checks + +``` +GET /health +GET /ready +``` + +--- + +# ๐Ÿง  Internal Mapping (Ollama) + +| Wrapper Endpoint | Ollama Endpoint | +| ------------------------- | --------------- | +| /v1/chat/completions | /api/chat | +| /v1/completions | /api/generate | +| /v1/embeddings | /api/embeddings | +| /v1/models | /api/tags | +| /v1/models/pull | /api/pull | +| DELETE /v1/models/{model} | /api/delete | +| load/unload | /api/generate | + +--- + +# โš™๏ธ Configuration + +### Docker (optional default) + +```yaml +environment: + - OLLAMA_KEEP_ALIVE=10m +``` + +> Note: Request-level `keep_alive` overrides this value. + +--- + +# ๐Ÿ”ง Advanced Features + +## ๐Ÿ”€ Model Routing + +Use abstract model names: + +```json +{ + "model": "fast" +} +``` + +Example mapping: + +``` +fast โ†’ llama3:8b +smart โ†’ llama3:70b +code โ†’ deepseek-coder +``` + +--- + +## ๐Ÿ“Š Usage Tracking + +``` +GET /v1/usage +``` + +Tracks: + +* request count +* latency +* per-model usage + +--- + +## ๐Ÿ” Authentication + +``` +Authorization: Bearer sk-xxxx +``` + +Endpoints: + +``` +POST /v1/keys +DELETE /v1/keys/{id} +``` + +--- + +## ๐Ÿšฆ Rate Limiting + +* Requests per minute +* Tokens per minute + +Returns: + +``` +429 Too Many Requests +``` + +--- + +## ๐Ÿง  Sessions (Context Management) + +``` +POST /v1/sessions +POST /v1/sessions/{id}/chat +``` + +Stores conversation history server-side. + +--- + +## โšก Caching + +* Embeddings +* Deterministic prompts (temperature = 0) + +--- + +## ๐Ÿงฉ Tool / Function Calling + +Supports structured tool execution: + +```json +{ + "tools": [ + { + "name": "function_name", + "parameters": {} + } + ] +} +``` + +--- + +## ๐Ÿ“ฆ Batch Requests + +``` +POST /v1/batch +``` + +--- + +## ๐Ÿง  Auto Eviction + +``` +POST /v1/runtime/evict +``` + +Strategies: + +* LRU +* memory threshold + +--- + +## ๐Ÿงพ Logs + +``` +GET /v1/logs +``` + +--- + +## ๐Ÿ“š Embedding Store (Optional) + +``` +POST /v1/documents +POST /v1/search +``` + +--- + +## ๐Ÿ”” Async Jobs / Webhooks + +``` +POST /v1/jobs +``` + +--- + +# ๐Ÿ—๏ธ Architecture + +``` +Client โ†’ Rust API โ†’ Ollama โ†’ Response +``` + +### Layers: + +* HTTP (Axum) +* Service layer (business logic) +* Provider abstraction +* Ollama client + +--- + +# ๐Ÿ”Œ Provider Abstraction (Future-Proof) + +```rust +trait LlmProvider { + async fn chat(...); + async fn embeddings(...); +} +``` + +Supports: + +* Ollama (current) +* OpenAI (future) +* Others + +--- + +# โš ๏ธ Notes + +* Ollama has no native unload endpoint โ†’ simulated via `keep_alive = 0` +* Streaming uses NDJSON โ†’ converted to SSE +* Chunk handling must be robust (partial JSON) + +--- + +# ๐ŸŽฏ Roadmap + +* [ ] Full OpenAI compatibility +* [ ] Multi-node routing +* [ ] GPU-aware scheduling +* [ ] Web UI dashboard +* [ ] Distributed inference + +--- + +# ๐Ÿง  Summary + +This project turns Ollama into: + +๐Ÿ‘‰ A local OpenAI-compatible API +๐Ÿ‘‰ A controllable model runtime +๐Ÿ‘‰ A foundation for a full LLM gateway + +--- + +# ๐Ÿ“œ License + +MIT diff --git a/src/main.rs b/src/main.rs index 9b0c378..0b7e930 100644 --- a/src/main.rs +++ b/src/main.rs @@ -1,34 +1,14 @@ mod auth; +mod providers; +mod routes; +mod state; -use crate::auth::jwt::Claims; -use crate::auth::middleware::auth_middleware; +use crate::providers::ollama::OllamaProvider; +use crate::state::app_state::AppState; -use axum::extract::Extension; -use axum::{Router, middleware, routing::get}; +use axum::Router; use std::net::SocketAddr; - -pub async fn protected_route(Extension(claims): Extension) -> String { - format!( - "Hello {}, your user id is {}", - claims.preferred_username.unwrap_or("unknown".to_string()), - claims.sub - ) -} - -pub async fn public_route() -> &'static str { - println!("Public route hit"); - "Public endpoint: no authentication required" -} - -pub fn app() -> Router { - let public_routes = Router::new().route("/", get(public_route)); - - let protected_routes = Router::new() - .route("/protected", get(protected_route)) - .layer(middleware::from_fn(auth_middleware)); - - Router::new().merge(public_routes).merge(protected_routes) -} +use std::sync::Arc; #[tokio::main] async fn main() { @@ -37,23 +17,14 @@ async fn main() { dotenvy::dotenv().ok(); } - // let jwks = keycloak::get_jwks() - // .await - // .expect("Failed to fetch JWKS"); + let state = AppState { + ollama: Arc::new(OllamaProvider::new("http://localhost:11434")), + }; - // // println!("{:?}", jwks); + let app = Router::new() + .nest("/v1", routes::v1::router()) + .with_state(state); - // let token = "eyJhbGciOiJSUzI1NiIsInR5cCIgOiAiSldUIiwia2lkIiA6ICJublpLek04TkZHVmpWbGFPRXZpMUtFSTVHQWRwaGlsYjh3RHRLeG5JOENZIn0.eyJleHAiOjE3NzU3Mzg2MDUsImlhdCI6MTc3NTczODMwNSwianRpIjoiNjk2OTY4NzQtZWMwNi00NGFkLTg0MDYtYmY3YWM4MjI5MjkxIiwiaXNzIjoiaHR0cHM6Ly9hdXRoLmljZWJlcmcuYmxhY2svcmVhbG1zL2ljZWJlcmciLCJzdWIiOiJmZGRiN2FjZC1kMmE5LTRmMTctOWIxNi1kZjVlN2EzNDI4YjciLCJ0eXAiOiJCZWFyZXIiLCJhenAiOiJjaGF0LWFwaSIsInNjb3BlIjoiIiwiY2xpZW50SG9zdCI6Ijg2LjIxMi44NC4xOTEiLCJjbGllbnRBZGRyZXNzIjoiODYuMjEyLjg0LjE5MSIsImNsaWVudF9pZCI6ImNoYXQtYXBpIn0.BS7ohLWiMDxAUz_Q-Qi2UoLYbNn8AUrYeSWeO-602SQ-AYBW3gfYxXOSeRgWyn4VfObpVfK7QfqQBUxorXxi1JVld-4fGXL8NXQNyq5Ip_JHNG1p02Z39Pe9MmC9MXOwA_GQF2PIkLIdOJ_W_guXVhl2ptEWPPSiXM5Z5CNg8lyOiKPI0g2JWV6FBRG-HMXzqnxAb1j8wGUpC9JzGwAU3sjWBGhT1AAovs-XLmm5hZEPxI-Ia3SmUnF-QjFMmebPVxLdxL7OszzVEhKipsZRiwQxjY6eJhJFFa8uycBigHPSzu_HqqkK6AjNlyExvR0EGvl9zUWdOfMPDiVX2Sg92g"; - // match jwt::validate_token(token, &jwks) { - // Ok(claims) => { - // println!("Valid token for user: {:?}", claims); - // } - // Err(err) => { - // println!("Invalid token: {}", err); - // } - // } - - let app = app(); let addr = SocketAddr::from(([0, 0, 0, 0], 3000)); println!("Server running on {}", addr); diff --git a/src/providers/mod.rs b/src/providers/mod.rs new file mode 100644 index 0000000..eb9349e --- /dev/null +++ b/src/providers/mod.rs @@ -0,0 +1 @@ +pub mod ollama; diff --git a/src/providers/ollama.rs b/src/providers/ollama.rs new file mode 100644 index 0000000..0cd93b1 --- /dev/null +++ b/src/providers/ollama.rs @@ -0,0 +1,25 @@ +use reqwest::Client; +use serde_json::Value; + +#[derive(Clone)] +pub struct OllamaProvider { + pub client: Client, + pub base_url: String, +} + +impl OllamaProvider { + pub fn new(base_url: impl Into) -> Self { + Self { + client: Client::new(), + base_url: base_url.into(), + } + } + + pub async fn list_models(&self) -> Result { + let url = format!("{}/api/tags", self.base_url); + + let res = self.client.get(url).send().await?.json::().await?; + + Ok(res) + } +} diff --git a/src/routes/mod.rs b/src/routes/mod.rs new file mode 100644 index 0000000..a3a6d96 --- /dev/null +++ b/src/routes/mod.rs @@ -0,0 +1 @@ +pub mod v1; diff --git a/src/routes/v1/mod.rs b/src/routes/v1/mod.rs new file mode 100644 index 0000000..0e98fa6 --- /dev/null +++ b/src/routes/v1/mod.rs @@ -0,0 +1,12 @@ +pub mod models; + +use crate::auth::middleware::auth_middleware; +use crate::routes::v1::models::list_models; +use crate::state::app_state::AppState; +use axum::{Router, middleware, routing::get}; + +pub fn router() -> Router { + Router::new() + .route("/models", get(list_models)) + .layer(middleware::from_fn(auth_middleware)) +} diff --git a/src/routes/v1/models.rs b/src/routes/v1/models.rs new file mode 100644 index 0000000..d2c966b --- /dev/null +++ b/src/routes/v1/models.rs @@ -0,0 +1,16 @@ +use axum::{Json, extract::State}; +use serde_json::Value; + +use crate::state::app_state::AppState; + +pub async fn list_models( + State(state): State, +) -> Result, (axum::http::StatusCode, String)> { + match state.ollama.list_models().await { + Ok(models) => Ok(Json(models)), + Err(err) => Err(( + axum::http::StatusCode::INTERNAL_SERVER_ERROR, + err.to_string(), + )), + } +} diff --git a/src/state/app_state.rs b/src/state/app_state.rs new file mode 100644 index 0000000..1fb247b --- /dev/null +++ b/src/state/app_state.rs @@ -0,0 +1,7 @@ +use crate::providers::ollama::OllamaProvider; +use std::sync::Arc; + +#[derive(Clone)] +pub struct AppState { + pub ollama: Arc, +} diff --git a/src/state/mod.rs b/src/state/mod.rs new file mode 100644 index 0000000..384f36c --- /dev/null +++ b/src/state/mod.rs @@ -0,0 +1 @@ +pub mod app_state;