feat: add models list endpoint

This commit is contained in:
2026-04-09 20:51:20 +02:00
parent b429f6256b
commit 8d26907c89
9 changed files with 467 additions and 43 deletions
+390
View File
@@ -1,6 +1,396 @@
# TODO # TODO
- Race condition on jwks token refresh - Race condition on jwks token refresh
- Rate Limiting
git tag -d v1.0.0; git push origin :refs/tags/v1.0.0; git tag -a v1.0.0 -m "Release v1.0.0"; git push origin v1.0.0 git tag -d v1.0.0; git push origin :refs/tags/v1.0.0; git tag -a v1.0.0 -m "Release v1.0.0"; git push origin v1.0.0
# 🦙 Ollama Rust API Wrapper
A high-performance Rust API wrapper around Ollama, providing an OpenAI-compatible interface, model lifecycle management, and advanced runtime features.
---
# 🚀 Features
* ✅ OpenAI-compatible API (`/v1/...`)
* ⚡ Streaming (Server-Sent Events)
* 🧠 Model lifecycle management (load/unload)
* 🔐 API key authentication (optional)
* 📊 Usage tracking & observability
* 🔀 Model routing & abstraction
* 🧩 Extensible architecture (multi-provider ready)
---
# 📡 API Endpoints
## 1. Core LLM API (OpenAI-compatible)
### Chat Completions
```
POST /v1/chat/completions
```
### Text Completions
```
POST /v1/completions
```
### Embeddings
```
POST /v1/embeddings
```
### List Models
```
GET /v1/models
```
---
## 2. Model Lifecycle Management
### Load (Warmup)
```
POST /v1/models/{model}/load
```
```json
{
"keep_alive": "10m"
}
```
---
### Unload (Free Memory)
```
POST /v1/models/{model}/unload
```
Internally uses:
```json
{
"model": "...",
"keep_alive": 0
}
```
---
### Reload (Optional)
```
POST /v1/models/{model}/reload
```
---
## 3. Model Management
### Pull Model
```
POST /v1/models/pull
```
### Delete Model
```
DELETE /v1/models/{model}
```
---
## 4. Runtime & Observability
### Model Status
```
GET /v1/models/{model}/status
```
### List Loaded Models
```
GET /v1/runtime/models
```
---
## 5. Streaming
Enable streaming with:
```json
{
"stream": true
}
```
Response format (SSE):
```
data: {"choices":[{"delta":{"content":"Hello"}}]}
data: {"choices":[{"delta":{"content":" world"}}]}
data: [DONE]
```
---
## 6. Health Checks
```
GET /health
GET /ready
```
---
# 🧠 Internal Mapping (Ollama)
| Wrapper Endpoint | Ollama Endpoint |
| ------------------------- | --------------- |
| /v1/chat/completions | /api/chat |
| /v1/completions | /api/generate |
| /v1/embeddings | /api/embeddings |
| /v1/models | /api/tags |
| /v1/models/pull | /api/pull |
| DELETE /v1/models/{model} | /api/delete |
| load/unload | /api/generate |
---
# ⚙️ Configuration
### Docker (optional default)
```yaml
environment:
- OLLAMA_KEEP_ALIVE=10m
```
> Note: Request-level `keep_alive` overrides this value.
---
# 🔧 Advanced Features
## 🔀 Model Routing
Use abstract model names:
```json
{
"model": "fast"
}
```
Example mapping:
```
fast → llama3:8b
smart → llama3:70b
code → deepseek-coder
```
---
## 📊 Usage Tracking
```
GET /v1/usage
```
Tracks:
* request count
* latency
* per-model usage
---
## 🔐 Authentication
```
Authorization: Bearer sk-xxxx
```
Endpoints:
```
POST /v1/keys
DELETE /v1/keys/{id}
```
---
## 🚦 Rate Limiting
* Requests per minute
* Tokens per minute
Returns:
```
429 Too Many Requests
```
---
## 🧠 Sessions (Context Management)
```
POST /v1/sessions
POST /v1/sessions/{id}/chat
```
Stores conversation history server-side.
---
## ⚡ Caching
* Embeddings
* Deterministic prompts (temperature = 0)
---
## 🧩 Tool / Function Calling
Supports structured tool execution:
```json
{
"tools": [
{
"name": "function_name",
"parameters": {}
}
]
}
```
---
## 📦 Batch Requests
```
POST /v1/batch
```
---
## 🧠 Auto Eviction
```
POST /v1/runtime/evict
```
Strategies:
* LRU
* memory threshold
---
## 🧾 Logs
```
GET /v1/logs
```
---
## 📚 Embedding Store (Optional)
```
POST /v1/documents
POST /v1/search
```
---
## 🔔 Async Jobs / Webhooks
```
POST /v1/jobs
```
---
# 🏗️ Architecture
```
Client → Rust API → Ollama → Response
```
### Layers:
* HTTP (Axum)
* Service layer (business logic)
* Provider abstraction
* Ollama client
---
# 🔌 Provider Abstraction (Future-Proof)
```rust
trait LlmProvider {
async fn chat(...);
async fn embeddings(...);
}
```
Supports:
* Ollama (current)
* OpenAI (future)
* Others
---
# ⚠️ Notes
* Ollama has no native unload endpoint → simulated via `keep_alive = 0`
* Streaming uses NDJSON → converted to SSE
* Chunk handling must be robust (partial JSON)
---
# 🎯 Roadmap
* [ ] Full OpenAI compatibility
* [ ] Multi-node routing
* [ ] GPU-aware scheduling
* [ ] Web UI dashboard
* [ ] Distributed inference
---
# 🧠 Summary
This project turns Ollama into:
👉 A local OpenAI-compatible API
👉 A controllable model runtime
👉 A foundation for a full LLM gateway
---
# 📜 License
MIT
+13 -42
View File
@@ -1,34 +1,14 @@
mod auth; mod auth;
mod providers;
mod routes;
mod state;
use crate::auth::jwt::Claims; use crate::providers::ollama::OllamaProvider;
use crate::auth::middleware::auth_middleware; use crate::state::app_state::AppState;
use axum::extract::Extension; use axum::Router;
use axum::{Router, middleware, routing::get};
use std::net::SocketAddr; use std::net::SocketAddr;
use std::sync::Arc;
pub async fn protected_route(Extension(claims): Extension<Claims>) -> String {
format!(
"Hello {}, your user id is {}",
claims.preferred_username.unwrap_or("unknown".to_string()),
claims.sub
)
}
pub async fn public_route() -> &'static str {
println!("Public route hit");
"Public endpoint: no authentication required"
}
pub fn app() -> Router {
let public_routes = Router::new().route("/", get(public_route));
let protected_routes = Router::new()
.route("/protected", get(protected_route))
.layer(middleware::from_fn(auth_middleware));
Router::new().merge(public_routes).merge(protected_routes)
}
#[tokio::main] #[tokio::main]
async fn main() { async fn main() {
@@ -37,23 +17,14 @@ async fn main() {
dotenvy::dotenv().ok(); dotenvy::dotenv().ok();
} }
// let jwks = keycloak::get_jwks() let state = AppState {
// .await ollama: Arc::new(OllamaProvider::new("http://localhost:11434")),
// .expect("Failed to fetch JWKS"); };
// // println!("{:?}", jwks); let app = Router::new()
.nest("/v1", routes::v1::router())
.with_state(state);
// let token = "eyJhbGciOiJSUzI1NiIsInR5cCIgOiAiSldUIiwia2lkIiA6ICJublpLek04TkZHVmpWbGFPRXZpMUtFSTVHQWRwaGlsYjh3RHRLeG5JOENZIn0.eyJleHAiOjE3NzU3Mzg2MDUsImlhdCI6MTc3NTczODMwNSwianRpIjoiNjk2OTY4NzQtZWMwNi00NGFkLTg0MDYtYmY3YWM4MjI5MjkxIiwiaXNzIjoiaHR0cHM6Ly9hdXRoLmljZWJlcmcuYmxhY2svcmVhbG1zL2ljZWJlcmciLCJzdWIiOiJmZGRiN2FjZC1kMmE5LTRmMTctOWIxNi1kZjVlN2EzNDI4YjciLCJ0eXAiOiJCZWFyZXIiLCJhenAiOiJjaGF0LWFwaSIsInNjb3BlIjoiIiwiY2xpZW50SG9zdCI6Ijg2LjIxMi44NC4xOTEiLCJjbGllbnRBZGRyZXNzIjoiODYuMjEyLjg0LjE5MSIsImNsaWVudF9pZCI6ImNoYXQtYXBpIn0.BS7ohLWiMDxAUz_Q-Qi2UoLYbNn8AUrYeSWeO-602SQ-AYBW3gfYxXOSeRgWyn4VfObpVfK7QfqQBUxorXxi1JVld-4fGXL8NXQNyq5Ip_JHNG1p02Z39Pe9MmC9MXOwA_GQF2PIkLIdOJ_W_guXVhl2ptEWPPSiXM5Z5CNg8lyOiKPI0g2JWV6FBRG-HMXzqnxAb1j8wGUpC9JzGwAU3sjWBGhT1AAovs-XLmm5hZEPxI-Ia3SmUnF-QjFMmebPVxLdxL7OszzVEhKipsZRiwQxjY6eJhJFFa8uycBigHPSzu_HqqkK6AjNlyExvR0EGvl9zUWdOfMPDiVX2Sg92g";
// match jwt::validate_token(token, &jwks) {
// Ok(claims) => {
// println!("Valid token for user: {:?}", claims);
// }
// Err(err) => {
// println!("Invalid token: {}", err);
// }
// }
let app = app();
let addr = SocketAddr::from(([0, 0, 0, 0], 3000)); let addr = SocketAddr::from(([0, 0, 0, 0], 3000));
println!("Server running on {}", addr); println!("Server running on {}", addr);
+1
View File
@@ -0,0 +1 @@
pub mod ollama;
+25
View File
@@ -0,0 +1,25 @@
use reqwest::Client;
use serde_json::Value;
#[derive(Clone)]
pub struct OllamaProvider {
pub client: Client,
pub base_url: String,
}
impl OllamaProvider {
pub fn new(base_url: impl Into<String>) -> Self {
Self {
client: Client::new(),
base_url: base_url.into(),
}
}
pub async fn list_models(&self) -> Result<Value, reqwest::Error> {
let url = format!("{}/api/tags", self.base_url);
let res = self.client.get(url).send().await?.json::<Value>().await?;
Ok(res)
}
}
+1
View File
@@ -0,0 +1 @@
pub mod v1;
+12
View File
@@ -0,0 +1,12 @@
pub mod models;
use crate::auth::middleware::auth_middleware;
use crate::routes::v1::models::list_models;
use crate::state::app_state::AppState;
use axum::{Router, middleware, routing::get};
pub fn router() -> Router<AppState> {
Router::new()
.route("/models", get(list_models))
.layer(middleware::from_fn(auth_middleware))
}
+16
View File
@@ -0,0 +1,16 @@
use axum::{Json, extract::State};
use serde_json::Value;
use crate::state::app_state::AppState;
pub async fn list_models(
State(state): State<AppState>,
) -> Result<Json<Value>, (axum::http::StatusCode, String)> {
match state.ollama.list_models().await {
Ok(models) => Ok(Json(models)),
Err(err) => Err((
axum::http::StatusCode::INTERNAL_SERVER_ERROR,
err.to_string(),
)),
}
}
+7
View File
@@ -0,0 +1,7 @@
use crate::providers::ollama::OllamaProvider;
use std::sync::Arc;
#[derive(Clone)]
pub struct AppState {
pub ollama: Arc<OllamaProvider>,
}
+1
View File
@@ -0,0 +1 @@
pub mod app_state;