feat: add models list endpoint
This commit is contained in:
@@ -1,6 +1,396 @@
|
|||||||
# TODO
|
# TODO
|
||||||
|
|
||||||
- Race condition on jwks token refresh
|
- Race condition on jwks token refresh
|
||||||
|
- Rate Limiting
|
||||||
|
|
||||||
|
|
||||||
git tag -d v1.0.0; git push origin :refs/tags/v1.0.0; git tag -a v1.0.0 -m "Release v1.0.0"; git push origin v1.0.0
|
git tag -d v1.0.0; git push origin :refs/tags/v1.0.0; git tag -a v1.0.0 -m "Release v1.0.0"; git push origin v1.0.0
|
||||||
|
|
||||||
|
|
||||||
|
# 🦙 Ollama Rust API Wrapper
|
||||||
|
|
||||||
|
A high-performance Rust API wrapper around Ollama, providing an OpenAI-compatible interface, model lifecycle management, and advanced runtime features.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
# 🚀 Features
|
||||||
|
|
||||||
|
* ✅ OpenAI-compatible API (`/v1/...`)
|
||||||
|
* ⚡ Streaming (Server-Sent Events)
|
||||||
|
* 🧠 Model lifecycle management (load/unload)
|
||||||
|
* 🔐 API key authentication (optional)
|
||||||
|
* 📊 Usage tracking & observability
|
||||||
|
* 🔀 Model routing & abstraction
|
||||||
|
* 🧩 Extensible architecture (multi-provider ready)
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
# 📡 API Endpoints
|
||||||
|
|
||||||
|
## 1. Core LLM API (OpenAI-compatible)
|
||||||
|
|
||||||
|
### Chat Completions
|
||||||
|
|
||||||
|
```
|
||||||
|
POST /v1/chat/completions
|
||||||
|
```
|
||||||
|
|
||||||
|
### Text Completions
|
||||||
|
|
||||||
|
```
|
||||||
|
POST /v1/completions
|
||||||
|
```
|
||||||
|
|
||||||
|
### Embeddings
|
||||||
|
|
||||||
|
```
|
||||||
|
POST /v1/embeddings
|
||||||
|
```
|
||||||
|
|
||||||
|
### List Models
|
||||||
|
|
||||||
|
```
|
||||||
|
GET /v1/models
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 2. Model Lifecycle Management
|
||||||
|
|
||||||
|
### Load (Warmup)
|
||||||
|
|
||||||
|
```
|
||||||
|
POST /v1/models/{model}/load
|
||||||
|
```
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"keep_alive": "10m"
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
### Unload (Free Memory)
|
||||||
|
|
||||||
|
```
|
||||||
|
POST /v1/models/{model}/unload
|
||||||
|
```
|
||||||
|
|
||||||
|
Internally uses:
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"model": "...",
|
||||||
|
"keep_alive": 0
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
### Reload (Optional)
|
||||||
|
|
||||||
|
```
|
||||||
|
POST /v1/models/{model}/reload
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 3. Model Management
|
||||||
|
|
||||||
|
### Pull Model
|
||||||
|
|
||||||
|
```
|
||||||
|
POST /v1/models/pull
|
||||||
|
```
|
||||||
|
|
||||||
|
### Delete Model
|
||||||
|
|
||||||
|
```
|
||||||
|
DELETE /v1/models/{model}
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 4. Runtime & Observability
|
||||||
|
|
||||||
|
### Model Status
|
||||||
|
|
||||||
|
```
|
||||||
|
GET /v1/models/{model}/status
|
||||||
|
```
|
||||||
|
|
||||||
|
### List Loaded Models
|
||||||
|
|
||||||
|
```
|
||||||
|
GET /v1/runtime/models
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 5. Streaming
|
||||||
|
|
||||||
|
Enable streaming with:
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"stream": true
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
Response format (SSE):
|
||||||
|
|
||||||
|
```
|
||||||
|
data: {"choices":[{"delta":{"content":"Hello"}}]}
|
||||||
|
|
||||||
|
data: {"choices":[{"delta":{"content":" world"}}]}
|
||||||
|
|
||||||
|
data: [DONE]
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 6. Health Checks
|
||||||
|
|
||||||
|
```
|
||||||
|
GET /health
|
||||||
|
GET /ready
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
# 🧠 Internal Mapping (Ollama)
|
||||||
|
|
||||||
|
| Wrapper Endpoint | Ollama Endpoint |
|
||||||
|
| ------------------------- | --------------- |
|
||||||
|
| /v1/chat/completions | /api/chat |
|
||||||
|
| /v1/completions | /api/generate |
|
||||||
|
| /v1/embeddings | /api/embeddings |
|
||||||
|
| /v1/models | /api/tags |
|
||||||
|
| /v1/models/pull | /api/pull |
|
||||||
|
| DELETE /v1/models/{model} | /api/delete |
|
||||||
|
| load/unload | /api/generate |
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
# ⚙️ Configuration
|
||||||
|
|
||||||
|
### Docker (optional default)
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
environment:
|
||||||
|
- OLLAMA_KEEP_ALIVE=10m
|
||||||
|
```
|
||||||
|
|
||||||
|
> Note: Request-level `keep_alive` overrides this value.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
# 🔧 Advanced Features
|
||||||
|
|
||||||
|
## 🔀 Model Routing
|
||||||
|
|
||||||
|
Use abstract model names:
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"model": "fast"
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
Example mapping:
|
||||||
|
|
||||||
|
```
|
||||||
|
fast → llama3:8b
|
||||||
|
smart → llama3:70b
|
||||||
|
code → deepseek-coder
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 📊 Usage Tracking
|
||||||
|
|
||||||
|
```
|
||||||
|
GET /v1/usage
|
||||||
|
```
|
||||||
|
|
||||||
|
Tracks:
|
||||||
|
|
||||||
|
* request count
|
||||||
|
* latency
|
||||||
|
* per-model usage
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 🔐 Authentication
|
||||||
|
|
||||||
|
```
|
||||||
|
Authorization: Bearer sk-xxxx
|
||||||
|
```
|
||||||
|
|
||||||
|
Endpoints:
|
||||||
|
|
||||||
|
```
|
||||||
|
POST /v1/keys
|
||||||
|
DELETE /v1/keys/{id}
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 🚦 Rate Limiting
|
||||||
|
|
||||||
|
* Requests per minute
|
||||||
|
* Tokens per minute
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
|
||||||
|
```
|
||||||
|
429 Too Many Requests
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 🧠 Sessions (Context Management)
|
||||||
|
|
||||||
|
```
|
||||||
|
POST /v1/sessions
|
||||||
|
POST /v1/sessions/{id}/chat
|
||||||
|
```
|
||||||
|
|
||||||
|
Stores conversation history server-side.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## ⚡ Caching
|
||||||
|
|
||||||
|
* Embeddings
|
||||||
|
* Deterministic prompts (temperature = 0)
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 🧩 Tool / Function Calling
|
||||||
|
|
||||||
|
Supports structured tool execution:
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"tools": [
|
||||||
|
{
|
||||||
|
"name": "function_name",
|
||||||
|
"parameters": {}
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 📦 Batch Requests
|
||||||
|
|
||||||
|
```
|
||||||
|
POST /v1/batch
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 🧠 Auto Eviction
|
||||||
|
|
||||||
|
```
|
||||||
|
POST /v1/runtime/evict
|
||||||
|
```
|
||||||
|
|
||||||
|
Strategies:
|
||||||
|
|
||||||
|
* LRU
|
||||||
|
* memory threshold
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 🧾 Logs
|
||||||
|
|
||||||
|
```
|
||||||
|
GET /v1/logs
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 📚 Embedding Store (Optional)
|
||||||
|
|
||||||
|
```
|
||||||
|
POST /v1/documents
|
||||||
|
POST /v1/search
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 🔔 Async Jobs / Webhooks
|
||||||
|
|
||||||
|
```
|
||||||
|
POST /v1/jobs
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
# 🏗️ Architecture
|
||||||
|
|
||||||
|
```
|
||||||
|
Client → Rust API → Ollama → Response
|
||||||
|
```
|
||||||
|
|
||||||
|
### Layers:
|
||||||
|
|
||||||
|
* HTTP (Axum)
|
||||||
|
* Service layer (business logic)
|
||||||
|
* Provider abstraction
|
||||||
|
* Ollama client
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
# 🔌 Provider Abstraction (Future-Proof)
|
||||||
|
|
||||||
|
```rust
|
||||||
|
trait LlmProvider {
|
||||||
|
async fn chat(...);
|
||||||
|
async fn embeddings(...);
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
Supports:
|
||||||
|
|
||||||
|
* Ollama (current)
|
||||||
|
* OpenAI (future)
|
||||||
|
* Others
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
# ⚠️ Notes
|
||||||
|
|
||||||
|
* Ollama has no native unload endpoint → simulated via `keep_alive = 0`
|
||||||
|
* Streaming uses NDJSON → converted to SSE
|
||||||
|
* Chunk handling must be robust (partial JSON)
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
# 🎯 Roadmap
|
||||||
|
|
||||||
|
* [ ] Full OpenAI compatibility
|
||||||
|
* [ ] Multi-node routing
|
||||||
|
* [ ] GPU-aware scheduling
|
||||||
|
* [ ] Web UI dashboard
|
||||||
|
* [ ] Distributed inference
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
# 🧠 Summary
|
||||||
|
|
||||||
|
This project turns Ollama into:
|
||||||
|
|
||||||
|
👉 A local OpenAI-compatible API
|
||||||
|
👉 A controllable model runtime
|
||||||
|
👉 A foundation for a full LLM gateway
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
# 📜 License
|
||||||
|
|
||||||
|
MIT
|
||||||
|
|||||||
+13
-42
@@ -1,34 +1,14 @@
|
|||||||
mod auth;
|
mod auth;
|
||||||
|
mod providers;
|
||||||
|
mod routes;
|
||||||
|
mod state;
|
||||||
|
|
||||||
use crate::auth::jwt::Claims;
|
use crate::providers::ollama::OllamaProvider;
|
||||||
use crate::auth::middleware::auth_middleware;
|
use crate::state::app_state::AppState;
|
||||||
|
|
||||||
use axum::extract::Extension;
|
use axum::Router;
|
||||||
use axum::{Router, middleware, routing::get};
|
|
||||||
use std::net::SocketAddr;
|
use std::net::SocketAddr;
|
||||||
|
use std::sync::Arc;
|
||||||
pub async fn protected_route(Extension(claims): Extension<Claims>) -> String {
|
|
||||||
format!(
|
|
||||||
"Hello {}, your user id is {}",
|
|
||||||
claims.preferred_username.unwrap_or("unknown".to_string()),
|
|
||||||
claims.sub
|
|
||||||
)
|
|
||||||
}
|
|
||||||
|
|
||||||
pub async fn public_route() -> &'static str {
|
|
||||||
println!("Public route hit");
|
|
||||||
"Public endpoint: no authentication required"
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn app() -> Router {
|
|
||||||
let public_routes = Router::new().route("/", get(public_route));
|
|
||||||
|
|
||||||
let protected_routes = Router::new()
|
|
||||||
.route("/protected", get(protected_route))
|
|
||||||
.layer(middleware::from_fn(auth_middleware));
|
|
||||||
|
|
||||||
Router::new().merge(public_routes).merge(protected_routes)
|
|
||||||
}
|
|
||||||
|
|
||||||
#[tokio::main]
|
#[tokio::main]
|
||||||
async fn main() {
|
async fn main() {
|
||||||
@@ -37,23 +17,14 @@ async fn main() {
|
|||||||
dotenvy::dotenv().ok();
|
dotenvy::dotenv().ok();
|
||||||
}
|
}
|
||||||
|
|
||||||
// let jwks = keycloak::get_jwks()
|
let state = AppState {
|
||||||
// .await
|
ollama: Arc::new(OllamaProvider::new("http://localhost:11434")),
|
||||||
// .expect("Failed to fetch JWKS");
|
};
|
||||||
|
|
||||||
// // println!("{:?}", jwks);
|
let app = Router::new()
|
||||||
|
.nest("/v1", routes::v1::router())
|
||||||
|
.with_state(state);
|
||||||
|
|
||||||
// let token = "eyJhbGciOiJSUzI1NiIsInR5cCIgOiAiSldUIiwia2lkIiA6ICJublpLek04TkZHVmpWbGFPRXZpMUtFSTVHQWRwaGlsYjh3RHRLeG5JOENZIn0.eyJleHAiOjE3NzU3Mzg2MDUsImlhdCI6MTc3NTczODMwNSwianRpIjoiNjk2OTY4NzQtZWMwNi00NGFkLTg0MDYtYmY3YWM4MjI5MjkxIiwiaXNzIjoiaHR0cHM6Ly9hdXRoLmljZWJlcmcuYmxhY2svcmVhbG1zL2ljZWJlcmciLCJzdWIiOiJmZGRiN2FjZC1kMmE5LTRmMTctOWIxNi1kZjVlN2EzNDI4YjciLCJ0eXAiOiJCZWFyZXIiLCJhenAiOiJjaGF0LWFwaSIsInNjb3BlIjoiIiwiY2xpZW50SG9zdCI6Ijg2LjIxMi44NC4xOTEiLCJjbGllbnRBZGRyZXNzIjoiODYuMjEyLjg0LjE5MSIsImNsaWVudF9pZCI6ImNoYXQtYXBpIn0.BS7ohLWiMDxAUz_Q-Qi2UoLYbNn8AUrYeSWeO-602SQ-AYBW3gfYxXOSeRgWyn4VfObpVfK7QfqQBUxorXxi1JVld-4fGXL8NXQNyq5Ip_JHNG1p02Z39Pe9MmC9MXOwA_GQF2PIkLIdOJ_W_guXVhl2ptEWPPSiXM5Z5CNg8lyOiKPI0g2JWV6FBRG-HMXzqnxAb1j8wGUpC9JzGwAU3sjWBGhT1AAovs-XLmm5hZEPxI-Ia3SmUnF-QjFMmebPVxLdxL7OszzVEhKipsZRiwQxjY6eJhJFFa8uycBigHPSzu_HqqkK6AjNlyExvR0EGvl9zUWdOfMPDiVX2Sg92g";
|
|
||||||
// match jwt::validate_token(token, &jwks) {
|
|
||||||
// Ok(claims) => {
|
|
||||||
// println!("Valid token for user: {:?}", claims);
|
|
||||||
// }
|
|
||||||
// Err(err) => {
|
|
||||||
// println!("Invalid token: {}", err);
|
|
||||||
// }
|
|
||||||
// }
|
|
||||||
|
|
||||||
let app = app();
|
|
||||||
let addr = SocketAddr::from(([0, 0, 0, 0], 3000));
|
let addr = SocketAddr::from(([0, 0, 0, 0], 3000));
|
||||||
println!("Server running on {}", addr);
|
println!("Server running on {}", addr);
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1 @@
|
|||||||
|
pub mod ollama;
|
||||||
@@ -0,0 +1,25 @@
|
|||||||
|
use reqwest::Client;
|
||||||
|
use serde_json::Value;
|
||||||
|
|
||||||
|
#[derive(Clone)]
|
||||||
|
pub struct OllamaProvider {
|
||||||
|
pub client: Client,
|
||||||
|
pub base_url: String,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl OllamaProvider {
|
||||||
|
pub fn new(base_url: impl Into<String>) -> Self {
|
||||||
|
Self {
|
||||||
|
client: Client::new(),
|
||||||
|
base_url: base_url.into(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub async fn list_models(&self) -> Result<Value, reqwest::Error> {
|
||||||
|
let url = format!("{}/api/tags", self.base_url);
|
||||||
|
|
||||||
|
let res = self.client.get(url).send().await?.json::<Value>().await?;
|
||||||
|
|
||||||
|
Ok(res)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1 @@
|
|||||||
|
pub mod v1;
|
||||||
@@ -0,0 +1,12 @@
|
|||||||
|
pub mod models;
|
||||||
|
|
||||||
|
use crate::auth::middleware::auth_middleware;
|
||||||
|
use crate::routes::v1::models::list_models;
|
||||||
|
use crate::state::app_state::AppState;
|
||||||
|
use axum::{Router, middleware, routing::get};
|
||||||
|
|
||||||
|
pub fn router() -> Router<AppState> {
|
||||||
|
Router::new()
|
||||||
|
.route("/models", get(list_models))
|
||||||
|
.layer(middleware::from_fn(auth_middleware))
|
||||||
|
}
|
||||||
@@ -0,0 +1,16 @@
|
|||||||
|
use axum::{Json, extract::State};
|
||||||
|
use serde_json::Value;
|
||||||
|
|
||||||
|
use crate::state::app_state::AppState;
|
||||||
|
|
||||||
|
pub async fn list_models(
|
||||||
|
State(state): State<AppState>,
|
||||||
|
) -> Result<Json<Value>, (axum::http::StatusCode, String)> {
|
||||||
|
match state.ollama.list_models().await {
|
||||||
|
Ok(models) => Ok(Json(models)),
|
||||||
|
Err(err) => Err((
|
||||||
|
axum::http::StatusCode::INTERNAL_SERVER_ERROR,
|
||||||
|
err.to_string(),
|
||||||
|
)),
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,7 @@
|
|||||||
|
use crate::providers::ollama::OllamaProvider;
|
||||||
|
use std::sync::Arc;
|
||||||
|
|
||||||
|
#[derive(Clone)]
|
||||||
|
pub struct AppState {
|
||||||
|
pub ollama: Arc<OllamaProvider>,
|
||||||
|
}
|
||||||
@@ -0,0 +1 @@
|
|||||||
|
pub mod app_state;
|
||||||
Reference in New Issue
Block a user