feat: add cors
CI / Rust CI (push) Successful in 4m37s

This commit is contained in:
2026-04-20 21:03:18 +02:00
parent 81129d6c9c
commit 57a3d04625
5 changed files with 292 additions and 11 deletions
+2 -1
View File
@@ -1,3 +1,4 @@
JWKS_URL=https://auth.iceberg.black/realms/iceberg/protocol/openid-connect/certs JWKS_URL=https://auth.iceberg.black/realms/iceberg/protocol/openid-connect/certs
ISSUER=https://auth.iceberg.black/realms/iceberg ISSUER=https://auth.iceberg.black/realms/iceberg
OLLAMA_URL=... OLLAMA_URL=...
CORS_ORIGIN=
Generated
+7 -6
View File
@@ -73,9 +73,9 @@ dependencies = [
[[package]] [[package]]
name = "axum" name = "axum"
version = "0.8.8" version = "0.8.9"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "8b52af3cb4058c895d37317bb27508dccc8e5f2d39454016b297bf4a400597b8" checksum = "31b698c5f9a010f6573133b09e0de5408834d0c82f8d7475a89fc1867a71cd90"
dependencies = [ dependencies = [
"axum-core", "axum-core",
"bytes", "bytes",
@@ -193,6 +193,7 @@ dependencies = [
"thiserror 2.0.18", "thiserror 2.0.18",
"tokio", "tokio",
"tokio-stream", "tokio-stream",
"tower-http",
"utoipa", "utoipa",
"uuid", "uuid",
"wiremock", "wiremock",
@@ -1758,9 +1759,9 @@ checksum = "1f3ccbac311fea05f86f61904b462b55fb3df8837a366dfc601a0161d0532f20"
[[package]] [[package]]
name = "tokio" name = "tokio"
version = "1.51.1" version = "1.52.1"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "f66bf9585cda4b724d3e78ab34b73fb2bbaba9011b9bfdf69dc836382ea13b8c" checksum = "b67dee974fe86fd92cc45b7a95fdd2f99a36a6d7b0d431a231178d3d670bbcc6"
dependencies = [ dependencies = [
"bytes", "bytes",
"libc", "libc",
@@ -1959,9 +1960,9 @@ dependencies = [
[[package]] [[package]]
name = "uuid" name = "uuid"
version = "1.23.0" version = "1.23.1"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "5ac8b6f42ead25368cf5b098aeb3dc8a1a2c05a3eee8a9a1a68c640edbfc79d9" checksum = "ddd74a9687298c6858e9b88ec8935ec45d22e8fd5e6394fa1bd4e99a87789c76"
dependencies = [ dependencies = [
"getrandom 0.4.2", "getrandom 0.4.2",
"js-sys", "js-sys",
+4 -3
View File
@@ -5,10 +5,10 @@ edition = "2024"
[dev-dependencies] [dev-dependencies]
wiremock = "0.6" wiremock = "0.6"
tokio = { version = "1", features = ["macros", "rt-multi-thread"] } tokio = { version = "1.52.1", features = ["macros", "rt-multi-thread"] }
[dependencies] [dependencies]
axum = "0.8.8" axum = "0.8.9"
utoipa = { version = "5.4.0", features = ["axum_extras"] } utoipa = { version = "5.4.0", features = ["axum_extras"] }
tokio = { version = "1", features = ["full"] } tokio = { version = "1", features = ["full"] }
serde = { version = "1", features = ["derive"] } serde = { version = "1", features = ["derive"] }
@@ -21,4 +21,5 @@ thiserror = "2.0.18"
tokio-stream = "0.1" tokio-stream = "0.1"
futures = "0.3" futures = "0.3"
chrono = { version = "0.4.44", features = ["serde"] } chrono = { version = "0.4.44", features = ["serde"] }
uuid = { version = "1", features = ["v4", "serde"] } uuid = { version = "1.23.1", features = ["v4", "serde"] }
tower-http = { version = "0.6.8", features = ["cors"] }
+266
View File
@@ -0,0 +1,266 @@
# TODO
- Race condition on jwks token refresh
- Rate Limiting
git tag -d v1.0.0; git push origin :refs/tags/v1.0.0; git tag -a v1.0.0 -m "Release v1.0.0"; git push origin v1.0.0
curl https://chat.iceberg.black/api/v1/models -H "Authorization: Bearer eyJhbGciOiJSUzI1NiIsInR5cCIgOiAiSldUIiwia2lkIiA6ICJublpLek04TkZHVmpWbGFPRXZpMUtFSTVHQWRwaGlsYjh3RHRLeG5JOENZIn0.eyJleHAiOjE3NzU4MTUyODksImlhdCI6MTc3NTgxNDk4OSwianRpIjoiMTgwYzA2NDUtYzZiMC00MDRmLTgyYjEtMzA3YmY4ZmJlNmJlIiwiaXNzIjoiaHR0cHM6Ly9hdXRoLmljZWJlcmcuYmxhY2svcmVhbG1zL2ljZWJlcmciLCJzdWIiOiJmZGRiN2FjZC1kMmE5LTRmMTctOWIxNi1kZjVlN2EzNDI4YjciLCJ0eXAiOiJCZWFyZXIiLCJhenAiOiJjaGF0LWFwaSIsInNjb3BlIjoiIiwiY2xpZW50SG9zdCI6Ijg2LjIxMi44NC4xOTEiLCJjbGllbnRBZGRyZXNzIjoiODYuMjEyLjg0LjE5MSIsImNsaWVudF9pZCI6ImNoYXQtYXBpIn0.abjHABcjCJNiB6vriRw60nfzabEfD7CXwyRhkahFkC8ATgfy4fn0T8PnFfsaRpsXuhamIhWwNskrA7L9V3mbWWKW-JEOvImDhTc8sIX0E5fDTbk8O5wa_2yzNdLpRxdSjLqgL544rB8I-LZ8bl5SxtdN3gHfrnWr5ef8bbLgPRzZIylT3QUpah0uywDM_cfrrve9SMHHOUUItyzmOHLw0Igit1EzyFNyjbWf6OAU6TMjOF_eFTc5sakyBwsdJnGy0nhj5R-wxLpr1ug3iEd3Y-jDzHO4m6jariXVJ7Vvbz71i7sadDEKdSyOKcCeQbU4T0Tf7IxqW4c2DNnk3BFFuw"
curl -X POST "https://auth.iceberg.black/realms/iceberg/protocol/openid-connect/token" -H "Content-Type: application/x-www-form-urlencoded" -d "grant_type=client_credentials" -d "client_id=chat-api" -d "client_secret=5fHUp8Z5GoNM70MVOGuQTfKFkaAE48Za"
curl -s -X POST https://chat.iceberg.black/api/v1/chat/completions \
-H "Content-Type: application/json" \
-H "Authorization: Bearer eyJhbGciOiJSUzI1NiIsInR5cCIgOiAiSldUIiwia2lkIiA6ICJublpLek04TkZHVmpWbGFPRXZpMUtFSTVHQWRwaGlsYjh3RHRLeG5JOENZIn0.eyJleHAiOjE3NzU4MTU0NTEsImlhdCI6MTc3NTgxNTE1MSwianRpIjoiOGNjMmM5MjUtZmMzNy00MDFhLWE1ZjUtODFlOGJhNjcxNzcwIiwiaXNzIjoiaHR0cHM6Ly9hdXRoLmljZWJlcmcuYmxhY2svcmVhbG1zL2ljZWJlcmciLCJzdWIiOiJmZGRiN2FjZC1kMmE5LTRmMTctOWIxNi1kZjVlN2EzNDI4YjciLCJ0eXAiOiJCZWFyZXIiLCJhenAiOiJjaGF0LWFwaSIsInNjb3BlIjoiIiwiY2xpZW50SG9zdCI6Ijg2LjIxMi44NC4xOTEiLCJjbGllbnRBZGRyZXNzIjoiODYuMjEyLjg0LjE5MSIsImNsaWVudF9pZCI6ImNoYXQtYXBpIn0.Exv34aoAqwzSqQziZH3zScAnK_xiNCXqpOQl54x7NFMdO0gVgacJgMxoW82Ym8aCNfBRMFtWclZJ9RFh1b9uSEUgsdVtqvMX-2kCAhiMSjmIBPU-L0gr5N63c3bScOoAYa37baR4mXQ5LxHjfkLo_rHDJ74adg2JMo359Wbuu1_OR708_q8yjgO_4fQeYbIffxADwRDvPSIwQ8Y2qRjaAIOZs0xmA9p128CUxlUyvhxwfFquYKRaDs5QE9pIAtvm_KuoBydYopm8j8cbm4ixDhUwYjkzniIIapY77NxpmMosfh8BhK0W3ieK2gTKfmiFDhYgkGybU6b1BnJMy1_Kxg" \
-d '{
"model": "llama3:latest",
"messages": [
{"role": "user", "content": "What is Rust?"}
]
}' | jq .
# 🦙 Ollama Rust API Wrapper
A high-performance Rust API wrapper around Ollama, providing an OpenAI-compatible interface, model lifecycle management, and advanced runtime features.
---
# 🚀 Features
* ✅ OpenAI-compatible API (`/v1/...`)
* ⚡ Streaming (Server-Sent Events)
* 🧠 Model lifecycle management (load/unload)
* 🔐 API key authentication (optional)
* 📊 Usage tracking & observability
* 🔀 Model routing & abstraction
* 🧩 Extensible architecture (multi-provider ready)
---
# 📡 API Endpoints
## 1. Core LLM API (OpenAI-compatible)
- [x] `POST /v1/chat/completions` + streaming
- [x] `POST /v1/completions` + streaming
- [ ] `POST /v1/embeddings`
- [x] `GET /v1/models`
## 2. Model Lifecycle Management
- [x] `POST /v1/models/{model}/load`
- [x] `POST /v1/models/{model}/unload`
## 3. Model Management
- [ ] `POST /v1/models/pull`
- [ ] `DELETE /v1/models/{model}`
## 4. Runtime & Observability
### Model Status
```
GET /v1/models/{model}/status
```
### List Loaded Models
```
GET /v1/runtime/models
```
---
## 6. Health Checks
```
GET /health
GET /ready
```
---
# 🧠 Internal Mapping (Ollama)
| Wrapper Endpoint | Ollama Endpoint |
| ------------------------- | --------------- |
| /v1/chat/completions | /api/chat |
| /v1/completions | /api/generate |
| /v1/embeddings | /api/embeddings |
| /v1/models | /api/tags |
| /v1/models/pull | /api/pull |
| DELETE /v1/models/{model} | /api/delete |
| load/unload | /api/generate |
---
# 🔧 Advanced Features
## 🔀 Model Routing
Use abstract model names:
```json
{
"model": "fast"
}
```
Example mapping:
```
fast → llama3:8b
smart → llama3:70b
code → deepseek-coder
```
---
## 📊 Usage Tracking
```
GET /v1/usage
```
Tracks:
* request count
* latency
* per-model usage
---
## 🚦 Rate Limiting
* Requests per minute
* Tokens per minute
Returns:
```
429 Too Many Requests
```
---
## 🧠 Sessions (Context Management)
```
POST /v1/sessions
POST /v1/sessions/{id}/chat
```
Stores conversation history server-side.
---
## ⚡ Caching
* Embeddings
* Deterministic prompts (temperature = 0)
---
## 🧩 Tool / Function Calling
Supports structured tool execution:
```json
{
"tools": [
{
"name": "function_name",
"parameters": {}
}
]
}
```
---
## 📦 Batch Requests
```
POST /v1/batch
```
---
## 🧠 Auto Eviction
```
POST /v1/runtime/evict
```
Strategies:
* LRU
* memory threshold
---
## 🧾 Logs
```
GET /v1/logs
```
## 🔔 Async Jobs / Webhooks
```
POST /v1/jobs
```
---
# 🏗️ Architecture
```
Client → Rust API → Ollama → Response
```
### Layers:
* HTTP (Axum)
* Service layer (business logic)
* Provider abstraction
* Ollama client
---
# 🔌 Provider Abstraction (Future-Proof)
```rust
trait LlmProvider {
async fn chat(...);
async fn embeddings(...);
}
```
Supports:
* Ollama (current)
* OpenAI (future)
* Others
---
# 🎯 Roadmap
* [ ] Full OpenAI compatibility
* [ ] Multi-node routing
* [ ] GPU-aware scheduling
* [ ] Web UI dashboard
* [ ] Distributed inference
---
# 🧠 Summary
This project turns Ollama into:
👉 A local OpenAI-compatible API
👉 A controllable model runtime
👉 A foundation for a full LLM gateway
+13 -1
View File
@@ -10,10 +10,12 @@ use crate::providers::ollama::client::OllamaProvider;
use crate::state::app_state::AppState; use crate::state::app_state::AppState;
use axum::Router; use axum::Router;
use axum::http::{HeaderValue, Method, header};
use once_cell::sync::Lazy; use once_cell::sync::Lazy;
use std::env; use std::env;
use std::net::SocketAddr; use std::net::SocketAddr;
use std::sync::Arc; use std::sync::Arc;
use tower_http::cors::CorsLayer;
static OLLAMA_URL: Lazy<String> = Lazy::new(|| env::var("OLLAMA_URL").expect("OLLAMA_URL not set")); static OLLAMA_URL: Lazy<String> = Lazy::new(|| env::var("OLLAMA_URL").expect("OLLAMA_URL not set"));
@@ -28,11 +30,21 @@ async fn main() {
ollama: Arc::new(OllamaProvider::new(OLLAMA_URL.as_str())), ollama: Arc::new(OllamaProvider::new(OLLAMA_URL.as_str())),
}; };
let cors_origin =
env::var("CORS_ORIGIN").unwrap_or_else(|_| "http://localhost:3000".to_string());
let cors = CorsLayer::new()
.allow_origin(cors_origin.parse::<HeaderValue>().unwrap())
.allow_methods([Method::GET, Method::POST, Method::PUT, Method::DELETE])
.allow_headers([header::CONTENT_TYPE, header::AUTHORIZATION, header::ACCEPT])
.allow_credentials(true);
let app = Router::new() let app = Router::new()
.nest("/v1", routes::v1::router()) .nest("/v1", routes::v1::router())
.layer(cors)
.with_state(state); .with_state(state);
let addr = SocketAddr::from(([0, 0, 0, 0], 3000)); let addr = SocketAddr::from(([0, 0, 0, 0], 3001));
println!("Server running on {}", addr); println!("Server running on {}", addr);
axum::serve(tokio::net::TcpListener::bind(addr).await.unwrap(), app) axum::serve(tokio::net::TcpListener::bind(addr).await.unwrap(), app)