Compare commits

...
30 Commits
Author SHA1 Message Date
LucasDLTG eaf197f86e feat: secure genrate key endpoint
CI / Rust CI (push) Successful in 2m9s
Publish & Deploy / Build and Push to Registry (push) Successful in 28s
Publish & Deploy / Deploy via SSH (push) Successful in 18s
2026-05-07 20:59:01 +02:00
LucasDLTG 7b98fab2c8 fix: add cors option http
CI / Rust CI (push) Successful in 2m8s
Publish & Deploy / Build and Push to Registry (push) Successful in 28s
Publish & Deploy / Deploy via SSH (push) Successful in 17s
2026-05-07 20:12:24 +02:00
LucasDLTG d7ab571bbe fix: release ci with sqlx
CI / Rust CI (push) Successful in 2m8s
Publish & Deploy / Build and Push to Registry (push) Successful in 27s
Publish & Deploy / Deploy via SSH (push) Successful in 7s
2026-05-07 19:56:17 +02:00
LucasDLTG 87fcd522c4 fix: release ci with sqlx
CI / Rust CI (push) Successful in 2m9s
Publish & Deploy / Build and Push to Registry (push) Failing after 14s
Publish & Deploy / Deploy via SSH (push) Has been skipped
2026-05-07 19:50:37 +02:00
LucasDLTG 5884d840e7 fix: cargo audit no fatal
CI / Rust CI (push) Successful in 5m12s
Publish & Deploy / Build and Push to Registry (push) Failing after 14s
Publish & Deploy / Deploy via SSH (push) Has been skipped
2026-05-07 19:43:20 +02:00
LucasDLTG a631b8cd1e feat: add .sqlx 2026-05-07 19:39:50 +02:00
LucasDLTG 8bc5200f77 feat: readme 2026-05-07 19:36:44 +02:00
LucasDLTG 8878dbb454 fix: pre commit
CI / Rust CI (push) Failing after 1m12s
2026-05-07 19:34:43 +02:00
LucasDLTG ecf275dedd feat: readme 2026-05-07 19:33:25 +02:00
LucasDLTG a6b9ee5b09 fix: ci with sqlx
CI / Rust CI (push) Failing after 1m10s
Publish & Deploy / Build and Push to Registry (push) Failing after 13s
Publish & Deploy / Deploy via SSH (push) Has been skipped
2026-05-07 19:17:01 +02:00
LucasDLTG 2e81132f2d fix: use nex table name
CI / Rust CI (push) Failing after 1m12s
Publish & Deploy / Build and Push to Registry (push) Failing after 1m48s
Publish & Deploy / Deploy via SSH (push) Has been skipped
2026-05-07 19:07:14 +02:00
LucasDLTG c534299c8e feat: update last time used api key 2026-05-07 16:15:27 +02:00
LucasDLTG 4b428ec32a feat: add api key verification 2026-05-07 14:13:29 +02:00
LucasDLTG 752373c7b7 feat: add api key generation endpint + role authorization 2026-05-07 12:32:29 +02:00
LucasDLTG d7ddc087a6 feat: add user in db 2026-05-06 15:22:49 +02:00
LucasDLTG 7f77bfef4e fix: port
CI / Rust CI (push) Successful in 1m37s
Publish & Deploy / Build and Push to Registry (push) Successful in 13s
Publish & Deploy / Deploy via SSH (push) Successful in 17s
2026-04-28 18:49:28 +02:00
LucasDLTG 4e64aab0a5 fix: format + tracing
CI / Rust CI (push) Successful in 1m39s
Publish & Deploy / Build and Push to Registry (push) Successful in 1m37s
Publish & Deploy / Deploy via SSH (push) Successful in 18s
2026-04-28 18:32:40 +02:00
LucasDLTG 4596dfe28b feat: add tracing
CI / Rust CI (push) Successful in 4m41s
2026-04-25 19:55:58 +02:00
LucasDLTG 57a3d04625 feat: add cors
CI / Rust CI (push) Successful in 4m37s
2026-04-20 21:03:18 +02:00
LucasDLTG 81129d6c9c feat: add metadata for api doc
CI / Rust CI (push) Successful in 1m33s
Publish & Deploy / Build and Push to Registry (push) Successful in 22s
Publish & Deploy / Deploy via SSH (push) Successful in 18s
2026-04-11 20:40:38 +02:00
LucasDLTG f4da65dbc2 refactor: middlewares folder
CI / Rust CI (push) Successful in 1m37s
Publish & Deploy / Build and Push to Registry (push) Successful in 24s
Publish & Deploy / Deploy via SSH (push) Successful in 17s
2026-04-11 20:35:36 +02:00
LucasDLTG e1f46c07b4 feat: add doc endpoint 2026-04-11 20:23:11 +02:00
LucasDLTG fc391b5d0e feat: update test
CI / Rust CI (push) Successful in 4m35s
Publish & Deploy / Build and Push to Registry (push) Successful in 1m30s
Publish & Deploy / Deploy via SSH (push) Successful in 18s
2026-04-10 22:26:45 +02:00
LucasDLTG e31cdf131f feat: chat typing 2026-04-10 21:33:52 +02:00
LucasDLTG d5856557b4 feat: strong typing for complete endpoints 2026-04-10 18:27:02 +02:00
LucasDLTG 562d154480 feat: add proper typing for api 2026-04-10 17:41:25 +02:00
LucasDLTG a22560c337 feat: add auto doc via utiopia
CI / Rust CI (push) Successful in 4m29s
2026-04-10 15:45:41 +02:00
LucasDLTG b3ad5249c0 update: .env.example
CI / Rust CI (push) Successful in 1m25s
2026-04-10 14:38:06 +02:00
LucasDLTG 1a99490e22 feat: add streaming
CI / Rust CI (push) Successful in 4m29s
2026-04-10 14:15:27 +02:00
LucasDLTG 686f9ff747 feat: add unload endpoint
CI / Rust CI (push) Successful in 1m24s
2026-04-10 12:48:14 +02:00
47 changed files with 3663 additions and 552 deletions
+2
View File
@@ -1,2 +1,4 @@
JWKS_URL=https://auth.iceberg.black/realms/iceberg/protocol/openid-connect/certs JWKS_URL=https://auth.iceberg.black/realms/iceberg/protocol/openid-connect/certs
ISSUER=https://auth.iceberg.black/realms/iceberg ISSUER=https://auth.iceberg.black/realms/iceberg
OLLAMA_URL=...
CORS_ORIGIN=
+2
View File
@@ -33,6 +33,8 @@ jobs:
${{ runner.os }}-cargo- ${{ runner.os }}-cargo-
- name: Run tests - name: Run tests
env:
SQLX_OFFLINE: true
run: cargo test --workspace --verbose --all run: cargo test --workspace --verbose --all
- name: Run Clippy (linter) - name: Run Clippy (linter)
+3 -2
View File
@@ -3,7 +3,7 @@ name: Publish & Deploy
on: on:
push: push:
tags: tags:
- 'v*' - "v*"
permissions: permissions:
contents: write contents: write
@@ -34,6 +34,8 @@ jobs:
- name: Build and Push - name: Build and Push
uses: https://github.com/docker/build-push-action@v5 uses: https://github.com/docker/build-push-action@v5
env:
SQLX_OFFLINE: true
with: with:
context: . context: .
push: true push: true
@@ -86,4 +88,3 @@ jobs:
REPO_LOWER=$REPO_LOWER TAG=$TAG docker compose up -d REPO_LOWER=$REPO_LOWER TAG=$TAG docker compose up -d
docker image prune -af docker image prune -af
+11 -4
View File
@@ -12,30 +12,37 @@ repos:
- id: clippy - id: clippy
name: clippy name: clippy
entry: cargo clippy -- -D warnings entry: bash -c 'SQLX_OFFLINE=true cargo clippy -- -D warnings'
language: system language: system
types: [rust] types: [rust]
pass_filenames: false pass_filenames: false
stages: [pre-commit] stages: [pre-commit]
# ─────────────── PRE-PUSH (heavy checks) ─────────────── # ─────────────── PRE-PUSH (heavy checks) ───────────────
- id: sqlx-prepare-check
name: sqlx prepare check
entry: cargo sqlx prepare --check
language: system
pass_filenames: false
stages: [pre-push]
- id: test - id: test
name: cargo test name: cargo test
entry: cargo test --all entry: bash -c 'SQLX_OFFLINE=true cargo test --all'
language: system language: system
pass_filenames: false pass_filenames: false
stages: [pre-push] stages: [pre-push]
- id: build - id: build
name: cargo build release name: cargo build release
entry: cargo build --release entry: bash -c 'SQLX_OFFLINE=true cargo build --release'
language: system language: system
pass_filenames: false pass_filenames: false
stages: [pre-push] stages: [pre-push]
- id: audit - id: audit
name: cargo audit name: cargo audit
entry: cargo audit entry: bash -c 'cargo audit || echo "cargo audit failed (non-blocking)"'
language: system language: system
pass_filenames: false pass_filenames: false
stages: [pre-push] stages: [pre-push]
@@ -0,0 +1,17 @@
{
"db_name": "PostgreSQL",
"query": "\n INSERT INTO auth.api_key (key_hash, name, created_by, scopes)\n VALUES ($1, $2, $3, $4)\n ",
"describe": {
"columns": [],
"parameters": {
"Left": [
"Text",
"Text",
"Uuid",
"TextArray"
]
},
"nullable": []
},
"hash": "14b38b0f3fca61893dc6e036dc68f643a724b729f73f8308ae0b3a70d87dc5de"
}
@@ -0,0 +1,14 @@
{
"db_name": "PostgreSQL",
"query": "\n INSERT INTO auth.app_user (id)\n VALUES ($1)\n ON CONFLICT (id) DO NOTHING\n ",
"describe": {
"columns": [],
"parameters": {
"Left": [
"Uuid"
]
},
"nullable": []
},
"hash": "70117c16fc3efeebbcb40838d13dd427bf246f4eabd3e335077f47a7aa7882e4"
}
@@ -0,0 +1,14 @@
{
"db_name": "PostgreSQL",
"query": "\n UPDATE auth.api_key\n SET last_used_at = now()\n WHERE id = $1\n ",
"describe": {
"columns": [],
"parameters": {
"Left": [
"Uuid"
]
},
"nullable": []
},
"hash": "ab03067777ea2a8f12c3dec5cf99caf4ad700008d892a33c25408c34bebab83c"
}
@@ -0,0 +1,28 @@
{
"db_name": "PostgreSQL",
"query": "\n SELECT u.id AS user_id, ak.id as key_id\n FROM auth.api_key ak\n JOIN auth.app_user u ON u.id = ak.created_by\n WHERE ak.key_hash = $1\n AND ak.revoked_at IS NULL\n ",
"describe": {
"columns": [
{
"ordinal": 0,
"name": "user_id",
"type_info": "Uuid"
},
{
"ordinal": 1,
"name": "key_id",
"type_info": "Uuid"
}
],
"parameters": {
"Left": [
"Text"
]
},
"nullable": [
false,
false
]
},
"hash": "c0816078de6c25d441752500b9307900ff3d6c7781ebd9c0d8b47031539d8bdf"
}
Generated
+1322 -27
View File
File diff suppressed because it is too large Load Diff
+15 -3
View File
@@ -5,15 +5,27 @@ edition = "2024"
[dev-dependencies] [dev-dependencies]
wiremock = "0.6" wiremock = "0.6"
tokio = { version = "1", features = ["macros", "rt-multi-thread"] } tokio = { version = "1.52.2", features = ["macros", "rt-multi-thread"] }
[dependencies] [dependencies]
axum = "0.8.8" axum = "0.8.9"
utoipa = { version = "5.5.0", features = ["axum_extras"] }
tokio = { version = "1", features = ["full"] } tokio = { version = "1", features = ["full"] }
serde = { version = "1", features = ["derive"] } serde = { version = "1", features = ["derive"] }
serde_json = "1" serde_json = "1"
jsonwebtoken = { version = "10.3.0", features = ["aws_lc_rs"] } jsonwebtoken = { version = "10.3.0", features = ["aws_lc_rs"] }
reqwest = { version = "0.13.2", features = ["json"] } reqwest = { version = "0.13.3", features = ["json", "stream"] }
once_cell = "1" once_cell = "1"
dotenvy = "0.15" dotenvy = "0.15"
thiserror = "2.0.18" thiserror = "2.0.18"
tokio-stream = "0.1"
futures = "0.3"
chrono = { version = "0.4.44", features = ["serde"] }
uuid = { version = "1.23.1", features = ["v4", "serde"] }
tower-http = { version = "0.6.10", features = ["cors"] }
tracing = "0.1.44"
tracing-subscriber = { version = "0.3", features = ["env-filter"]}
sqlx = { version = "0.8.6", features = ["runtime-tokio-rustls", "postgres", "uuid", "chrono"] }
rand = "0.8"
base64 = "0.22.1"
sha2 = "0.11.0"
+1
View File
@@ -17,6 +17,7 @@ RUN rm -rf src
# Build actual app # Build actual app
COPY src ./src COPY src ./src
COPY .sqlx ./.sqlx
RUN cargo build --release RUN cargo build --release
# ── Runtime Stage ───────────────────────────────────────────────────────────── # ── Runtime Stage ─────────────────────────────────────────────────────────────
+1 -1
View File
@@ -7,7 +7,7 @@ services:
- .env - .env
ports: ports:
- "6066:3000" - "6066:3001"
restart: always restart: always
+269
View File
@@ -0,0 +1,269 @@
# TODO
- Race condition on jwks token refresh
- Rate Limiting
git tag -d v1.0.0; git push origin :refs/tags/v1.0.0; git tag -a v1.0.0 -m "Release v1.0.0"; git push origin v1.0.0
curl https://chat.iceberg.black/api/v1/models -H "Authorization: Bearer eyJhbGciOiJSUzI1NiIsInR5cCIgOiAiSldUIiwia2lkIiA6ICJublpLek04TkZHVmpWbGFPRXZpMUtFSTVHQWRwaGlsYjh3RHRLeG5JOENZIn0.eyJleHAiOjE3NzU4MTUyODksImlhdCI6MTc3NTgxNDk4OSwianRpIjoiMTgwYzA2NDUtYzZiMC00MDRmLTgyYjEtMzA3YmY4ZmJlNmJlIiwiaXNzIjoiaHR0cHM6Ly9hdXRoLmljZWJlcmcuYmxhY2svcmVhbG1zL2ljZWJlcmciLCJzdWIiOiJmZGRiN2FjZC1kMmE5LTRmMTctOWIxNi1kZjVlN2EzNDI4YjciLCJ0eXAiOiJCZWFyZXIiLCJhenAiOiJjaGF0LWFwaSIsInNjb3BlIjoiIiwiY2xpZW50SG9zdCI6Ijg2LjIxMi44NC4xOTEiLCJjbGllbnRBZGRyZXNzIjoiODYuMjEyLjg0LjE5MSIsImNsaWVudF9pZCI6ImNoYXQtYXBpIn0.abjHABcjCJNiB6vriRw60nfzabEfD7CXwyRhkahFkC8ATgfy4fn0T8PnFfsaRpsXuhamIhWwNskrA7L9V3mbWWKW-JEOvImDhTc8sIX0E5fDTbk8O5wa_2yzNdLpRxdSjLqgL544rB8I-LZ8bl5SxtdN3gHfrnWr5ef8bbLgPRzZIylT3QUpah0uywDM_cfrrve9SMHHOUUItyzmOHLw0Igit1EzyFNyjbWf6OAU6TMjOF_eFTc5sakyBwsdJnGy0nhj5R-wxLpr1ug3iEd3Y-jDzHO4m6jariXVJ7Vvbz71i7sadDEKdSyOKcCeQbU4T0Tf7IxqW4c2DNnk3BFFuw"
curl -X POST "https://auth.iceberg.black/realms/iceberg/protocol/openid-connect/token" -H "Content-Type: application/x-www-form-urlencoded" -d "grant_type=client_credentials" -d "client_id=chat-api" -d "client_secret=5fHUp8Z5GoNM70MVOGuQTfKFkaAE48Za"
curl -s -X POST https://chat.iceberg.black/api/v1/chat/completions \
-H "Content-Type: application/json" \
-H "Authorization: Bearer eyJhbGciOiJSUzI1NiIsInR5cCIgOiAiSldUIiwia2lkIiA6ICJublpLek04TkZHVmpWbGFPRXZpMUtFSTVHQWRwaGlsYjh3RHRLeG5JOENZIn0.eyJleHAiOjE3NzU4MTU0NTEsImlhdCI6MTc3NTgxNTE1MSwianRpIjoiOGNjMmM5MjUtZmMzNy00MDFhLWE1ZjUtODFlOGJhNjcxNzcwIiwiaXNzIjoiaHR0cHM6Ly9hdXRoLmljZWJlcmcuYmxhY2svcmVhbG1zL2ljZWJlcmciLCJzdWIiOiJmZGRiN2FjZC1kMmE5LTRmMTctOWIxNi1kZjVlN2EzNDI4YjciLCJ0eXAiOiJCZWFyZXIiLCJhenAiOiJjaGF0LWFwaSIsInNjb3BlIjoiIiwiY2xpZW50SG9zdCI6Ijg2LjIxMi44NC4xOTEiLCJjbGllbnRBZGRyZXNzIjoiODYuMjEyLjg0LjE5MSIsImNsaWVudF9pZCI6ImNoYXQtYXBpIn0.Exv34aoAqwzSqQziZH3zScAnK_xiNCXqpOQl54x7NFMdO0gVgacJgMxoW82Ym8aCNfBRMFtWclZJ9RFh1b9uSEUgsdVtqvMX-2kCAhiMSjmIBPU-L0gr5N63c3bScOoAYa37baR4mXQ5LxHjfkLo_rHDJ74adg2JMo359Wbuu1_OR708_q8yjgO_4fQeYbIffxADwRDvPSIwQ8Y2qRjaAIOZs0xmA9p128CUxlUyvhxwfFquYKRaDs5QE9pIAtvm_KuoBydYopm8j8cbm4ixDhUwYjkzniIIapY77NxpmMosfh8BhK0W3ieK2gTKfmiFDhYgkGybU6b1BnJMy1_Kxg" \
-d '{
"model": "llama3:latest",
"messages": [
{"role": "user", "content": "What is Rust?"}
]
}' | jq .
# 🦙 Ollama Rust API Wrapper
A high-performance Rust API wrapper around Ollama, providing an OpenAI-compatible interface, model lifecycle management, and advanced runtime features.
---
# 🚀 Features
- ✅ OpenAI-compatible API (`/v1/...`)
- ⚡ Streaming (Server-Sent Events)
- 🧠 Model lifecycle management (load/unload)
- 🔐 API key authentication (optional)
- 📊 Usage tracking & observability
- 🔀 Model routing & abstraction
- 🧩 Extensible architecture (multi-provider ready)
---
# 📡 API Endpoints
## 1. Core LLM API (OpenAI-compatible)
- [x] `POST /v1/chat/completions` + streaming
- [x] `POST /v1/completions` + streaming
- [ ] `POST /v1/embeddings`
- [x] `GET /v1/models`
## 2. Model Lifecycle Management
- [x] `POST /v1/models/{model}/load`
- [x] `POST /v1/models/{model}/unload`
## 3. Model Management
- [ ] `POST /v1/models/pull`
- [ ] `DELETE /v1/models/{model}`
## 4. Runtime & Observability
### Model Status
```
GET /v1/models/{model}/status
```
### List Loaded Models
```
GET /v1/runtime/models
```
---
## 6. Health Checks
```
GET /health
GET /ready
```
---
# 🧠 Internal Mapping (Ollama)
| Wrapper Endpoint | Ollama Endpoint |
| ------------------------- | --------------- |
| /v1/chat/completions | /api/chat |
| /v1/completions | /api/generate |
| /v1/embeddings | /api/embeddings |
| /v1/models | /api/tags |
| /v1/models/pull | /api/pull |
| DELETE /v1/models/{model} | /api/delete |
| load/unload | /api/generate |
---
# 🔧 Advanced Features
## 🔀 Model Routing
Use abstract model names:
```json
{
"model": "fast"
}
```
Example mapping:
```
fast → llama3:8b
smart → llama3:70b
code → deepseek-coder
```
---
## 📊 Usage Tracking
```
GET /v1/usage
```
Tracks:
- request count
- latency
- per-model usage
---
## 🚦 Rate Limiting
- Requests per minute
- Tokens per minute
Returns:
```
429 Too Many Requests
```
---
## 🧠 Sessions (Context Management)
```
POST /v1/sessions
POST /v1/sessions/{id}/chat
```
Stores conversation history server-side.
---
## ⚡ Caching
- Embeddings
- Deterministic prompts (temperature = 0)
---
## 🧩 Tool / Function Calling
Supports structured tool execution:
```json
{
"tools": [
{
"name": "function_name",
"parameters": {}
}
]
}
```
---
## 📦 Batch Requests
```
POST /v1/batch
```
---
## 🧠 Auto Eviction
```
POST /v1/runtime/evict
```
Strategies:
- LRU
- memory threshold
---
## 🧾 Logs
```
GET /v1/logs
```
## 🔔 Async Jobs / Webhooks
```
POST /v1/jobs
```
---
# 🏗️ Architecture
```
Client → Rust API → Ollama → Response
```
### Layers:
- HTTP (Axum)
- Service layer (business logic)
- Provider abstraction
- Ollama client
---
# 🔌 Provider Abstraction (Future-Proof)
```rust
trait LlmProvider {
async fn chat(...);
async fn embeddings(...);
}
```
Supports:
- Ollama (current)
- OpenAI (future)
- Others
---
# 🎯 Roadmap
- [ ] Full OpenAI compatibility
- [ ] Multi-node routing
- [ ] GPU-aware scheduling
- [ ] Web UI dashboard
- [ ] Distributed inference
---
# 🧠 Summary
This project turns Ollama into:
👉 A local OpenAI-compatible API
👉 A controllable model runtime
👉 A foundation for a full LLM gateway
# TODO
- open api doc for bearer token
- db
-59
View File
@@ -1,59 +0,0 @@
use jsonwebtoken::{Algorithm, DecodingKey, Validation, decode, decode_header};
use once_cell::sync::Lazy;
use serde::{Deserialize, Serialize};
use serde_json::Value;
use std::env;
#[derive(Debug, Deserialize, Serialize, Clone)]
pub struct Claims {
pub sub: String,
pub preferred_username: Option<String>,
pub exp: usize,
pub iss: String,
pub aud: Option<String>,
pub realm_access: Option<RealmAccess>,
}
#[derive(Debug, Deserialize, Serialize, Clone)]
pub struct RealmAccess {
pub roles: Vec<String>,
}
static ISSUER: Lazy<String> = Lazy::new(|| env::var("ISSUER").expect("ISSUER not set"));
pub fn validate_token(token: &str, jwks: &Value) -> Result<Claims, String> {
// 1. Decode header
let header = decode_header(token).map_err(|_| "Invalid header")?;
let kid = header.kid.ok_or("Missing kid")?;
// 2. Find matching key
let keys = jwks["keys"].as_array().ok_or("Invalid JWKS")?;
let key = keys
.iter()
.find(|k| k["kid"] == kid)
.ok_or("Matching key not found")?;
// 3. Extract RSA components
let n = key["n"].as_str().ok_or("Missing n")?;
let e = key["e"].as_str().ok_or("Missing e")?;
let decoding_key =
DecodingKey::from_rsa_components(n, e).map_err(|_| "Invalid decoding key")?;
// 4. Setup validation rules
let mut validation = Validation::new(Algorithm::RS256);
validation.set_issuer(&[ISSUER.as_str()]);
// Optional but recommended:
validation.validate_exp = true;
validation.validate_aud = false; // depends on your Keycloak config
// 5. Decode & verify
let token_data = decode::<Claims>(token, &decoding_key, &validation)
.map_err(|_| "Token validation failed")?;
Ok(token_data.claims)
}
-57
View File
@@ -1,57 +0,0 @@
use once_cell::sync::Lazy;
use serde_json::Value;
use std::env;
use std::sync::Arc;
use std::time::{Duration, Instant};
use tokio::sync::RwLock;
#[derive(Clone)]
struct JwksCache {
jwks: Value,
last_fetched: Instant,
}
static JWK_CACHE: Lazy<Arc<RwLock<Option<JwksCache>>>> = Lazy::new(|| Arc::new(RwLock::new(None)));
static JWKS_URL: Lazy<String> = Lazy::new(|| env::var("JWKS_URL").expect("JWKS_URL not set"));
async fn fetch_jwks() -> Result<Value, reqwest::Error> {
let jwks = reqwest::get(JWKS_URL.as_str())
.await?
.json::<Value>()
.await?;
Ok(jwks)
}
pub async fn refresh_jwks() -> Result<Value, reqwest::Error> {
let jwks = fetch_jwks().await?;
let mut write = JWK_CACHE.write().await;
*write = Some(JwksCache {
jwks: jwks.clone(),
last_fetched: Instant::now(),
});
Ok(jwks)
}
pub async fn get_jwks() -> Result<Value, reqwest::Error> {
let ttl = Duration::from_secs(3600); // 1 hour
{
// Read lock first (fast path)
let read = JWK_CACHE.read().await;
if let Some(cache) = read
.as_ref()
.filter(|cache| cache.last_fetched.elapsed() < ttl)
{
return Ok(cache.jwks.clone());
}
}
// Expired or empty → refresh
refresh_jwks().await
}
-51
View File
@@ -1,51 +0,0 @@
use axum::{extract::Request, http::StatusCode, middleware::Next, response::Response};
use crate::auth::{
jwt::validate_token,
keycloak::{get_jwks, refresh_jwks},
};
pub async fn auth_middleware(mut request: Request, next: Next) -> Result<Response, StatusCode> {
#[cfg(debug_assertions)]
println!("Middleware hit");
let headers = request.headers();
#[cfg(debug_assertions)]
println!("Headers extracted");
let auth_header = headers.get("authorization").and_then(|v| v.to_str().ok());
#[cfg(debug_assertions)]
println!("Auth header: {:?}", auth_header);
let auth_header = auth_header.ok_or(StatusCode::UNAUTHORIZED)?;
let token = auth_header
.strip_prefix("Bearer ")
.ok_or(StatusCode::UNAUTHORIZED)?;
let jwks = get_jwks().await.map_err(|_| StatusCode::UNAUTHORIZED)?;
match validate_token(token, &jwks) {
Ok(claims) => {
#[cfg(debug_assertions)]
println!("Token valid");
request.extensions_mut().insert(claims);
Ok(next.run(request).await)
}
Err(_) => {
let jwks = refresh_jwks().await.map_err(|_| StatusCode::UNAUTHORIZED)?;
match validate_token(token, &jwks) {
Ok(claims) => {
request.extensions_mut().insert(claims);
Ok(next.run(request).await)
}
Err(_) => Err(StatusCode::UNAUTHORIZED),
}
}
}
}
-3
View File
@@ -1,3 +0,0 @@
pub mod jwt;
pub mod keycloak;
pub mod middleware;
+1
View File
@@ -0,0 +1 @@
pub mod postgres;
+17
View File
@@ -0,0 +1,17 @@
use sqlx::PgPool;
use uuid::Uuid;
pub async fn update_last_access(pool: &PgPool, api_key_id: Uuid) -> Result<(), sqlx::Error> {
sqlx::query!(
r#"
UPDATE auth.api_key
SET last_used_at = now()
WHERE id = $1
"#,
api_key_id
)
.execute(pool)
.await?;
Ok(())
}
+3
View File
@@ -0,0 +1,3 @@
pub mod api_key;
pub mod pool;
pub mod user_repository;
+14
View File
@@ -0,0 +1,14 @@
use sqlx::{PgPool, postgres::PgPoolOptions};
use std::time::Duration;
pub async fn create_pool(database_url: &str) -> Result<PgPool, sqlx::Error> {
PgPoolOptions::new()
.max_connections(10)
.acquire_timeout(Duration::from_secs(5))
.connect(database_url)
.await
.map_err(|err| {
tracing::error!("Postgres connection error: {:?}", err);
err
})
}
+17
View File
@@ -0,0 +1,17 @@
use sqlx::PgPool;
use uuid::Uuid;
pub async fn ensure_user_exists(pool: &PgPool, user_id: Uuid) -> Result<(), sqlx::Error> {
sqlx::query!(
r#"
INSERT INTO auth.app_user (id)
VALUES ($1)
ON CONFLICT (id) DO NOTHING
"#,
user_id
)
.execute(pool)
.await?;
Ok(())
}
+55
View File
@@ -0,0 +1,55 @@
use utoipa::OpenApi;
use crate::dto::api;
use crate::routes;
#[derive(OpenApi)]
#[openapi(
info(
title = "Ollama Proxy",
description = "OpenAI-compatible proxy for local Ollama models",
version = "0.1.0",
license(
name = "MIT",
url = "https://opensource.org/licenses/MIT"
),
),
paths(
routes::v1::chat::completions,
routes::v1::chat::chat_completions,
routes::v1::models::list_models,
routes::v1::models::load_model,
routes::v1::models::unload_model,
),
components(
schemas(
api::ErrorResponse,
api::ModelsResponse,
api::ModelInfo,
api::LoadModelResponse,
api::LoadModelBody,
api::UnloadModelResponse,
api::BaseLLMRequest,
api::CompletionRequest,
api::CompletionObject,
api::FinishReason,
api::CompletionResponse,
api::Choice,
api::Usage,
api::CompletionChunk,
api::ChatRequest,
api::Message,
api::Role,
api::ChatCompletionResponse,
api::ChatChoice,
api::ChatCompletionChunk,
api::ChatChunkChoice,
api::ChatDelta,
)
),
tags(
(name = "chat", description = "Chat & completions"),
(name = "models", description = "Model management")
)
)]
pub struct ApiDoc;
+191
View File
@@ -0,0 +1,191 @@
use serde::{Deserialize, Serialize};
use utoipa::ToSchema;
#[derive(Debug, Serialize, ToSchema)]
pub struct ErrorResponse {
pub error: String,
}
impl ErrorResponse {
pub fn new(msg: impl Into<String>) -> Self {
Self { error: msg.into() }
}
}
#[derive(Debug, Serialize, Deserialize, ToSchema)]
pub struct ModelsResponse {
pub models: Vec<ModelInfo>,
}
#[derive(Debug, Serialize, Deserialize, ToSchema)]
pub struct ModelInfo {
pub name: String,
pub family: Option<String>,
pub parameter_size: Option<String>,
pub quantization: Option<String>,
}
#[derive(Debug, Serialize, Deserialize, ToSchema)]
pub struct LoadModelResponse {
pub model: String,
pub status: String,
pub keep_alive: String,
}
#[derive(Debug, Serialize, Deserialize, ToSchema)]
pub struct LoadModelBody {
pub keep_alive: Option<String>,
}
#[derive(Debug, Serialize, Deserialize, ToSchema)]
pub struct UnloadModelResponse {
pub model: String,
pub status: String,
}
#[derive(Debug, Deserialize, Serialize, ToSchema, Default)]
pub struct BaseLLMRequest {
pub model: String,
#[serde(default)]
pub stream: bool,
pub temperature: Option<f32>,
pub top_p: Option<f32>,
// Ollama-native
pub top_k: Option<u32>,
pub repeat_penalty: Option<f32>,
pub seed: Option<i64>,
pub num_ctx: Option<u32>,
pub num_predict: Option<u32>,
pub stop: Option<Vec<String>>,
pub keep_alive: Option<String>,
}
#[derive(Debug, Deserialize, Serialize, ToSchema)]
pub struct CompletionRequest {
#[serde(flatten)]
pub base: BaseLLMRequest,
pub prompt: String,
}
#[derive(Debug, Serialize, Deserialize, ToSchema, PartialEq, Eq)]
#[serde(rename_all = "snake_case")]
pub enum CompletionObject {
TextCompletion,
}
#[derive(Debug, Serialize, Deserialize, ToSchema, PartialEq, Eq)]
#[serde(rename_all = "snake_case")]
pub enum FinishReason {
Stop,
Length,
ContentFilter,
ToolCalls,
}
#[derive(Debug, Serialize, Deserialize, ToSchema)]
pub struct CompletionResponse {
pub id: String,
pub object: CompletionObject,
pub created: u64,
pub model: String,
pub choices: Vec<Choice>,
pub usage: Usage,
}
#[derive(Debug, Serialize, Deserialize, ToSchema)]
pub struct Choice {
pub text: String,
pub index: u32,
pub finish_reason: FinishReason,
}
#[derive(Debug, Serialize, Deserialize, ToSchema)]
pub struct Usage {
pub prompt_tokens: u32,
pub completion_tokens: u32,
pub total_tokens: u32,
}
#[derive(Debug, Serialize, Deserialize, ToSchema)]
pub struct CompletionChunk {
pub id: String,
pub object: String,
pub choices: Vec<Choice>,
}
#[derive(Debug, Deserialize, Serialize, ToSchema)]
pub struct ChatRequest {
#[serde(flatten)]
pub base: BaseLLMRequest,
pub messages: Vec<Message>,
}
#[derive(Debug, Deserialize, Serialize, ToSchema)]
pub struct Message {
pub role: Role,
pub content: String,
}
#[derive(Debug, Deserialize, Serialize, ToSchema, PartialEq, Eq)]
#[serde(rename_all = "lowercase")]
pub enum Role {
System,
User,
Assistant,
}
#[derive(Debug, Serialize, Deserialize, ToSchema)]
pub struct ChatCompletionResponse {
pub id: String,
pub object: String,
pub created: u64,
pub model: String,
pub choices: Vec<ChatChoice>,
pub usage: Option<Usage>, // optional (Ollama may not always provide)
}
#[derive(Debug, Serialize, Deserialize, ToSchema)]
pub struct ChatChoice {
pub index: u32,
pub message: Message,
pub finish_reason: FinishReason,
}
#[derive(Debug, Serialize, ToSchema)]
pub struct ChatCompletionChunk {
pub id: String,
pub object: String,
pub choices: Vec<ChatChunkChoice>,
}
#[derive(Debug, Serialize, ToSchema)]
pub struct ChatChunkChoice {
pub index: u32,
pub delta: ChatDelta,
pub finish_reason: Option<FinishReason>,
}
#[derive(Debug, Serialize, ToSchema)]
pub struct ChatDelta {
pub role: Option<Role>,
pub content: Option<String>,
}
#[derive(serde::Deserialize)]
pub struct CreateApiKeyRequest {
pub name: String,
pub scopes: Vec<String>,
}
#[derive(serde::Serialize)]
pub struct CreateApiKeyResponse {
pub api_key: String, // ONLY returned once
}
+2
View File
@@ -0,0 +1,2 @@
pub mod api;
pub mod ollama;
+80
View File
@@ -0,0 +1,80 @@
use serde::{Deserialize, Serialize};
use crate::dto::api;
#[derive(Debug, Serialize, Deserialize)]
pub struct OllamaModels {
pub models: Vec<OllamaModel>,
}
#[derive(Debug, Serialize, Deserialize)]
pub struct OllamaModel {
pub name: String,
pub details: Option<OllamaModelDetails>,
pub size: Option<u64>,
pub digest: Option<String>,
pub modified_at: Option<String>,
}
#[derive(Debug, Serialize, Deserialize)]
pub struct OllamaModelDetails {
pub family: Option<String>,
pub parameter_size: Option<String>,
pub quantization_level: Option<String>,
}
#[derive(Debug, Serialize, Deserialize)]
pub struct OllamaOptions {
pub temperature: Option<f32>,
pub top_p: Option<f32>,
pub top_k: Option<u32>,
pub repeat_penalty: Option<f32>,
pub seed: Option<i64>,
pub num_ctx: Option<u32>,
pub num_predict: Option<u32>,
}
#[derive(Debug, Serialize)]
pub struct OllamaGenerateRequest<'a> {
pub model: &'a str,
pub prompt: &'a str,
pub stream: bool,
pub options: OllamaOptions,
}
#[derive(Debug, Serialize, Deserialize)]
pub struct OllamaGenerateResponse {
pub model: String,
pub created_at: Option<String>,
pub response: String,
pub done: bool,
#[serde(default)]
pub context: Option<Vec<u64>>,
pub total_duration: Option<u64>,
pub load_duration: Option<u64>,
pub prompt_eval_count: Option<u32>,
pub eval_count: Option<u32>,
}
#[derive(Debug, Serialize)]
pub struct OllamaChatRequest<'a> {
pub model: &'a str,
pub messages: &'a [api::Message],
pub stream: bool,
pub options: OllamaOptions,
}
#[derive(Debug, Deserialize)]
pub struct OllamaChatResponse {
pub model: String,
pub message: api::Message,
pub done: bool,
pub prompt_eval_count: Option<u32>,
pub eval_count: Option<u32>,
}
+6
View File
@@ -6,12 +6,18 @@ pub enum OllamaError {
#[error("prompt is required and cannot be empty")] #[error("prompt is required and cannot be empty")]
MissingPrompt, MissingPrompt,
#[error("model is required and cannot be empty")]
MissingModel,
#[error("messages must be a non-empty array containing at least one user message")] #[error("messages must be a non-empty array containing at least one user message")]
MissingMessages, MissingMessages,
#[error("model '{0}' is not available — run `ollama pull {0}` first")] #[error("model '{0}' is not available — run `ollama pull {0}` first")]
ModelNotFound(String), ModelNotFound(String),
#[error("keep_alive is required and cannot be empty")]
MissingKeepAlive,
#[error( #[error(
"invalid keep_alive format '{0}' — expected <number><unit> (e.g. 30s, 10m, 2h), a plain integer (seconds), or -1" "invalid keep_alive format '{0}' — expected <number><unit> (e.g. 30s, 10m, 2h), a plain integer (seconds), or -1"
)] )]
+5
View File
@@ -1,2 +1,7 @@
pub mod databases;
pub mod dto;
pub mod errors; pub mod errors;
pub mod middlewares;
pub mod providers; pub mod providers;
pub mod state;
pub mod utils;
+50 -5
View File
@@ -1,20 +1,34 @@
mod auth; mod databases;
mod docs;
mod dto;
mod errors; mod errors;
mod middlewares;
mod providers; mod providers;
mod routes; mod routes;
mod state; mod state;
mod utils;
use crate::providers::ollama::OllamaProvider; use crate::databases::postgres;
use crate::providers::ollama::client::OllamaProvider;
use crate::state::app_state::AppState; use crate::state::app_state::AppState;
use axum::Router; use axum::Router;
use axum::http::{HeaderName, HeaderValue, Method, header};
use once_cell::sync::Lazy; use once_cell::sync::Lazy;
use std::env; use std::env;
use std::net::SocketAddr; use std::net::SocketAddr;
use std::sync::Arc; use std::sync::Arc;
use tower_http::cors::CorsLayer;
use tracing_subscriber::{EnvFilter, fmt};
static OLLAMA_URL: Lazy<String> = Lazy::new(|| env::var("OLLAMA_URL").expect("OLLAMA_URL not set")); static OLLAMA_URL: Lazy<String> = Lazy::new(|| env::var("OLLAMA_URL").expect("OLLAMA_URL not set"));
pub fn init_tracing() {
let filter = env::var("RUST_LOG").unwrap_or_else(|_| "info".to_string());
fmt().with_env_filter(EnvFilter::new(filter)).init();
}
#[tokio::main] #[tokio::main]
async fn main() { async fn main() {
#[cfg(debug_assertions)] #[cfg(debug_assertions)]
@@ -22,16 +36,47 @@ async fn main() {
dotenvy::dotenv().ok(); dotenvy::dotenv().ok();
} }
init_tracing();
// DB Connection
let database_url = env::var("DATABASE_URL").expect("DATABASE_URL must be set");
let pool = postgres::pool::create_pool(&database_url)
.await
.expect("Fatal error");
let state = AppState { let state = AppState {
ollama: Arc::new(OllamaProvider::new(OLLAMA_URL.as_str())), ollama: Arc::new(OllamaProvider::new(OLLAMA_URL.as_str())),
postgres: pool,
}; };
let cors_origin =
env::var("CORS_ORIGIN").unwrap_or_else(|_| "http://localhost:3000".to_string());
let cors = CorsLayer::new()
.allow_origin(cors_origin.parse::<HeaderValue>().unwrap())
.allow_methods([
Method::GET,
Method::POST,
Method::PUT,
Method::DELETE,
Method::OPTIONS,
])
.allow_headers([
header::CONTENT_TYPE,
header::AUTHORIZATION,
header::ACCEPT,
HeaderName::from_static("x-api-key"),
])
.allow_credentials(true);
let app = Router::new() let app = Router::new()
.nest("/v1", routes::v1::router()) .nest("/v1", routes::v1::router(state.clone()))
.layer(cors)
.with_state(state); .with_state(state);
let addr = SocketAddr::from(([0, 0, 0, 0], 3000)); let addr = SocketAddr::from(([0, 0, 0, 0], 3001));
println!("Server running on {}", addr); tracing::debug!("Server running on {}", addr);
axum::serve(tokio::net::TcpListener::bind(addr).await.unwrap(), app) axum::serve(tokio::net::TcpListener::bind(addr).await.unwrap(), app)
.await .await
+7
View File
@@ -0,0 +1,7 @@
use uuid::Uuid;
#[derive(Clone, Debug)]
pub struct ApiKeyClaims {
pub sub: Uuid,
pub api_key_id: Uuid,
}
+148
View File
@@ -0,0 +1,148 @@
use jsonwebtoken::{Algorithm, DecodingKey, Validation, decode, decode_header};
use once_cell::sync::Lazy;
use serde::{Deserialize, Serialize};
use serde_json::Value;
use std::collections::HashMap;
use std::env;
use std::sync::Arc;
use std::time::{Duration, Instant};
use tokio::sync::RwLock;
// ------ JWKS ------
#[derive(Clone)]
struct JwksCache {
jwks: Value,
last_fetched: Instant,
}
static JWK_CACHE: Lazy<Arc<RwLock<Option<JwksCache>>>> = Lazy::new(|| Arc::new(RwLock::new(None)));
static JWKS_URL: Lazy<String> = Lazy::new(|| env::var("JWKS_URL").expect("JWKS_URL not set"));
async fn fetch_jwks() -> Result<Value, reqwest::Error> {
let jwks = reqwest::get(JWKS_URL.as_str())
.await?
.json::<Value>()
.await?;
Ok(jwks)
}
pub async fn refresh_jwks() -> Result<Value, reqwest::Error> {
let jwks = fetch_jwks().await?;
let mut write = JWK_CACHE.write().await;
*write = Some(JwksCache {
jwks: jwks.clone(),
last_fetched: Instant::now(),
});
Ok(jwks)
}
pub async fn get_jwks() -> Result<Value, reqwest::Error> {
let ttl = Duration::from_secs(3600); // 1 hour
{
// Read lock first (fast path)
let read = JWK_CACHE.read().await;
if let Some(cache) = read
.as_ref()
.filter(|cache| cache.last_fetched.elapsed() < ttl)
{
return Ok(cache.jwks.clone());
}
}
// Expired or empty → refresh
refresh_jwks().await
}
// ------ Claims ------
#[derive(Debug, Deserialize, Serialize, Clone, Default)]
pub struct KeycloakClaims {
pub sub: String,
pub preferred_username: Option<String>,
pub exp: usize,
pub iss: String,
pub aud: Option<Vec<String>>,
pub realm_access: Option<RealmAccess>,
#[serde(default)]
pub resource_access: HashMap<String, ResourceAccess>,
}
#[derive(Debug, Deserialize, Serialize, Clone)]
pub struct RealmAccess {
pub roles: Vec<String>,
}
#[derive(Debug, Deserialize, Serialize, Clone)]
pub struct ResourceAccess {
pub roles: Vec<String>,
}
impl KeycloakClaims {
pub fn realm_roles(&self) -> &[String] {
self.realm_access
.as_ref()
.map_or(&[], |r| r.roles.as_slice())
}
pub fn has_realm_role(&self, role: &str) -> bool {
self.realm_roles().iter().any(|r| r == role)
}
pub fn client_roles(&self, client: &str) -> &[String] {
self.resource_access
.get(client)
.map_or(&[], |r| r.roles.as_slice())
}
pub fn has_client_role(&self, client: &str, role: &str) -> bool {
self.client_roles(client).iter().any(|r| r == role)
}
}
// ------ Validation ------
static ISSUER: Lazy<String> = Lazy::new(|| env::var("ISSUER").expect("ISSUER not set"));
pub fn validate_token(token: &str, jwks: &Value) -> Result<KeycloakClaims, String> {
// 1. Decode header
let header = decode_header(token).map_err(|_| "Invalid header")?;
let kid = header.kid.ok_or("Missing kid")?;
// 2. Find matching key
let keys = jwks["keys"].as_array().ok_or("Invalid JWKS")?;
let key = keys
.iter()
.find(|k| k["kid"] == kid)
.ok_or("Matching key not found")?;
// 3. Extract RSA components
let n = key["n"].as_str().ok_or("Missing n")?;
let e = key["e"].as_str().ok_or("Missing e")?;
let decoding_key =
DecodingKey::from_rsa_components(n, e).map_err(|_| "Invalid decoding key")?;
// 4. Setup validation rules
let mut validation = Validation::new(Algorithm::RS256);
validation.set_issuer(&[ISSUER.as_str()]);
validation.validate_exp = true;
validation.validate_aud = false;
// 5. Decode & verify
let token_data = decode::<KeycloakClaims>(token, &decoding_key, &validation)
.map_err(|_| "Token validation failed")?;
Ok(token_data.claims)
}
+215
View File
@@ -0,0 +1,215 @@
use axum::{
extract::{Request, State},
http::StatusCode,
middleware::Next,
response::{IntoResponse, Response},
};
use crate::databases::postgres::{
api_key::update_last_access, user_repository::ensure_user_exists,
};
use crate::middlewares::auth::apikey::ApiKeyClaims;
use crate::middlewares::auth::keycloak::{KeycloakClaims, get_jwks, refresh_jwks, validate_token};
use crate::state::app_state::AppState;
use crate::utils::crypto::hash_key;
use uuid::Uuid;
#[derive(Clone, Debug)]
pub enum Auth {
Jwt(KeycloakClaims),
ApiKey(ApiKeyClaims),
}
impl Auth {
pub fn user_id(&self) -> Uuid {
match self {
Auth::Jwt(c) => c.sub.parse().expect("sub is a valid UUID"),
Auth::ApiKey(c) => c.sub,
}
}
pub fn has_realm_role(&self, role: &str) -> bool {
match self {
Auth::Jwt(c) => c.has_realm_role(role),
Auth::ApiKey(_) => false, // API keys carry no roles
}
}
pub fn has_client_role(&self, client: &str, role: &str) -> bool {
match self {
Auth::Jwt(c) => c.has_client_role(client, role),
Auth::ApiKey(_) => false,
}
}
}
pub async fn auth_middleware(
State(state): State<AppState>,
request: Request,
next: Next,
) -> Result<Response, StatusCode> {
match try_jwt(&state, request, next).await {
Ok(response) => Ok(response),
Err((request, next)) => try_api_key(&state, request, next).await,
}
}
/// Returns Ok(Response) if JWT was valid and request handled.
/// Returns Err((request, next)) if no JWT was present (caller should try next method).
/// Returns a 401/500 response directly if JWT was present but invalid.
async fn try_jwt(
state: &AppState,
request: Request,
next: Next,
) -> Result<Response, (Request, Next)> {
let token = request
.headers()
.get("authorization")
.and_then(|v| v.to_str().ok())
.and_then(|v| v.strip_prefix("Bearer "))
.map(str::to_owned);
let Some(token) = token else {
// No Authorization header at all → let API key branch try
return Err((request, next));
};
let jwks = match get_jwks().await {
Ok(j) => j,
Err(e) => {
tracing::error!("Failed to fetch JWKS: {e}");
// Token was present but we can't validate → hard 500
return Ok(StatusCode::INTERNAL_SERVER_ERROR.into_response());
}
};
let claims = match validate_token(&token, &jwks) {
Ok(c) => c,
Err(_) => {
// Try refreshing JWKS once
match refresh_jwks().await {
Ok(fresh_jwks) => match validate_token(&token, &fresh_jwks) {
Ok(c) => c,
Err(_) => {
tracing::warn!("JWT validation failed after JWKS refresh");
return Ok(StatusCode::UNAUTHORIZED.into_response());
}
},
Err(e) => {
tracing::error!("Failed to refresh JWKS: {e}");
return Ok(StatusCode::INTERNAL_SERVER_ERROR.into_response());
}
}
}
};
tracing::debug!("JWT valid, sub={}", claims.sub);
handle_auth(state, request, next, Auth::Jwt(claims))
.await
.map_err(|_| unreachable!())
}
/// Returns Ok(Response) if API key was valid.
/// Returns Err(StatusCode) otherwise (UNAUTHORIZED or INTERNAL_SERVER_ERROR).
async fn try_api_key(
state: &AppState,
request: Request,
next: Next,
) -> Result<Response, StatusCode> {
let key = request
.headers()
.get("x-api-key")
.and_then(|v| v.to_str().ok())
.ok_or(StatusCode::UNAUTHORIZED)?
.to_owned();
let key_hash = hash_key(&key);
let row = sqlx::query!(
r#"
SELECT u.id AS user_id, ak.id as key_id
FROM auth.api_key ak
JOIN auth.app_user u ON u.id = ak.created_by
WHERE ak.key_hash = $1
AND ak.revoked_at IS NULL
"#,
key_hash
)
.fetch_optional(&state.postgres)
.await
.map_err(|e| {
tracing::error!("DB error during API key lookup: {e}");
StatusCode::INTERNAL_SERVER_ERROR
})?
.ok_or(StatusCode::UNAUTHORIZED)?;
tracing::debug!("API key valid, user_id={}", row.user_id);
handle_auth(
state,
request,
next,
Auth::ApiKey(ApiKeyClaims {
sub: row.user_id,
api_key_id: row.key_id,
}),
)
.await
}
// ── Shared post-auth logic ────────────────────────────────────────────────────
/// Ensures the user exists in the DB, inserts `Auth` into extensions, runs the next handler.
async fn handle_auth(
state: &AppState,
mut request: Request,
next: Next,
auth: Auth,
) -> Result<Response, StatusCode> {
match &auth {
Auth::Jwt(_) => {
ensure_user_exists(&state.postgres, auth.user_id())
.await
.map_err(|e| {
tracing::error!("ensure_user_exists failed: {e}");
StatusCode::INTERNAL_SERVER_ERROR
})?;
}
Auth::ApiKey(key) => {
update_last_access(&state.postgres, key.api_key_id)
.await
.map_err(|e| {
tracing::error!("update_last_access failed: {e}");
StatusCode::INTERNAL_SERVER_ERROR
})?;
}
}
request.extensions_mut().insert(auth);
Ok(next.run(request).await)
}
// ── Role guard ───────────────────────────────────────────────────────────────
/// Layer-level middleware that checks roles *after* `auth_middleware` has run.
pub async fn require_roles(
request: Request,
next: Next,
realm_role: Option<&'static str>,
client_role: Option<&'static str>,
) -> Result<Response, StatusCode> {
let auth = request
.extensions()
.get::<Auth>()
.ok_or(StatusCode::UNAUTHORIZED)?;
if realm_role.is_some_and(|role| !auth.has_realm_role(role)) {
return Err(StatusCode::FORBIDDEN);
}
if client_role.is_some_and(|role| !auth.has_client_role("chat-api", role)) {
return Err(StatusCode::FORBIDDEN);
}
Ok(next.run(request).await)
}
+5
View File
@@ -0,0 +1,5 @@
pub mod apikey;
pub mod keycloak;
pub mod middleware;
pub use middleware::auth_middleware;
+1
View File
@@ -0,0 +1 @@
pub mod auth;
-223
View File
@@ -1,223 +0,0 @@
use crate::errors::OllamaError;
use reqwest::Client;
use serde_json::{Value, json};
#[derive(Clone)]
pub struct OllamaProvider {
pub client: Client,
pub base_url: String,
}
impl OllamaProvider {
pub fn new(base_url: impl Into<String>) -> Self {
Self {
client: Client::new(),
base_url: base_url.into(),
}
}
// ── private helpers ──────────────────────────────────────────────────────
fn build_options(body: &Value) -> Value {
json!({
"temperature": body.get("temperature"),
"top_p": body.get("top_p"),
"num_predict": body.get("max_tokens"),
})
}
async fn validate_model(&self, model: &str) -> Result<(), OllamaError> {
let available = self.list_models().await?;
let exists = available
.get("models")
.and_then(|m| m.as_array())
.map(|arr| {
arr.iter()
.any(|m| m.get("name").and_then(|n| n.as_str()) == Some(model))
})
.unwrap_or(false);
if !exists {
return Err(OllamaError::ModelNotFound(model.to_string()));
}
Ok(())
}
fn parse_keep_alive(s: &str) -> Result<(), OllamaError> {
let s = s.trim();
// Ollama also accepts plain integers (seconds) or "-1" (load forever)
if s == "-1" || s.parse::<u64>().is_ok() {
return Ok(());
}
// Otherwise expect: <number><unit> e.g. "10m", "2h", "30s"
let (num, unit) = s
.find(|c: char| c.is_alphabetic())
.map(|i| s.split_at(i))
.ok_or_else(|| OllamaError::InvalidKeepAlive(s.to_string()))?;
num.parse::<u64>()
.map_err(|_| OllamaError::InvalidKeepAlive(s.to_string()))?;
match unit {
"s" | "m" | "h" => Ok(()),
_ => Err(OllamaError::InvalidKeepAlive(s.to_string())),
}
}
// ── public endpoints ─────────────────────────────────────────────────────
pub async fn list_models(&self) -> Result<Value, OllamaError> {
let url = format!("{}/api/tags", self.base_url);
let res = self.client.get(url).send().await?.json::<Value>().await?;
Ok(res)
}
pub async fn load_model(
&self,
model: &str,
keep_alive: Option<&str>,
) -> Result<Value, OllamaError> {
self.validate_model(model).await?;
let keep_alive = keep_alive.unwrap_or("5m");
Self::parse_keep_alive(keep_alive)?; // ← validated before any network call
let payload = json!({
"model": model,
"prompt": "",
"keep_alive": keep_alive,
"stream": false,
});
let res = self
.client
.post(format!("{}/api/generate", self.base_url))
.json(&payload)
.send()
.await?
.json::<Value>()
.await?;
Ok(json!({
"model": res.get("model"),
"status": "loaded",
"keep_alive": keep_alive,
}))
}
pub async fn completions(&self, body: Value) -> Result<Value, OllamaError> {
let prompt = body
.get("prompt")
.and_then(|v| v.as_str())
.filter(|s| !s.trim().is_empty())
.ok_or(OllamaError::MissingPrompt)?;
let model = body
.get("model")
.and_then(|v| v.as_str())
.unwrap_or("llama3");
self.validate_model(model).await?;
let ollama_payload = json!({
"model": model,
"prompt": prompt,
"stream": false,
"options": Self::build_options(&body),
});
let res = self
.client
.post(format!("{}/api/generate", self.base_url))
.json(&ollama_payload)
.send()
.await?
.json::<Value>()
.await?;
Ok(json!({
"id": "cmpl-ollama",
"object": "text_completion",
"model": res.get("model"),
"choices": [{
"text": res.get("response"),
"index": 0,
"finish_reason": if res.get("done").and_then(|v| v.as_bool()).unwrap_or(false) {
"stop"
} else {
"length"
},
}],
"usage": {
"prompt_tokens": res.get("prompt_eval_count"),
"completion_tokens": res.get("eval_count"),
"total_tokens": null,
}
}))
}
pub async fn chat_completions(&self, body: Value) -> Result<Value, OllamaError> {
let model = body
.get("model")
.and_then(|v| v.as_str())
.unwrap_or("llama3");
let messages = body
.get("messages")
.and_then(|v| v.as_array())
.filter(|arr| !arr.is_empty())
.ok_or(OllamaError::MissingMessages)?;
let has_user_msg = messages
.iter()
.any(|m| m.get("role").and_then(|r| r.as_str()) == Some("user"));
if !has_user_msg {
return Err(OllamaError::MissingMessages);
}
self.validate_model(model).await?;
let ollama_payload = json!({
"model": model,
"messages": messages,
"stream": false,
"options": Self::build_options(&body),
});
let res = self
.client
.post(format!("{}/api/chat", self.base_url))
.json(&ollama_payload)
.send()
.await?
.json::<Value>()
.await?;
Ok(json!({
"id": "chatcmpl-ollama",
"object": "chat.completion",
"model": res.get("model"),
"choices": [{
"index": 0,
"message": {
"role": res.get("message").and_then(|m| m.get("role")),
"content": res.get("message").and_then(|m| m.get("content")),
},
"finish_reason": res
.get("done_reason")
.and_then(|v| v.as_str())
.unwrap_or("stop"),
}],
"usage": {
"prompt_tokens": res.get("prompt_eval_count"),
"completion_tokens": res.get("eval_count"),
"total_tokens": res.get("prompt_eval_count")
.and_then(|p| p.as_u64())
.zip(res.get("eval_count").and_then(|e| e.as_u64()))
.map(|(p, e)| p + e),
}
}))
}
}
+432
View File
@@ -0,0 +1,432 @@
use crate::dto::{api, ollama};
use crate::errors::OllamaError;
use axum::response::sse::Event;
use futures::StreamExt;
use reqwest::Client;
use serde_json::json;
use tokio_stream::wrappers::ReceiverStream;
#[derive(Clone)]
pub struct OllamaProvider {
pub client: Client,
pub base_url: String,
}
impl OllamaProvider {
pub fn new(base_url: impl Into<String>) -> Self {
Self {
client: Client::new(),
base_url: base_url.into(),
}
}
// ── private helpers ──────────────────────────────────────────────────────
async fn model_exists(&self, model: &str) -> Result<bool, OllamaError> {
let url = format!("{}/api/tags", self.base_url);
let res = self
.client
.get(url)
.send()
.await?
.json::<ollama::OllamaModels>()
.await?;
Ok(res.models.iter().any(|m| m.name == model))
}
fn has_user_message(&self, messages: &[api::Message]) -> bool {
messages.iter().any(|m| matches!(m.role, api::Role::User))
}
fn extract_completion_params<'a>(
&self,
body: &'a api::CompletionRequest,
) -> Result<(&'a str, &'a str), OllamaError> {
let prompt = body.prompt.trim();
let model = body.base.model.as_str();
Ok((prompt, model))
}
fn extract_chat_params<'a>(
&self,
body: &'a api::ChatRequest,
) -> Result<(&'a [api::Message], &'a str), OllamaError> {
let model = body.base.model.as_str();
Ok((&body.messages, model))
}
pub fn parse_keep_alive(&self, s: &str) -> Result<(), OllamaError> {
let s = s.trim();
if s == "-1" || s.parse::<u64>().is_ok() {
return Ok(());
}
let split = s
.find(|c: char| c.is_alphabetic())
.ok_or_else(|| OllamaError::InvalidKeepAlive(s.to_string()))?;
let (num, unit) = s.split_at(split);
num.parse::<u64>()
.map_err(|_| OllamaError::InvalidKeepAlive(s.to_string()))?;
match unit {
"s" | "m" | "h" => Ok(()),
_ => Err(OllamaError::InvalidKeepAlive(s.to_string())),
}
}
// // ── public endpoints ─────────────────────────────────────────────────────
pub async fn list_models(&self) -> Result<api::ModelsResponse, OllamaError> {
let url = format!("{}/api/tags", self.base_url);
let res = self
.client
.get(url)
.send()
.await?
.json::<ollama::OllamaModels>()
.await?;
let models = res.models.into_iter().map(api::ModelInfo::from).collect();
Ok(api::ModelsResponse { models })
}
pub async fn load_model(
&self,
model: &str,
keep_alive: Option<&str>,
) -> Result<api::LoadModelResponse, OllamaError> {
let url = format!("{}/api/generate", self.base_url);
let keep_alive = keep_alive.ok_or(OllamaError::MissingKeepAlive)?;
self.parse_keep_alive(keep_alive)?;
let exists = self.model_exists(model).await?;
if !exists {
return Err(OllamaError::ModelNotFound(model.to_string()));
}
let payload = json!({
"model": model,
"prompt": "",
"keep_alive": keep_alive,
"stream": false,
});
let _res = self
.client
.post(url)
.json(&payload)
.send()
.await?
.text()
.await?;
Ok(api::LoadModelResponse {
model: model.to_string(),
status: "loaded".to_string(),
keep_alive: keep_alive.to_string(),
})
}
pub async fn unload_model(&self, model: &str) -> Result<api::UnloadModelResponse, OllamaError> {
let url = format!("{}/api/generate", self.base_url);
let exists = self.model_exists(model).await?;
if !exists {
return Err(OllamaError::ModelNotFound(model.to_string()));
}
let payload = json!({
"model": model,
"prompt": "",
"keep_alive": "0",
"stream": false,
});
let _res = self
.client
.post(url)
.json(&payload)
.send()
.await?
.text()
.await?;
Ok(api::UnloadModelResponse {
model: model.to_string(),
status: "unloaded".to_string(),
})
}
pub async fn completions(
&self,
body: &api::CompletionRequest,
) -> Result<api::CompletionResponse, OllamaError> {
let url = format!("{}/api/generate", self.base_url);
let (prompt, model) = self.extract_completion_params(body)?;
if prompt.is_empty() {
return Err(OllamaError::MissingPrompt);
}
if model.is_empty() {
return Err(OllamaError::MissingModel);
}
let exists = self.model_exists(model).await?;
if !exists {
return Err(OllamaError::ModelNotFound(model.to_string()));
}
let options = ollama::OllamaOptions::from(&body.base);
let payload = ollama::OllamaGenerateRequest {
model,
prompt,
stream: false,
options,
};
let res = self
.client
.post(url)
.json(&payload)
.send()
.await?
.json::<ollama::OllamaGenerateResponse>()
.await?;
Ok(api::CompletionResponse::from(res))
}
pub async fn completions_stream(
&self,
body: &api::CompletionRequest,
) -> Result<ReceiverStream<Result<Event, OllamaError>>, OllamaError> {
let url = format!("{}/api/generate", self.base_url);
let (prompt, model) = self.extract_completion_params(body)?;
if prompt.is_empty() {
return Err(OllamaError::MissingPrompt);
}
if model.is_empty() {
return Err(OllamaError::MissingModel);
}
let exists = self.model_exists(model).await?;
if !exists {
return Err(OllamaError::ModelNotFound(model.to_string()));
}
let options = ollama::OllamaOptions::from(&body.base);
let payload = ollama::OllamaGenerateRequest {
model,
prompt,
stream: true,
options,
};
let mut byte_stream = self
.client
.post(url)
.json(&payload)
.send()
.await?
.bytes_stream();
let (tx, rx) = tokio::sync::mpsc::channel(32);
tokio::spawn(async move {
while let Some(chunk) = byte_stream.next().await {
let chunk = match chunk {
Ok(b) => b,
Err(e) => {
let _ = tx.send(Err(OllamaError::Http(e))).await;
break;
}
};
// 🔥 IMPORTANT: typed deserialization
let parsed: ollama::OllamaGenerateResponse = match serde_json::from_slice(&chunk) {
Ok(v) => v,
Err(_) => continue,
};
// map → OpenAI chunk
let event_data = serde_json::to_string(&api::CompletionChunk {
id: "cmpl-ollama".to_string(),
object: "text_completion".to_string(),
choices: vec![api::Choice {
text: parsed.response,
index: 0,
finish_reason: if parsed.done {
api::FinishReason::Stop
} else {
api::FinishReason::Length
},
}],
})
.unwrap_or_default();
let _ = tx.send(Ok(Event::default().data(event_data))).await;
if parsed.done {
let _ = tx.send(Ok(Event::default().data("[DONE]"))).await;
break;
}
}
});
Ok(ReceiverStream::new(rx))
}
pub async fn chat_completions(
&self,
body: &api::ChatRequest,
) -> Result<api::ChatCompletionResponse, OllamaError> {
let url = format!("{}/api/chat", self.base_url);
let (messages, model) = self.extract_chat_params(body)?;
if body.messages.is_empty() {
return Err(OllamaError::MissingMessages);
}
if !self.has_user_message(&body.messages) {
return Err(OllamaError::MissingMessages);
}
if model.is_empty() {
return Err(OllamaError::MissingModel);
}
let exists = self.model_exists(model).await?;
if !exists {
return Err(OllamaError::ModelNotFound(model.to_string()));
}
// let options = ollama::OllamaOptions::from(body);
let options = ollama::OllamaOptions::from(&body.base);
let payload = ollama::OllamaChatRequest {
model,
messages,
stream: false,
options,
};
let res = self
.client
.post(url)
.json(&payload)
.send()
.await?
.json::<ollama::OllamaChatResponse>()
.await?;
Ok(api::ChatCompletionResponse::from(res))
}
pub async fn chat_completions_stream(
&self,
body: &api::ChatRequest,
) -> Result<ReceiverStream<Result<Event, OllamaError>>, OllamaError> {
let url = format!("{}/api/chat", self.base_url);
let (messages, model) = self.extract_chat_params(body)?;
if messages.is_empty() {
return Err(OllamaError::MissingMessages);
}
if model.is_empty() {
return Err(OllamaError::MissingModel);
}
let exists = self.model_exists(model).await?;
if !exists {
return Err(OllamaError::ModelNotFound(model.to_string()));
}
let options = ollama::OllamaOptions::from(&body.base);
let payload = ollama::OllamaChatRequest {
model,
messages,
stream: true,
options,
};
let mut byte_stream = self
.client
.post(url)
.json(&payload)
.send()
.await?
.bytes_stream();
let (tx, rx) = tokio::sync::mpsc::channel(32);
tokio::spawn(async move {
let stream_id = format!("chatcmpl-{}", uuid::Uuid::new_v4());
while let Some(chunk) = byte_stream.next().await {
let chunk = match chunk {
Ok(b) => b,
Err(e) => {
let _ = tx.send(Err(OllamaError::Http(e))).await;
break;
}
};
let parsed: ollama::OllamaChatResponse = match serde_json::from_slice(&chunk) {
Ok(v) => v,
Err(_) => continue,
};
let event = api::ChatCompletionChunk {
id: stream_id.clone(),
object: "chat.completion.chunk".to_string(),
choices: vec![api::ChatChunkChoice {
index: 0,
delta: api::ChatDelta {
role: Some(parsed.message.role),
content: Some(parsed.message.content),
},
finish_reason: if parsed.done {
Some(api::FinishReason::Stop)
} else {
None
},
}],
};
let event_data = serde_json::to_string(&event).unwrap_or_default();
let _ = tx.send(Ok(Event::default().data(event_data))).await;
if parsed.done {
let _ = tx.send(Ok(Event::default().data("[DONE]"))).await;
break;
}
}
});
Ok(ReceiverStream::new(rx))
}
}
+86
View File
@@ -0,0 +1,86 @@
use crate::dto::{api, ollama};
use chrono::Utc;
use uuid::Uuid;
impl From<ollama::OllamaModel> for api::ModelInfo {
fn from(m: ollama::OllamaModel) -> Self {
Self {
name: m.name,
family: m.details.as_ref().and_then(|d| d.family.clone()),
parameter_size: m.details.as_ref().and_then(|d| d.parameter_size.clone()),
quantization: m
.details
.as_ref()
.and_then(|d| d.quantization_level.clone()),
}
}
}
impl From<ollama::OllamaGenerateResponse> for api::CompletionResponse {
fn from(res: ollama::OllamaGenerateResponse) -> Self {
Self {
id: Uuid::new_v4().to_string(),
object: api::CompletionObject::TextCompletion,
model: res.model,
created: Utc::now().timestamp() as u64,
choices: vec![api::Choice {
text: res.response,
index: 0,
finish_reason: api::FinishReason::Stop,
}],
usage: api::Usage {
prompt_tokens: res.prompt_eval_count.unwrap_or(0),
completion_tokens: res.eval_count.unwrap_or(0),
total_tokens: res.prompt_eval_count.unwrap_or(0) + res.eval_count.unwrap_or(0),
},
}
}
}
impl From<&api::BaseLLMRequest> for ollama::OllamaOptions {
fn from(base: &api::BaseLLMRequest) -> Self {
Self {
temperature: base.temperature,
top_p: base.top_p,
top_k: base.top_k,
repeat_penalty: base.repeat_penalty,
seed: base.seed,
num_ctx: base.num_ctx,
num_predict: base.num_predict,
}
}
}
impl From<ollama::OllamaChatResponse> for api::ChatCompletionResponse {
fn from(res: ollama::OllamaChatResponse) -> Self {
let prompt_tokens = res.prompt_eval_count.unwrap_or(0);
let completion_tokens = res.eval_count.unwrap_or(0);
Self {
id: Uuid::new_v4().to_string(),
object: "chat.completion".to_string(),
created: Utc::now().timestamp() as u64,
model: res.model,
choices: vec![api::ChatChoice {
index: 0,
message: res.message,
finish_reason: if res.done {
api::FinishReason::Stop
} else {
api::FinishReason::Length
},
}],
usage: Some(api::Usage {
prompt_tokens,
completion_tokens,
total_tokens: prompt_tokens + completion_tokens,
}),
}
}
}
+2
View File
@@ -0,0 +1,2 @@
pub mod client;
pub mod mapper;
+50
View File
@@ -0,0 +1,50 @@
use axum::{
Json,
extract::{Extension, State},
http::StatusCode,
};
use base64::{Engine as _, engine::general_purpose};
use rand::RngCore;
use rand::rngs::OsRng;
use crate::dto::api::{CreateApiKeyRequest, CreateApiKeyResponse};
use crate::middlewares::auth::middleware::Auth;
use crate::state::app_state::AppState;
use crate::utils::crypto::hash_key;
fn generate_api_key() -> String {
let mut bytes = [0u8; 32];
OsRng.fill_bytes(&mut bytes);
general_purpose::URL_SAFE_NO_PAD.encode(bytes)
}
pub async fn create_api_key(
State(state): State<AppState>,
Extension(claims): Extension<Auth>,
Json(body): Json<CreateApiKeyRequest>,
) -> Result<Json<CreateApiKeyResponse>, StatusCode> {
if matches!(claims, Auth::ApiKey(_)) {
return Err(StatusCode::FORBIDDEN);
}
let raw_key = generate_api_key();
let key_hash = hash_key(&raw_key);
dbg!(&claims);
sqlx::query!(
r#"
INSERT INTO auth.api_key (key_hash, name, created_by, scopes)
VALUES ($1, $2, $3, $4)
"#,
key_hash,
body.name,
claims.user_id(),
&body.scopes
)
.execute(&state.postgres)
.await
.map_err(|_| StatusCode::INTERNAL_SERVER_ERROR)?;
Ok(Json(CreateApiKeyResponse { api_key: raw_key }))
}
+154 -28
View File
@@ -1,44 +1,170 @@
use axum::{Json, extract::State}; use axum::{
use serde_json::Value; Json,
extract::State,
response::{
IntoResponse, Response,
sse::{KeepAlive, Sse},
},
};
use crate::dto::api;
use crate::errors::OllamaError; use crate::errors::OllamaError;
use crate::state::app_state::AppState; use crate::state::app_state::AppState;
#[utoipa::path(
post,
path = "/completions",
tag = "chat",
request_body(
content = api::CompletionRequest,
description = "Text completion request",
content_type = "application/json"
),
responses(
(
status = 200,
description = "Text completion response. If stream=true, response is SSE stream of chunks ending in [DONE].",
body = api::CompletionResponse,
content_type = "application/json"
),
(
status = 400,
description = "Invalid request: missing prompt, model, or invalid format",
body = api::ErrorResponse,
example = json!({ "error": "prompt is required and cannot be empty" })
),
(
status = 422,
description = "Model not found or not available locally",
body = api::ErrorResponse,
example = json!({ "error": "model 'llama3' is not available — run `ollama pull llama3` first" })
),
(
status = 500,
description = "Internal server error (Ollama or network failure)",
body = api::ErrorResponse,
example = json!({ "error": "connection refused" })
)
)
)]
pub async fn completions( pub async fn completions(
State(state): State<AppState>, State(state): State<AppState>,
Json(body): Json<Value>, Json(body): Json<api::CompletionRequest>,
) -> Result<Json<Value>, (axum::http::StatusCode, String)> { ) -> Result<Response, (axum::http::StatusCode, Json<api::ErrorResponse>)> {
match state.ollama.completions(body).await { tracing::debug!("Received /completion with body {:?}", body);
Ok(response) => Ok(Json(response)), if body.base.stream {
Err(OllamaError::MissingPrompt) => Err(( let stream = state.ollama.completions_stream(&body).await.map_err(|e| {
axum::http::StatusCode::BAD_REQUEST, let (code, msg) = ollama_err(e);
"prompt is required and cannot be empty".to_string(), (code, Json(api::ErrorResponse::new(msg)))
)), })?;
Err(OllamaError::ModelNotFound(m)) => Err((
axum::http::StatusCode::UNPROCESSABLE_ENTITY, Ok(Sse::new(stream)
format!("model '{m}' is not available — run `ollama pull {m}` first"), .keep_alive(KeepAlive::default())
)), .into_response())
Err(e) => Err((axum::http::StatusCode::INTERNAL_SERVER_ERROR, e.to_string())), } else {
let response = state.ollama.completions(&body).await.map_err(|e| {
let (code, msg) = ollama_err(e);
(code, Json(api::ErrorResponse::new(msg)))
})?;
Ok(Json(response).into_response())
} }
} }
#[utoipa::path(
post,
path = "/chat/completions",
tag = "chat",
request_body(
content = api::ChatRequest,
description = "Chat completion request with message history",
content_type = "application/json"
),
responses(
(
status = 200,
description = "Chat completion response. If stream=false returns JSON. If stream=true returns SSE stream of chunks ending with [DONE].",
body = api::ChatCompletionResponse,
content_type = "application/json"
),
(
status = 400,
description = "Invalid request",
body = api::ErrorResponse,
example = json!({ "error": "messages array with at least one user message is required" })
),
(
status = 401,
description = "Unauthorized",
body = api::ErrorResponse,
example = json!({ "error": "missing or invalid token" })
),
(
status = 422,
description = "Model not found or unavailable",
body = api::ErrorResponse,
example = json!({ "error": "model 'llama3' is not available — run `ollama pull llama3` first" })
),
(
status = 500,
description = "Internal server error",
body = api::ErrorResponse,
example = json!({ "error": "connection refused" })
)
)
)]
pub async fn chat_completions( pub async fn chat_completions(
State(state): State<AppState>, State(state): State<AppState>,
Json(body): Json<Value>, Json(body): Json<api::ChatRequest>,
) -> Result<Json<Value>, (axum::http::StatusCode, String)> { ) -> Result<Response, (axum::http::StatusCode, String)> {
match state.ollama.chat_completions(body).await { tracing::debug!("Received /completion with body {:?}", body);
Ok(response) => Ok(Json(response)), if body.base.stream {
Err(OllamaError::MissingMessages) => Err(( let stream = state
.ollama
.chat_completions_stream(&body)
.await
.map_err(ollama_err)?;
Ok(Sse::new(stream)
.keep_alive(KeepAlive::default())
.into_response())
} else {
let response = state
.ollama
.chat_completions(&body)
.await
.map_err(ollama_err)?;
Ok(Json(response).into_response())
}
}
fn ollama_err(e: OllamaError) -> (axum::http::StatusCode, String) {
match e {
OllamaError::MissingPrompt => (
axum::http::StatusCode::BAD_REQUEST,
"prompt is required and cannot be empty".to_string(),
),
OllamaError::MissingModel => (
axum::http::StatusCode::BAD_REQUEST, axum::http::StatusCode::BAD_REQUEST,
"messages must be a non-empty array with at least one user message".to_string(), "model is required and cannot be empty".to_string(),
)), ),
Err(OllamaError::ModelNotFound(m)) => Err(( OllamaError::ModelNotFound(m) => (
axum::http::StatusCode::UNPROCESSABLE_ENTITY, axum::http::StatusCode::UNPROCESSABLE_ENTITY,
format!("model '{m}' is not available — run `ollama pull {m}` first"), format!("model '{m}' is not available — run `ollama pull {m}` first"),
)), ),
Err(OllamaError::Http(e)) => { OllamaError::MissingKeepAlive => (
Err((axum::http::StatusCode::INTERNAL_SERVER_ERROR, e.to_string())) axum::http::StatusCode::BAD_REQUEST,
} "keep alive is required and cannot be empty".to_string(),
Err(e) => Err((axum::http::StatusCode::BAD_REQUEST, e.to_string())), ),
OllamaError::InvalidKeepAlive(v) => (
axum::http::StatusCode::BAD_REQUEST,
format!("invalid keep_alive '{v}'"),
),
OllamaError::MissingMessages => (
axum::http::StatusCode::BAD_REQUEST,
"messages array with at least one user message is required".to_string(),
),
OllamaError::Http(e) => (axum::http::StatusCode::INTERNAL_SERVER_ERROR, e.to_string()),
} }
} }
+28 -4
View File
@@ -1,15 +1,39 @@
pub mod apikey;
pub mod chat; pub mod chat;
pub mod models; pub mod models;
use crate::auth::middleware::auth_middleware; use crate::docs::ApiDoc;
use crate::middlewares::auth::{auth_middleware, middleware::require_roles};
use crate::state::app_state::AppState; use crate::state::app_state::AppState;
use axum::{Router, middleware, routing::get, routing::post};
pub fn router() -> Router<AppState> { use axum::{Json, Router, middleware, routing::get, routing::post};
use utoipa::OpenApi;
async fn openapi_json() -> Json<utoipa::openapi::OpenApi> {
Json(ApiDoc::openapi())
}
fn public_router() -> Router<AppState> {
Router::new().route("/docs.json", get(openapi_json))
}
pub fn protected_router() -> Router<AppState> {
Router::new() Router::new()
.route("/models", get(models::list_models)) .route("/models", get(models::list_models))
.route("/completions", post(chat::completions)) .route("/completions", post(chat::completions))
.route("/chat/completions", post(chat::chat_completions)) .route("/chat/completions", post(chat::chat_completions))
.route("/models/{model}/load", post(models::load_model)) .route("/models/{model}/load", post(models::load_model))
.layer(middleware::from_fn(auth_middleware)) .route("/models/{model}/unload", post(models::unload_model))
.route(
"/keys/generate",
post(apikey::create_api_key).route_layer(middleware::from_fn(|req, next| {
require_roles(req, next, None, Some("admin"))
})),
)
}
pub fn router(state: AppState) -> Router<AppState> {
Router::new()
.merge(public_router())
.merge(protected_router().layer(middleware::from_fn_with_state(state, auth_middleware)))
} }
+119 -15
View File
@@ -2,38 +2,134 @@ use axum::{
Json, Json,
extract::{Path, State}, extract::{Path, State},
}; };
use serde::Deserialize;
use serde_json::Value;
use crate::dto::api;
use crate::{errors::OllamaError, state::app_state::AppState}; use crate::{errors::OllamaError, state::app_state::AppState};
#[derive(Deserialize)] #[utoipa::path(
pub struct LoadModelBody { get,
pub keep_alive: Option<String>, path = "/models",
} tag = "models",
responses(
(
status = 200,
description = "List of locally available Ollama models",
body = api::ModelsResponse,
content_type = "application/json",
),
(
status = 500,
description = "Internal server error (Ollama or network failure)",
body = api::ErrorResponse,
example = json!({ "error": "connection refused" })
)
)
)]
pub async fn list_models( pub async fn list_models(
State(state): State<AppState>, State(state): State<AppState>,
) -> Result<Json<Value>, (axum::http::StatusCode, String)> { ) -> Result<Json<api::ModelsResponse>, (axum::http::StatusCode, String)> {
match state.ollama.list_models().await { match state.ollama.list_models().await {
Ok(models) => Ok(Json(models)), Ok(models) => Ok(Json(models)),
Err(e) => Err(ollama_err(e)), Err(e) => Err(ollama_err(e)),
} }
} }
#[utoipa::path(
post,
path = "/models/{model}/load",
tag = "models",
params(
("model" = String, Path, description = "Name of the model to load into memory (e.g. 'llama3')")
),
request_body(
content = api::LoadModelBody,
description = "Load model request",
content_type = "application/json",
example = json!({ "keep_alive": "10m" })
),
responses(
(
status = 200,
description = "Model successfully loaded into memory",
body = api::LoadModelResponse,
content_type = "application/json",
),
(
status = 400,
description = "Invalid or missing keep_alive format",
body = api::ErrorResponse,
examples(
("Missing" = (value = json!({ "error": "keep alive is required and cannot be empty" }))),
("Invalid" = (value = json!({ "error": "invalid keep_alive '10x' — use 30s / 10m / 2h, a plain integer, or -1" })))
)
),
(
status = 404,
description = "Model not found locally",
body = api::ErrorResponse,
example = json!({ "error": "model 'llama3' not found — run `ollama pull llama3`" })
),
(
status = 500,
description = "Internal server error (Ollama or network failure)",
body = api::ErrorResponse,
example = json!({ "error": "connection refused" })
)
)
)]
pub async fn load_model( pub async fn load_model(
State(state): State<AppState>, State(state): State<AppState>,
Path(model): Path<String>, Path(model): Path<String>,
Json(body): Json<LoadModelBody>, Json(body): Json<api::LoadModelBody>,
) -> Result<Json<Value>, (axum::http::StatusCode, String)> { ) -> Result<Json<api::LoadModelResponse>, (axum::http::StatusCode, String)> {
match state let response = state
.ollama .ollama
.load_model(&model, body.keep_alive.as_deref()) .load_model(&model, body.keep_alive.as_deref())
.await .await
{ .map_err(ollama_err)?;
Ok(response) => Ok(Json(response)),
Err(e) => Err(ollama_err(e)), Ok(Json(response))
} }
#[utoipa::path(
delete,
path = "/models/{model}/load",
tag = "models",
params(
("model" = String, Path, description = "Name of the model to unload from memory (e.g. 'llama3')")
),
responses(
(
status = 200,
description = "Model successfully unloaded from memory",
body = api::UnloadModelResponse,
content_type = "application/json",
),
(
status = 404,
description = "Model not found locally",
body = api::ErrorResponse,
example = json!({ "error": "model 'llama3' not found — run `ollama pull llama3`" })
),
(
status = 500,
description = "Internal server error (Ollama or network failure)",
body = api::ErrorResponse,
example = json!({ "error": "connection refused" })
)
)
)]
pub async fn unload_model(
State(state): State<AppState>,
Path(model): Path<String>,
) -> Result<Json<api::UnloadModelResponse>, (axum::http::StatusCode, String)> {
let response = state
.ollama
.unload_model(&model)
.await
.map_err(ollama_err)?;
Ok(Json(response))
} }
fn ollama_err(e: OllamaError) -> (axum::http::StatusCode, String) { fn ollama_err(e: OllamaError) -> (axum::http::StatusCode, String) {
@@ -42,6 +138,10 @@ fn ollama_err(e: OllamaError) -> (axum::http::StatusCode, String) {
axum::http::StatusCode::NOT_FOUND, axum::http::StatusCode::NOT_FOUND,
format!("model '{m}' not found — run `ollama pull {m}`"), format!("model '{m}' not found — run `ollama pull {m}`"),
), ),
OllamaError::MissingModel => (
axum::http::StatusCode::BAD_REQUEST,
"model is required and cannot be empty".to_string(),
),
OllamaError::MissingPrompt => ( OllamaError::MissingPrompt => (
axum::http::StatusCode::BAD_REQUEST, axum::http::StatusCode::BAD_REQUEST,
"prompt is required".to_string(), "prompt is required".to_string(),
@@ -50,6 +150,10 @@ fn ollama_err(e: OllamaError) -> (axum::http::StatusCode, String) {
axum::http::StatusCode::BAD_REQUEST, axum::http::StatusCode::BAD_REQUEST,
"messages array with at least one user message is required".to_string(), "messages array with at least one user message is required".to_string(),
), ),
OllamaError::MissingKeepAlive => (
axum::http::StatusCode::BAD_REQUEST,
"keep alive is required and cannot be empty".to_string(),
),
OllamaError::InvalidKeepAlive(v) => ( OllamaError::InvalidKeepAlive(v) => (
axum::http::StatusCode::BAD_REQUEST, axum::http::StatusCode::BAD_REQUEST,
format!("invalid keep_alive '{v}' — use 30s / 10m / 2h, a plain integer, or -1"), format!("invalid keep_alive '{v}' — use 30s / 10m / 2h, a plain integer, or -1"),
+8
View File
@@ -0,0 +1,8 @@
// use axum::Json;
// use utoipa::OpenApi;
// use crate::openapi::V1ApiDoc;
// pub async fn openapi_json() -> Json<utoipa::openapi::OpenApi> {
// Json(V1ApiDoc::openapi())
// }
+3 -1
View File
@@ -1,7 +1,9 @@
use crate::providers::ollama::OllamaProvider; use crate::providers::ollama::client::OllamaProvider;
use sqlx::PgPool;
use std::sync::Arc; use std::sync::Arc;
#[derive(Clone)] #[derive(Clone)]
pub struct AppState { pub struct AppState {
pub ollama: Arc<OllamaProvider>, pub ollama: Arc<OllamaProvider>,
pub postgres: PgPool,
} }
+11
View File
@@ -0,0 +1,11 @@
use sha2::{Digest, Sha256};
pub fn hash_key(key: &str) -> String {
let mut hasher = Sha256::new();
hasher.update(key.as_bytes());
hasher
.finalize()
.iter()
.map(|b| format!("{:02x}", b))
.collect()
}
+1
View File
@@ -0,0 +1 @@
pub mod crypto;
+249 -65
View File
@@ -2,7 +2,9 @@ use serde_json::json;
use wiremock::matchers::{method, path}; use wiremock::matchers::{method, path};
use wiremock::{Mock, MockServer, ResponseTemplate}; use wiremock::{Mock, MockServer, ResponseTemplate};
use chat::providers::ollama::OllamaProvider; use chat::dto::api;
use chat::errors::OllamaError;
use chat::providers::ollama::client::OllamaProvider;
// ── helpers ────────────────────────────────────────────────────────────────── // ── helpers ──────────────────────────────────────────────────────────────────
@@ -31,7 +33,7 @@ async fn test_list_models_ok() {
.await; .await;
let res = provider.list_models().await.unwrap(); let res = provider.list_models().await.unwrap();
assert_eq!(res["models"][0]["name"], "llama3"); assert_eq!(res.models[0].name, "llama3");
} }
// ── completions ─────────────────────────────────────────────────────────────── // ── completions ───────────────────────────────────────────────────────────────
@@ -58,19 +60,25 @@ async fn test_completions_ok() {
.mount(&server) .mount(&server)
.await; .await;
let res = provider let req = api::CompletionRequest {
.completions(json!({ base: api::BaseLLMRequest {
"model": "llama3", model: "llama3".to_string(),
"prompt": "Who are you?" ..Default::default()
})) },
.await prompt: "Hello".to_string(),
.unwrap(); };
assert_eq!(res["object"], "text_completion"); let res = provider.completions(&req).await.unwrap();
assert_eq!(res["choices"][0]["text"], "I am a helpful assistant.");
assert_eq!(res["choices"][0]["finish_reason"], "stop"); assert_eq!(res.object, api::CompletionObject::TextCompletion);
assert_eq!(res["usage"]["prompt_tokens"], 10);
assert_eq!(res["usage"]["completion_tokens"], 8); assert_eq!(res.choices.len(), 1);
assert_eq!(res.choices[0].text, "I am a helpful assistant.");
assert_eq!(res.choices[0].finish_reason, api::FinishReason::Stop);
assert_eq!(res.usage.prompt_tokens, 10);
assert_eq!(res.usage.completion_tokens, 8);
assert_eq!(res.usage.total_tokens, 18);
} }
#[tokio::test] #[tokio::test]
@@ -83,11 +91,17 @@ async fn test_completions_missing_prompt() {
.mount(&server) .mount(&server)
.await; .await;
let err = provider let req = api::CompletionRequest {
.completions(json!({ "model": "llama3" })) base: api::BaseLLMRequest {
.await model: "llama3".to_string(),
.unwrap_err(); ..Default::default()
assert!(matches!(err, chat::errors::OllamaError::MissingPrompt)); },
prompt: "".to_string(),
};
let err = provider.completions(&req).await.unwrap_err();
assert!(matches!(err, OllamaError::MissingPrompt));
} }
#[tokio::test] #[tokio::test]
@@ -100,13 +114,17 @@ async fn test_completions_empty_prompt() {
.mount(&server) .mount(&server)
.await; .await;
let err = provider let req = api::CompletionRequest {
.completions(json!({ base: api::BaseLLMRequest {
"model": "llama3", "prompt": " " model: "llama3".to_string(),
})) ..Default::default()
.await },
.unwrap_err(); prompt: " ".to_string(),
assert!(matches!(err, chat::errors::OllamaError::MissingPrompt)); };
let err = provider.completions(&req).await.unwrap_err();
assert!(matches!(err, OllamaError::MissingPrompt));
} }
#[tokio::test] #[tokio::test]
@@ -119,12 +137,16 @@ async fn test_completions_model_not_found() {
.mount(&server) .mount(&server)
.await; .await;
let err = provider let req = api::CompletionRequest {
.completions(json!({ base: api::BaseLLMRequest {
"model": "gpt-4", "prompt": "hello" model: "gpt-4".to_string(),
})) ..Default::default()
.await },
.unwrap_err(); prompt: "hello".to_string(),
};
let err = provider.completions(&req).await.unwrap_err();
assert!(matches!(err, chat::errors::OllamaError::ModelNotFound(_))); assert!(matches!(err, chat::errors::OllamaError::ModelNotFound(_)));
} }
@@ -146,28 +168,37 @@ async fn test_chat_completions_ok() {
"model": "llama3", "model": "llama3",
"message": { "role": "assistant", "content": "4." }, "message": { "role": "assistant", "content": "4." },
"done": true, "done": true,
"done_reason": "stop",
"prompt_eval_count": 5, "prompt_eval_count": 5,
"eval_count": 2, "eval_count": 2
}))) })))
.mount(&server) .mount(&server)
.await; .await;
let res = provider let req = api::ChatRequest {
.chat_completions(json!({ base: api::BaseLLMRequest {
"model": "llama3", model: "llama3".to_string(),
"messages": [ ..Default::default()
{ "role": "user", "content": "What is 2+2?" } },
] messages: vec![api::Message {
})) role: api::Role::User,
.await content: "What is 2+2?".to_string(),
.unwrap(); }],
};
assert_eq!(res["object"], "chat.completion"); let res = provider.chat_completions(&req).await.unwrap();
assert_eq!(res["choices"][0]["message"]["role"], "assistant");
assert_eq!(res["choices"][0]["message"]["content"], "4."); assert_eq!(res.object, "chat.completion");
assert_eq!(res["choices"][0]["finish_reason"], "stop"); assert_eq!(res.choices.len(), 1);
assert_eq!(res["usage"]["total_tokens"], 7);
assert_eq!(res.choices[0].message.role, api::Role::Assistant);
assert_eq!(res.choices[0].message.content, "4.");
assert_eq!(res.choices[0].finish_reason, api::FinishReason::Stop);
let usage = res.usage.unwrap();
assert_eq!(usage.prompt_tokens, 5);
assert_eq!(usage.completion_tokens, 2);
assert_eq!(usage.total_tokens, 7);
} }
#[tokio::test] #[tokio::test]
@@ -180,12 +211,16 @@ async fn test_chat_completions_missing_messages() {
.mount(&server) .mount(&server)
.await; .await;
let err = provider let req = api::ChatRequest {
.chat_completions(json!({ base: api::BaseLLMRequest {
"model": "llama3", "messages": [] model: "llama3".to_string(),
})) ..Default::default()
.await },
.unwrap_err(); messages: vec![],
};
let err = provider.chat_completions(&req).await.unwrap_err();
assert!(matches!(err, chat::errors::OllamaError::MissingMessages)); assert!(matches!(err, chat::errors::OllamaError::MissingMessages));
} }
@@ -199,13 +234,19 @@ async fn test_chat_completions_no_user_message() {
.mount(&server) .mount(&server)
.await; .await;
let err = provider let req = api::ChatRequest {
.chat_completions(json!({ base: api::BaseLLMRequest {
"model": "llama3", model: "llama3".to_string(),
"messages": [{ "role": "system", "content": "be helpful" }] ..Default::default()
})) },
.await messages: vec![api::Message {
.unwrap_err(); role: api::Role::System,
content: "be helpful".to_string(),
}],
};
let err = provider.chat_completions(&req).await.unwrap_err();
assert!(matches!(err, chat::errors::OllamaError::MissingMessages)); assert!(matches!(err, chat::errors::OllamaError::MissingMessages));
} }
@@ -219,12 +260,155 @@ async fn test_chat_completions_model_not_found() {
.mount(&server) .mount(&server)
.await; .await;
let req = api::ChatRequest {
base: api::BaseLLMRequest {
model: "gpt-4".to_string(),
..Default::default()
},
messages: vec![api::Message {
role: api::Role::User,
content: "hi".to_string(),
}],
};
let err = provider.chat_completions(&req).await.unwrap_err();
assert!(matches!(err, chat::errors::OllamaError::ModelNotFound(_)));
}
// // ── load_model ────────────────────────────────────────────────────────────────
#[tokio::test]
async fn test_load_model_ok() {
let (server, provider) = setup().await;
Mock::given(method("GET"))
.and(path("/api/tags"))
.respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
.mount(&server)
.await;
Mock::given(method("POST"))
.and(path("/api/generate"))
.respond_with(ResponseTemplate::new(200).set_body_json(json!({
"model": "llama3",
"response": "ok",
"done": true,
})))
.mount(&server)
.await;
let res = provider.load_model("llama3", Some("10m")).await.unwrap();
assert_eq!(res.model, "llama3");
assert_eq!(res.status, "loaded");
assert_eq!(res.keep_alive, "10m");
}
#[tokio::test]
async fn test_load_model_not_found() {
let (server, provider) = setup().await;
Mock::given(method("GET"))
.and(path("/api/tags"))
.respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
.mount(&server)
.await;
let err = provider.load_model("gpt-4", Some("10m")).await.unwrap_err();
assert!(matches!(err, chat::errors::OllamaError::ModelNotFound(_)));
}
#[tokio::test]
async fn test_load_model_invalid_keep_alive() {
let (server, provider) = setup().await;
Mock::given(method("GET"))
.and(path("/api/tags"))
.respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
.mount(&server)
.await;
let err = provider let err = provider
.chat_completions(json!({ .load_model("llama3", Some("10x"))
"model": "gpt-4",
"messages": [{ "role": "user", "content": "hi" }]
}))
.await .await
.unwrap_err(); .unwrap_err();
assert!(matches!(
err,
chat::errors::OllamaError::InvalidKeepAlive(_)
));
}
#[tokio::test]
async fn test_load_model_keep_alive_plain_integer() {
let (server, provider) = setup().await;
Mock::given(method("GET"))
.and(path("/api/tags"))
.respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
.mount(&server)
.await;
Mock::given(method("POST"))
.and(path("/api/generate"))
.respond_with(ResponseTemplate::new(200).set_body_json(json!({
"model": "llama3",
"done": true,
})))
.mount(&server)
.await;
let res = provider.load_model("llama3", Some("3600")).await.unwrap();
assert_eq!(res.status, "loaded");
assert_eq!(res.keep_alive, "3600");
let res = provider.load_model("llama3", Some("-1")).await.unwrap();
assert_eq!(res.status, "loaded");
assert_eq!(res.keep_alive, "-1");
}
// // ── unload_model ──────────────────────────────────────────────────────────────
#[tokio::test]
async fn test_unload_model_ok() {
let (server, provider) = setup().await;
Mock::given(method("GET"))
.and(path("/api/tags"))
.respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
.mount(&server)
.await;
Mock::given(method("POST"))
.and(path("/api/generate"))
.respond_with(ResponseTemplate::new(200).set_body_json(json!({
"model": "llama3",
"response": "ok",
"done": true,
})))
.mount(&server)
.await;
let res = provider.unload_model("llama3").await.unwrap();
assert_eq!(res.model, "llama3");
assert_eq!(res.status, "unloaded");
}
#[tokio::test]
async fn test_unload_model_not_found() {
let (server, provider) = setup().await;
Mock::given(method("GET"))
.and(path("/api/tags"))
.respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
.mount(&server)
.await;
let err = provider.unload_model("gpt-4").await.unwrap_err();
assert!(matches!(err, chat::errors::OllamaError::ModelNotFound(_))); assert!(matches!(err, chat::errors::OllamaError::ModelNotFound(_)));
} }