Compare commits
36
Commits
46e455fa07
...
v1.0.0
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
eaf197f86e | ||
|
|
7b98fab2c8 | ||
|
|
d7ab571bbe | ||
|
|
87fcd522c4 | ||
|
|
5884d840e7 | ||
|
|
a631b8cd1e | ||
|
|
8bc5200f77 | ||
|
|
8878dbb454 | ||
|
|
ecf275dedd | ||
|
|
a6b9ee5b09 | ||
|
|
2e81132f2d | ||
|
|
c534299c8e | ||
|
|
4b428ec32a | ||
|
|
752373c7b7 | ||
|
|
d7ddc087a6 | ||
|
|
7f77bfef4e | ||
|
|
4e64aab0a5 | ||
|
|
4596dfe28b | ||
|
|
57a3d04625 | ||
|
|
81129d6c9c | ||
|
|
f4da65dbc2 | ||
|
|
e1f46c07b4 | ||
|
|
fc391b5d0e | ||
|
|
e31cdf131f | ||
|
|
d5856557b4 | ||
|
|
562d154480 | ||
|
|
a22560c337 | ||
|
|
b3ad5249c0 | ||
|
|
1a99490e22 | ||
|
|
686f9ff747 | ||
|
|
7e430bcf00 | ||
|
|
1a06577fa3 | ||
|
|
db14d816cb | ||
|
|
6dc7231309 | ||
|
|
70fe9bc4da | ||
|
|
fa09e75a09 |
@@ -1,2 +1,4 @@
|
|||||||
JWKS_URL=https://auth.iceberg.black/realms/iceberg/protocol/openid-connect/certs
|
JWKS_URL=https://auth.iceberg.black/realms/iceberg/protocol/openid-connect/certs
|
||||||
ISSUER=https://auth.iceberg.black/realms/iceberg
|
ISSUER=https://auth.iceberg.black/realms/iceberg
|
||||||
|
OLLAMA_URL=...
|
||||||
|
CORS_ORIGIN=
|
||||||
@@ -32,8 +32,10 @@ jobs:
|
|||||||
restore-keys: |
|
restore-keys: |
|
||||||
${{ runner.os }}-cargo-
|
${{ runner.os }}-cargo-
|
||||||
|
|
||||||
# - name: Run tests
|
- name: Run tests
|
||||||
# run: cargo test --workspace --verbose
|
env:
|
||||||
|
SQLX_OFFLINE: true
|
||||||
|
run: cargo test --workspace --verbose --all
|
||||||
|
|
||||||
- name: Run Clippy (linter)
|
- name: Run Clippy (linter)
|
||||||
run: cargo clippy --all-targets --all-features -- -D warnings
|
run: cargo clippy --all-targets --all-features -- -D warnings
|
||||||
|
|||||||
@@ -3,7 +3,7 @@ name: Publish & Deploy
|
|||||||
on:
|
on:
|
||||||
push:
|
push:
|
||||||
tags:
|
tags:
|
||||||
- 'v*'
|
- "v*"
|
||||||
|
|
||||||
permissions:
|
permissions:
|
||||||
contents: write
|
contents: write
|
||||||
@@ -34,6 +34,8 @@ jobs:
|
|||||||
|
|
||||||
- name: Build and Push
|
- name: Build and Push
|
||||||
uses: https://github.com/docker/build-push-action@v5
|
uses: https://github.com/docker/build-push-action@v5
|
||||||
|
env:
|
||||||
|
SQLX_OFFLINE: true
|
||||||
with:
|
with:
|
||||||
context: .
|
context: .
|
||||||
push: true
|
push: true
|
||||||
@@ -86,4 +88,3 @@ jobs:
|
|||||||
REPO_LOWER=$REPO_LOWER TAG=$TAG docker compose up -d
|
REPO_LOWER=$REPO_LOWER TAG=$TAG docker compose up -d
|
||||||
|
|
||||||
docker image prune -af
|
docker image prune -af
|
||||||
|
|
||||||
|
|||||||
+11
-4
@@ -12,30 +12,37 @@ repos:
|
|||||||
|
|
||||||
- id: clippy
|
- id: clippy
|
||||||
name: clippy
|
name: clippy
|
||||||
entry: cargo clippy -- -D warnings
|
entry: bash -c 'SQLX_OFFLINE=true cargo clippy -- -D warnings'
|
||||||
language: system
|
language: system
|
||||||
types: [rust]
|
types: [rust]
|
||||||
pass_filenames: false
|
pass_filenames: false
|
||||||
stages: [pre-commit]
|
stages: [pre-commit]
|
||||||
|
|
||||||
# ─────────────── PRE-PUSH (heavy checks) ───────────────
|
# ─────────────── PRE-PUSH (heavy checks) ───────────────
|
||||||
|
- id: sqlx-prepare-check
|
||||||
|
name: sqlx prepare check
|
||||||
|
entry: cargo sqlx prepare --check
|
||||||
|
language: system
|
||||||
|
pass_filenames: false
|
||||||
|
stages: [pre-push]
|
||||||
|
|
||||||
- id: test
|
- id: test
|
||||||
name: cargo test
|
name: cargo test
|
||||||
entry: cargo test --all
|
entry: bash -c 'SQLX_OFFLINE=true cargo test --all'
|
||||||
language: system
|
language: system
|
||||||
pass_filenames: false
|
pass_filenames: false
|
||||||
stages: [pre-push]
|
stages: [pre-push]
|
||||||
|
|
||||||
- id: build
|
- id: build
|
||||||
name: cargo build release
|
name: cargo build release
|
||||||
entry: cargo build --release
|
entry: bash -c 'SQLX_OFFLINE=true cargo build --release'
|
||||||
language: system
|
language: system
|
||||||
pass_filenames: false
|
pass_filenames: false
|
||||||
stages: [pre-push]
|
stages: [pre-push]
|
||||||
|
|
||||||
- id: audit
|
- id: audit
|
||||||
name: cargo audit
|
name: cargo audit
|
||||||
entry: cargo audit
|
entry: bash -c 'cargo audit || echo "cargo audit failed (non-blocking)"'
|
||||||
language: system
|
language: system
|
||||||
pass_filenames: false
|
pass_filenames: false
|
||||||
stages: [pre-push]
|
stages: [pre-push]
|
||||||
+17
@@ -0,0 +1,17 @@
|
|||||||
|
{
|
||||||
|
"db_name": "PostgreSQL",
|
||||||
|
"query": "\n INSERT INTO auth.api_key (key_hash, name, created_by, scopes)\n VALUES ($1, $2, $3, $4)\n ",
|
||||||
|
"describe": {
|
||||||
|
"columns": [],
|
||||||
|
"parameters": {
|
||||||
|
"Left": [
|
||||||
|
"Text",
|
||||||
|
"Text",
|
||||||
|
"Uuid",
|
||||||
|
"TextArray"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"nullable": []
|
||||||
|
},
|
||||||
|
"hash": "14b38b0f3fca61893dc6e036dc68f643a724b729f73f8308ae0b3a70d87dc5de"
|
||||||
|
}
|
||||||
+14
@@ -0,0 +1,14 @@
|
|||||||
|
{
|
||||||
|
"db_name": "PostgreSQL",
|
||||||
|
"query": "\n INSERT INTO auth.app_user (id)\n VALUES ($1)\n ON CONFLICT (id) DO NOTHING\n ",
|
||||||
|
"describe": {
|
||||||
|
"columns": [],
|
||||||
|
"parameters": {
|
||||||
|
"Left": [
|
||||||
|
"Uuid"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"nullable": []
|
||||||
|
},
|
||||||
|
"hash": "70117c16fc3efeebbcb40838d13dd427bf246f4eabd3e335077f47a7aa7882e4"
|
||||||
|
}
|
||||||
+14
@@ -0,0 +1,14 @@
|
|||||||
|
{
|
||||||
|
"db_name": "PostgreSQL",
|
||||||
|
"query": "\n UPDATE auth.api_key\n SET last_used_at = now()\n WHERE id = $1\n ",
|
||||||
|
"describe": {
|
||||||
|
"columns": [],
|
||||||
|
"parameters": {
|
||||||
|
"Left": [
|
||||||
|
"Uuid"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"nullable": []
|
||||||
|
},
|
||||||
|
"hash": "ab03067777ea2a8f12c3dec5cf99caf4ad700008d892a33c25408c34bebab83c"
|
||||||
|
}
|
||||||
+28
@@ -0,0 +1,28 @@
|
|||||||
|
{
|
||||||
|
"db_name": "PostgreSQL",
|
||||||
|
"query": "\n SELECT u.id AS user_id, ak.id as key_id\n FROM auth.api_key ak\n JOIN auth.app_user u ON u.id = ak.created_by\n WHERE ak.key_hash = $1\n AND ak.revoked_at IS NULL\n ",
|
||||||
|
"describe": {
|
||||||
|
"columns": [
|
||||||
|
{
|
||||||
|
"ordinal": 0,
|
||||||
|
"name": "user_id",
|
||||||
|
"type_info": "Uuid"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"ordinal": 1,
|
||||||
|
"name": "key_id",
|
||||||
|
"type_info": "Uuid"
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"parameters": {
|
||||||
|
"Left": [
|
||||||
|
"Text"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"nullable": [
|
||||||
|
false,
|
||||||
|
false
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"hash": "c0816078de6c25d441752500b9307900ff3d6c7781ebd9c0d8b47031539d8bdf"
|
||||||
|
}
|
||||||
Generated
+1485
-28
File diff suppressed because it is too large
Load Diff
+19
-2
@@ -3,12 +3,29 @@ name = "chat"
|
|||||||
version = "0.1.0"
|
version = "0.1.0"
|
||||||
edition = "2024"
|
edition = "2024"
|
||||||
|
|
||||||
|
[dev-dependencies]
|
||||||
|
wiremock = "0.6"
|
||||||
|
tokio = { version = "1.52.2", features = ["macros", "rt-multi-thread"] }
|
||||||
|
|
||||||
[dependencies]
|
[dependencies]
|
||||||
axum = "0.8.8"
|
axum = "0.8.9"
|
||||||
|
utoipa = { version = "5.5.0", features = ["axum_extras"] }
|
||||||
tokio = { version = "1", features = ["full"] }
|
tokio = { version = "1", features = ["full"] }
|
||||||
serde = { version = "1", features = ["derive"] }
|
serde = { version = "1", features = ["derive"] }
|
||||||
serde_json = "1"
|
serde_json = "1"
|
||||||
jsonwebtoken = { version = "10.3.0", features = ["aws_lc_rs"] }
|
jsonwebtoken = { version = "10.3.0", features = ["aws_lc_rs"] }
|
||||||
reqwest = { version = "0.13.2", features = ["json"] }
|
reqwest = { version = "0.13.3", features = ["json", "stream"] }
|
||||||
once_cell = "1"
|
once_cell = "1"
|
||||||
dotenvy = "0.15"
|
dotenvy = "0.15"
|
||||||
|
thiserror = "2.0.18"
|
||||||
|
tokio-stream = "0.1"
|
||||||
|
futures = "0.3"
|
||||||
|
chrono = { version = "0.4.44", features = ["serde"] }
|
||||||
|
uuid = { version = "1.23.1", features = ["v4", "serde"] }
|
||||||
|
tower-http = { version = "0.6.10", features = ["cors"] }
|
||||||
|
tracing = "0.1.44"
|
||||||
|
tracing-subscriber = { version = "0.3", features = ["env-filter"]}
|
||||||
|
sqlx = { version = "0.8.6", features = ["runtime-tokio-rustls", "postgres", "uuid", "chrono"] }
|
||||||
|
rand = "0.8"
|
||||||
|
base64 = "0.22.1"
|
||||||
|
sha2 = "0.11.0"
|
||||||
@@ -17,6 +17,7 @@ RUN rm -rf src
|
|||||||
|
|
||||||
# Build actual app
|
# Build actual app
|
||||||
COPY src ./src
|
COPY src ./src
|
||||||
|
COPY .sqlx ./.sqlx
|
||||||
RUN cargo build --release
|
RUN cargo build --release
|
||||||
|
|
||||||
# ── Runtime Stage ─────────────────────────────────────────────────────────────
|
# ── Runtime Stage ─────────────────────────────────────────────────────────────
|
||||||
|
|||||||
+1
-1
@@ -7,7 +7,7 @@ services:
|
|||||||
- .env
|
- .env
|
||||||
|
|
||||||
ports:
|
ports:
|
||||||
- "6066:3000"
|
- "6066:3001"
|
||||||
|
|
||||||
restart: always
|
restart: always
|
||||||
|
|
||||||
|
|||||||
@@ -3,9 +3,21 @@
|
|||||||
- Race condition on jwks token refresh
|
- Race condition on jwks token refresh
|
||||||
- Rate Limiting
|
- Rate Limiting
|
||||||
|
|
||||||
|
|
||||||
git tag -d v1.0.0; git push origin :refs/tags/v1.0.0; git tag -a v1.0.0 -m "Release v1.0.0"; git push origin v1.0.0
|
git tag -d v1.0.0; git push origin :refs/tags/v1.0.0; git tag -a v1.0.0 -m "Release v1.0.0"; git push origin v1.0.0
|
||||||
|
|
||||||
|
curl https://chat.iceberg.black/api/v1/models -H "Authorization: Bearer eyJhbGciOiJSUzI1NiIsInR5cCIgOiAiSldUIiwia2lkIiA6ICJublpLek04TkZHVmpWbGFPRXZpMUtFSTVHQWRwaGlsYjh3RHRLeG5JOENZIn0.eyJleHAiOjE3NzU4MTUyODksImlhdCI6MTc3NTgxNDk4OSwianRpIjoiMTgwYzA2NDUtYzZiMC00MDRmLTgyYjEtMzA3YmY4ZmJlNmJlIiwiaXNzIjoiaHR0cHM6Ly9hdXRoLmljZWJlcmcuYmxhY2svcmVhbG1zL2ljZWJlcmciLCJzdWIiOiJmZGRiN2FjZC1kMmE5LTRmMTctOWIxNi1kZjVlN2EzNDI4YjciLCJ0eXAiOiJCZWFyZXIiLCJhenAiOiJjaGF0LWFwaSIsInNjb3BlIjoiIiwiY2xpZW50SG9zdCI6Ijg2LjIxMi44NC4xOTEiLCJjbGllbnRBZGRyZXNzIjoiODYuMjEyLjg0LjE5MSIsImNsaWVudF9pZCI6ImNoYXQtYXBpIn0.abjHABcjCJNiB6vriRw60nfzabEfD7CXwyRhkahFkC8ATgfy4fn0T8PnFfsaRpsXuhamIhWwNskrA7L9V3mbWWKW-JEOvImDhTc8sIX0E5fDTbk8O5wa_2yzNdLpRxdSjLqgL544rB8I-LZ8bl5SxtdN3gHfrnWr5ef8bbLgPRzZIylT3QUpah0uywDM_cfrrve9SMHHOUUItyzmOHLw0Igit1EzyFNyjbWf6OAU6TMjOF_eFTc5sakyBwsdJnGy0nhj5R-wxLpr1ug3iEd3Y-jDzHO4m6jariXVJ7Vvbz71i7sadDEKdSyOKcCeQbU4T0Tf7IxqW4c2DNnk3BFFuw"
|
||||||
|
|
||||||
|
curl -X POST "https://auth.iceberg.black/realms/iceberg/protocol/openid-connect/token" -H "Content-Type: application/x-www-form-urlencoded" -d "grant_type=client_credentials" -d "client_id=chat-api" -d "client_secret=5fHUp8Z5GoNM70MVOGuQTfKFkaAE48Za"
|
||||||
|
|
||||||
|
curl -s -X POST https://chat.iceberg.black/api/v1/chat/completions \
|
||||||
|
-H "Content-Type: application/json" \
|
||||||
|
-H "Authorization: Bearer eyJhbGciOiJSUzI1NiIsInR5cCIgOiAiSldUIiwia2lkIiA6ICJublpLek04TkZHVmpWbGFPRXZpMUtFSTVHQWRwaGlsYjh3RHRLeG5JOENZIn0.eyJleHAiOjE3NzU4MTU0NTEsImlhdCI6MTc3NTgxNTE1MSwianRpIjoiOGNjMmM5MjUtZmMzNy00MDFhLWE1ZjUtODFlOGJhNjcxNzcwIiwiaXNzIjoiaHR0cHM6Ly9hdXRoLmljZWJlcmcuYmxhY2svcmVhbG1zL2ljZWJlcmciLCJzdWIiOiJmZGRiN2FjZC1kMmE5LTRmMTctOWIxNi1kZjVlN2EzNDI4YjciLCJ0eXAiOiJCZWFyZXIiLCJhenAiOiJjaGF0LWFwaSIsInNjb3BlIjoiIiwiY2xpZW50SG9zdCI6Ijg2LjIxMi44NC4xOTEiLCJjbGllbnRBZGRyZXNzIjoiODYuMjEyLjg0LjE5MSIsImNsaWVudF9pZCI6ImNoYXQtYXBpIn0.Exv34aoAqwzSqQziZH3zScAnK_xiNCXqpOQl54x7NFMdO0gVgacJgMxoW82Ym8aCNfBRMFtWclZJ9RFh1b9uSEUgsdVtqvMX-2kCAhiMSjmIBPU-L0gr5N63c3bScOoAYa37baR4mXQ5LxHjfkLo_rHDJ74adg2JMo359Wbuu1_OR708_q8yjgO_4fQeYbIffxADwRDvPSIwQ8Y2qRjaAIOZs0xmA9p128CUxlUyvhxwfFquYKRaDs5QE9pIAtvm_KuoBydYopm8j8cbm4ixDhUwYjkzniIIapY77NxpmMosfh8BhK0W3ieK2gTKfmiFDhYgkGybU6b1BnJMy1_Kxg" \
|
||||||
|
-d '{
|
||||||
|
"model": "llama3:latest",
|
||||||
|
"messages": [
|
||||||
|
{"role": "user", "content": "What is Rust?"}
|
||||||
|
]
|
||||||
|
}' | jq .
|
||||||
|
|
||||||
# 🦙 Ollama Rust API Wrapper
|
# 🦙 Ollama Rust API Wrapper
|
||||||
|
|
||||||
@@ -15,13 +27,13 @@ A high-performance Rust API wrapper around Ollama, providing an OpenAI-compatibl
|
|||||||
|
|
||||||
# 🚀 Features
|
# 🚀 Features
|
||||||
|
|
||||||
* ✅ OpenAI-compatible API (`/v1/...`)
|
- ✅ OpenAI-compatible API (`/v1/...`)
|
||||||
* ⚡ Streaming (Server-Sent Events)
|
- ⚡ Streaming (Server-Sent Events)
|
||||||
* 🧠 Model lifecycle management (load/unload)
|
- 🧠 Model lifecycle management (load/unload)
|
||||||
* 🔐 API key authentication (optional)
|
- 🔐 API key authentication (optional)
|
||||||
* 📊 Usage tracking & observability
|
- 📊 Usage tracking & observability
|
||||||
* 🔀 Model routing & abstraction
|
- 🔀 Model routing & abstraction
|
||||||
* 🧩 Extensible architecture (multi-provider ready)
|
- 🧩 Extensible architecture (multi-provider ready)
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
@@ -29,88 +41,20 @@ A high-performance Rust API wrapper around Ollama, providing an OpenAI-compatibl
|
|||||||
|
|
||||||
## 1. Core LLM API (OpenAI-compatible)
|
## 1. Core LLM API (OpenAI-compatible)
|
||||||
|
|
||||||
### Chat Completions
|
- [x] `POST /v1/chat/completions` + streaming
|
||||||
|
- [x] `POST /v1/completions` + streaming
|
||||||
```
|
- [ ] `POST /v1/embeddings`
|
||||||
POST /v1/chat/completions
|
- [x] `GET /v1/models`
|
||||||
```
|
|
||||||
|
|
||||||
### Text Completions
|
|
||||||
|
|
||||||
```
|
|
||||||
POST /v1/completions
|
|
||||||
```
|
|
||||||
|
|
||||||
### Embeddings
|
|
||||||
|
|
||||||
```
|
|
||||||
POST /v1/embeddings
|
|
||||||
```
|
|
||||||
|
|
||||||
### List Models
|
|
||||||
|
|
||||||
```
|
|
||||||
GET /v1/models
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## 2. Model Lifecycle Management
|
## 2. Model Lifecycle Management
|
||||||
|
|
||||||
### Load (Warmup)
|
- [x] `POST /v1/models/{model}/load`
|
||||||
|
- [x] `POST /v1/models/{model}/unload`
|
||||||
```
|
|
||||||
POST /v1/models/{model}/load
|
|
||||||
```
|
|
||||||
|
|
||||||
```json
|
|
||||||
{
|
|
||||||
"keep_alive": "10m"
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
### Unload (Free Memory)
|
|
||||||
|
|
||||||
```
|
|
||||||
POST /v1/models/{model}/unload
|
|
||||||
```
|
|
||||||
|
|
||||||
Internally uses:
|
|
||||||
|
|
||||||
```json
|
|
||||||
{
|
|
||||||
"model": "...",
|
|
||||||
"keep_alive": 0
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
### Reload (Optional)
|
|
||||||
|
|
||||||
```
|
|
||||||
POST /v1/models/{model}/reload
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## 3. Model Management
|
## 3. Model Management
|
||||||
|
|
||||||
### Pull Model
|
- [ ] `POST /v1/models/pull`
|
||||||
|
- [ ] `DELETE /v1/models/{model}`
|
||||||
```
|
|
||||||
POST /v1/models/pull
|
|
||||||
```
|
|
||||||
|
|
||||||
### Delete Model
|
|
||||||
|
|
||||||
```
|
|
||||||
DELETE /v1/models/{model}
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## 4. Runtime & Observability
|
## 4. Runtime & Observability
|
||||||
|
|
||||||
@@ -128,28 +72,6 @@ GET /v1/runtime/models
|
|||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
## 5. Streaming
|
|
||||||
|
|
||||||
Enable streaming with:
|
|
||||||
|
|
||||||
```json
|
|
||||||
{
|
|
||||||
"stream": true
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
Response format (SSE):
|
|
||||||
|
|
||||||
```
|
|
||||||
data: {"choices":[{"delta":{"content":"Hello"}}]}
|
|
||||||
|
|
||||||
data: {"choices":[{"delta":{"content":" world"}}]}
|
|
||||||
|
|
||||||
data: [DONE]
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## 6. Health Checks
|
## 6. Health Checks
|
||||||
|
|
||||||
```
|
```
|
||||||
@@ -173,19 +95,6 @@ GET /ready
|
|||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
# ⚙️ Configuration
|
|
||||||
|
|
||||||
### Docker (optional default)
|
|
||||||
|
|
||||||
```yaml
|
|
||||||
environment:
|
|
||||||
- OLLAMA_KEEP_ALIVE=10m
|
|
||||||
```
|
|
||||||
|
|
||||||
> Note: Request-level `keep_alive` overrides this value.
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
# 🔧 Advanced Features
|
# 🔧 Advanced Features
|
||||||
|
|
||||||
## 🔀 Model Routing
|
## 🔀 Model Routing
|
||||||
@@ -216,31 +125,16 @@ GET /v1/usage
|
|||||||
|
|
||||||
Tracks:
|
Tracks:
|
||||||
|
|
||||||
* request count
|
- request count
|
||||||
* latency
|
- latency
|
||||||
* per-model usage
|
- per-model usage
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## 🔐 Authentication
|
|
||||||
|
|
||||||
```
|
|
||||||
Authorization: Bearer sk-xxxx
|
|
||||||
```
|
|
||||||
|
|
||||||
Endpoints:
|
|
||||||
|
|
||||||
```
|
|
||||||
POST /v1/keys
|
|
||||||
DELETE /v1/keys/{id}
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
## 🚦 Rate Limiting
|
## 🚦 Rate Limiting
|
||||||
|
|
||||||
* Requests per minute
|
- Requests per minute
|
||||||
* Tokens per minute
|
- Tokens per minute
|
||||||
|
|
||||||
Returns:
|
Returns:
|
||||||
|
|
||||||
@@ -263,8 +157,8 @@ Stores conversation history server-side.
|
|||||||
|
|
||||||
## ⚡ Caching
|
## ⚡ Caching
|
||||||
|
|
||||||
* Embeddings
|
- Embeddings
|
||||||
* Deterministic prompts (temperature = 0)
|
- Deterministic prompts (temperature = 0)
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
@@ -301,8 +195,8 @@ POST /v1/runtime/evict
|
|||||||
|
|
||||||
Strategies:
|
Strategies:
|
||||||
|
|
||||||
* LRU
|
- LRU
|
||||||
* memory threshold
|
- memory threshold
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
@@ -312,17 +206,6 @@ Strategies:
|
|||||||
GET /v1/logs
|
GET /v1/logs
|
||||||
```
|
```
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## 📚 Embedding Store (Optional)
|
|
||||||
|
|
||||||
```
|
|
||||||
POST /v1/documents
|
|
||||||
POST /v1/search
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## 🔔 Async Jobs / Webhooks
|
## 🔔 Async Jobs / Webhooks
|
||||||
|
|
||||||
```
|
```
|
||||||
@@ -339,10 +222,10 @@ Client → Rust API → Ollama → Response
|
|||||||
|
|
||||||
### Layers:
|
### Layers:
|
||||||
|
|
||||||
* HTTP (Axum)
|
- HTTP (Axum)
|
||||||
* Service layer (business logic)
|
- Service layer (business logic)
|
||||||
* Provider abstraction
|
- Provider abstraction
|
||||||
* Ollama client
|
- Ollama client
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
@@ -357,27 +240,19 @@ trait LlmProvider {
|
|||||||
|
|
||||||
Supports:
|
Supports:
|
||||||
|
|
||||||
* Ollama (current)
|
- Ollama (current)
|
||||||
* OpenAI (future)
|
- OpenAI (future)
|
||||||
* Others
|
- Others
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
# ⚠️ Notes
|
|
||||||
|
|
||||||
* Ollama has no native unload endpoint → simulated via `keep_alive = 0`
|
|
||||||
* Streaming uses NDJSON → converted to SSE
|
|
||||||
* Chunk handling must be robust (partial JSON)
|
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
# 🎯 Roadmap
|
# 🎯 Roadmap
|
||||||
|
|
||||||
* [ ] Full OpenAI compatibility
|
- [ ] Full OpenAI compatibility
|
||||||
* [ ] Multi-node routing
|
- [ ] Multi-node routing
|
||||||
* [ ] GPU-aware scheduling
|
- [ ] GPU-aware scheduling
|
||||||
* [ ] Web UI dashboard
|
- [ ] Web UI dashboard
|
||||||
* [ ] Distributed inference
|
- [ ] Distributed inference
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
@@ -389,8 +264,6 @@ This project turns Ollama into:
|
|||||||
👉 A controllable model runtime
|
👉 A controllable model runtime
|
||||||
👉 A foundation for a full LLM gateway
|
👉 A foundation for a full LLM gateway
|
||||||
|
|
||||||
---
|
# TODO
|
||||||
|
- open api doc for bearer token
|
||||||
# 📜 License
|
- db
|
||||||
|
|
||||||
MIT
|
|
||||||
|
|||||||
@@ -1,59 +0,0 @@
|
|||||||
use jsonwebtoken::{Algorithm, DecodingKey, Validation, decode, decode_header};
|
|
||||||
use once_cell::sync::Lazy;
|
|
||||||
use serde::{Deserialize, Serialize};
|
|
||||||
use serde_json::Value;
|
|
||||||
use std::env;
|
|
||||||
|
|
||||||
#[derive(Debug, Deserialize, Serialize, Clone)]
|
|
||||||
pub struct Claims {
|
|
||||||
pub sub: String,
|
|
||||||
pub preferred_username: Option<String>,
|
|
||||||
pub exp: usize,
|
|
||||||
pub iss: String,
|
|
||||||
pub aud: Option<String>,
|
|
||||||
pub realm_access: Option<RealmAccess>,
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Debug, Deserialize, Serialize, Clone)]
|
|
||||||
pub struct RealmAccess {
|
|
||||||
pub roles: Vec<String>,
|
|
||||||
}
|
|
||||||
|
|
||||||
static ISSUER: Lazy<String> = Lazy::new(|| env::var("ISSUER").expect("ISSUER not set"));
|
|
||||||
|
|
||||||
pub fn validate_token(token: &str, jwks: &Value) -> Result<Claims, String> {
|
|
||||||
// 1. Decode header
|
|
||||||
let header = decode_header(token).map_err(|_| "Invalid header")?;
|
|
||||||
|
|
||||||
let kid = header.kid.ok_or("Missing kid")?;
|
|
||||||
|
|
||||||
// 2. Find matching key
|
|
||||||
let keys = jwks["keys"].as_array().ok_or("Invalid JWKS")?;
|
|
||||||
|
|
||||||
let key = keys
|
|
||||||
.iter()
|
|
||||||
.find(|k| k["kid"] == kid)
|
|
||||||
.ok_or("Matching key not found")?;
|
|
||||||
|
|
||||||
// 3. Extract RSA components
|
|
||||||
let n = key["n"].as_str().ok_or("Missing n")?;
|
|
||||||
let e = key["e"].as_str().ok_or("Missing e")?;
|
|
||||||
|
|
||||||
let decoding_key =
|
|
||||||
DecodingKey::from_rsa_components(n, e).map_err(|_| "Invalid decoding key")?;
|
|
||||||
|
|
||||||
// 4. Setup validation rules
|
|
||||||
let mut validation = Validation::new(Algorithm::RS256);
|
|
||||||
|
|
||||||
validation.set_issuer(&[ISSUER.as_str()]);
|
|
||||||
|
|
||||||
// Optional but recommended:
|
|
||||||
validation.validate_exp = true;
|
|
||||||
validation.validate_aud = false; // depends on your Keycloak config
|
|
||||||
|
|
||||||
// 5. Decode & verify
|
|
||||||
let token_data = decode::<Claims>(token, &decoding_key, &validation)
|
|
||||||
.map_err(|_| "Token validation failed")?;
|
|
||||||
|
|
||||||
Ok(token_data.claims)
|
|
||||||
}
|
|
||||||
@@ -1,57 +0,0 @@
|
|||||||
use once_cell::sync::Lazy;
|
|
||||||
use serde_json::Value;
|
|
||||||
use std::env;
|
|
||||||
use std::sync::Arc;
|
|
||||||
use std::time::{Duration, Instant};
|
|
||||||
use tokio::sync::RwLock;
|
|
||||||
|
|
||||||
#[derive(Clone)]
|
|
||||||
struct JwksCache {
|
|
||||||
jwks: Value,
|
|
||||||
last_fetched: Instant,
|
|
||||||
}
|
|
||||||
|
|
||||||
static JWK_CACHE: Lazy<Arc<RwLock<Option<JwksCache>>>> = Lazy::new(|| Arc::new(RwLock::new(None)));
|
|
||||||
|
|
||||||
static JWKS_URL: Lazy<String> = Lazy::new(|| env::var("JWKS_URL").expect("JWKS_URL not set"));
|
|
||||||
|
|
||||||
async fn fetch_jwks() -> Result<Value, reqwest::Error> {
|
|
||||||
let jwks = reqwest::get(JWKS_URL.as_str())
|
|
||||||
.await?
|
|
||||||
.json::<Value>()
|
|
||||||
.await?;
|
|
||||||
|
|
||||||
Ok(jwks)
|
|
||||||
}
|
|
||||||
|
|
||||||
pub async fn refresh_jwks() -> Result<Value, reqwest::Error> {
|
|
||||||
let jwks = fetch_jwks().await?;
|
|
||||||
|
|
||||||
let mut write = JWK_CACHE.write().await;
|
|
||||||
|
|
||||||
*write = Some(JwksCache {
|
|
||||||
jwks: jwks.clone(),
|
|
||||||
last_fetched: Instant::now(),
|
|
||||||
});
|
|
||||||
|
|
||||||
Ok(jwks)
|
|
||||||
}
|
|
||||||
|
|
||||||
pub async fn get_jwks() -> Result<Value, reqwest::Error> {
|
|
||||||
let ttl = Duration::from_secs(3600); // 1 hour
|
|
||||||
|
|
||||||
{
|
|
||||||
// Read lock first (fast path)
|
|
||||||
let read = JWK_CACHE.read().await;
|
|
||||||
|
|
||||||
if let Some(cache) = read
|
|
||||||
.as_ref()
|
|
||||||
.filter(|cache| cache.last_fetched.elapsed() < ttl)
|
|
||||||
{
|
|
||||||
return Ok(cache.jwks.clone());
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// Expired or empty → refresh
|
|
||||||
refresh_jwks().await
|
|
||||||
}
|
|
||||||
@@ -1,47 +0,0 @@
|
|||||||
use axum::{extract::Request, http::StatusCode, middleware::Next, response::Response};
|
|
||||||
|
|
||||||
use crate::auth::{
|
|
||||||
jwt::validate_token,
|
|
||||||
keycloak::{get_jwks, refresh_jwks},
|
|
||||||
};
|
|
||||||
|
|
||||||
pub async fn auth_middleware(mut request: Request, next: Next) -> Result<Response, StatusCode> {
|
|
||||||
dbg!("Middleware hit");
|
|
||||||
|
|
||||||
let headers = request.headers();
|
|
||||||
|
|
||||||
dbg!("Headers extracted");
|
|
||||||
|
|
||||||
let auth_header = headers.get("authorization").and_then(|v| v.to_str().ok());
|
|
||||||
|
|
||||||
dbg!("Auth header: {:?}", auth_header);
|
|
||||||
|
|
||||||
let auth_header = auth_header.ok_or(StatusCode::UNAUTHORIZED)?;
|
|
||||||
|
|
||||||
let token = auth_header
|
|
||||||
.strip_prefix("Bearer ")
|
|
||||||
.ok_or(StatusCode::UNAUTHORIZED)?;
|
|
||||||
|
|
||||||
let jwks = get_jwks().await.map_err(|_| StatusCode::UNAUTHORIZED)?;
|
|
||||||
|
|
||||||
match validate_token(token, &jwks) {
|
|
||||||
Ok(claims) => {
|
|
||||||
dbg!("Token valid");
|
|
||||||
|
|
||||||
request.extensions_mut().insert(claims);
|
|
||||||
|
|
||||||
Ok(next.run(request).await)
|
|
||||||
}
|
|
||||||
Err(_) => {
|
|
||||||
let jwks = refresh_jwks().await.map_err(|_| StatusCode::UNAUTHORIZED)?;
|
|
||||||
|
|
||||||
match validate_token(token, &jwks) {
|
|
||||||
Ok(claims) => {
|
|
||||||
request.extensions_mut().insert(claims);
|
|
||||||
Ok(next.run(request).await)
|
|
||||||
}
|
|
||||||
Err(_) => Err(StatusCode::UNAUTHORIZED),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -1,3 +0,0 @@
|
|||||||
pub mod jwt;
|
|
||||||
pub mod keycloak;
|
|
||||||
pub mod middleware;
|
|
||||||
@@ -0,0 +1 @@
|
|||||||
|
pub mod postgres;
|
||||||
@@ -0,0 +1,17 @@
|
|||||||
|
use sqlx::PgPool;
|
||||||
|
use uuid::Uuid;
|
||||||
|
|
||||||
|
pub async fn update_last_access(pool: &PgPool, api_key_id: Uuid) -> Result<(), sqlx::Error> {
|
||||||
|
sqlx::query!(
|
||||||
|
r#"
|
||||||
|
UPDATE auth.api_key
|
||||||
|
SET last_used_at = now()
|
||||||
|
WHERE id = $1
|
||||||
|
"#,
|
||||||
|
api_key_id
|
||||||
|
)
|
||||||
|
.execute(pool)
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
@@ -0,0 +1,3 @@
|
|||||||
|
pub mod api_key;
|
||||||
|
pub mod pool;
|
||||||
|
pub mod user_repository;
|
||||||
@@ -0,0 +1,14 @@
|
|||||||
|
use sqlx::{PgPool, postgres::PgPoolOptions};
|
||||||
|
use std::time::Duration;
|
||||||
|
|
||||||
|
pub async fn create_pool(database_url: &str) -> Result<PgPool, sqlx::Error> {
|
||||||
|
PgPoolOptions::new()
|
||||||
|
.max_connections(10)
|
||||||
|
.acquire_timeout(Duration::from_secs(5))
|
||||||
|
.connect(database_url)
|
||||||
|
.await
|
||||||
|
.map_err(|err| {
|
||||||
|
tracing::error!("Postgres connection error: {:?}", err);
|
||||||
|
err
|
||||||
|
})
|
||||||
|
}
|
||||||
@@ -0,0 +1,17 @@
|
|||||||
|
use sqlx::PgPool;
|
||||||
|
use uuid::Uuid;
|
||||||
|
|
||||||
|
pub async fn ensure_user_exists(pool: &PgPool, user_id: Uuid) -> Result<(), sqlx::Error> {
|
||||||
|
sqlx::query!(
|
||||||
|
r#"
|
||||||
|
INSERT INTO auth.app_user (id)
|
||||||
|
VALUES ($1)
|
||||||
|
ON CONFLICT (id) DO NOTHING
|
||||||
|
"#,
|
||||||
|
user_id
|
||||||
|
)
|
||||||
|
.execute(pool)
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
+55
@@ -0,0 +1,55 @@
|
|||||||
|
use utoipa::OpenApi;
|
||||||
|
|
||||||
|
use crate::dto::api;
|
||||||
|
use crate::routes;
|
||||||
|
|
||||||
|
#[derive(OpenApi)]
|
||||||
|
#[openapi(
|
||||||
|
info(
|
||||||
|
title = "Ollama Proxy",
|
||||||
|
description = "OpenAI-compatible proxy for local Ollama models",
|
||||||
|
version = "0.1.0",
|
||||||
|
license(
|
||||||
|
name = "MIT",
|
||||||
|
url = "https://opensource.org/licenses/MIT"
|
||||||
|
),
|
||||||
|
),
|
||||||
|
paths(
|
||||||
|
routes::v1::chat::completions,
|
||||||
|
routes::v1::chat::chat_completions,
|
||||||
|
routes::v1::models::list_models,
|
||||||
|
routes::v1::models::load_model,
|
||||||
|
routes::v1::models::unload_model,
|
||||||
|
),
|
||||||
|
components(
|
||||||
|
schemas(
|
||||||
|
api::ErrorResponse,
|
||||||
|
api::ModelsResponse,
|
||||||
|
api::ModelInfo,
|
||||||
|
api::LoadModelResponse,
|
||||||
|
api::LoadModelBody,
|
||||||
|
api::UnloadModelResponse,
|
||||||
|
api::BaseLLMRequest,
|
||||||
|
api::CompletionRequest,
|
||||||
|
api::CompletionObject,
|
||||||
|
api::FinishReason,
|
||||||
|
api::CompletionResponse,
|
||||||
|
api::Choice,
|
||||||
|
api::Usage,
|
||||||
|
api::CompletionChunk,
|
||||||
|
api::ChatRequest,
|
||||||
|
api::Message,
|
||||||
|
api::Role,
|
||||||
|
api::ChatCompletionResponse,
|
||||||
|
api::ChatChoice,
|
||||||
|
api::ChatCompletionChunk,
|
||||||
|
api::ChatChunkChoice,
|
||||||
|
api::ChatDelta,
|
||||||
|
)
|
||||||
|
),
|
||||||
|
tags(
|
||||||
|
(name = "chat", description = "Chat & completions"),
|
||||||
|
(name = "models", description = "Model management")
|
||||||
|
)
|
||||||
|
)]
|
||||||
|
pub struct ApiDoc;
|
||||||
+191
@@ -0,0 +1,191 @@
|
|||||||
|
use serde::{Deserialize, Serialize};
|
||||||
|
use utoipa::ToSchema;
|
||||||
|
|
||||||
|
#[derive(Debug, Serialize, ToSchema)]
|
||||||
|
pub struct ErrorResponse {
|
||||||
|
pub error: String,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl ErrorResponse {
|
||||||
|
pub fn new(msg: impl Into<String>) -> Self {
|
||||||
|
Self { error: msg.into() }
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Serialize, Deserialize, ToSchema)]
|
||||||
|
pub struct ModelsResponse {
|
||||||
|
pub models: Vec<ModelInfo>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Serialize, Deserialize, ToSchema)]
|
||||||
|
pub struct ModelInfo {
|
||||||
|
pub name: String,
|
||||||
|
pub family: Option<String>,
|
||||||
|
pub parameter_size: Option<String>,
|
||||||
|
pub quantization: Option<String>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Serialize, Deserialize, ToSchema)]
|
||||||
|
pub struct LoadModelResponse {
|
||||||
|
pub model: String,
|
||||||
|
pub status: String,
|
||||||
|
pub keep_alive: String,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Serialize, Deserialize, ToSchema)]
|
||||||
|
pub struct LoadModelBody {
|
||||||
|
pub keep_alive: Option<String>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Serialize, Deserialize, ToSchema)]
|
||||||
|
pub struct UnloadModelResponse {
|
||||||
|
pub model: String,
|
||||||
|
pub status: String,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Deserialize, Serialize, ToSchema, Default)]
|
||||||
|
pub struct BaseLLMRequest {
|
||||||
|
pub model: String,
|
||||||
|
|
||||||
|
#[serde(default)]
|
||||||
|
pub stream: bool,
|
||||||
|
|
||||||
|
pub temperature: Option<f32>,
|
||||||
|
pub top_p: Option<f32>,
|
||||||
|
|
||||||
|
// Ollama-native
|
||||||
|
pub top_k: Option<u32>,
|
||||||
|
pub repeat_penalty: Option<f32>,
|
||||||
|
pub seed: Option<i64>,
|
||||||
|
|
||||||
|
pub num_ctx: Option<u32>,
|
||||||
|
pub num_predict: Option<u32>,
|
||||||
|
|
||||||
|
pub stop: Option<Vec<String>>,
|
||||||
|
|
||||||
|
pub keep_alive: Option<String>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Deserialize, Serialize, ToSchema)]
|
||||||
|
pub struct CompletionRequest {
|
||||||
|
#[serde(flatten)]
|
||||||
|
pub base: BaseLLMRequest,
|
||||||
|
|
||||||
|
pub prompt: String,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Serialize, Deserialize, ToSchema, PartialEq, Eq)]
|
||||||
|
#[serde(rename_all = "snake_case")]
|
||||||
|
pub enum CompletionObject {
|
||||||
|
TextCompletion,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Serialize, Deserialize, ToSchema, PartialEq, Eq)]
|
||||||
|
#[serde(rename_all = "snake_case")]
|
||||||
|
pub enum FinishReason {
|
||||||
|
Stop,
|
||||||
|
Length,
|
||||||
|
ContentFilter,
|
||||||
|
ToolCalls,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Serialize, Deserialize, ToSchema)]
|
||||||
|
pub struct CompletionResponse {
|
||||||
|
pub id: String,
|
||||||
|
pub object: CompletionObject,
|
||||||
|
pub created: u64,
|
||||||
|
pub model: String,
|
||||||
|
pub choices: Vec<Choice>,
|
||||||
|
pub usage: Usage,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Serialize, Deserialize, ToSchema)]
|
||||||
|
pub struct Choice {
|
||||||
|
pub text: String,
|
||||||
|
pub index: u32,
|
||||||
|
pub finish_reason: FinishReason,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Serialize, Deserialize, ToSchema)]
|
||||||
|
pub struct Usage {
|
||||||
|
pub prompt_tokens: u32,
|
||||||
|
pub completion_tokens: u32,
|
||||||
|
pub total_tokens: u32,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Serialize, Deserialize, ToSchema)]
|
||||||
|
pub struct CompletionChunk {
|
||||||
|
pub id: String,
|
||||||
|
pub object: String,
|
||||||
|
pub choices: Vec<Choice>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Deserialize, Serialize, ToSchema)]
|
||||||
|
pub struct ChatRequest {
|
||||||
|
#[serde(flatten)]
|
||||||
|
pub base: BaseLLMRequest,
|
||||||
|
|
||||||
|
pub messages: Vec<Message>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Deserialize, Serialize, ToSchema)]
|
||||||
|
pub struct Message {
|
||||||
|
pub role: Role,
|
||||||
|
pub content: String,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Deserialize, Serialize, ToSchema, PartialEq, Eq)]
|
||||||
|
#[serde(rename_all = "lowercase")]
|
||||||
|
pub enum Role {
|
||||||
|
System,
|
||||||
|
User,
|
||||||
|
Assistant,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Serialize, Deserialize, ToSchema)]
|
||||||
|
pub struct ChatCompletionResponse {
|
||||||
|
pub id: String,
|
||||||
|
pub object: String,
|
||||||
|
pub created: u64,
|
||||||
|
pub model: String,
|
||||||
|
pub choices: Vec<ChatChoice>,
|
||||||
|
pub usage: Option<Usage>, // optional (Ollama may not always provide)
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Serialize, Deserialize, ToSchema)]
|
||||||
|
pub struct ChatChoice {
|
||||||
|
pub index: u32,
|
||||||
|
pub message: Message,
|
||||||
|
pub finish_reason: FinishReason,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Serialize, ToSchema)]
|
||||||
|
pub struct ChatCompletionChunk {
|
||||||
|
pub id: String,
|
||||||
|
pub object: String,
|
||||||
|
pub choices: Vec<ChatChunkChoice>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Serialize, ToSchema)]
|
||||||
|
pub struct ChatChunkChoice {
|
||||||
|
pub index: u32,
|
||||||
|
pub delta: ChatDelta,
|
||||||
|
pub finish_reason: Option<FinishReason>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Serialize, ToSchema)]
|
||||||
|
pub struct ChatDelta {
|
||||||
|
pub role: Option<Role>,
|
||||||
|
pub content: Option<String>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(serde::Deserialize)]
|
||||||
|
pub struct CreateApiKeyRequest {
|
||||||
|
pub name: String,
|
||||||
|
pub scopes: Vec<String>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(serde::Serialize)]
|
||||||
|
pub struct CreateApiKeyResponse {
|
||||||
|
pub api_key: String, // ONLY returned once
|
||||||
|
}
|
||||||
@@ -0,0 +1,2 @@
|
|||||||
|
pub mod api;
|
||||||
|
pub mod ollama;
|
||||||
@@ -0,0 +1,80 @@
|
|||||||
|
use serde::{Deserialize, Serialize};
|
||||||
|
|
||||||
|
use crate::dto::api;
|
||||||
|
|
||||||
|
#[derive(Debug, Serialize, Deserialize)]
|
||||||
|
pub struct OllamaModels {
|
||||||
|
pub models: Vec<OllamaModel>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Serialize, Deserialize)]
|
||||||
|
pub struct OllamaModel {
|
||||||
|
pub name: String,
|
||||||
|
|
||||||
|
pub details: Option<OllamaModelDetails>,
|
||||||
|
|
||||||
|
pub size: Option<u64>,
|
||||||
|
pub digest: Option<String>,
|
||||||
|
pub modified_at: Option<String>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Serialize, Deserialize)]
|
||||||
|
pub struct OllamaModelDetails {
|
||||||
|
pub family: Option<String>,
|
||||||
|
pub parameter_size: Option<String>,
|
||||||
|
pub quantization_level: Option<String>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Serialize, Deserialize)]
|
||||||
|
pub struct OllamaOptions {
|
||||||
|
pub temperature: Option<f32>,
|
||||||
|
pub top_p: Option<f32>,
|
||||||
|
pub top_k: Option<u32>,
|
||||||
|
pub repeat_penalty: Option<f32>,
|
||||||
|
pub seed: Option<i64>,
|
||||||
|
|
||||||
|
pub num_ctx: Option<u32>,
|
||||||
|
pub num_predict: Option<u32>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Serialize)]
|
||||||
|
pub struct OllamaGenerateRequest<'a> {
|
||||||
|
pub model: &'a str,
|
||||||
|
pub prompt: &'a str,
|
||||||
|
pub stream: bool,
|
||||||
|
pub options: OllamaOptions,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Serialize, Deserialize)]
|
||||||
|
pub struct OllamaGenerateResponse {
|
||||||
|
pub model: String,
|
||||||
|
pub created_at: Option<String>,
|
||||||
|
pub response: String,
|
||||||
|
pub done: bool,
|
||||||
|
|
||||||
|
#[serde(default)]
|
||||||
|
pub context: Option<Vec<u64>>,
|
||||||
|
|
||||||
|
pub total_duration: Option<u64>,
|
||||||
|
pub load_duration: Option<u64>,
|
||||||
|
pub prompt_eval_count: Option<u32>,
|
||||||
|
pub eval_count: Option<u32>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Serialize)]
|
||||||
|
pub struct OllamaChatRequest<'a> {
|
||||||
|
pub model: &'a str,
|
||||||
|
pub messages: &'a [api::Message],
|
||||||
|
pub stream: bool,
|
||||||
|
pub options: OllamaOptions,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Deserialize)]
|
||||||
|
pub struct OllamaChatResponse {
|
||||||
|
pub model: String,
|
||||||
|
pub message: api::Message,
|
||||||
|
pub done: bool,
|
||||||
|
|
||||||
|
pub prompt_eval_count: Option<u32>,
|
||||||
|
pub eval_count: Option<u32>,
|
||||||
|
}
|
||||||
@@ -0,0 +1,28 @@
|
|||||||
|
// errors.rs
|
||||||
|
use thiserror::Error;
|
||||||
|
|
||||||
|
#[derive(Debug, Error)]
|
||||||
|
pub enum OllamaError {
|
||||||
|
#[error("prompt is required and cannot be empty")]
|
||||||
|
MissingPrompt,
|
||||||
|
|
||||||
|
#[error("model is required and cannot be empty")]
|
||||||
|
MissingModel,
|
||||||
|
|
||||||
|
#[error("messages must be a non-empty array containing at least one user message")]
|
||||||
|
MissingMessages,
|
||||||
|
|
||||||
|
#[error("model '{0}' is not available — run `ollama pull {0}` first")]
|
||||||
|
ModelNotFound(String),
|
||||||
|
|
||||||
|
#[error("keep_alive is required and cannot be empty")]
|
||||||
|
MissingKeepAlive,
|
||||||
|
|
||||||
|
#[error(
|
||||||
|
"invalid keep_alive format '{0}' — expected <number><unit> (e.g. 30s, 10m, 2h), a plain integer (seconds), or -1"
|
||||||
|
)]
|
||||||
|
InvalidKeepAlive(String),
|
||||||
|
|
||||||
|
#[error(transparent)]
|
||||||
|
Http(#[from] reqwest::Error),
|
||||||
|
}
|
||||||
@@ -0,0 +1,7 @@
|
|||||||
|
pub mod databases;
|
||||||
|
pub mod dto;
|
||||||
|
pub mod errors;
|
||||||
|
pub mod middlewares;
|
||||||
|
pub mod providers;
|
||||||
|
pub mod state;
|
||||||
|
pub mod utils;
|
||||||
+56
-6
@@ -1,14 +1,33 @@
|
|||||||
mod auth;
|
mod databases;
|
||||||
|
mod docs;
|
||||||
|
mod dto;
|
||||||
|
mod errors;
|
||||||
|
mod middlewares;
|
||||||
mod providers;
|
mod providers;
|
||||||
mod routes;
|
mod routes;
|
||||||
mod state;
|
mod state;
|
||||||
|
mod utils;
|
||||||
|
|
||||||
use crate::providers::ollama::OllamaProvider;
|
use crate::databases::postgres;
|
||||||
|
use crate::providers::ollama::client::OllamaProvider;
|
||||||
use crate::state::app_state::AppState;
|
use crate::state::app_state::AppState;
|
||||||
|
|
||||||
use axum::Router;
|
use axum::Router;
|
||||||
|
use axum::http::{HeaderName, HeaderValue, Method, header};
|
||||||
|
use once_cell::sync::Lazy;
|
||||||
|
use std::env;
|
||||||
use std::net::SocketAddr;
|
use std::net::SocketAddr;
|
||||||
use std::sync::Arc;
|
use std::sync::Arc;
|
||||||
|
use tower_http::cors::CorsLayer;
|
||||||
|
use tracing_subscriber::{EnvFilter, fmt};
|
||||||
|
|
||||||
|
static OLLAMA_URL: Lazy<String> = Lazy::new(|| env::var("OLLAMA_URL").expect("OLLAMA_URL not set"));
|
||||||
|
|
||||||
|
pub fn init_tracing() {
|
||||||
|
let filter = env::var("RUST_LOG").unwrap_or_else(|_| "info".to_string());
|
||||||
|
|
||||||
|
fmt().with_env_filter(EnvFilter::new(filter)).init();
|
||||||
|
}
|
||||||
|
|
||||||
#[tokio::main]
|
#[tokio::main]
|
||||||
async fn main() {
|
async fn main() {
|
||||||
@@ -17,16 +36,47 @@ async fn main() {
|
|||||||
dotenvy::dotenv().ok();
|
dotenvy::dotenv().ok();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
init_tracing();
|
||||||
|
|
||||||
|
// DB Connection
|
||||||
|
let database_url = env::var("DATABASE_URL").expect("DATABASE_URL must be set");
|
||||||
|
|
||||||
|
let pool = postgres::pool::create_pool(&database_url)
|
||||||
|
.await
|
||||||
|
.expect("Fatal error");
|
||||||
|
|
||||||
let state = AppState {
|
let state = AppState {
|
||||||
ollama: Arc::new(OllamaProvider::new("http://localhost:11434")),
|
ollama: Arc::new(OllamaProvider::new(OLLAMA_URL.as_str())),
|
||||||
|
postgres: pool,
|
||||||
};
|
};
|
||||||
|
|
||||||
|
let cors_origin =
|
||||||
|
env::var("CORS_ORIGIN").unwrap_or_else(|_| "http://localhost:3000".to_string());
|
||||||
|
|
||||||
|
let cors = CorsLayer::new()
|
||||||
|
.allow_origin(cors_origin.parse::<HeaderValue>().unwrap())
|
||||||
|
.allow_methods([
|
||||||
|
Method::GET,
|
||||||
|
Method::POST,
|
||||||
|
Method::PUT,
|
||||||
|
Method::DELETE,
|
||||||
|
Method::OPTIONS,
|
||||||
|
])
|
||||||
|
.allow_headers([
|
||||||
|
header::CONTENT_TYPE,
|
||||||
|
header::AUTHORIZATION,
|
||||||
|
header::ACCEPT,
|
||||||
|
HeaderName::from_static("x-api-key"),
|
||||||
|
])
|
||||||
|
.allow_credentials(true);
|
||||||
|
|
||||||
let app = Router::new()
|
let app = Router::new()
|
||||||
.nest("/v1", routes::v1::router())
|
.nest("/v1", routes::v1::router(state.clone()))
|
||||||
|
.layer(cors)
|
||||||
.with_state(state);
|
.with_state(state);
|
||||||
|
|
||||||
let addr = SocketAddr::from(([0, 0, 0, 0], 3000));
|
let addr = SocketAddr::from(([0, 0, 0, 0], 3001));
|
||||||
println!("Server running on {}", addr);
|
tracing::debug!("Server running on {}", addr);
|
||||||
|
|
||||||
axum::serve(tokio::net::TcpListener::bind(addr).await.unwrap(), app)
|
axum::serve(tokio::net::TcpListener::bind(addr).await.unwrap(), app)
|
||||||
.await
|
.await
|
||||||
|
|||||||
@@ -0,0 +1,7 @@
|
|||||||
|
use uuid::Uuid;
|
||||||
|
|
||||||
|
#[derive(Clone, Debug)]
|
||||||
|
pub struct ApiKeyClaims {
|
||||||
|
pub sub: Uuid,
|
||||||
|
pub api_key_id: Uuid,
|
||||||
|
}
|
||||||
@@ -0,0 +1,148 @@
|
|||||||
|
use jsonwebtoken::{Algorithm, DecodingKey, Validation, decode, decode_header};
|
||||||
|
use once_cell::sync::Lazy;
|
||||||
|
use serde::{Deserialize, Serialize};
|
||||||
|
use serde_json::Value;
|
||||||
|
use std::collections::HashMap;
|
||||||
|
use std::env;
|
||||||
|
use std::sync::Arc;
|
||||||
|
use std::time::{Duration, Instant};
|
||||||
|
use tokio::sync::RwLock;
|
||||||
|
|
||||||
|
// ------ JWKS ------
|
||||||
|
|
||||||
|
#[derive(Clone)]
|
||||||
|
struct JwksCache {
|
||||||
|
jwks: Value,
|
||||||
|
last_fetched: Instant,
|
||||||
|
}
|
||||||
|
|
||||||
|
static JWK_CACHE: Lazy<Arc<RwLock<Option<JwksCache>>>> = Lazy::new(|| Arc::new(RwLock::new(None)));
|
||||||
|
|
||||||
|
static JWKS_URL: Lazy<String> = Lazy::new(|| env::var("JWKS_URL").expect("JWKS_URL not set"));
|
||||||
|
|
||||||
|
async fn fetch_jwks() -> Result<Value, reqwest::Error> {
|
||||||
|
let jwks = reqwest::get(JWKS_URL.as_str())
|
||||||
|
.await?
|
||||||
|
.json::<Value>()
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
Ok(jwks)
|
||||||
|
}
|
||||||
|
|
||||||
|
pub async fn refresh_jwks() -> Result<Value, reqwest::Error> {
|
||||||
|
let jwks = fetch_jwks().await?;
|
||||||
|
|
||||||
|
let mut write = JWK_CACHE.write().await;
|
||||||
|
|
||||||
|
*write = Some(JwksCache {
|
||||||
|
jwks: jwks.clone(),
|
||||||
|
last_fetched: Instant::now(),
|
||||||
|
});
|
||||||
|
|
||||||
|
Ok(jwks)
|
||||||
|
}
|
||||||
|
|
||||||
|
pub async fn get_jwks() -> Result<Value, reqwest::Error> {
|
||||||
|
let ttl = Duration::from_secs(3600); // 1 hour
|
||||||
|
|
||||||
|
{
|
||||||
|
// Read lock first (fast path)
|
||||||
|
let read = JWK_CACHE.read().await;
|
||||||
|
|
||||||
|
if let Some(cache) = read
|
||||||
|
.as_ref()
|
||||||
|
.filter(|cache| cache.last_fetched.elapsed() < ttl)
|
||||||
|
{
|
||||||
|
return Ok(cache.jwks.clone());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Expired or empty → refresh
|
||||||
|
refresh_jwks().await
|
||||||
|
}
|
||||||
|
|
||||||
|
// ------ Claims ------
|
||||||
|
|
||||||
|
#[derive(Debug, Deserialize, Serialize, Clone, Default)]
|
||||||
|
pub struct KeycloakClaims {
|
||||||
|
pub sub: String,
|
||||||
|
pub preferred_username: Option<String>,
|
||||||
|
pub exp: usize,
|
||||||
|
pub iss: String,
|
||||||
|
pub aud: Option<Vec<String>>,
|
||||||
|
pub realm_access: Option<RealmAccess>,
|
||||||
|
#[serde(default)]
|
||||||
|
pub resource_access: HashMap<String, ResourceAccess>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Deserialize, Serialize, Clone)]
|
||||||
|
pub struct RealmAccess {
|
||||||
|
pub roles: Vec<String>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Deserialize, Serialize, Clone)]
|
||||||
|
pub struct ResourceAccess {
|
||||||
|
pub roles: Vec<String>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl KeycloakClaims {
|
||||||
|
pub fn realm_roles(&self) -> &[String] {
|
||||||
|
self.realm_access
|
||||||
|
.as_ref()
|
||||||
|
.map_or(&[], |r| r.roles.as_slice())
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn has_realm_role(&self, role: &str) -> bool {
|
||||||
|
self.realm_roles().iter().any(|r| r == role)
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn client_roles(&self, client: &str) -> &[String] {
|
||||||
|
self.resource_access
|
||||||
|
.get(client)
|
||||||
|
.map_or(&[], |r| r.roles.as_slice())
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn has_client_role(&self, client: &str, role: &str) -> bool {
|
||||||
|
self.client_roles(client).iter().any(|r| r == role)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ------ Validation ------
|
||||||
|
|
||||||
|
static ISSUER: Lazy<String> = Lazy::new(|| env::var("ISSUER").expect("ISSUER not set"));
|
||||||
|
|
||||||
|
pub fn validate_token(token: &str, jwks: &Value) -> Result<KeycloakClaims, String> {
|
||||||
|
// 1. Decode header
|
||||||
|
let header = decode_header(token).map_err(|_| "Invalid header")?;
|
||||||
|
|
||||||
|
let kid = header.kid.ok_or("Missing kid")?;
|
||||||
|
|
||||||
|
// 2. Find matching key
|
||||||
|
let keys = jwks["keys"].as_array().ok_or("Invalid JWKS")?;
|
||||||
|
|
||||||
|
let key = keys
|
||||||
|
.iter()
|
||||||
|
.find(|k| k["kid"] == kid)
|
||||||
|
.ok_or("Matching key not found")?;
|
||||||
|
|
||||||
|
// 3. Extract RSA components
|
||||||
|
let n = key["n"].as_str().ok_or("Missing n")?;
|
||||||
|
let e = key["e"].as_str().ok_or("Missing e")?;
|
||||||
|
|
||||||
|
let decoding_key =
|
||||||
|
DecodingKey::from_rsa_components(n, e).map_err(|_| "Invalid decoding key")?;
|
||||||
|
|
||||||
|
// 4. Setup validation rules
|
||||||
|
let mut validation = Validation::new(Algorithm::RS256);
|
||||||
|
|
||||||
|
validation.set_issuer(&[ISSUER.as_str()]);
|
||||||
|
|
||||||
|
validation.validate_exp = true;
|
||||||
|
validation.validate_aud = false;
|
||||||
|
|
||||||
|
// 5. Decode & verify
|
||||||
|
let token_data = decode::<KeycloakClaims>(token, &decoding_key, &validation)
|
||||||
|
.map_err(|_| "Token validation failed")?;
|
||||||
|
|
||||||
|
Ok(token_data.claims)
|
||||||
|
}
|
||||||
@@ -0,0 +1,215 @@
|
|||||||
|
use axum::{
|
||||||
|
extract::{Request, State},
|
||||||
|
http::StatusCode,
|
||||||
|
middleware::Next,
|
||||||
|
response::{IntoResponse, Response},
|
||||||
|
};
|
||||||
|
|
||||||
|
use crate::databases::postgres::{
|
||||||
|
api_key::update_last_access, user_repository::ensure_user_exists,
|
||||||
|
};
|
||||||
|
use crate::middlewares::auth::apikey::ApiKeyClaims;
|
||||||
|
use crate::middlewares::auth::keycloak::{KeycloakClaims, get_jwks, refresh_jwks, validate_token};
|
||||||
|
use crate::state::app_state::AppState;
|
||||||
|
use crate::utils::crypto::hash_key;
|
||||||
|
use uuid::Uuid;
|
||||||
|
|
||||||
|
#[derive(Clone, Debug)]
|
||||||
|
pub enum Auth {
|
||||||
|
Jwt(KeycloakClaims),
|
||||||
|
ApiKey(ApiKeyClaims),
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Auth {
|
||||||
|
pub fn user_id(&self) -> Uuid {
|
||||||
|
match self {
|
||||||
|
Auth::Jwt(c) => c.sub.parse().expect("sub is a valid UUID"),
|
||||||
|
Auth::ApiKey(c) => c.sub,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn has_realm_role(&self, role: &str) -> bool {
|
||||||
|
match self {
|
||||||
|
Auth::Jwt(c) => c.has_realm_role(role),
|
||||||
|
Auth::ApiKey(_) => false, // API keys carry no roles
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn has_client_role(&self, client: &str, role: &str) -> bool {
|
||||||
|
match self {
|
||||||
|
Auth::Jwt(c) => c.has_client_role(client, role),
|
||||||
|
Auth::ApiKey(_) => false,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub async fn auth_middleware(
|
||||||
|
State(state): State<AppState>,
|
||||||
|
request: Request,
|
||||||
|
next: Next,
|
||||||
|
) -> Result<Response, StatusCode> {
|
||||||
|
match try_jwt(&state, request, next).await {
|
||||||
|
Ok(response) => Ok(response),
|
||||||
|
Err((request, next)) => try_api_key(&state, request, next).await,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Returns Ok(Response) if JWT was valid and request handled.
|
||||||
|
/// Returns Err((request, next)) if no JWT was present (caller should try next method).
|
||||||
|
/// Returns a 401/500 response directly if JWT was present but invalid.
|
||||||
|
async fn try_jwt(
|
||||||
|
state: &AppState,
|
||||||
|
request: Request,
|
||||||
|
next: Next,
|
||||||
|
) -> Result<Response, (Request, Next)> {
|
||||||
|
let token = request
|
||||||
|
.headers()
|
||||||
|
.get("authorization")
|
||||||
|
.and_then(|v| v.to_str().ok())
|
||||||
|
.and_then(|v| v.strip_prefix("Bearer "))
|
||||||
|
.map(str::to_owned);
|
||||||
|
|
||||||
|
let Some(token) = token else {
|
||||||
|
// No Authorization header at all → let API key branch try
|
||||||
|
return Err((request, next));
|
||||||
|
};
|
||||||
|
|
||||||
|
let jwks = match get_jwks().await {
|
||||||
|
Ok(j) => j,
|
||||||
|
Err(e) => {
|
||||||
|
tracing::error!("Failed to fetch JWKS: {e}");
|
||||||
|
// Token was present but we can't validate → hard 500
|
||||||
|
return Ok(StatusCode::INTERNAL_SERVER_ERROR.into_response());
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
let claims = match validate_token(&token, &jwks) {
|
||||||
|
Ok(c) => c,
|
||||||
|
Err(_) => {
|
||||||
|
// Try refreshing JWKS once
|
||||||
|
match refresh_jwks().await {
|
||||||
|
Ok(fresh_jwks) => match validate_token(&token, &fresh_jwks) {
|
||||||
|
Ok(c) => c,
|
||||||
|
Err(_) => {
|
||||||
|
tracing::warn!("JWT validation failed after JWKS refresh");
|
||||||
|
return Ok(StatusCode::UNAUTHORIZED.into_response());
|
||||||
|
}
|
||||||
|
},
|
||||||
|
Err(e) => {
|
||||||
|
tracing::error!("Failed to refresh JWKS: {e}");
|
||||||
|
return Ok(StatusCode::INTERNAL_SERVER_ERROR.into_response());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
tracing::debug!("JWT valid, sub={}", claims.sub);
|
||||||
|
handle_auth(state, request, next, Auth::Jwt(claims))
|
||||||
|
.await
|
||||||
|
.map_err(|_| unreachable!())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Returns Ok(Response) if API key was valid.
|
||||||
|
/// Returns Err(StatusCode) otherwise (UNAUTHORIZED or INTERNAL_SERVER_ERROR).
|
||||||
|
async fn try_api_key(
|
||||||
|
state: &AppState,
|
||||||
|
request: Request,
|
||||||
|
next: Next,
|
||||||
|
) -> Result<Response, StatusCode> {
|
||||||
|
let key = request
|
||||||
|
.headers()
|
||||||
|
.get("x-api-key")
|
||||||
|
.and_then(|v| v.to_str().ok())
|
||||||
|
.ok_or(StatusCode::UNAUTHORIZED)?
|
||||||
|
.to_owned();
|
||||||
|
|
||||||
|
let key_hash = hash_key(&key);
|
||||||
|
|
||||||
|
let row = sqlx::query!(
|
||||||
|
r#"
|
||||||
|
SELECT u.id AS user_id, ak.id as key_id
|
||||||
|
FROM auth.api_key ak
|
||||||
|
JOIN auth.app_user u ON u.id = ak.created_by
|
||||||
|
WHERE ak.key_hash = $1
|
||||||
|
AND ak.revoked_at IS NULL
|
||||||
|
"#,
|
||||||
|
key_hash
|
||||||
|
)
|
||||||
|
.fetch_optional(&state.postgres)
|
||||||
|
.await
|
||||||
|
.map_err(|e| {
|
||||||
|
tracing::error!("DB error during API key lookup: {e}");
|
||||||
|
StatusCode::INTERNAL_SERVER_ERROR
|
||||||
|
})?
|
||||||
|
.ok_or(StatusCode::UNAUTHORIZED)?;
|
||||||
|
|
||||||
|
tracing::debug!("API key valid, user_id={}", row.user_id);
|
||||||
|
|
||||||
|
handle_auth(
|
||||||
|
state,
|
||||||
|
request,
|
||||||
|
next,
|
||||||
|
Auth::ApiKey(ApiKeyClaims {
|
||||||
|
sub: row.user_id,
|
||||||
|
api_key_id: row.key_id,
|
||||||
|
}),
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
}
|
||||||
|
|
||||||
|
// ── Shared post-auth logic ────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
/// Ensures the user exists in the DB, inserts `Auth` into extensions, runs the next handler.
|
||||||
|
async fn handle_auth(
|
||||||
|
state: &AppState,
|
||||||
|
mut request: Request,
|
||||||
|
next: Next,
|
||||||
|
auth: Auth,
|
||||||
|
) -> Result<Response, StatusCode> {
|
||||||
|
match &auth {
|
||||||
|
Auth::Jwt(_) => {
|
||||||
|
ensure_user_exists(&state.postgres, auth.user_id())
|
||||||
|
.await
|
||||||
|
.map_err(|e| {
|
||||||
|
tracing::error!("ensure_user_exists failed: {e}");
|
||||||
|
StatusCode::INTERNAL_SERVER_ERROR
|
||||||
|
})?;
|
||||||
|
}
|
||||||
|
Auth::ApiKey(key) => {
|
||||||
|
update_last_access(&state.postgres, key.api_key_id)
|
||||||
|
.await
|
||||||
|
.map_err(|e| {
|
||||||
|
tracing::error!("update_last_access failed: {e}");
|
||||||
|
StatusCode::INTERNAL_SERVER_ERROR
|
||||||
|
})?;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
request.extensions_mut().insert(auth);
|
||||||
|
Ok(next.run(request).await)
|
||||||
|
}
|
||||||
|
|
||||||
|
// ── Role guard ───────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
/// Layer-level middleware that checks roles *after* `auth_middleware` has run.
|
||||||
|
pub async fn require_roles(
|
||||||
|
request: Request,
|
||||||
|
next: Next,
|
||||||
|
realm_role: Option<&'static str>,
|
||||||
|
client_role: Option<&'static str>,
|
||||||
|
) -> Result<Response, StatusCode> {
|
||||||
|
let auth = request
|
||||||
|
.extensions()
|
||||||
|
.get::<Auth>()
|
||||||
|
.ok_or(StatusCode::UNAUTHORIZED)?;
|
||||||
|
|
||||||
|
if realm_role.is_some_and(|role| !auth.has_realm_role(role)) {
|
||||||
|
return Err(StatusCode::FORBIDDEN);
|
||||||
|
}
|
||||||
|
|
||||||
|
if client_role.is_some_and(|role| !auth.has_client_role("chat-api", role)) {
|
||||||
|
return Err(StatusCode::FORBIDDEN);
|
||||||
|
}
|
||||||
|
|
||||||
|
Ok(next.run(request).await)
|
||||||
|
}
|
||||||
@@ -0,0 +1,5 @@
|
|||||||
|
pub mod apikey;
|
||||||
|
pub mod keycloak;
|
||||||
|
pub mod middleware;
|
||||||
|
|
||||||
|
pub use middleware::auth_middleware;
|
||||||
@@ -0,0 +1 @@
|
|||||||
|
pub mod auth;
|
||||||
@@ -1,25 +0,0 @@
|
|||||||
use reqwest::Client;
|
|
||||||
use serde_json::Value;
|
|
||||||
|
|
||||||
#[derive(Clone)]
|
|
||||||
pub struct OllamaProvider {
|
|
||||||
pub client: Client,
|
|
||||||
pub base_url: String,
|
|
||||||
}
|
|
||||||
|
|
||||||
impl OllamaProvider {
|
|
||||||
pub fn new(base_url: impl Into<String>) -> Self {
|
|
||||||
Self {
|
|
||||||
client: Client::new(),
|
|
||||||
base_url: base_url.into(),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
pub async fn list_models(&self) -> Result<Value, reqwest::Error> {
|
|
||||||
let url = format!("{}/api/tags", self.base_url);
|
|
||||||
|
|
||||||
let res = self.client.get(url).send().await?.json::<Value>().await?;
|
|
||||||
|
|
||||||
Ok(res)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -0,0 +1,432 @@
|
|||||||
|
use crate::dto::{api, ollama};
|
||||||
|
use crate::errors::OllamaError;
|
||||||
|
use axum::response::sse::Event;
|
||||||
|
use futures::StreamExt;
|
||||||
|
use reqwest::Client;
|
||||||
|
use serde_json::json;
|
||||||
|
use tokio_stream::wrappers::ReceiverStream;
|
||||||
|
|
||||||
|
#[derive(Clone)]
|
||||||
|
pub struct OllamaProvider {
|
||||||
|
pub client: Client,
|
||||||
|
pub base_url: String,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl OllamaProvider {
|
||||||
|
pub fn new(base_url: impl Into<String>) -> Self {
|
||||||
|
Self {
|
||||||
|
client: Client::new(),
|
||||||
|
base_url: base_url.into(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ── private helpers ──────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
async fn model_exists(&self, model: &str) -> Result<bool, OllamaError> {
|
||||||
|
let url = format!("{}/api/tags", self.base_url);
|
||||||
|
|
||||||
|
let res = self
|
||||||
|
.client
|
||||||
|
.get(url)
|
||||||
|
.send()
|
||||||
|
.await?
|
||||||
|
.json::<ollama::OllamaModels>()
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
Ok(res.models.iter().any(|m| m.name == model))
|
||||||
|
}
|
||||||
|
|
||||||
|
fn has_user_message(&self, messages: &[api::Message]) -> bool {
|
||||||
|
messages.iter().any(|m| matches!(m.role, api::Role::User))
|
||||||
|
}
|
||||||
|
|
||||||
|
fn extract_completion_params<'a>(
|
||||||
|
&self,
|
||||||
|
body: &'a api::CompletionRequest,
|
||||||
|
) -> Result<(&'a str, &'a str), OllamaError> {
|
||||||
|
let prompt = body.prompt.trim();
|
||||||
|
|
||||||
|
let model = body.base.model.as_str();
|
||||||
|
|
||||||
|
Ok((prompt, model))
|
||||||
|
}
|
||||||
|
|
||||||
|
fn extract_chat_params<'a>(
|
||||||
|
&self,
|
||||||
|
body: &'a api::ChatRequest,
|
||||||
|
) -> Result<(&'a [api::Message], &'a str), OllamaError> {
|
||||||
|
let model = body.base.model.as_str();
|
||||||
|
|
||||||
|
Ok((&body.messages, model))
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn parse_keep_alive(&self, s: &str) -> Result<(), OllamaError> {
|
||||||
|
let s = s.trim();
|
||||||
|
|
||||||
|
if s == "-1" || s.parse::<u64>().is_ok() {
|
||||||
|
return Ok(());
|
||||||
|
}
|
||||||
|
|
||||||
|
let split = s
|
||||||
|
.find(|c: char| c.is_alphabetic())
|
||||||
|
.ok_or_else(|| OllamaError::InvalidKeepAlive(s.to_string()))?;
|
||||||
|
|
||||||
|
let (num, unit) = s.split_at(split);
|
||||||
|
|
||||||
|
num.parse::<u64>()
|
||||||
|
.map_err(|_| OllamaError::InvalidKeepAlive(s.to_string()))?;
|
||||||
|
|
||||||
|
match unit {
|
||||||
|
"s" | "m" | "h" => Ok(()),
|
||||||
|
_ => Err(OllamaError::InvalidKeepAlive(s.to_string())),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// // ── public endpoints ─────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
pub async fn list_models(&self) -> Result<api::ModelsResponse, OllamaError> {
|
||||||
|
let url = format!("{}/api/tags", self.base_url);
|
||||||
|
|
||||||
|
let res = self
|
||||||
|
.client
|
||||||
|
.get(url)
|
||||||
|
.send()
|
||||||
|
.await?
|
||||||
|
.json::<ollama::OllamaModels>()
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
let models = res.models.into_iter().map(api::ModelInfo::from).collect();
|
||||||
|
|
||||||
|
Ok(api::ModelsResponse { models })
|
||||||
|
}
|
||||||
|
|
||||||
|
pub async fn load_model(
|
||||||
|
&self,
|
||||||
|
model: &str,
|
||||||
|
keep_alive: Option<&str>,
|
||||||
|
) -> Result<api::LoadModelResponse, OllamaError> {
|
||||||
|
let url = format!("{}/api/generate", self.base_url);
|
||||||
|
|
||||||
|
let keep_alive = keep_alive.ok_or(OllamaError::MissingKeepAlive)?;
|
||||||
|
|
||||||
|
self.parse_keep_alive(keep_alive)?;
|
||||||
|
|
||||||
|
let exists = self.model_exists(model).await?;
|
||||||
|
if !exists {
|
||||||
|
return Err(OllamaError::ModelNotFound(model.to_string()));
|
||||||
|
}
|
||||||
|
|
||||||
|
let payload = json!({
|
||||||
|
"model": model,
|
||||||
|
"prompt": "",
|
||||||
|
"keep_alive": keep_alive,
|
||||||
|
"stream": false,
|
||||||
|
});
|
||||||
|
|
||||||
|
let _res = self
|
||||||
|
.client
|
||||||
|
.post(url)
|
||||||
|
.json(&payload)
|
||||||
|
.send()
|
||||||
|
.await?
|
||||||
|
.text()
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
Ok(api::LoadModelResponse {
|
||||||
|
model: model.to_string(),
|
||||||
|
status: "loaded".to_string(),
|
||||||
|
keep_alive: keep_alive.to_string(),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
pub async fn unload_model(&self, model: &str) -> Result<api::UnloadModelResponse, OllamaError> {
|
||||||
|
let url = format!("{}/api/generate", self.base_url);
|
||||||
|
|
||||||
|
let exists = self.model_exists(model).await?;
|
||||||
|
if !exists {
|
||||||
|
return Err(OllamaError::ModelNotFound(model.to_string()));
|
||||||
|
}
|
||||||
|
|
||||||
|
let payload = json!({
|
||||||
|
"model": model,
|
||||||
|
"prompt": "",
|
||||||
|
"keep_alive": "0",
|
||||||
|
"stream": false,
|
||||||
|
});
|
||||||
|
|
||||||
|
let _res = self
|
||||||
|
.client
|
||||||
|
.post(url)
|
||||||
|
.json(&payload)
|
||||||
|
.send()
|
||||||
|
.await?
|
||||||
|
.text()
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
Ok(api::UnloadModelResponse {
|
||||||
|
model: model.to_string(),
|
||||||
|
status: "unloaded".to_string(),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
pub async fn completions(
|
||||||
|
&self,
|
||||||
|
body: &api::CompletionRequest,
|
||||||
|
) -> Result<api::CompletionResponse, OllamaError> {
|
||||||
|
let url = format!("{}/api/generate", self.base_url);
|
||||||
|
|
||||||
|
let (prompt, model) = self.extract_completion_params(body)?;
|
||||||
|
|
||||||
|
if prompt.is_empty() {
|
||||||
|
return Err(OllamaError::MissingPrompt);
|
||||||
|
}
|
||||||
|
|
||||||
|
if model.is_empty() {
|
||||||
|
return Err(OllamaError::MissingModel);
|
||||||
|
}
|
||||||
|
|
||||||
|
let exists = self.model_exists(model).await?;
|
||||||
|
if !exists {
|
||||||
|
return Err(OllamaError::ModelNotFound(model.to_string()));
|
||||||
|
}
|
||||||
|
|
||||||
|
let options = ollama::OllamaOptions::from(&body.base);
|
||||||
|
|
||||||
|
let payload = ollama::OllamaGenerateRequest {
|
||||||
|
model,
|
||||||
|
prompt,
|
||||||
|
stream: false,
|
||||||
|
options,
|
||||||
|
};
|
||||||
|
|
||||||
|
let res = self
|
||||||
|
.client
|
||||||
|
.post(url)
|
||||||
|
.json(&payload)
|
||||||
|
.send()
|
||||||
|
.await?
|
||||||
|
.json::<ollama::OllamaGenerateResponse>()
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
Ok(api::CompletionResponse::from(res))
|
||||||
|
}
|
||||||
|
|
||||||
|
pub async fn completions_stream(
|
||||||
|
&self,
|
||||||
|
body: &api::CompletionRequest,
|
||||||
|
) -> Result<ReceiverStream<Result<Event, OllamaError>>, OllamaError> {
|
||||||
|
let url = format!("{}/api/generate", self.base_url);
|
||||||
|
|
||||||
|
let (prompt, model) = self.extract_completion_params(body)?;
|
||||||
|
|
||||||
|
if prompt.is_empty() {
|
||||||
|
return Err(OllamaError::MissingPrompt);
|
||||||
|
}
|
||||||
|
|
||||||
|
if model.is_empty() {
|
||||||
|
return Err(OllamaError::MissingModel);
|
||||||
|
}
|
||||||
|
|
||||||
|
let exists = self.model_exists(model).await?;
|
||||||
|
if !exists {
|
||||||
|
return Err(OllamaError::ModelNotFound(model.to_string()));
|
||||||
|
}
|
||||||
|
|
||||||
|
let options = ollama::OllamaOptions::from(&body.base);
|
||||||
|
|
||||||
|
let payload = ollama::OllamaGenerateRequest {
|
||||||
|
model,
|
||||||
|
prompt,
|
||||||
|
stream: true,
|
||||||
|
options,
|
||||||
|
};
|
||||||
|
|
||||||
|
let mut byte_stream = self
|
||||||
|
.client
|
||||||
|
.post(url)
|
||||||
|
.json(&payload)
|
||||||
|
.send()
|
||||||
|
.await?
|
||||||
|
.bytes_stream();
|
||||||
|
|
||||||
|
let (tx, rx) = tokio::sync::mpsc::channel(32);
|
||||||
|
|
||||||
|
tokio::spawn(async move {
|
||||||
|
while let Some(chunk) = byte_stream.next().await {
|
||||||
|
let chunk = match chunk {
|
||||||
|
Ok(b) => b,
|
||||||
|
Err(e) => {
|
||||||
|
let _ = tx.send(Err(OllamaError::Http(e))).await;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
// 🔥 IMPORTANT: typed deserialization
|
||||||
|
let parsed: ollama::OllamaGenerateResponse = match serde_json::from_slice(&chunk) {
|
||||||
|
Ok(v) => v,
|
||||||
|
Err(_) => continue,
|
||||||
|
};
|
||||||
|
|
||||||
|
// map → OpenAI chunk
|
||||||
|
let event_data = serde_json::to_string(&api::CompletionChunk {
|
||||||
|
id: "cmpl-ollama".to_string(),
|
||||||
|
object: "text_completion".to_string(),
|
||||||
|
choices: vec![api::Choice {
|
||||||
|
text: parsed.response,
|
||||||
|
index: 0,
|
||||||
|
finish_reason: if parsed.done {
|
||||||
|
api::FinishReason::Stop
|
||||||
|
} else {
|
||||||
|
api::FinishReason::Length
|
||||||
|
},
|
||||||
|
}],
|
||||||
|
})
|
||||||
|
.unwrap_or_default();
|
||||||
|
|
||||||
|
let _ = tx.send(Ok(Event::default().data(event_data))).await;
|
||||||
|
|
||||||
|
if parsed.done {
|
||||||
|
let _ = tx.send(Ok(Event::default().data("[DONE]"))).await;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
});
|
||||||
|
|
||||||
|
Ok(ReceiverStream::new(rx))
|
||||||
|
}
|
||||||
|
|
||||||
|
pub async fn chat_completions(
|
||||||
|
&self,
|
||||||
|
body: &api::ChatRequest,
|
||||||
|
) -> Result<api::ChatCompletionResponse, OllamaError> {
|
||||||
|
let url = format!("{}/api/chat", self.base_url);
|
||||||
|
|
||||||
|
let (messages, model) = self.extract_chat_params(body)?;
|
||||||
|
|
||||||
|
if body.messages.is_empty() {
|
||||||
|
return Err(OllamaError::MissingMessages);
|
||||||
|
}
|
||||||
|
|
||||||
|
if !self.has_user_message(&body.messages) {
|
||||||
|
return Err(OllamaError::MissingMessages);
|
||||||
|
}
|
||||||
|
|
||||||
|
if model.is_empty() {
|
||||||
|
return Err(OllamaError::MissingModel);
|
||||||
|
}
|
||||||
|
|
||||||
|
let exists = self.model_exists(model).await?;
|
||||||
|
if !exists {
|
||||||
|
return Err(OllamaError::ModelNotFound(model.to_string()));
|
||||||
|
}
|
||||||
|
|
||||||
|
// let options = ollama::OllamaOptions::from(body);
|
||||||
|
let options = ollama::OllamaOptions::from(&body.base);
|
||||||
|
|
||||||
|
let payload = ollama::OllamaChatRequest {
|
||||||
|
model,
|
||||||
|
messages,
|
||||||
|
stream: false,
|
||||||
|
options,
|
||||||
|
};
|
||||||
|
|
||||||
|
let res = self
|
||||||
|
.client
|
||||||
|
.post(url)
|
||||||
|
.json(&payload)
|
||||||
|
.send()
|
||||||
|
.await?
|
||||||
|
.json::<ollama::OllamaChatResponse>()
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
Ok(api::ChatCompletionResponse::from(res))
|
||||||
|
}
|
||||||
|
|
||||||
|
pub async fn chat_completions_stream(
|
||||||
|
&self,
|
||||||
|
body: &api::ChatRequest,
|
||||||
|
) -> Result<ReceiverStream<Result<Event, OllamaError>>, OllamaError> {
|
||||||
|
let url = format!("{}/api/chat", self.base_url);
|
||||||
|
|
||||||
|
let (messages, model) = self.extract_chat_params(body)?;
|
||||||
|
|
||||||
|
if messages.is_empty() {
|
||||||
|
return Err(OllamaError::MissingMessages);
|
||||||
|
}
|
||||||
|
|
||||||
|
if model.is_empty() {
|
||||||
|
return Err(OllamaError::MissingModel);
|
||||||
|
}
|
||||||
|
|
||||||
|
let exists = self.model_exists(model).await?;
|
||||||
|
if !exists {
|
||||||
|
return Err(OllamaError::ModelNotFound(model.to_string()));
|
||||||
|
}
|
||||||
|
|
||||||
|
let options = ollama::OllamaOptions::from(&body.base);
|
||||||
|
|
||||||
|
let payload = ollama::OllamaChatRequest {
|
||||||
|
model,
|
||||||
|
messages,
|
||||||
|
stream: true,
|
||||||
|
options,
|
||||||
|
};
|
||||||
|
|
||||||
|
let mut byte_stream = self
|
||||||
|
.client
|
||||||
|
.post(url)
|
||||||
|
.json(&payload)
|
||||||
|
.send()
|
||||||
|
.await?
|
||||||
|
.bytes_stream();
|
||||||
|
|
||||||
|
let (tx, rx) = tokio::sync::mpsc::channel(32);
|
||||||
|
|
||||||
|
tokio::spawn(async move {
|
||||||
|
let stream_id = format!("chatcmpl-{}", uuid::Uuid::new_v4());
|
||||||
|
|
||||||
|
while let Some(chunk) = byte_stream.next().await {
|
||||||
|
let chunk = match chunk {
|
||||||
|
Ok(b) => b,
|
||||||
|
Err(e) => {
|
||||||
|
let _ = tx.send(Err(OllamaError::Http(e))).await;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
let parsed: ollama::OllamaChatResponse = match serde_json::from_slice(&chunk) {
|
||||||
|
Ok(v) => v,
|
||||||
|
Err(_) => continue,
|
||||||
|
};
|
||||||
|
|
||||||
|
let event = api::ChatCompletionChunk {
|
||||||
|
id: stream_id.clone(),
|
||||||
|
object: "chat.completion.chunk".to_string(),
|
||||||
|
choices: vec![api::ChatChunkChoice {
|
||||||
|
index: 0,
|
||||||
|
delta: api::ChatDelta {
|
||||||
|
role: Some(parsed.message.role),
|
||||||
|
content: Some(parsed.message.content),
|
||||||
|
},
|
||||||
|
finish_reason: if parsed.done {
|
||||||
|
Some(api::FinishReason::Stop)
|
||||||
|
} else {
|
||||||
|
None
|
||||||
|
},
|
||||||
|
}],
|
||||||
|
};
|
||||||
|
|
||||||
|
let event_data = serde_json::to_string(&event).unwrap_or_default();
|
||||||
|
|
||||||
|
let _ = tx.send(Ok(Event::default().data(event_data))).await;
|
||||||
|
|
||||||
|
if parsed.done {
|
||||||
|
let _ = tx.send(Ok(Event::default().data("[DONE]"))).await;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
});
|
||||||
|
|
||||||
|
Ok(ReceiverStream::new(rx))
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,86 @@
|
|||||||
|
use crate::dto::{api, ollama};
|
||||||
|
|
||||||
|
use chrono::Utc;
|
||||||
|
use uuid::Uuid;
|
||||||
|
|
||||||
|
impl From<ollama::OllamaModel> for api::ModelInfo {
|
||||||
|
fn from(m: ollama::OllamaModel) -> Self {
|
||||||
|
Self {
|
||||||
|
name: m.name,
|
||||||
|
|
||||||
|
family: m.details.as_ref().and_then(|d| d.family.clone()),
|
||||||
|
parameter_size: m.details.as_ref().and_then(|d| d.parameter_size.clone()),
|
||||||
|
quantization: m
|
||||||
|
.details
|
||||||
|
.as_ref()
|
||||||
|
.and_then(|d| d.quantization_level.clone()),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<ollama::OllamaGenerateResponse> for api::CompletionResponse {
|
||||||
|
fn from(res: ollama::OllamaGenerateResponse) -> Self {
|
||||||
|
Self {
|
||||||
|
id: Uuid::new_v4().to_string(),
|
||||||
|
object: api::CompletionObject::TextCompletion,
|
||||||
|
model: res.model,
|
||||||
|
created: Utc::now().timestamp() as u64,
|
||||||
|
|
||||||
|
choices: vec![api::Choice {
|
||||||
|
text: res.response,
|
||||||
|
index: 0,
|
||||||
|
finish_reason: api::FinishReason::Stop,
|
||||||
|
}],
|
||||||
|
|
||||||
|
usage: api::Usage {
|
||||||
|
prompt_tokens: res.prompt_eval_count.unwrap_or(0),
|
||||||
|
completion_tokens: res.eval_count.unwrap_or(0),
|
||||||
|
total_tokens: res.prompt_eval_count.unwrap_or(0) + res.eval_count.unwrap_or(0),
|
||||||
|
},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<&api::BaseLLMRequest> for ollama::OllamaOptions {
|
||||||
|
fn from(base: &api::BaseLLMRequest) -> Self {
|
||||||
|
Self {
|
||||||
|
temperature: base.temperature,
|
||||||
|
top_p: base.top_p,
|
||||||
|
top_k: base.top_k,
|
||||||
|
repeat_penalty: base.repeat_penalty,
|
||||||
|
seed: base.seed,
|
||||||
|
num_ctx: base.num_ctx,
|
||||||
|
num_predict: base.num_predict,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<ollama::OllamaChatResponse> for api::ChatCompletionResponse {
|
||||||
|
fn from(res: ollama::OllamaChatResponse) -> Self {
|
||||||
|
let prompt_tokens = res.prompt_eval_count.unwrap_or(0);
|
||||||
|
let completion_tokens = res.eval_count.unwrap_or(0);
|
||||||
|
|
||||||
|
Self {
|
||||||
|
id: Uuid::new_v4().to_string(),
|
||||||
|
object: "chat.completion".to_string(),
|
||||||
|
created: Utc::now().timestamp() as u64,
|
||||||
|
model: res.model,
|
||||||
|
|
||||||
|
choices: vec![api::ChatChoice {
|
||||||
|
index: 0,
|
||||||
|
message: res.message,
|
||||||
|
finish_reason: if res.done {
|
||||||
|
api::FinishReason::Stop
|
||||||
|
} else {
|
||||||
|
api::FinishReason::Length
|
||||||
|
},
|
||||||
|
}],
|
||||||
|
|
||||||
|
usage: Some(api::Usage {
|
||||||
|
prompt_tokens,
|
||||||
|
completion_tokens,
|
||||||
|
total_tokens: prompt_tokens + completion_tokens,
|
||||||
|
}),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,2 @@
|
|||||||
|
pub mod client;
|
||||||
|
pub mod mapper;
|
||||||
@@ -0,0 +1,50 @@
|
|||||||
|
use axum::{
|
||||||
|
Json,
|
||||||
|
extract::{Extension, State},
|
||||||
|
http::StatusCode,
|
||||||
|
};
|
||||||
|
use base64::{Engine as _, engine::general_purpose};
|
||||||
|
use rand::RngCore;
|
||||||
|
use rand::rngs::OsRng;
|
||||||
|
|
||||||
|
use crate::dto::api::{CreateApiKeyRequest, CreateApiKeyResponse};
|
||||||
|
use crate::middlewares::auth::middleware::Auth;
|
||||||
|
use crate::state::app_state::AppState;
|
||||||
|
use crate::utils::crypto::hash_key;
|
||||||
|
|
||||||
|
fn generate_api_key() -> String {
|
||||||
|
let mut bytes = [0u8; 32];
|
||||||
|
OsRng.fill_bytes(&mut bytes);
|
||||||
|
general_purpose::URL_SAFE_NO_PAD.encode(bytes)
|
||||||
|
}
|
||||||
|
|
||||||
|
pub async fn create_api_key(
|
||||||
|
State(state): State<AppState>,
|
||||||
|
Extension(claims): Extension<Auth>,
|
||||||
|
Json(body): Json<CreateApiKeyRequest>,
|
||||||
|
) -> Result<Json<CreateApiKeyResponse>, StatusCode> {
|
||||||
|
if matches!(claims, Auth::ApiKey(_)) {
|
||||||
|
return Err(StatusCode::FORBIDDEN);
|
||||||
|
}
|
||||||
|
|
||||||
|
let raw_key = generate_api_key();
|
||||||
|
let key_hash = hash_key(&raw_key);
|
||||||
|
|
||||||
|
dbg!(&claims);
|
||||||
|
|
||||||
|
sqlx::query!(
|
||||||
|
r#"
|
||||||
|
INSERT INTO auth.api_key (key_hash, name, created_by, scopes)
|
||||||
|
VALUES ($1, $2, $3, $4)
|
||||||
|
"#,
|
||||||
|
key_hash,
|
||||||
|
body.name,
|
||||||
|
claims.user_id(),
|
||||||
|
&body.scopes
|
||||||
|
)
|
||||||
|
.execute(&state.postgres)
|
||||||
|
.await
|
||||||
|
.map_err(|_| StatusCode::INTERNAL_SERVER_ERROR)?;
|
||||||
|
|
||||||
|
Ok(Json(CreateApiKeyResponse { api_key: raw_key }))
|
||||||
|
}
|
||||||
@@ -0,0 +1,170 @@
|
|||||||
|
use axum::{
|
||||||
|
Json,
|
||||||
|
extract::State,
|
||||||
|
response::{
|
||||||
|
IntoResponse, Response,
|
||||||
|
sse::{KeepAlive, Sse},
|
||||||
|
},
|
||||||
|
};
|
||||||
|
|
||||||
|
use crate::dto::api;
|
||||||
|
use crate::errors::OllamaError;
|
||||||
|
use crate::state::app_state::AppState;
|
||||||
|
|
||||||
|
#[utoipa::path(
|
||||||
|
post,
|
||||||
|
path = "/completions",
|
||||||
|
tag = "chat",
|
||||||
|
request_body(
|
||||||
|
content = api::CompletionRequest,
|
||||||
|
description = "Text completion request",
|
||||||
|
content_type = "application/json"
|
||||||
|
),
|
||||||
|
responses(
|
||||||
|
(
|
||||||
|
status = 200,
|
||||||
|
description = "Text completion response. If stream=true, response is SSE stream of chunks ending in [DONE].",
|
||||||
|
body = api::CompletionResponse,
|
||||||
|
content_type = "application/json"
|
||||||
|
),
|
||||||
|
(
|
||||||
|
status = 400,
|
||||||
|
description = "Invalid request: missing prompt, model, or invalid format",
|
||||||
|
body = api::ErrorResponse,
|
||||||
|
example = json!({ "error": "prompt is required and cannot be empty" })
|
||||||
|
),
|
||||||
|
(
|
||||||
|
status = 422,
|
||||||
|
description = "Model not found or not available locally",
|
||||||
|
body = api::ErrorResponse,
|
||||||
|
example = json!({ "error": "model 'llama3' is not available — run `ollama pull llama3` first" })
|
||||||
|
),
|
||||||
|
(
|
||||||
|
status = 500,
|
||||||
|
description = "Internal server error (Ollama or network failure)",
|
||||||
|
body = api::ErrorResponse,
|
||||||
|
example = json!({ "error": "connection refused" })
|
||||||
|
)
|
||||||
|
)
|
||||||
|
)]
|
||||||
|
pub async fn completions(
|
||||||
|
State(state): State<AppState>,
|
||||||
|
Json(body): Json<api::CompletionRequest>,
|
||||||
|
) -> Result<Response, (axum::http::StatusCode, Json<api::ErrorResponse>)> {
|
||||||
|
tracing::debug!("Received /completion with body {:?}", body);
|
||||||
|
if body.base.stream {
|
||||||
|
let stream = state.ollama.completions_stream(&body).await.map_err(|e| {
|
||||||
|
let (code, msg) = ollama_err(e);
|
||||||
|
(code, Json(api::ErrorResponse::new(msg)))
|
||||||
|
})?;
|
||||||
|
|
||||||
|
Ok(Sse::new(stream)
|
||||||
|
.keep_alive(KeepAlive::default())
|
||||||
|
.into_response())
|
||||||
|
} else {
|
||||||
|
let response = state.ollama.completions(&body).await.map_err(|e| {
|
||||||
|
let (code, msg) = ollama_err(e);
|
||||||
|
(code, Json(api::ErrorResponse::new(msg)))
|
||||||
|
})?;
|
||||||
|
|
||||||
|
Ok(Json(response).into_response())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[utoipa::path(
|
||||||
|
post,
|
||||||
|
path = "/chat/completions",
|
||||||
|
tag = "chat",
|
||||||
|
request_body(
|
||||||
|
content = api::ChatRequest,
|
||||||
|
description = "Chat completion request with message history",
|
||||||
|
content_type = "application/json"
|
||||||
|
),
|
||||||
|
responses(
|
||||||
|
(
|
||||||
|
status = 200,
|
||||||
|
description = "Chat completion response. If stream=false returns JSON. If stream=true returns SSE stream of chunks ending with [DONE].",
|
||||||
|
body = api::ChatCompletionResponse,
|
||||||
|
content_type = "application/json"
|
||||||
|
),
|
||||||
|
(
|
||||||
|
status = 400,
|
||||||
|
description = "Invalid request",
|
||||||
|
body = api::ErrorResponse,
|
||||||
|
example = json!({ "error": "messages array with at least one user message is required" })
|
||||||
|
),
|
||||||
|
(
|
||||||
|
status = 401,
|
||||||
|
description = "Unauthorized",
|
||||||
|
body = api::ErrorResponse,
|
||||||
|
example = json!({ "error": "missing or invalid token" })
|
||||||
|
),
|
||||||
|
(
|
||||||
|
status = 422,
|
||||||
|
description = "Model not found or unavailable",
|
||||||
|
body = api::ErrorResponse,
|
||||||
|
example = json!({ "error": "model 'llama3' is not available — run `ollama pull llama3` first" })
|
||||||
|
),
|
||||||
|
(
|
||||||
|
status = 500,
|
||||||
|
description = "Internal server error",
|
||||||
|
body = api::ErrorResponse,
|
||||||
|
example = json!({ "error": "connection refused" })
|
||||||
|
)
|
||||||
|
)
|
||||||
|
)]
|
||||||
|
pub async fn chat_completions(
|
||||||
|
State(state): State<AppState>,
|
||||||
|
Json(body): Json<api::ChatRequest>,
|
||||||
|
) -> Result<Response, (axum::http::StatusCode, String)> {
|
||||||
|
tracing::debug!("Received /completion with body {:?}", body);
|
||||||
|
if body.base.stream {
|
||||||
|
let stream = state
|
||||||
|
.ollama
|
||||||
|
.chat_completions_stream(&body)
|
||||||
|
.await
|
||||||
|
.map_err(ollama_err)?;
|
||||||
|
|
||||||
|
Ok(Sse::new(stream)
|
||||||
|
.keep_alive(KeepAlive::default())
|
||||||
|
.into_response())
|
||||||
|
} else {
|
||||||
|
let response = state
|
||||||
|
.ollama
|
||||||
|
.chat_completions(&body)
|
||||||
|
.await
|
||||||
|
.map_err(ollama_err)?;
|
||||||
|
|
||||||
|
Ok(Json(response).into_response())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn ollama_err(e: OllamaError) -> (axum::http::StatusCode, String) {
|
||||||
|
match e {
|
||||||
|
OllamaError::MissingPrompt => (
|
||||||
|
axum::http::StatusCode::BAD_REQUEST,
|
||||||
|
"prompt is required and cannot be empty".to_string(),
|
||||||
|
),
|
||||||
|
OllamaError::MissingModel => (
|
||||||
|
axum::http::StatusCode::BAD_REQUEST,
|
||||||
|
"model is required and cannot be empty".to_string(),
|
||||||
|
),
|
||||||
|
OllamaError::ModelNotFound(m) => (
|
||||||
|
axum::http::StatusCode::UNPROCESSABLE_ENTITY,
|
||||||
|
format!("model '{m}' is not available — run `ollama pull {m}` first"),
|
||||||
|
),
|
||||||
|
OllamaError::MissingKeepAlive => (
|
||||||
|
axum::http::StatusCode::BAD_REQUEST,
|
||||||
|
"keep alive is required and cannot be empty".to_string(),
|
||||||
|
),
|
||||||
|
OllamaError::InvalidKeepAlive(v) => (
|
||||||
|
axum::http::StatusCode::BAD_REQUEST,
|
||||||
|
format!("invalid keep_alive '{v}'"),
|
||||||
|
),
|
||||||
|
OllamaError::MissingMessages => (
|
||||||
|
axum::http::StatusCode::BAD_REQUEST,
|
||||||
|
"messages array with at least one user message is required".to_string(),
|
||||||
|
),
|
||||||
|
OllamaError::Http(e) => (axum::http::StatusCode::INTERNAL_SERVER_ERROR, e.to_string()),
|
||||||
|
}
|
||||||
|
}
|
||||||
+33
-6
@@ -1,12 +1,39 @@
|
|||||||
|
pub mod apikey;
|
||||||
|
pub mod chat;
|
||||||
pub mod models;
|
pub mod models;
|
||||||
|
|
||||||
use crate::auth::middleware::auth_middleware;
|
use crate::docs::ApiDoc;
|
||||||
use crate::routes::v1::models::list_models;
|
use crate::middlewares::auth::{auth_middleware, middleware::require_roles};
|
||||||
use crate::state::app_state::AppState;
|
use crate::state::app_state::AppState;
|
||||||
use axum::{Router, middleware, routing::get};
|
|
||||||
|
|
||||||
pub fn router() -> Router<AppState> {
|
use axum::{Json, Router, middleware, routing::get, routing::post};
|
||||||
|
use utoipa::OpenApi;
|
||||||
|
|
||||||
|
async fn openapi_json() -> Json<utoipa::openapi::OpenApi> {
|
||||||
|
Json(ApiDoc::openapi())
|
||||||
|
}
|
||||||
|
|
||||||
|
fn public_router() -> Router<AppState> {
|
||||||
|
Router::new().route("/docs.json", get(openapi_json))
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn protected_router() -> Router<AppState> {
|
||||||
Router::new()
|
Router::new()
|
||||||
.route("/models", get(list_models))
|
.route("/models", get(models::list_models))
|
||||||
.layer(middleware::from_fn(auth_middleware))
|
.route("/completions", post(chat::completions))
|
||||||
|
.route("/chat/completions", post(chat::chat_completions))
|
||||||
|
.route("/models/{model}/load", post(models::load_model))
|
||||||
|
.route("/models/{model}/unload", post(models::unload_model))
|
||||||
|
.route(
|
||||||
|
"/keys/generate",
|
||||||
|
post(apikey::create_api_key).route_layer(middleware::from_fn(|req, next| {
|
||||||
|
require_roles(req, next, None, Some("admin"))
|
||||||
|
})),
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn router(state: AppState) -> Router<AppState> {
|
||||||
|
Router::new()
|
||||||
|
.merge(public_router())
|
||||||
|
.merge(protected_router().layer(middleware::from_fn_with_state(state, auth_middleware)))
|
||||||
}
|
}
|
||||||
|
|||||||
+155
-8
@@ -1,16 +1,163 @@
|
|||||||
use axum::{Json, extract::State};
|
use axum::{
|
||||||
use serde_json::Value;
|
Json,
|
||||||
|
extract::{Path, State},
|
||||||
|
};
|
||||||
|
|
||||||
use crate::state::app_state::AppState;
|
use crate::dto::api;
|
||||||
|
use crate::{errors::OllamaError, state::app_state::AppState};
|
||||||
|
|
||||||
|
#[utoipa::path(
|
||||||
|
get,
|
||||||
|
path = "/models",
|
||||||
|
tag = "models",
|
||||||
|
responses(
|
||||||
|
(
|
||||||
|
status = 200,
|
||||||
|
description = "List of locally available Ollama models",
|
||||||
|
body = api::ModelsResponse,
|
||||||
|
content_type = "application/json",
|
||||||
|
),
|
||||||
|
(
|
||||||
|
status = 500,
|
||||||
|
description = "Internal server error (Ollama or network failure)",
|
||||||
|
body = api::ErrorResponse,
|
||||||
|
example = json!({ "error": "connection refused" })
|
||||||
|
)
|
||||||
|
)
|
||||||
|
)]
|
||||||
pub async fn list_models(
|
pub async fn list_models(
|
||||||
State(state): State<AppState>,
|
State(state): State<AppState>,
|
||||||
) -> Result<Json<Value>, (axum::http::StatusCode, String)> {
|
) -> Result<Json<api::ModelsResponse>, (axum::http::StatusCode, String)> {
|
||||||
match state.ollama.list_models().await {
|
match state.ollama.list_models().await {
|
||||||
Ok(models) => Ok(Json(models)),
|
Ok(models) => Ok(Json(models)),
|
||||||
Err(err) => Err((
|
Err(e) => Err(ollama_err(e)),
|
||||||
axum::http::StatusCode::INTERNAL_SERVER_ERROR,
|
}
|
||||||
err.to_string(),
|
}
|
||||||
)),
|
|
||||||
|
#[utoipa::path(
|
||||||
|
post,
|
||||||
|
path = "/models/{model}/load",
|
||||||
|
tag = "models",
|
||||||
|
params(
|
||||||
|
("model" = String, Path, description = "Name of the model to load into memory (e.g. 'llama3')")
|
||||||
|
),
|
||||||
|
request_body(
|
||||||
|
content = api::LoadModelBody,
|
||||||
|
description = "Load model request",
|
||||||
|
content_type = "application/json",
|
||||||
|
example = json!({ "keep_alive": "10m" })
|
||||||
|
),
|
||||||
|
responses(
|
||||||
|
(
|
||||||
|
status = 200,
|
||||||
|
description = "Model successfully loaded into memory",
|
||||||
|
body = api::LoadModelResponse,
|
||||||
|
content_type = "application/json",
|
||||||
|
),
|
||||||
|
(
|
||||||
|
status = 400,
|
||||||
|
description = "Invalid or missing keep_alive format",
|
||||||
|
body = api::ErrorResponse,
|
||||||
|
examples(
|
||||||
|
("Missing" = (value = json!({ "error": "keep alive is required and cannot be empty" }))),
|
||||||
|
("Invalid" = (value = json!({ "error": "invalid keep_alive '10x' — use 30s / 10m / 2h, a plain integer, or -1" })))
|
||||||
|
)
|
||||||
|
),
|
||||||
|
(
|
||||||
|
status = 404,
|
||||||
|
description = "Model not found locally",
|
||||||
|
body = api::ErrorResponse,
|
||||||
|
example = json!({ "error": "model 'llama3' not found — run `ollama pull llama3`" })
|
||||||
|
),
|
||||||
|
(
|
||||||
|
status = 500,
|
||||||
|
description = "Internal server error (Ollama or network failure)",
|
||||||
|
body = api::ErrorResponse,
|
||||||
|
example = json!({ "error": "connection refused" })
|
||||||
|
)
|
||||||
|
)
|
||||||
|
)]
|
||||||
|
pub async fn load_model(
|
||||||
|
State(state): State<AppState>,
|
||||||
|
Path(model): Path<String>,
|
||||||
|
Json(body): Json<api::LoadModelBody>,
|
||||||
|
) -> Result<Json<api::LoadModelResponse>, (axum::http::StatusCode, String)> {
|
||||||
|
let response = state
|
||||||
|
.ollama
|
||||||
|
.load_model(&model, body.keep_alive.as_deref())
|
||||||
|
.await
|
||||||
|
.map_err(ollama_err)?;
|
||||||
|
|
||||||
|
Ok(Json(response))
|
||||||
|
}
|
||||||
|
|
||||||
|
#[utoipa::path(
|
||||||
|
delete,
|
||||||
|
path = "/models/{model}/load",
|
||||||
|
tag = "models",
|
||||||
|
params(
|
||||||
|
("model" = String, Path, description = "Name of the model to unload from memory (e.g. 'llama3')")
|
||||||
|
),
|
||||||
|
responses(
|
||||||
|
(
|
||||||
|
status = 200,
|
||||||
|
description = "Model successfully unloaded from memory",
|
||||||
|
body = api::UnloadModelResponse,
|
||||||
|
content_type = "application/json",
|
||||||
|
),
|
||||||
|
(
|
||||||
|
status = 404,
|
||||||
|
description = "Model not found locally",
|
||||||
|
body = api::ErrorResponse,
|
||||||
|
example = json!({ "error": "model 'llama3' not found — run `ollama pull llama3`" })
|
||||||
|
),
|
||||||
|
(
|
||||||
|
status = 500,
|
||||||
|
description = "Internal server error (Ollama or network failure)",
|
||||||
|
body = api::ErrorResponse,
|
||||||
|
example = json!({ "error": "connection refused" })
|
||||||
|
)
|
||||||
|
)
|
||||||
|
)]
|
||||||
|
pub async fn unload_model(
|
||||||
|
State(state): State<AppState>,
|
||||||
|
Path(model): Path<String>,
|
||||||
|
) -> Result<Json<api::UnloadModelResponse>, (axum::http::StatusCode, String)> {
|
||||||
|
let response = state
|
||||||
|
.ollama
|
||||||
|
.unload_model(&model)
|
||||||
|
.await
|
||||||
|
.map_err(ollama_err)?;
|
||||||
|
|
||||||
|
Ok(Json(response))
|
||||||
|
}
|
||||||
|
|
||||||
|
fn ollama_err(e: OllamaError) -> (axum::http::StatusCode, String) {
|
||||||
|
match e {
|
||||||
|
OllamaError::ModelNotFound(m) => (
|
||||||
|
axum::http::StatusCode::NOT_FOUND,
|
||||||
|
format!("model '{m}' not found — run `ollama pull {m}`"),
|
||||||
|
),
|
||||||
|
OllamaError::MissingModel => (
|
||||||
|
axum::http::StatusCode::BAD_REQUEST,
|
||||||
|
"model is required and cannot be empty".to_string(),
|
||||||
|
),
|
||||||
|
OllamaError::MissingPrompt => (
|
||||||
|
axum::http::StatusCode::BAD_REQUEST,
|
||||||
|
"prompt is required".to_string(),
|
||||||
|
),
|
||||||
|
OllamaError::MissingMessages => (
|
||||||
|
axum::http::StatusCode::BAD_REQUEST,
|
||||||
|
"messages array with at least one user message is required".to_string(),
|
||||||
|
),
|
||||||
|
OllamaError::MissingKeepAlive => (
|
||||||
|
axum::http::StatusCode::BAD_REQUEST,
|
||||||
|
"keep alive is required and cannot be empty".to_string(),
|
||||||
|
),
|
||||||
|
OllamaError::InvalidKeepAlive(v) => (
|
||||||
|
axum::http::StatusCode::BAD_REQUEST,
|
||||||
|
format!("invalid keep_alive '{v}' — use 30s / 10m / 2h, a plain integer, or -1"),
|
||||||
|
),
|
||||||
|
OllamaError::Http(e) => (axum::http::StatusCode::INTERNAL_SERVER_ERROR, e.to_string()),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,8 @@
|
|||||||
|
// use axum::Json;
|
||||||
|
// use utoipa::OpenApi;
|
||||||
|
|
||||||
|
// use crate::openapi::V1ApiDoc;
|
||||||
|
|
||||||
|
// pub async fn openapi_json() -> Json<utoipa::openapi::OpenApi> {
|
||||||
|
// Json(V1ApiDoc::openapi())
|
||||||
|
// }
|
||||||
@@ -1,7 +1,9 @@
|
|||||||
use crate::providers::ollama::OllamaProvider;
|
use crate::providers::ollama::client::OllamaProvider;
|
||||||
|
use sqlx::PgPool;
|
||||||
use std::sync::Arc;
|
use std::sync::Arc;
|
||||||
|
|
||||||
#[derive(Clone)]
|
#[derive(Clone)]
|
||||||
pub struct AppState {
|
pub struct AppState {
|
||||||
pub ollama: Arc<OllamaProvider>,
|
pub ollama: Arc<OllamaProvider>,
|
||||||
|
pub postgres: PgPool,
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,11 @@
|
|||||||
|
use sha2::{Digest, Sha256};
|
||||||
|
|
||||||
|
pub fn hash_key(key: &str) -> String {
|
||||||
|
let mut hasher = Sha256::new();
|
||||||
|
hasher.update(key.as_bytes());
|
||||||
|
hasher
|
||||||
|
.finalize()
|
||||||
|
.iter()
|
||||||
|
.map(|b| format!("{:02x}", b))
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
@@ -0,0 +1 @@
|
|||||||
|
pub mod crypto;
|
||||||
@@ -0,0 +1,414 @@
|
|||||||
|
use serde_json::json;
|
||||||
|
use wiremock::matchers::{method, path};
|
||||||
|
use wiremock::{Mock, MockServer, ResponseTemplate};
|
||||||
|
|
||||||
|
use chat::dto::api;
|
||||||
|
use chat::errors::OllamaError;
|
||||||
|
use chat::providers::ollama::client::OllamaProvider;
|
||||||
|
|
||||||
|
// ── helpers ──────────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
async fn setup() -> (MockServer, OllamaProvider) {
|
||||||
|
let server = MockServer::start().await;
|
||||||
|
let provider = OllamaProvider::new(server.uri());
|
||||||
|
(server, provider)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn models_response(names: &[&str]) -> serde_json::Value {
|
||||||
|
json!({
|
||||||
|
"models": names.iter().map(|n| json!({ "name": n })).collect::<Vec<_>>()
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
// ── list_models ───────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_list_models_ok() {
|
||||||
|
let (server, provider) = setup().await;
|
||||||
|
|
||||||
|
Mock::given(method("GET"))
|
||||||
|
.and(path("/api/tags"))
|
||||||
|
.respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
||||||
|
.mount(&server)
|
||||||
|
.await;
|
||||||
|
|
||||||
|
let res = provider.list_models().await.unwrap();
|
||||||
|
assert_eq!(res.models[0].name, "llama3");
|
||||||
|
}
|
||||||
|
|
||||||
|
// ── completions ───────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_completions_ok() {
|
||||||
|
let (server, provider) = setup().await;
|
||||||
|
|
||||||
|
Mock::given(method("GET"))
|
||||||
|
.and(path("/api/tags"))
|
||||||
|
.respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
||||||
|
.mount(&server)
|
||||||
|
.await;
|
||||||
|
|
||||||
|
Mock::given(method("POST"))
|
||||||
|
.and(path("/api/generate"))
|
||||||
|
.respond_with(ResponseTemplate::new(200).set_body_json(json!({
|
||||||
|
"model": "llama3",
|
||||||
|
"response": "I am a helpful assistant.",
|
||||||
|
"done": true,
|
||||||
|
"prompt_eval_count": 10,
|
||||||
|
"eval_count": 8,
|
||||||
|
})))
|
||||||
|
.mount(&server)
|
||||||
|
.await;
|
||||||
|
|
||||||
|
let req = api::CompletionRequest {
|
||||||
|
base: api::BaseLLMRequest {
|
||||||
|
model: "llama3".to_string(),
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
prompt: "Hello".to_string(),
|
||||||
|
};
|
||||||
|
|
||||||
|
let res = provider.completions(&req).await.unwrap();
|
||||||
|
|
||||||
|
assert_eq!(res.object, api::CompletionObject::TextCompletion);
|
||||||
|
|
||||||
|
assert_eq!(res.choices.len(), 1);
|
||||||
|
assert_eq!(res.choices[0].text, "I am a helpful assistant.");
|
||||||
|
assert_eq!(res.choices[0].finish_reason, api::FinishReason::Stop);
|
||||||
|
|
||||||
|
assert_eq!(res.usage.prompt_tokens, 10);
|
||||||
|
assert_eq!(res.usage.completion_tokens, 8);
|
||||||
|
assert_eq!(res.usage.total_tokens, 18);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_completions_missing_prompt() {
|
||||||
|
let (server, provider) = setup().await;
|
||||||
|
|
||||||
|
Mock::given(method("GET"))
|
||||||
|
.and(path("/api/tags"))
|
||||||
|
.respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
||||||
|
.mount(&server)
|
||||||
|
.await;
|
||||||
|
|
||||||
|
let req = api::CompletionRequest {
|
||||||
|
base: api::BaseLLMRequest {
|
||||||
|
model: "llama3".to_string(),
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
prompt: "".to_string(),
|
||||||
|
};
|
||||||
|
|
||||||
|
let err = provider.completions(&req).await.unwrap_err();
|
||||||
|
|
||||||
|
assert!(matches!(err, OllamaError::MissingPrompt));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_completions_empty_prompt() {
|
||||||
|
let (server, provider) = setup().await;
|
||||||
|
|
||||||
|
Mock::given(method("GET"))
|
||||||
|
.and(path("/api/tags"))
|
||||||
|
.respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
||||||
|
.mount(&server)
|
||||||
|
.await;
|
||||||
|
|
||||||
|
let req = api::CompletionRequest {
|
||||||
|
base: api::BaseLLMRequest {
|
||||||
|
model: "llama3".to_string(),
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
prompt: " ".to_string(),
|
||||||
|
};
|
||||||
|
|
||||||
|
let err = provider.completions(&req).await.unwrap_err();
|
||||||
|
|
||||||
|
assert!(matches!(err, OllamaError::MissingPrompt));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_completions_model_not_found() {
|
||||||
|
let (server, provider) = setup().await;
|
||||||
|
|
||||||
|
Mock::given(method("GET"))
|
||||||
|
.and(path("/api/tags"))
|
||||||
|
.respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
||||||
|
.mount(&server)
|
||||||
|
.await;
|
||||||
|
|
||||||
|
let req = api::CompletionRequest {
|
||||||
|
base: api::BaseLLMRequest {
|
||||||
|
model: "gpt-4".to_string(),
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
prompt: "hello".to_string(),
|
||||||
|
};
|
||||||
|
|
||||||
|
let err = provider.completions(&req).await.unwrap_err();
|
||||||
|
|
||||||
|
assert!(matches!(err, chat::errors::OllamaError::ModelNotFound(_)));
|
||||||
|
}
|
||||||
|
|
||||||
|
// ── chat_completions ──────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_chat_completions_ok() {
|
||||||
|
let (server, provider) = setup().await;
|
||||||
|
|
||||||
|
Mock::given(method("GET"))
|
||||||
|
.and(path("/api/tags"))
|
||||||
|
.respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
||||||
|
.mount(&server)
|
||||||
|
.await;
|
||||||
|
|
||||||
|
Mock::given(method("POST"))
|
||||||
|
.and(path("/api/chat"))
|
||||||
|
.respond_with(ResponseTemplate::new(200).set_body_json(json!({
|
||||||
|
"model": "llama3",
|
||||||
|
"message": { "role": "assistant", "content": "4." },
|
||||||
|
"done": true,
|
||||||
|
"prompt_eval_count": 5,
|
||||||
|
"eval_count": 2
|
||||||
|
})))
|
||||||
|
.mount(&server)
|
||||||
|
.await;
|
||||||
|
|
||||||
|
let req = api::ChatRequest {
|
||||||
|
base: api::BaseLLMRequest {
|
||||||
|
model: "llama3".to_string(),
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
messages: vec![api::Message {
|
||||||
|
role: api::Role::User,
|
||||||
|
content: "What is 2+2?".to_string(),
|
||||||
|
}],
|
||||||
|
};
|
||||||
|
|
||||||
|
let res = provider.chat_completions(&req).await.unwrap();
|
||||||
|
|
||||||
|
assert_eq!(res.object, "chat.completion");
|
||||||
|
assert_eq!(res.choices.len(), 1);
|
||||||
|
|
||||||
|
assert_eq!(res.choices[0].message.role, api::Role::Assistant);
|
||||||
|
assert_eq!(res.choices[0].message.content, "4.");
|
||||||
|
|
||||||
|
assert_eq!(res.choices[0].finish_reason, api::FinishReason::Stop);
|
||||||
|
|
||||||
|
let usage = res.usage.unwrap();
|
||||||
|
assert_eq!(usage.prompt_tokens, 5);
|
||||||
|
assert_eq!(usage.completion_tokens, 2);
|
||||||
|
assert_eq!(usage.total_tokens, 7);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_chat_completions_missing_messages() {
|
||||||
|
let (server, provider) = setup().await;
|
||||||
|
|
||||||
|
Mock::given(method("GET"))
|
||||||
|
.and(path("/api/tags"))
|
||||||
|
.respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
||||||
|
.mount(&server)
|
||||||
|
.await;
|
||||||
|
|
||||||
|
let req = api::ChatRequest {
|
||||||
|
base: api::BaseLLMRequest {
|
||||||
|
model: "llama3".to_string(),
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
messages: vec![],
|
||||||
|
};
|
||||||
|
|
||||||
|
let err = provider.chat_completions(&req).await.unwrap_err();
|
||||||
|
|
||||||
|
assert!(matches!(err, chat::errors::OllamaError::MissingMessages));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_chat_completions_no_user_message() {
|
||||||
|
let (server, provider) = setup().await;
|
||||||
|
|
||||||
|
Mock::given(method("GET"))
|
||||||
|
.and(path("/api/tags"))
|
||||||
|
.respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
||||||
|
.mount(&server)
|
||||||
|
.await;
|
||||||
|
|
||||||
|
let req = api::ChatRequest {
|
||||||
|
base: api::BaseLLMRequest {
|
||||||
|
model: "llama3".to_string(),
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
messages: vec![api::Message {
|
||||||
|
role: api::Role::System,
|
||||||
|
content: "be helpful".to_string(),
|
||||||
|
}],
|
||||||
|
};
|
||||||
|
|
||||||
|
let err = provider.chat_completions(&req).await.unwrap_err();
|
||||||
|
|
||||||
|
assert!(matches!(err, chat::errors::OllamaError::MissingMessages));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_chat_completions_model_not_found() {
|
||||||
|
let (server, provider) = setup().await;
|
||||||
|
|
||||||
|
Mock::given(method("GET"))
|
||||||
|
.and(path("/api/tags"))
|
||||||
|
.respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
||||||
|
.mount(&server)
|
||||||
|
.await;
|
||||||
|
|
||||||
|
let req = api::ChatRequest {
|
||||||
|
base: api::BaseLLMRequest {
|
||||||
|
model: "gpt-4".to_string(),
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
messages: vec![api::Message {
|
||||||
|
role: api::Role::User,
|
||||||
|
content: "hi".to_string(),
|
||||||
|
}],
|
||||||
|
};
|
||||||
|
|
||||||
|
let err = provider.chat_completions(&req).await.unwrap_err();
|
||||||
|
|
||||||
|
assert!(matches!(err, chat::errors::OllamaError::ModelNotFound(_)));
|
||||||
|
}
|
||||||
|
|
||||||
|
// // ── load_model ────────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_load_model_ok() {
|
||||||
|
let (server, provider) = setup().await;
|
||||||
|
|
||||||
|
Mock::given(method("GET"))
|
||||||
|
.and(path("/api/tags"))
|
||||||
|
.respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
||||||
|
.mount(&server)
|
||||||
|
.await;
|
||||||
|
|
||||||
|
Mock::given(method("POST"))
|
||||||
|
.and(path("/api/generate"))
|
||||||
|
.respond_with(ResponseTemplate::new(200).set_body_json(json!({
|
||||||
|
"model": "llama3",
|
||||||
|
"response": "ok",
|
||||||
|
"done": true,
|
||||||
|
})))
|
||||||
|
.mount(&server)
|
||||||
|
.await;
|
||||||
|
|
||||||
|
let res = provider.load_model("llama3", Some("10m")).await.unwrap();
|
||||||
|
|
||||||
|
assert_eq!(res.model, "llama3");
|
||||||
|
assert_eq!(res.status, "loaded");
|
||||||
|
assert_eq!(res.keep_alive, "10m");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_load_model_not_found() {
|
||||||
|
let (server, provider) = setup().await;
|
||||||
|
|
||||||
|
Mock::given(method("GET"))
|
||||||
|
.and(path("/api/tags"))
|
||||||
|
.respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
||||||
|
.mount(&server)
|
||||||
|
.await;
|
||||||
|
|
||||||
|
let err = provider.load_model("gpt-4", Some("10m")).await.unwrap_err();
|
||||||
|
|
||||||
|
assert!(matches!(err, chat::errors::OllamaError::ModelNotFound(_)));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_load_model_invalid_keep_alive() {
|
||||||
|
let (server, provider) = setup().await;
|
||||||
|
|
||||||
|
Mock::given(method("GET"))
|
||||||
|
.and(path("/api/tags"))
|
||||||
|
.respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
||||||
|
.mount(&server)
|
||||||
|
.await;
|
||||||
|
|
||||||
|
let err = provider
|
||||||
|
.load_model("llama3", Some("10x"))
|
||||||
|
.await
|
||||||
|
.unwrap_err();
|
||||||
|
|
||||||
|
assert!(matches!(
|
||||||
|
err,
|
||||||
|
chat::errors::OllamaError::InvalidKeepAlive(_)
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_load_model_keep_alive_plain_integer() {
|
||||||
|
let (server, provider) = setup().await;
|
||||||
|
|
||||||
|
Mock::given(method("GET"))
|
||||||
|
.and(path("/api/tags"))
|
||||||
|
.respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
||||||
|
.mount(&server)
|
||||||
|
.await;
|
||||||
|
|
||||||
|
Mock::given(method("POST"))
|
||||||
|
.and(path("/api/generate"))
|
||||||
|
.respond_with(ResponseTemplate::new(200).set_body_json(json!({
|
||||||
|
"model": "llama3",
|
||||||
|
"done": true,
|
||||||
|
})))
|
||||||
|
.mount(&server)
|
||||||
|
.await;
|
||||||
|
|
||||||
|
let res = provider.load_model("llama3", Some("3600")).await.unwrap();
|
||||||
|
|
||||||
|
assert_eq!(res.status, "loaded");
|
||||||
|
assert_eq!(res.keep_alive, "3600");
|
||||||
|
|
||||||
|
let res = provider.load_model("llama3", Some("-1")).await.unwrap();
|
||||||
|
|
||||||
|
assert_eq!(res.status, "loaded");
|
||||||
|
assert_eq!(res.keep_alive, "-1");
|
||||||
|
}
|
||||||
|
|
||||||
|
// // ── unload_model ──────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_unload_model_ok() {
|
||||||
|
let (server, provider) = setup().await;
|
||||||
|
|
||||||
|
Mock::given(method("GET"))
|
||||||
|
.and(path("/api/tags"))
|
||||||
|
.respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
||||||
|
.mount(&server)
|
||||||
|
.await;
|
||||||
|
|
||||||
|
Mock::given(method("POST"))
|
||||||
|
.and(path("/api/generate"))
|
||||||
|
.respond_with(ResponseTemplate::new(200).set_body_json(json!({
|
||||||
|
"model": "llama3",
|
||||||
|
"response": "ok",
|
||||||
|
"done": true,
|
||||||
|
})))
|
||||||
|
.mount(&server)
|
||||||
|
.await;
|
||||||
|
|
||||||
|
let res = provider.unload_model("llama3").await.unwrap();
|
||||||
|
|
||||||
|
assert_eq!(res.model, "llama3");
|
||||||
|
assert_eq!(res.status, "unloaded");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_unload_model_not_found() {
|
||||||
|
let (server, provider) = setup().await;
|
||||||
|
|
||||||
|
Mock::given(method("GET"))
|
||||||
|
.and(path("/api/tags"))
|
||||||
|
.respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
||||||
|
.mount(&server)
|
||||||
|
.await;
|
||||||
|
|
||||||
|
let err = provider.unload_model("gpt-4").await.unwrap_err();
|
||||||
|
assert!(matches!(err, chat::errors::OllamaError::ModelNotFound(_)));
|
||||||
|
}
|
||||||
Reference in New Issue
Block a user