reafctor: all code without stream
This commit is contained in:
-17
@@ -1,17 +0,0 @@
|
|||||||
{
|
|
||||||
"db_name": "PostgreSQL",
|
|
||||||
"query": "\n INSERT INTO auth.api_key (key_hash, name, created_by, scopes)\n VALUES ($1, $2, $3, $4)\n ",
|
|
||||||
"describe": {
|
|
||||||
"columns": [],
|
|
||||||
"parameters": {
|
|
||||||
"Left": [
|
|
||||||
"Text",
|
|
||||||
"Text",
|
|
||||||
"Uuid",
|
|
||||||
"TextArray"
|
|
||||||
]
|
|
||||||
},
|
|
||||||
"nullable": []
|
|
||||||
},
|
|
||||||
"hash": "14b38b0f3fca61893dc6e036dc68f643a724b729f73f8308ae0b3a70d87dc5de"
|
|
||||||
}
|
|
||||||
+51
@@ -0,0 +1,51 @@
|
|||||||
|
{
|
||||||
|
"db_name": "PostgreSQL",
|
||||||
|
"query": "\n SELECT\n u.id AS user_id,\n ak.id AS api_key_id,\n ak.scopes AS \"roles!: Vec<Role>\"\n FROM auth.api_key ak\n JOIN auth.app_user u ON u.id = ak.created_by\n WHERE ak.key_hash = $1\n AND ak.revoked_at IS NULL\n ",
|
||||||
|
"describe": {
|
||||||
|
"columns": [
|
||||||
|
{
|
||||||
|
"ordinal": 0,
|
||||||
|
"name": "user_id",
|
||||||
|
"type_info": "Uuid"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"ordinal": 1,
|
||||||
|
"name": "api_key_id",
|
||||||
|
"type_info": "Uuid"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"ordinal": 2,
|
||||||
|
"name": "roles!: Vec<Role>",
|
||||||
|
"type_info": {
|
||||||
|
"Custom": {
|
||||||
|
"name": "auth.role[]",
|
||||||
|
"kind": {
|
||||||
|
"Array": {
|
||||||
|
"Custom": {
|
||||||
|
"name": "auth.role",
|
||||||
|
"kind": {
|
||||||
|
"Enum": [
|
||||||
|
"user",
|
||||||
|
"admin"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"parameters": {
|
||||||
|
"Left": [
|
||||||
|
"Text"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"nullable": [
|
||||||
|
false,
|
||||||
|
false,
|
||||||
|
false
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"hash": "52a8daad9aa79a9c391bfa52b92b268f3910e4e3266f029f17f460c9f5362dd4"
|
||||||
|
}
|
||||||
-14
@@ -1,14 +0,0 @@
|
|||||||
{
|
|
||||||
"db_name": "PostgreSQL",
|
|
||||||
"query": "\n INSERT INTO auth.app_user (id)\n VALUES ($1)\n ON CONFLICT (id) DO NOTHING\n ",
|
|
||||||
"describe": {
|
|
||||||
"columns": [],
|
|
||||||
"parameters": {
|
|
||||||
"Left": [
|
|
||||||
"Uuid"
|
|
||||||
]
|
|
||||||
},
|
|
||||||
"nullable": []
|
|
||||||
},
|
|
||||||
"hash": "70117c16fc3efeebbcb40838d13dd427bf246f4eabd3e335077f47a7aa7882e4"
|
|
||||||
}
|
|
||||||
-34
@@ -1,34 +0,0 @@
|
|||||||
{
|
|
||||||
"db_name": "PostgreSQL",
|
|
||||||
"query": "\n SELECT u.id AS user_id, ak.id as key_id, ak.scopes as roles\n FROM auth.api_key ak\n JOIN auth.app_user u ON u.id = ak.created_by\n WHERE ak.key_hash = $1\n AND ak.revoked_at IS NULL\n ",
|
|
||||||
"describe": {
|
|
||||||
"columns": [
|
|
||||||
{
|
|
||||||
"ordinal": 0,
|
|
||||||
"name": "user_id",
|
|
||||||
"type_info": "Uuid"
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"ordinal": 1,
|
|
||||||
"name": "key_id",
|
|
||||||
"type_info": "Uuid"
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"ordinal": 2,
|
|
||||||
"name": "roles",
|
|
||||||
"type_info": "TextArray"
|
|
||||||
}
|
|
||||||
],
|
|
||||||
"parameters": {
|
|
||||||
"Left": [
|
|
||||||
"Text"
|
|
||||||
]
|
|
||||||
},
|
|
||||||
"nullable": [
|
|
||||||
false,
|
|
||||||
false,
|
|
||||||
false
|
|
||||||
]
|
|
||||||
},
|
|
||||||
"hash": "931f64701ca90a159a54dab717b9cb4004548fc2efc7cbcb056e11e2534eb441"
|
|
||||||
}
|
|
||||||
+34
@@ -0,0 +1,34 @@
|
|||||||
|
{
|
||||||
|
"db_name": "PostgreSQL",
|
||||||
|
"query": "\n INSERT INTO auth.api_key (key_hash, name, created_by, scopes)\n VALUES ($1, $2, $3, $4::auth.role[])\n ",
|
||||||
|
"describe": {
|
||||||
|
"columns": [],
|
||||||
|
"parameters": {
|
||||||
|
"Left": [
|
||||||
|
"Text",
|
||||||
|
"Text",
|
||||||
|
"Uuid",
|
||||||
|
{
|
||||||
|
"Custom": {
|
||||||
|
"name": "auth.role[]",
|
||||||
|
"kind": {
|
||||||
|
"Array": {
|
||||||
|
"Custom": {
|
||||||
|
"name": "auth.role",
|
||||||
|
"kind": {
|
||||||
|
"Enum": [
|
||||||
|
"user",
|
||||||
|
"admin"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"nullable": []
|
||||||
|
},
|
||||||
|
"hash": "ac5047be40cf3c686d3a3f877fc7af232e6a9e9010c947422f0497ff76b06c58"
|
||||||
|
}
|
||||||
+15
-4
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"db_name": "PostgreSQL",
|
"db_name": "PostgreSQL",
|
||||||
"query": "\n SELECT id, parent_id, role, content, created_at, tokens\n FROM chat.message\n WHERE conversation_id = $1\n AND ($2::timestamptz IS NULL OR created_at < $2)\n ORDER BY created_at ASC\n LIMIT $3\n ",
|
"query": "\n SELECT id, parent_id, role as \"role: MessageRole\", content, created_at, tokens\n FROM chat.message\n WHERE conversation_id = $1\n AND ($2::timestamptz IS NULL OR created_at < $2)\n ORDER BY created_at DESC\n LIMIT $3\n ",
|
||||||
"describe": {
|
"describe": {
|
||||||
"columns": [
|
"columns": [
|
||||||
{
|
{
|
||||||
@@ -15,8 +15,19 @@
|
|||||||
},
|
},
|
||||||
{
|
{
|
||||||
"ordinal": 2,
|
"ordinal": 2,
|
||||||
"name": "role",
|
"name": "role: MessageRole",
|
||||||
"type_info": "Text"
|
"type_info": {
|
||||||
|
"Custom": {
|
||||||
|
"name": "chat.role",
|
||||||
|
"kind": {
|
||||||
|
"Enum": [
|
||||||
|
"user",
|
||||||
|
"assistant",
|
||||||
|
"system"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"ordinal": 3,
|
"ordinal": 3,
|
||||||
@@ -50,5 +61,5 @@
|
|||||||
true
|
true
|
||||||
]
|
]
|
||||||
},
|
},
|
||||||
"hash": "aa2722be6d9aec100c73bf65ef14ce3e979afe85a6d9d2b6e0005b9848efbd77"
|
"hash": "ac5b689f154241e193cff087a552af61028fd0ae9f34c13504950be6742f107f"
|
||||||
}
|
}
|
||||||
+12
-1
@@ -13,7 +13,18 @@
|
|||||||
"Left": [
|
"Left": [
|
||||||
"Uuid",
|
"Uuid",
|
||||||
"Uuid",
|
"Uuid",
|
||||||
"Text",
|
{
|
||||||
|
"Custom": {
|
||||||
|
"name": "chat.role",
|
||||||
|
"kind": {
|
||||||
|
"Enum": [
|
||||||
|
"user",
|
||||||
|
"assistant",
|
||||||
|
"system"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
},
|
||||||
"Text",
|
"Text",
|
||||||
"Int4"
|
"Int4"
|
||||||
]
|
]
|
||||||
|
|||||||
+14
@@ -0,0 +1,14 @@
|
|||||||
|
{
|
||||||
|
"db_name": "PostgreSQL",
|
||||||
|
"query": "\n INSERT INTO auth.app_user (id, last_seen_at)\n VALUES ($1, now())\n ON CONFLICT (id)\n DO UPDATE SET last_seen_at = now()\n ",
|
||||||
|
"describe": {
|
||||||
|
"columns": [],
|
||||||
|
"parameters": {
|
||||||
|
"Left": [
|
||||||
|
"Uuid"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"nullable": []
|
||||||
|
},
|
||||||
|
"hash": "f389b37680c97698323c81cb7eeb9bcbafad49d7e1df83cb15a1b86bbdeb4fdf"
|
||||||
|
}
|
||||||
Generated
+37
-10
@@ -93,6 +93,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
|
|||||||
checksum = "31b698c5f9a010f6573133b09e0de5408834d0c82f8d7475a89fc1867a71cd90"
|
checksum = "31b698c5f9a010f6573133b09e0de5408834d0c82f8d7475a89fc1867a71cd90"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"axum-core",
|
"axum-core",
|
||||||
|
"axum-macros",
|
||||||
"bytes",
|
"bytes",
|
||||||
"form_urlencoded",
|
"form_urlencoded",
|
||||||
"futures-util",
|
"futures-util",
|
||||||
@@ -138,6 +139,17 @@ dependencies = [
|
|||||||
"tracing",
|
"tracing",
|
||||||
]
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "axum-macros"
|
||||||
|
version = "0.5.1"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "7aa268c23bfbbd2c4363b9cd302a4f504fb2a9dfe7e3451d66f35dd392e20aca"
|
||||||
|
dependencies = [
|
||||||
|
"proc-macro2",
|
||||||
|
"quote",
|
||||||
|
"syn",
|
||||||
|
]
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "base64"
|
name = "base64"
|
||||||
version = "0.22.1"
|
version = "0.22.1"
|
||||||
@@ -1165,9 +1177,9 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "jsonwebtoken"
|
name = "jsonwebtoken"
|
||||||
version = "10.3.0"
|
version = "10.4.0"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "0529410abe238729a60b108898784df8984c87f6054c9c4fcacc47e4803c1ce1"
|
checksum = "eba32bfb4ffdeaca3e34431072faf01745c9b26d25504aa7a6cf5684334fc4fc"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"aws-lc-rs",
|
"aws-lc-rs",
|
||||||
"base64",
|
"base64",
|
||||||
@@ -1178,6 +1190,7 @@ dependencies = [
|
|||||||
"serde_json",
|
"serde_json",
|
||||||
"signature",
|
"signature",
|
||||||
"simple_asn1",
|
"simple_asn1",
|
||||||
|
"zeroize",
|
||||||
]
|
]
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
@@ -1718,9 +1731,9 @@ checksum = "dc897dd8d9e8bd1ed8cdad82b5966c3e0ecae09fb1907d58efaa013543185d0a"
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "reqwest"
|
name = "reqwest"
|
||||||
version = "0.13.3"
|
version = "0.13.4"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "62e0021ea2c22aed41653bc7e1419abb2c97e038ff2c33d0e1309e49a97deec0"
|
checksum = "219c5811de6525e5416c7d5d53bb656d3afdbc6c5af816e0802bcfa42dbdc1c3"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"base64",
|
"base64",
|
||||||
"bytes",
|
"bytes",
|
||||||
@@ -1972,9 +1985,9 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "serde_json"
|
name = "serde_json"
|
||||||
version = "1.0.149"
|
version = "1.0.150"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "83fc039473c5595ace860d8c4fafa220ff474b3fc6bfdb4293327f1a37e94d86"
|
checksum = "e8014e44b4736ed0538adeecded0fce2a272f22dc9578a7eb6b2d9993c74cfb9"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"itoa",
|
"itoa",
|
||||||
"memchr",
|
"memchr",
|
||||||
@@ -2588,9 +2601,9 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "tower-http"
|
name = "tower-http"
|
||||||
version = "0.6.10"
|
version = "0.6.11"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "68d6fdd9f81c2819c9a8b0e0cd91660e7746a8e6ea2ba7c6b2b057985f6bcb51"
|
checksum = "4cfcf7e2740e6fc6d4d688b4ef00650406bb94adf4731e43c096c3a19fe40840"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"bitflags",
|
"bitflags",
|
||||||
"bytes",
|
"bytes",
|
||||||
@@ -2780,9 +2793,9 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "uuid"
|
name = "uuid"
|
||||||
version = "1.23.1"
|
version = "1.23.2"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "ddd74a9687298c6858e9b88ec8935ec45d22e8fd5e6394fa1bd4e99a87789c76"
|
checksum = "d258b83ceec21034727ecee8c382cfa6c3e133699b0742c64571814fb420c9f7"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"getrandom 0.4.2",
|
"getrandom 0.4.2",
|
||||||
"js-sys",
|
"js-sys",
|
||||||
@@ -3569,6 +3582,20 @@ name = "zeroize"
|
|||||||
version = "1.8.2"
|
version = "1.8.2"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "b97154e67e32c85465826e8bcc1c59429aaaf107c1e4a9e53c8d8ccd5eff88d0"
|
checksum = "b97154e67e32c85465826e8bcc1c59429aaaf107c1e4a9e53c8d8ccd5eff88d0"
|
||||||
|
dependencies = [
|
||||||
|
"zeroize_derive",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "zeroize_derive"
|
||||||
|
version = "1.4.3"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "85a5b4158499876c763cb03bc4e49185d3cccbabb15b33c627f7884f43db852e"
|
||||||
|
dependencies = [
|
||||||
|
"proc-macro2",
|
||||||
|
"quote",
|
||||||
|
"syn",
|
||||||
|
]
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "zerotrie"
|
name = "zerotrie"
|
||||||
|
|||||||
+7
-7
@@ -8,24 +8,24 @@ wiremock = "0.6"
|
|||||||
tokio = { version = "1.52.3", features = ["macros", "rt-multi-thread"] }
|
tokio = { version = "1.52.3", features = ["macros", "rt-multi-thread"] }
|
||||||
|
|
||||||
[dependencies]
|
[dependencies]
|
||||||
axum = "0.8.9"
|
axum = { version = "0.8.9", features = ["macros"] }
|
||||||
utoipa = { version = "5.5.0", features = ["axum_extras", "uuid"] }
|
utoipa = { version = "5.5.0", features = ["axum_extras", "uuid"] }
|
||||||
tokio = { version = "1", features = ["full"] }
|
tokio = { version = "1", features = ["full"] }
|
||||||
serde = { version = "1", features = ["derive"] }
|
serde = { version = "1", features = ["derive"] }
|
||||||
serde_json = "1"
|
serde_json = "1.0.150"
|
||||||
jsonwebtoken = { version = "10.3.0", features = ["aws_lc_rs"] }
|
jsonwebtoken = { version = "10.4.0", features = ["aws_lc_rs"] }
|
||||||
reqwest = { version = "0.13.3", features = ["json", "stream"] }
|
reqwest = { version = "0.13.4", features = ["json", "stream"] }
|
||||||
once_cell = "1"
|
once_cell = "1"
|
||||||
dotenvy = "0.15"
|
dotenvy = "0.15"
|
||||||
thiserror = "2.0.18"
|
thiserror = "2.0.18"
|
||||||
tokio-stream = "0.1"
|
tokio-stream = "0.1"
|
||||||
futures = "0.3"
|
futures = "0.3"
|
||||||
chrono = { version = "0.4.44", features = ["serde"] }
|
chrono = { version = "0.4.44", features = ["serde"] }
|
||||||
uuid = { version = "1.23.1", features = ["v4", "serde"] }
|
uuid = { version = "1.23.2", features = ["v4", "serde"] }
|
||||||
tower-http = { version = "0.6.10", features = ["cors"] }
|
tower-http = { version = "0.6.11", features = ["cors"] }
|
||||||
tracing = "0.1.44"
|
tracing = "0.1.44"
|
||||||
tracing-subscriber = { version = "0.3", features = ["env-filter"]}
|
tracing-subscriber = { version = "0.3", features = ["env-filter"]}
|
||||||
sqlx = { version = "0.8.6", features = ["runtime-tokio-rustls", "postgres", "uuid", "chrono"] }
|
sqlx = { version = "0.8.6", features = ["runtime-tokio-rustls", "postgres", "uuid", "chrono", "macros"] }
|
||||||
rand = "0.8"
|
rand = "0.8"
|
||||||
base64 = "0.22.1"
|
base64 = "0.22.1"
|
||||||
sha2 = "0.11.0"
|
sha2 = "0.11.0"
|
||||||
@@ -19,6 +19,17 @@ curl -s -X POST https://chat.iceberg.black/api/v1/chat/completions \
|
|||||||
]
|
]
|
||||||
}' | jq .
|
}' | jq .
|
||||||
|
|
||||||
|
curl -X POST http://localhost:3001/v1/completions \
|
||||||
|
-H "Content-Type: application/json" \
|
||||||
|
-H "x-api-key: UXVqi1Jazl_-A0TuBudw2Y3PeUiNwCMYayXwBWwuMf0" \
|
||||||
|
-H "Accept: text/event-stream" \
|
||||||
|
-d '{"model": "llama3:latest", "prompt": "hello", "stream": true}'
|
||||||
|
|
||||||
|
|
||||||
|
providers → LlmError ┐
|
||||||
|
├── services → ServiceError → api → HTTP
|
||||||
|
databases → DbError ┘
|
||||||
|
|
||||||
# 🦙 Ollama Rust API Wrapper
|
# 🦙 Ollama Rust API Wrapper
|
||||||
|
|
||||||
A high-performance Rust API wrapper around Ollama, providing an OpenAI-compatible interface, model lifecycle management, and advanced runtime features.
|
A high-performance Rust API wrapper around Ollama, providing an OpenAI-compatible interface, model lifecycle management, and advanced runtime features.
|
||||||
@@ -266,3 +277,5 @@ This project turns Ollama into:
|
|||||||
|
|
||||||
# TODO
|
# TODO
|
||||||
- open api doc for bearer token
|
- open api doc for bearer token
|
||||||
|
- Unify check before sending to ollama payload
|
||||||
|
- load/unload model functions
|
||||||
@@ -0,0 +1,57 @@
|
|||||||
|
use axum::{
|
||||||
|
Router,
|
||||||
|
http::{HeaderName, HeaderValue, Method, header},
|
||||||
|
};
|
||||||
|
use std::{env, sync::Arc};
|
||||||
|
use tower_http::cors::CorsLayer;
|
||||||
|
|
||||||
|
use crate::services::{AuthService, ChatService, ConversationService};
|
||||||
|
use crate::{api, databases::postgres, providers::ollama::client::OllamaProvider};
|
||||||
|
|
||||||
|
use once_cell::sync::Lazy;
|
||||||
|
|
||||||
|
static OLLAMA_URL: Lazy<String> = Lazy::new(|| env::var("OLLAMA_URL").expect("OLLAMA_URL not set"));
|
||||||
|
|
||||||
|
pub async fn build_app() -> Router {
|
||||||
|
let database_url = env::var("DATABASE_URL").expect("DATABASE_URL must be set");
|
||||||
|
|
||||||
|
let pool = postgres::pool::create_pool(&database_url)
|
||||||
|
.await
|
||||||
|
.expect("Fatal error");
|
||||||
|
let conversation_service = ConversationService::new(pool.clone());
|
||||||
|
let api_key_service = AuthService::new(pool.clone());
|
||||||
|
|
||||||
|
let ollama = OllamaProvider::new(OLLAMA_URL.as_str());
|
||||||
|
let chat_service = ChatService::new(ollama, conversation_service.clone());
|
||||||
|
|
||||||
|
let state = Arc::new(api::state::AppState {
|
||||||
|
conversation_service,
|
||||||
|
auth_service: api_key_service,
|
||||||
|
chat_service,
|
||||||
|
});
|
||||||
|
|
||||||
|
let cors_origin =
|
||||||
|
env::var("CORS_ORIGIN").unwrap_or_else(|_| "http://localhost:3000".to_string());
|
||||||
|
|
||||||
|
let cors = CorsLayer::new()
|
||||||
|
.allow_origin(cors_origin.parse::<HeaderValue>().unwrap())
|
||||||
|
.allow_methods([
|
||||||
|
Method::GET,
|
||||||
|
Method::POST,
|
||||||
|
Method::PUT,
|
||||||
|
Method::DELETE,
|
||||||
|
Method::OPTIONS,
|
||||||
|
])
|
||||||
|
.allow_headers([
|
||||||
|
header::CONTENT_TYPE,
|
||||||
|
header::AUTHORIZATION,
|
||||||
|
header::ACCEPT,
|
||||||
|
HeaderName::from_static("x-api-key"),
|
||||||
|
])
|
||||||
|
.allow_credentials(true);
|
||||||
|
|
||||||
|
Router::new()
|
||||||
|
.nest("/v1", api::routes::v1::router(state.clone()))
|
||||||
|
.layer(cors)
|
||||||
|
.with_state(state)
|
||||||
|
}
|
||||||
@@ -0,0 +1,55 @@
|
|||||||
|
use utoipa::OpenApi;
|
||||||
|
|
||||||
|
use crate::api;
|
||||||
|
use crate::api::routes;
|
||||||
|
|
||||||
|
#[derive(OpenApi)]
|
||||||
|
#[openapi(
|
||||||
|
info(
|
||||||
|
title = "Ollama Proxy",
|
||||||
|
description = "OpenAI-compatible proxy for local Ollama models",
|
||||||
|
version = "0.1.0",
|
||||||
|
license(
|
||||||
|
name = "MIT",
|
||||||
|
url = "https://opensource.org/licenses/MIT"
|
||||||
|
),
|
||||||
|
),
|
||||||
|
paths(
|
||||||
|
// routes::v1::chat::completions,
|
||||||
|
// routes::v1::chat::chat_completions,
|
||||||
|
routes::v1::models::list_models,
|
||||||
|
// routes::v1::models::load_model,
|
||||||
|
// routes::v1::models::unload_model,
|
||||||
|
),
|
||||||
|
components(
|
||||||
|
schemas(
|
||||||
|
api::errors::ErrorResponse,
|
||||||
|
api::types::ModelsResponse,
|
||||||
|
api::types::ModelInfo,
|
||||||
|
api::types::LoadModelResponse,
|
||||||
|
api::types::LoadModelRequest,
|
||||||
|
api::types::UnloadModelResponse,
|
||||||
|
api::types::LLMOptions,
|
||||||
|
api::types::CompletionRequest,
|
||||||
|
api::types::CompletionObject,
|
||||||
|
api::types::FinishReason,
|
||||||
|
api::types::CompletionResponse,
|
||||||
|
api::types::Choice,
|
||||||
|
api::types::Usage,
|
||||||
|
api::types::CompletionChunk,
|
||||||
|
api::types::ChatRequest,
|
||||||
|
api::types::Message,
|
||||||
|
api::types::Role,
|
||||||
|
api::types::ChatResponse,
|
||||||
|
api::types::ChatChoice,
|
||||||
|
api::types::ChatCompletionChunk,
|
||||||
|
api::types::ChatChunkChoice,
|
||||||
|
api::types::Delta,
|
||||||
|
)
|
||||||
|
),
|
||||||
|
tags(
|
||||||
|
(name = "chat", description = "Chat & completions"),
|
||||||
|
(name = "models", description = "Model management")
|
||||||
|
)
|
||||||
|
)]
|
||||||
|
pub struct ApiDoc;
|
||||||
+123
-89
@@ -1,114 +1,148 @@
|
|||||||
use crate::databases::postgres::errors::DbError;
|
use crate::databases::errors::DbError;
|
||||||
use crate::providers::ollama::errors::OllamaError;
|
use crate::providers::keycloak::errors::AuthError;
|
||||||
|
use crate::providers::ollama::errors::LlmError;
|
||||||
|
use crate::services::errors::ServiceError;
|
||||||
|
|
||||||
use axum::Json;
|
use axum::Json;
|
||||||
use axum::http::StatusCode;
|
use axum::http::StatusCode;
|
||||||
use axum::response::{IntoResponse, Response};
|
use axum::response::{IntoResponse, Response};
|
||||||
use serde::Serialize;
|
use serde::Serialize;
|
||||||
|
use thiserror::Error;
|
||||||
|
use utoipa::ToSchema;
|
||||||
|
|
||||||
#[derive(Serialize)]
|
#[derive(Serialize, ToSchema)]
|
||||||
pub struct ErrorResponse {
|
pub struct ErrorResponse {
|
||||||
|
pub status: u16,
|
||||||
pub error: String,
|
pub error: String,
|
||||||
pub code: String,
|
pub code: String,
|
||||||
}
|
}
|
||||||
|
|
||||||
pub enum ApiError {
|
pub struct ApiError {
|
||||||
Db(DbError),
|
pub status: StatusCode,
|
||||||
Ollama(OllamaError),
|
pub code: &'static str,
|
||||||
|
pub message: String,
|
||||||
}
|
}
|
||||||
|
|
||||||
impl From<DbError> for ApiError {
|
#[derive(Debug, Error)]
|
||||||
fn from(e: DbError) -> Self {
|
pub enum AuthMiddlewareError {
|
||||||
ApiError::Db(e)
|
#[error("invalid authorization format")]
|
||||||
|
InvalidAuthorizationFormat,
|
||||||
|
|
||||||
|
#[error("authentication required")]
|
||||||
|
AuthenticationRequired,
|
||||||
|
|
||||||
|
#[error("insufficient permissions")]
|
||||||
|
Forbidden,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<AuthMiddlewareError> for ApiError {
|
||||||
|
fn from(err: AuthMiddlewareError) -> Self {
|
||||||
|
match err {
|
||||||
|
AuthMiddlewareError::InvalidAuthorizationFormat => Self {
|
||||||
|
status: StatusCode::UNAUTHORIZED,
|
||||||
|
code: "AUTH_INVALID_FORMAT",
|
||||||
|
message: "invalid authorization format".into(),
|
||||||
|
},
|
||||||
|
AuthMiddlewareError::AuthenticationRequired => Self {
|
||||||
|
status: StatusCode::UNAUTHORIZED,
|
||||||
|
code: "AUTH_REQUIRED",
|
||||||
|
message: "missing authorization header or api key".into(),
|
||||||
|
},
|
||||||
|
AuthMiddlewareError::Forbidden => Self {
|
||||||
|
status: StatusCode::FORBIDDEN,
|
||||||
|
code: "AUTH_FORBIDDEN",
|
||||||
|
message: "insufficient permissions".into(),
|
||||||
|
},
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
impl From<OllamaError> for ApiError {
|
impl From<ServiceError> for ApiError {
|
||||||
fn from(e: OllamaError) -> Self {
|
fn from(err: ServiceError) -> Self {
|
||||||
ApiError::Ollama(e)
|
match err {
|
||||||
|
ServiceError::Db(e) => match e {
|
||||||
|
DbError::Connection(_) => Self {
|
||||||
|
status: StatusCode::INTERNAL_SERVER_ERROR,
|
||||||
|
code: "DB_CONNECTION",
|
||||||
|
message: "database connection error".into(),
|
||||||
|
},
|
||||||
|
DbError::Timeout => Self {
|
||||||
|
status: StatusCode::REQUEST_TIMEOUT,
|
||||||
|
code: "DB_TIMEOUT",
|
||||||
|
message: "database timeout".into(),
|
||||||
|
},
|
||||||
|
DbError::NotFound => Self {
|
||||||
|
status: StatusCode::NOT_FOUND,
|
||||||
|
code: "DB_NOT_FOUND",
|
||||||
|
message: "not found".into(),
|
||||||
|
},
|
||||||
|
DbError::Unauthorized => Self {
|
||||||
|
status: StatusCode::UNAUTHORIZED,
|
||||||
|
code: "DB_UNAUTHORIZED",
|
||||||
|
message: "unauthorized".into(),
|
||||||
|
},
|
||||||
|
},
|
||||||
|
|
||||||
|
ServiceError::Llm(e) => match e {
|
||||||
|
LlmError::MissingPrompt => Self {
|
||||||
|
status: StatusCode::BAD_REQUEST,
|
||||||
|
code: "OLLAMA_MISSING_PROMPT",
|
||||||
|
message: "prompt is required".into(),
|
||||||
|
},
|
||||||
|
LlmError::Http(_) => Self {
|
||||||
|
status: StatusCode::BAD_GATEWAY,
|
||||||
|
code: "OLLAMA_HTTP_ERROR",
|
||||||
|
message: "upstream error".into(),
|
||||||
|
},
|
||||||
|
_ => Self {
|
||||||
|
status: StatusCode::BAD_REQUEST,
|
||||||
|
code: "OLLAMA_ERROR",
|
||||||
|
message: "llm error".into(),
|
||||||
|
},
|
||||||
|
},
|
||||||
|
|
||||||
|
ServiceError::Auth(e) => match e {
|
||||||
|
AuthError::InvalidToken | AuthError::TokenValidationFailed => Self {
|
||||||
|
status: StatusCode::UNAUTHORIZED,
|
||||||
|
code: "AUTH_INVALID_TOKEN",
|
||||||
|
message: "invalid or expired token".into(),
|
||||||
|
},
|
||||||
|
AuthError::InvalidHeader
|
||||||
|
// | AuthError::InvalidHeaderDecode
|
||||||
|
| AuthError::MissingKid => Self {
|
||||||
|
status: StatusCode::UNAUTHORIZED,
|
||||||
|
code: "AUTH_INVALID_HEADER",
|
||||||
|
message: "invalid authorization header".into(),
|
||||||
|
},
|
||||||
|
AuthError::JwkNotFound
|
||||||
|
| AuthError::InvalidJwks
|
||||||
|
| AuthError::MissingModulus
|
||||||
|
| AuthError::MissingExponent
|
||||||
|
| AuthError::InvalidDecodingKey => Self {
|
||||||
|
status: StatusCode::INTERNAL_SERVER_ERROR,
|
||||||
|
code: "AUTH_JWKS_ERROR",
|
||||||
|
message: "key validation error".into(),
|
||||||
|
},
|
||||||
|
AuthError::JwksFetchFailed
|
||||||
|
| AuthError::JwksRefreshFailed
|
||||||
|
| AuthError::Reqwest(_) => Self {
|
||||||
|
status: StatusCode::SERVICE_UNAVAILABLE,
|
||||||
|
code: "AUTH_JWKS_FETCH",
|
||||||
|
message: "failed to fetch authorization keys".into(),
|
||||||
|
},
|
||||||
|
},
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
impl IntoResponse for ApiError {
|
impl IntoResponse for ApiError {
|
||||||
fn into_response(self) -> Response {
|
fn into_response(self) -> Response {
|
||||||
let (status, body) = match self {
|
let body = ErrorResponse {
|
||||||
ApiError::Db(db_err) => match db_err {
|
status: self.status.as_u16(),
|
||||||
DbError::Connection(_) => (
|
error: self.message,
|
||||||
StatusCode::INTERNAL_SERVER_ERROR,
|
code: self.code.to_string(),
|
||||||
ErrorResponse {
|
|
||||||
error: "database connection error".to_string(),
|
|
||||||
code: "DB_CONNECTION".to_string(),
|
|
||||||
},
|
|
||||||
),
|
|
||||||
DbError::Timeout => (
|
|
||||||
StatusCode::REQUEST_TIMEOUT,
|
|
||||||
ErrorResponse {
|
|
||||||
error: "database timeout".to_string(),
|
|
||||||
code: "DB_TIMEOUT".to_string(),
|
|
||||||
},
|
|
||||||
),
|
|
||||||
DbError::NotFound => (
|
|
||||||
StatusCode::NOT_FOUND,
|
|
||||||
ErrorResponse {
|
|
||||||
error: "not found".to_string(),
|
|
||||||
code: "DB_NOT_FOUND".to_string(),
|
|
||||||
},
|
|
||||||
),
|
|
||||||
},
|
|
||||||
|
|
||||||
ApiError::Ollama(err) => match err {
|
|
||||||
OllamaError::MissingPrompt => (
|
|
||||||
StatusCode::BAD_REQUEST,
|
|
||||||
ErrorResponse {
|
|
||||||
error: "prompt is required and cannot be empty".to_string(),
|
|
||||||
code: "OLLAMA_MISSING_PROMPT".to_string(),
|
|
||||||
},
|
|
||||||
),
|
|
||||||
OllamaError::MissingModel => (
|
|
||||||
StatusCode::BAD_REQUEST,
|
|
||||||
ErrorResponse {
|
|
||||||
error: "model is required and cannot be empty".to_string(),
|
|
||||||
code: "OLLAMA_MISSING_MODEL".to_string(),
|
|
||||||
},
|
|
||||||
),
|
|
||||||
OllamaError::ModelNotFound(m) => (
|
|
||||||
StatusCode::UNPROCESSABLE_ENTITY,
|
|
||||||
ErrorResponse {
|
|
||||||
error: format!("model '{m}' is not available"),
|
|
||||||
code: "OLLAMA_MODEL_NOT_FOUND".to_string(),
|
|
||||||
},
|
|
||||||
),
|
|
||||||
OllamaError::MissingKeepAlive => (
|
|
||||||
StatusCode::BAD_REQUEST,
|
|
||||||
ErrorResponse {
|
|
||||||
error: "keep_alive is required".to_string(),
|
|
||||||
code: "OLLAMA_MISSING_KEEPALIVE".to_string(),
|
|
||||||
},
|
|
||||||
),
|
|
||||||
OllamaError::InvalidKeepAlive(v) => (
|
|
||||||
StatusCode::BAD_REQUEST,
|
|
||||||
ErrorResponse {
|
|
||||||
error: format!("invalid keep_alive '{v}'"),
|
|
||||||
code: "OLLAMA_INVALID_KEEPALIVE".to_string(),
|
|
||||||
},
|
|
||||||
),
|
|
||||||
OllamaError::MissingMessages => (
|
|
||||||
StatusCode::BAD_REQUEST,
|
|
||||||
ErrorResponse {
|
|
||||||
error: "messages array required".to_string(),
|
|
||||||
code: "OLLAMA_MISSING_MESSAGES".to_string(),
|
|
||||||
},
|
|
||||||
),
|
|
||||||
OllamaError::Http(e) => (
|
|
||||||
StatusCode::INTERNAL_SERVER_ERROR,
|
|
||||||
ErrorResponse {
|
|
||||||
error: e.to_string(),
|
|
||||||
code: "OLLAMA_HTTP_ERROR".to_string(),
|
|
||||||
},
|
|
||||||
),
|
|
||||||
},
|
|
||||||
};
|
};
|
||||||
|
|
||||||
(status, Json(body)).into_response()
|
(self.status, Json(body)).into_response()
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,162 @@
|
|||||||
|
use crate::api::errors::{ApiError, AuthMiddlewareError};
|
||||||
|
use crate::api::state::SharedState;
|
||||||
|
|
||||||
|
use axum::{
|
||||||
|
extract::{Request, State},
|
||||||
|
http::{HeaderMap, StatusCode},
|
||||||
|
middleware::Next,
|
||||||
|
response::{IntoResponse, Response},
|
||||||
|
};
|
||||||
|
|
||||||
|
enum JwtError {
|
||||||
|
MissingHeader,
|
||||||
|
InvalidFormat,
|
||||||
|
Service(ApiError),
|
||||||
|
}
|
||||||
|
|
||||||
|
enum ApiKeyError {
|
||||||
|
MissingHeader,
|
||||||
|
Service(ApiError),
|
||||||
|
}
|
||||||
|
|
||||||
|
pub async fn auth_middleware(
|
||||||
|
State(state): State<SharedState>,
|
||||||
|
req: Request,
|
||||||
|
next: Next,
|
||||||
|
) -> Response {
|
||||||
|
let headers = req.headers();
|
||||||
|
|
||||||
|
let auth = match try_jwt(&state, headers).await {
|
||||||
|
Ok(auth) => auth,
|
||||||
|
Err(jwt_err) => match try_api_key(&state, headers).await {
|
||||||
|
Ok(auth) => auth,
|
||||||
|
Err(api_key_err) => {
|
||||||
|
let err = match (jwt_err, api_key_err) {
|
||||||
|
// Both headers absent
|
||||||
|
(JwtError::MissingHeader, ApiKeyError::MissingHeader) => {
|
||||||
|
AuthMiddlewareError::AuthenticationRequired
|
||||||
|
}
|
||||||
|
// JWT header present but malformed — surface it, api key result irrelevant
|
||||||
|
(JwtError::InvalidFormat, _) => AuthMiddlewareError::InvalidAuthorizationFormat,
|
||||||
|
// JWT service failure — api key header was missing, so JWT was the intended method
|
||||||
|
(JwtError::Service(e), ApiKeyError::MissingHeader) => {
|
||||||
|
return e.into_response();
|
||||||
|
}
|
||||||
|
// Both services failed
|
||||||
|
(JwtError::Service(e), ApiKeyError::Service(_)) => {
|
||||||
|
return e.into_response();
|
||||||
|
}
|
||||||
|
// JWT missing, api key service failed
|
||||||
|
(JwtError::MissingHeader, ApiKeyError::Service(e)) => {
|
||||||
|
return e.into_response();
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
return ApiError::from(err).into_response();
|
||||||
|
}
|
||||||
|
},
|
||||||
|
};
|
||||||
|
|
||||||
|
match handle_auth(&state, req, next, auth).await {
|
||||||
|
Ok(response) => response,
|
||||||
|
Err(err) => err.into_response(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn try_jwt(
|
||||||
|
state: &SharedState,
|
||||||
|
headers: &HeaderMap,
|
||||||
|
) -> Result<crate::core::auth::Auth, JwtError> {
|
||||||
|
let token = headers
|
||||||
|
.get("authorization")
|
||||||
|
.and_then(|v| v.to_str().ok())
|
||||||
|
.ok_or(JwtError::MissingHeader)?
|
||||||
|
.strip_prefix("Bearer ")
|
||||||
|
.ok_or(JwtError::InvalidFormat)?;
|
||||||
|
|
||||||
|
let claims = state
|
||||||
|
.auth_service
|
||||||
|
.validate_jwt(token)
|
||||||
|
.await
|
||||||
|
.map_err(|e| JwtError::Service(e.into()))?;
|
||||||
|
|
||||||
|
Ok(crate::core::auth::Auth::Jwt(claims))
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn try_api_key(
|
||||||
|
state: &SharedState,
|
||||||
|
headers: &HeaderMap,
|
||||||
|
) -> Result<crate::core::auth::Auth, ApiKeyError> {
|
||||||
|
let key = headers
|
||||||
|
.get("x-api-key")
|
||||||
|
.and_then(|v| v.to_str().ok())
|
||||||
|
.ok_or(ApiKeyError::MissingHeader)?;
|
||||||
|
|
||||||
|
let auth = state
|
||||||
|
.auth_service
|
||||||
|
.validate_api_key(key)
|
||||||
|
.await
|
||||||
|
.map_err(|e| ApiKeyError::Service(e.into()))?;
|
||||||
|
|
||||||
|
Ok(crate::core::auth::Auth::ApiKey(auth))
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn handle_auth(
|
||||||
|
state: &SharedState,
|
||||||
|
mut request: Request,
|
||||||
|
next: Next,
|
||||||
|
auth: crate::core::auth::Auth,
|
||||||
|
) -> Result<Response, ApiError> {
|
||||||
|
match &auth {
|
||||||
|
crate::core::auth::Auth::Jwt(_) => {
|
||||||
|
state.auth_service.create_user(&auth.user_id()).await?;
|
||||||
|
}
|
||||||
|
crate::core::auth::Auth::ApiKey(key) => {
|
||||||
|
state
|
||||||
|
.auth_service
|
||||||
|
.update_last_access_api_key(&key.user_id)
|
||||||
|
.await?;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
request.extensions_mut().insert(auth);
|
||||||
|
|
||||||
|
Ok(next.run(request).await)
|
||||||
|
}
|
||||||
|
|
||||||
|
// ── Role guard ───────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
#[macro_export]
|
||||||
|
macro_rules! role_guard {
|
||||||
|
($jwt:expr, $api:expr) => {
|
||||||
|
middleware::from_fn(move |req, next| {
|
||||||
|
$crate::api::middlewares::auth::require_roles(req, next, $jwt, $api)
|
||||||
|
})
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
pub async fn require_roles(
|
||||||
|
request: Request,
|
||||||
|
next: Next,
|
||||||
|
jwt_role: Option<&'static str>,
|
||||||
|
api_key_role: Option<&crate::core::auth::api_key::KeyRole>,
|
||||||
|
) -> Result<Response, ApiError> {
|
||||||
|
let auth = request
|
||||||
|
.extensions()
|
||||||
|
.get::<crate::core::auth::Auth>()
|
||||||
|
.ok_or(ApiError {
|
||||||
|
status: StatusCode::UNAUTHORIZED,
|
||||||
|
code: "MISSING_AUTH_HEADER",
|
||||||
|
message: "API key missing".into(),
|
||||||
|
})?;
|
||||||
|
|
||||||
|
if jwt_role.is_some_and(|role| !auth.has_jwt_role(role)) {
|
||||||
|
return Err(AuthMiddlewareError::Forbidden.into());
|
||||||
|
}
|
||||||
|
|
||||||
|
if api_key_role.is_some_and(|role| !auth.has_apikey_role(role)) {
|
||||||
|
return Err(AuthMiddlewareError::Forbidden.into());
|
||||||
|
}
|
||||||
|
|
||||||
|
Ok(next.run(request).await)
|
||||||
|
}
|
||||||
@@ -1 +1,7 @@
|
|||||||
|
pub mod app;
|
||||||
|
pub mod docs;
|
||||||
pub mod errors;
|
pub mod errors;
|
||||||
|
pub mod middlewares;
|
||||||
|
pub mod routes;
|
||||||
|
pub mod state;
|
||||||
|
pub mod types;
|
||||||
|
|||||||
@@ -0,0 +1,25 @@
|
|||||||
|
use crate::api::errors::ApiError;
|
||||||
|
use crate::api::state::SharedState;
|
||||||
|
use crate::api::types::{CreateApiKeyRequest, CreateApiKeyResponse};
|
||||||
|
use crate::core::auth::Auth;
|
||||||
|
|
||||||
|
use axum::{
|
||||||
|
Json,
|
||||||
|
extract::{Extension, State},
|
||||||
|
};
|
||||||
|
|
||||||
|
pub async fn create_api_key(
|
||||||
|
State(state): State<SharedState>,
|
||||||
|
Extension(claims): Extension<Auth>,
|
||||||
|
Json(body): Json<CreateApiKeyRequest>,
|
||||||
|
) -> Result<Json<CreateApiKeyResponse>, ApiError> {
|
||||||
|
let payload = crate::core::auth::api_key::CreateApiKeyRequest {
|
||||||
|
user_id: claims.user_id(),
|
||||||
|
name: body.name,
|
||||||
|
roles: body.scopes.into_iter().map(Into::into).collect(),
|
||||||
|
};
|
||||||
|
|
||||||
|
let key = state.auth_service.create_api_key(payload).await?;
|
||||||
|
|
||||||
|
Ok(Json(CreateApiKeyResponse { api_key: key }))
|
||||||
|
}
|
||||||
@@ -0,0 +1,337 @@
|
|||||||
|
use crate::api;
|
||||||
|
use crate::api::errors::{ApiError, ErrorResponse};
|
||||||
|
use crate::api::state::SharedState;
|
||||||
|
use crate::core::auth::Auth;
|
||||||
|
use crate::core::llm::completions::CompletionResult;
|
||||||
|
|
||||||
|
use axum::{
|
||||||
|
Json,
|
||||||
|
extract::{Extension, Path, Query, State},
|
||||||
|
response::{
|
||||||
|
IntoResponse,
|
||||||
|
sse::{Event, KeepAlive, Sse},
|
||||||
|
},
|
||||||
|
};
|
||||||
|
use futures::StreamExt;
|
||||||
|
use uuid::Uuid;
|
||||||
|
|
||||||
|
#[utoipa::path(
|
||||||
|
post,
|
||||||
|
path = "/completions",
|
||||||
|
tag = "chat",
|
||||||
|
request_body(
|
||||||
|
content = api::types::CompletionRequest,
|
||||||
|
description = "Text completion request",
|
||||||
|
content_type = "application/json"
|
||||||
|
),
|
||||||
|
responses(
|
||||||
|
(
|
||||||
|
status = 200,
|
||||||
|
description = "Text completion response. If stream=true, response is SSE stream of chunks ending in [DONE].",
|
||||||
|
body = api::types::CompletionResponse,
|
||||||
|
content_type = "application/json"
|
||||||
|
),
|
||||||
|
(
|
||||||
|
status = 400,
|
||||||
|
description = "Invalid request: missing prompt, model, or invalid format",
|
||||||
|
body = ErrorResponse,
|
||||||
|
example = json!({ "error": "prompt is required and cannot be empty" })
|
||||||
|
),
|
||||||
|
(
|
||||||
|
status = 422,
|
||||||
|
description = "Model not found or not available locally",
|
||||||
|
body = ErrorResponse,
|
||||||
|
example = json!({ "error": "model 'llama3' is not available — run `ollama pull llama3` first" })
|
||||||
|
),
|
||||||
|
(
|
||||||
|
status = 500,
|
||||||
|
description = "Internal server error (Ollama or network failure)",
|
||||||
|
body = ErrorResponse,
|
||||||
|
example = json!({ "error": "connection refused" })
|
||||||
|
)
|
||||||
|
)
|
||||||
|
)]
|
||||||
|
pub async fn completions(
|
||||||
|
State(state): State<SharedState>,
|
||||||
|
Json(body): Json<api::types::CompletionRequest>,
|
||||||
|
) -> Result<impl IntoResponse, ApiError> {
|
||||||
|
tracing::debug!("Received /completion with body {:?}", body);
|
||||||
|
|
||||||
|
let response = state.chat_service.complete(body.into()).await?;
|
||||||
|
|
||||||
|
match response {
|
||||||
|
CompletionResult::NoStream(res) => Ok(Json::<
|
||||||
|
crate::core::llm::completions::CompletionResultNoStream,
|
||||||
|
>(res)
|
||||||
|
.into_response()),
|
||||||
|
CompletionResult::Stream(stream) => {
|
||||||
|
let sse_stream = stream.map(|item| match item {
|
||||||
|
Ok(event) => {
|
||||||
|
let data = serde_json::to_string(&event).unwrap_or_default();
|
||||||
|
Ok(Event::default().data(data))
|
||||||
|
}
|
||||||
|
Err(e) => Err(e),
|
||||||
|
});
|
||||||
|
|
||||||
|
Ok(Sse::new(sse_stream)
|
||||||
|
.keep_alive(KeepAlive::default())
|
||||||
|
.into_response())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[utoipa::path(
|
||||||
|
post,
|
||||||
|
path = "/chat/completions",
|
||||||
|
tag = "chat",
|
||||||
|
request_body(
|
||||||
|
content = api::types::ChatRequest,
|
||||||
|
description = "Chat completion request with message history",
|
||||||
|
content_type = "application/json"
|
||||||
|
),
|
||||||
|
responses(
|
||||||
|
(
|
||||||
|
status = 200,
|
||||||
|
description = "Chat completion response. If stream=false returns JSON. If stream=true returns SSE stream of chunks ending with [DONE].",
|
||||||
|
body = api::types::ChatResponse,
|
||||||
|
content_type = "application/json"
|
||||||
|
),
|
||||||
|
(
|
||||||
|
status = 400,
|
||||||
|
description = "Invalid request",
|
||||||
|
body = api::errors::ErrorResponse,
|
||||||
|
example = json!({ "error": "messages array with at least one user message is required" })
|
||||||
|
),
|
||||||
|
(
|
||||||
|
status = 401,
|
||||||
|
description = "Unauthorized",
|
||||||
|
body = api::errors::ErrorResponse,
|
||||||
|
example = json!({ "error": "missing or invalid token" })
|
||||||
|
),
|
||||||
|
(
|
||||||
|
status = 422,
|
||||||
|
description = "Model not found or unavailable",
|
||||||
|
body = api::errors::ErrorResponse,
|
||||||
|
example = json!({ "error": "model 'llama3' is not available — run `ollama pull llama3` first" })
|
||||||
|
),
|
||||||
|
(
|
||||||
|
status = 500,
|
||||||
|
description = "Internal server error",
|
||||||
|
body = api::errors::ErrorResponse,
|
||||||
|
example = json!({ "error": "connection refused" })
|
||||||
|
)
|
||||||
|
)
|
||||||
|
)]
|
||||||
|
pub async fn chat_completions(
|
||||||
|
State(state): State<SharedState>,
|
||||||
|
Extension(auth): Extension<Auth>,
|
||||||
|
Json(body): Json<api::types::ChatRequest>,
|
||||||
|
) -> Result<impl IntoResponse, ApiError> {
|
||||||
|
tracing::debug!("Received /chat/completion with body {:?}", body);
|
||||||
|
|
||||||
|
let response = state.chat_service.chat_complete(body.into(), &auth).await?;
|
||||||
|
|
||||||
|
match response {
|
||||||
|
crate::core::llm::chat::ChatCompletionResult::NoStream(res) => {
|
||||||
|
Ok(Json::<crate::core::llm::chat::ChatCompletionResultNoStream>(res).into_response())
|
||||||
|
}
|
||||||
|
crate::core::llm::chat::ChatCompletionResult::Stream(stream) => {
|
||||||
|
let sse_stream = stream.map(|item| match item {
|
||||||
|
Ok(event) => {
|
||||||
|
let data = serde_json::to_string(&event).unwrap_or_default();
|
||||||
|
Ok(Event::default().data(data))
|
||||||
|
}
|
||||||
|
Err(e) => Err(e),
|
||||||
|
});
|
||||||
|
|
||||||
|
Ok(Sse::new(sse_stream)
|
||||||
|
.keep_alive(KeepAlive::default())
|
||||||
|
.into_response())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Conversation retrieveing
|
||||||
|
pub async fn get_conversations(
|
||||||
|
State(state): State<SharedState>,
|
||||||
|
Extension(auth): Extension<Auth>,
|
||||||
|
Query(params): Query<api::types::CursorPage>,
|
||||||
|
) -> Result<Json<api::types::ConversationListResponse>, ApiError> {
|
||||||
|
tracing::debug!("Conversation hit: {:?}", auth);
|
||||||
|
|
||||||
|
let pointer = crate::core::databases::conversations::CursorPage {
|
||||||
|
limit: params.limit.unwrap_or(10),
|
||||||
|
before: params.before,
|
||||||
|
};
|
||||||
|
|
||||||
|
let conversations = state
|
||||||
|
.conversation_service
|
||||||
|
.get_conversations_entries(auth.user_id(), pointer)
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
Ok(Json(conversations.into()))
|
||||||
|
}
|
||||||
|
|
||||||
|
pub async fn get_messages(
|
||||||
|
State(state): State<SharedState>,
|
||||||
|
Extension(auth): Extension<Auth>,
|
||||||
|
Path(conversation_id): Path<Uuid>,
|
||||||
|
Query(params): Query<api::types::CursorPage>,
|
||||||
|
) -> Result<Json<api::types::MessageListResponse>, ApiError> {
|
||||||
|
tracing::debug!("Messages hit: {:?}", auth);
|
||||||
|
|
||||||
|
let pointer = crate::core::databases::conversations::CursorPage {
|
||||||
|
limit: params.limit.unwrap_or(15),
|
||||||
|
before: params.before,
|
||||||
|
};
|
||||||
|
|
||||||
|
let messages = state
|
||||||
|
.conversation_service
|
||||||
|
.get_messages_entries(auth.user_id(), conversation_id, pointer)
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
Ok(Json(messages.into()))
|
||||||
|
}
|
||||||
|
|
||||||
|
// async fn handle_stream(
|
||||||
|
// state: AppState,
|
||||||
|
// auth: Auth,
|
||||||
|
// body: api::ChatRequest,
|
||||||
|
// conversation_id: Option<Uuid>,
|
||||||
|
// ) -> Result<Response, ApiError> {
|
||||||
|
// // Handle anonymous (API key) path early — no DB logging
|
||||||
|
// let Some(conv_id) = conversation_id else {
|
||||||
|
// let stream = state.ollama.chat_completions_stream(&body).await?;
|
||||||
|
|
||||||
|
// let plain_stream = stream.map(
|
||||||
|
// |item| -> Result<Event, crate::providers::ollama::errors::OllamaError> {
|
||||||
|
// match item {
|
||||||
|
// Ok(chunk) => Ok(
|
||||||
|
// Event::default().data(serde_json::to_string(&chunk).unwrap_or_default())
|
||||||
|
// ),
|
||||||
|
// Err(e) => Err(e),
|
||||||
|
// }
|
||||||
|
// },
|
||||||
|
// );
|
||||||
|
// return Ok(Sse::new(plain_stream)
|
||||||
|
// .keep_alive(KeepAlive::default())
|
||||||
|
// .into_response());
|
||||||
|
// };
|
||||||
|
|
||||||
|
// // From here conv_id is a plain Uuid — all variables stay in scope
|
||||||
|
// let user_msg_id = log_user_message(
|
||||||
|
// &state.postgres,
|
||||||
|
// auth.user_id(),
|
||||||
|
// conv_id,
|
||||||
|
// body.parent_id,
|
||||||
|
// body.messages
|
||||||
|
// .last()
|
||||||
|
// .map(|m| m.content.as_str())
|
||||||
|
// .unwrap_or(""),
|
||||||
|
// None,
|
||||||
|
// )
|
||||||
|
// .await?;
|
||||||
|
|
||||||
|
// let start_event = api::StreamEvent::Start(api::StartEventData {
|
||||||
|
// conversation_id: conv_id,
|
||||||
|
// created: chrono::Utc::now().timestamp() as u64,
|
||||||
|
// id: user_msg_id,
|
||||||
|
// });
|
||||||
|
|
||||||
|
// let (tx, rx) = tokio::sync::mpsc::channel::<
|
||||||
|
// Result<Event, crate::providers::ollama::errors::OllamaError>,
|
||||||
|
// >(32);
|
||||||
|
|
||||||
|
// // Send start event immediately, before Ollama is contacted
|
||||||
|
// let _ = tx
|
||||||
|
// .send(Ok(Event::default()
|
||||||
|
// .event("metadata")
|
||||||
|
// .data(serde_json::to_string(&start_event).unwrap())))
|
||||||
|
// .await;
|
||||||
|
|
||||||
|
// let pool = state.postgres.clone();
|
||||||
|
// let user_id = auth.user_id();
|
||||||
|
|
||||||
|
// tokio::spawn(async move {
|
||||||
|
// // Ollama called inside spawn — start event already queued
|
||||||
|
// let stream = match state.ollama.chat_completions_stream(&body).await {
|
||||||
|
// Ok(s) => s,
|
||||||
|
// Err(e) => {
|
||||||
|
// let _ = tx.send(Err(e)).await;
|
||||||
|
// return;
|
||||||
|
// }
|
||||||
|
// };
|
||||||
|
|
||||||
|
// let mut stream = stream;
|
||||||
|
// let mut accumulated = String::new();
|
||||||
|
|
||||||
|
// while let Some(item) = futures::StreamExt::next(&mut stream).await {
|
||||||
|
// match item {
|
||||||
|
// Ok(chunk) => {
|
||||||
|
// let is_done = chunk.choices[0].finish_reason == Some(api::FinishReason::Stop);
|
||||||
|
|
||||||
|
// if let Some(content) = chunk.choices[0].delta.content.as_ref() {
|
||||||
|
// accumulated.push_str(content);
|
||||||
|
// }
|
||||||
|
|
||||||
|
// if is_done {
|
||||||
|
// let prompt_tokens = chunk.usage.as_ref().map(|u| u.prompt_tokens);
|
||||||
|
// let completion_tokens = chunk.usage.as_ref().map(|u| u.completion_tokens);
|
||||||
|
|
||||||
|
// if let Some(pt) = prompt_tokens {
|
||||||
|
// let _ = update_message_tokens(&pool, user_id, user_msg_id, pt).await;
|
||||||
|
// }
|
||||||
|
|
||||||
|
// let assistant_msg_id = log_assistant_message(
|
||||||
|
// &pool,
|
||||||
|
// user_id,
|
||||||
|
// conv_id,
|
||||||
|
// user_msg_id,
|
||||||
|
// &accumulated,
|
||||||
|
// completion_tokens,
|
||||||
|
// )
|
||||||
|
// .await;
|
||||||
|
|
||||||
|
// if let Ok(msg_id) = assistant_msg_id {
|
||||||
|
// let end_event = api::StreamEvent::End(api::EndEventData {
|
||||||
|
// usage: api::Usage {
|
||||||
|
// prompt_tokens: prompt_tokens.unwrap_or(0),
|
||||||
|
// completion_tokens: completion_tokens.unwrap_or(0),
|
||||||
|
// total_tokens: chunk
|
||||||
|
// .usage
|
||||||
|
// .as_ref()
|
||||||
|
// .map(|u| u.total_tokens)
|
||||||
|
// .unwrap_or(0),
|
||||||
|
// },
|
||||||
|
// id: msg_id,
|
||||||
|
// created: chrono::Utc::now().timestamp() as u64,
|
||||||
|
// });
|
||||||
|
|
||||||
|
// let _ = tx
|
||||||
|
// .send(Ok(Event::default()
|
||||||
|
// .event("metadata")
|
||||||
|
// .data(serde_json::to_string(&end_event).unwrap())))
|
||||||
|
// .await;
|
||||||
|
// }
|
||||||
|
|
||||||
|
// break;
|
||||||
|
// }
|
||||||
|
|
||||||
|
// let data = api::StreamEvent::Delta(chunk);
|
||||||
|
// let json = serde_json::to_string(&data).unwrap();
|
||||||
|
// if tx.send(Ok(Event::default().data(json))).await.is_err() {
|
||||||
|
// break;
|
||||||
|
// }
|
||||||
|
// }
|
||||||
|
// Err(e) => {
|
||||||
|
// let _ = tx.send(Err(e)).await;
|
||||||
|
// break;
|
||||||
|
// }
|
||||||
|
// }
|
||||||
|
// }
|
||||||
|
// });
|
||||||
|
|
||||||
|
// Ok(Sse::new(tokio_stream::wrappers::ReceiverStream::new(rx))
|
||||||
|
// .keep_alive(KeepAlive::default())
|
||||||
|
// .into_response())
|
||||||
|
// }
|
||||||
@@ -2,33 +2,36 @@ pub mod apikey;
|
|||||||
pub mod chat;
|
pub mod chat;
|
||||||
pub mod models;
|
pub mod models;
|
||||||
|
|
||||||
use crate::docs::ApiDoc;
|
use crate::api::docs::ApiDoc;
|
||||||
use crate::middlewares::auth::{auth_middleware, middleware::require_roles};
|
use crate::api::middlewares::auth::auth_middleware;
|
||||||
use crate::state::app_state::AppState;
|
use crate::api::state::app_state::SharedState;
|
||||||
|
use crate::role_guard;
|
||||||
|
|
||||||
use axum::{Json, Router, middleware, routing::get, routing::post};
|
use axum::{
|
||||||
|
Json, Router, middleware,
|
||||||
|
routing::{get, post},
|
||||||
|
};
|
||||||
use utoipa::OpenApi;
|
use utoipa::OpenApi;
|
||||||
|
|
||||||
async fn openapi_json() -> Json<utoipa::openapi::OpenApi> {
|
async fn openapi_json() -> Json<utoipa::openapi::OpenApi> {
|
||||||
Json(ApiDoc::openapi())
|
Json(ApiDoc::openapi())
|
||||||
}
|
}
|
||||||
|
|
||||||
fn public_router() -> Router<AppState> {
|
fn public_router() -> Router<SharedState> {
|
||||||
Router::new().route("/docs.json", get(openapi_json))
|
Router::new().route("/docs.json", get(openapi_json))
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn protected_router() -> Router<AppState> {
|
pub fn protected_router() -> Router<SharedState> {
|
||||||
Router::new()
|
Router::new()
|
||||||
.route("/models", get(models::list_models))
|
.route("/models", get(models::list_models))
|
||||||
.route("/completions", post(chat::completions))
|
.route("/completions", post(chat::completions))
|
||||||
.route("/chat/completions", post(chat::chat_completions))
|
.route("/chat/completions", post(chat::chat_completions))
|
||||||
.route("/models/{model}/load", post(models::load_model))
|
.route("/models/{model}/load", post(models::load_model))
|
||||||
.route("/models/{model}/unload", post(models::unload_model))
|
// .route("/models/{model}/unload", post(models::unload_model))
|
||||||
.route(
|
.route(
|
||||||
"/keys/generate",
|
"/keys/generate",
|
||||||
post(apikey::create_api_key).route_layer(middleware::from_fn(|req, next| {
|
post(apikey::create_api_key) // Usage
|
||||||
require_roles(req, next, None, Some("admin"))
|
.route_layer(role_guard!(Some("admin"), None)),
|
||||||
})),
|
|
||||||
)
|
)
|
||||||
.route("/conversations", get(chat::get_conversations))
|
.route("/conversations", get(chat::get_conversations))
|
||||||
.route(
|
.route(
|
||||||
@@ -37,8 +40,12 @@ pub fn protected_router() -> Router<AppState> {
|
|||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn router(state: AppState) -> Router<AppState> {
|
pub fn router(state: SharedState) -> Router<SharedState> {
|
||||||
Router::new()
|
Router::new()
|
||||||
.merge(public_router())
|
.merge(public_router())
|
||||||
.merge(protected_router().layer(middleware::from_fn_with_state(state, auth_middleware)))
|
.merge(protected_router().layer(middleware::from_fn_with_state(
|
||||||
|
state.clone(),
|
||||||
|
auth_middleware,
|
||||||
|
)))
|
||||||
|
.with_state(state)
|
||||||
}
|
}
|
||||||
@@ -0,0 +1,138 @@
|
|||||||
|
use crate::api::{self, state::SharedState};
|
||||||
|
|
||||||
|
use axum::{
|
||||||
|
Json,
|
||||||
|
extract::{Path, State},
|
||||||
|
};
|
||||||
|
|
||||||
|
#[utoipa::path(
|
||||||
|
get,
|
||||||
|
path = "/models",
|
||||||
|
tag = "models",
|
||||||
|
responses(
|
||||||
|
(
|
||||||
|
status = 200,
|
||||||
|
description = "List of locally available Ollama models",
|
||||||
|
body = api::types::ModelsResponse,
|
||||||
|
content_type = "application/json",
|
||||||
|
),
|
||||||
|
(
|
||||||
|
status = 500,
|
||||||
|
description = "Internal server error (Ollama or network failure)",
|
||||||
|
body = api::errors::ErrorResponse,
|
||||||
|
example = json!({ "error": "connection refused" })
|
||||||
|
)
|
||||||
|
)
|
||||||
|
)]
|
||||||
|
// #[axum::debug_handler]
|
||||||
|
pub async fn list_models(
|
||||||
|
State(state): State<SharedState>,
|
||||||
|
) -> Result<Json<api::types::ModelsResponse>, api::errors::ApiError> {
|
||||||
|
let models = state.chat_service.list_models().await?;
|
||||||
|
|
||||||
|
Ok(Json(models.into()))
|
||||||
|
}
|
||||||
|
|
||||||
|
#[utoipa::path(
|
||||||
|
post,
|
||||||
|
path = "/models/{model}/load",
|
||||||
|
tag = "models",
|
||||||
|
params(
|
||||||
|
("model" = String, Path, description = "Name of the model to load into memory (e.g. 'llama3')")
|
||||||
|
),
|
||||||
|
request_body(
|
||||||
|
content = api::types::LoadModelRequest,
|
||||||
|
description = "Load model request",
|
||||||
|
content_type = "application/json",
|
||||||
|
example = json!({ "keep_alive": "10m" })
|
||||||
|
),
|
||||||
|
responses(
|
||||||
|
(
|
||||||
|
status = 200,
|
||||||
|
description = "Model successfully loaded into memory",
|
||||||
|
body = api::types::LoadModelResponse,
|
||||||
|
content_type = "application/json",
|
||||||
|
),
|
||||||
|
(
|
||||||
|
status = 400,
|
||||||
|
description = "Invalid or missing keep_alive format",
|
||||||
|
body = api::errors::ErrorResponse,
|
||||||
|
examples(
|
||||||
|
("Missing" = (value = json!({ "error": "keep alive is required and cannot be empty" }))),
|
||||||
|
("Invalid" = (value = json!({ "error": "invalid keep_alive '10x' — use 30s / 10m / 2h, a plain integer, or -1" })))
|
||||||
|
)
|
||||||
|
),
|
||||||
|
(
|
||||||
|
status = 404,
|
||||||
|
description = "Model not found locally",
|
||||||
|
body = api::errors::ErrorResponse,
|
||||||
|
example = json!({ "error": "model 'llama3' not found — run `ollama pull llama3`" })
|
||||||
|
),
|
||||||
|
(
|
||||||
|
status = 500,
|
||||||
|
description = "Internal server error (Ollama or network failure)",
|
||||||
|
body = api::errors::ErrorResponse,
|
||||||
|
example = json!({ "error": "connection refused" })
|
||||||
|
)
|
||||||
|
)
|
||||||
|
)]
|
||||||
|
pub async fn load_model(
|
||||||
|
State(state): State<SharedState>,
|
||||||
|
Path(model): Path<String>,
|
||||||
|
Json(body): Json<api::types::LoadModelRequest>,
|
||||||
|
) -> Result<Json<api::types::LoadModelResponse>, api::errors::ApiError> {
|
||||||
|
let response = state
|
||||||
|
.chat_service
|
||||||
|
.load_model(crate::core::llm::models::LoadModelRequest {
|
||||||
|
model,
|
||||||
|
keep_alive: body.keep_alive.clone(),
|
||||||
|
})
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
Ok(Json(api::types::LoadModelResponse {
|
||||||
|
model: response.model,
|
||||||
|
keep_alive: body.keep_alive,
|
||||||
|
status: "loaded".to_string(),
|
||||||
|
}))
|
||||||
|
}
|
||||||
|
|
||||||
|
// #[utoipa::path(
|
||||||
|
// delete,
|
||||||
|
// path = "/models/{model}/load",
|
||||||
|
// tag = "models",
|
||||||
|
// params(
|
||||||
|
// ("model" = String, Path, description = "Name of the model to unload from memory (e.g. 'llama3')")
|
||||||
|
// ),
|
||||||
|
// responses(
|
||||||
|
// (
|
||||||
|
// status = 200,
|
||||||
|
// description = "Model successfully unloaded from memory",
|
||||||
|
// body = api::types::UnloadModelResponse,
|
||||||
|
// content_type = "application/json",
|
||||||
|
// ),
|
||||||
|
// (
|
||||||
|
// status = 404,
|
||||||
|
// description = "Model not found locally",
|
||||||
|
// body = api::errors::ErrorResponse,
|
||||||
|
// example = json!({ "error": "model 'llama3' not found — run `ollama pull llama3`" })
|
||||||
|
// ),
|
||||||
|
// (
|
||||||
|
// status = 500,
|
||||||
|
// description = "Internal server error (Ollama or network failure)",
|
||||||
|
// body = api::errors::ErrorResponse,
|
||||||
|
// example = json!({ "error": "connection refused" })
|
||||||
|
// )
|
||||||
|
// )
|
||||||
|
// )]
|
||||||
|
// pub async fn unload_model(
|
||||||
|
// State(state): State<AppState>,
|
||||||
|
// Path(model): Path<String>,
|
||||||
|
// ) -> Result<Json<api::types::UnloadModelResponse>, (axum::http::StatusCode, String)> {
|
||||||
|
// let response = state
|
||||||
|
// .ollama
|
||||||
|
// .unload_model(&model)
|
||||||
|
// .await
|
||||||
|
// .map_err(into_http_response)?;
|
||||||
|
|
||||||
|
// Ok(Json(response))
|
||||||
|
// }
|
||||||
@@ -0,0 +1,12 @@
|
|||||||
|
use crate::services::{AuthService, ChatService, ConversationService};
|
||||||
|
|
||||||
|
use std::sync::Arc;
|
||||||
|
|
||||||
|
#[derive(Clone)]
|
||||||
|
pub struct AppState {
|
||||||
|
pub conversation_service: ConversationService,
|
||||||
|
pub auth_service: AuthService,
|
||||||
|
pub chat_service: ChatService,
|
||||||
|
}
|
||||||
|
|
||||||
|
pub type SharedState = Arc<AppState>;
|
||||||
@@ -0,0 +1,4 @@
|
|||||||
|
pub mod app_state;
|
||||||
|
|
||||||
|
pub use app_state::AppState;
|
||||||
|
pub use app_state::SharedState;
|
||||||
@@ -1,31 +1,32 @@
|
|||||||
use serde::{Deserialize, Serialize};
|
use serde::{Deserialize, Serialize};
|
||||||
|
use std::collections::HashMap;
|
||||||
use utoipa::ToSchema;
|
use utoipa::ToSchema;
|
||||||
use uuid::Uuid;
|
use uuid::Uuid;
|
||||||
|
|
||||||
|
// ------ Models ------
|
||||||
|
|
||||||
#[derive(Debug, Serialize, ToSchema)]
|
#[derive(Debug, Serialize, ToSchema)]
|
||||||
pub struct ErrorResponse {
|
|
||||||
pub error: String,
|
|
||||||
}
|
|
||||||
|
|
||||||
impl ErrorResponse {
|
|
||||||
pub fn new(msg: impl Into<String>) -> Self {
|
|
||||||
Self { error: msg.into() }
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Debug, Serialize, Deserialize, ToSchema)]
|
|
||||||
pub struct ModelsResponse {
|
|
||||||
pub models: Vec<ModelInfo>,
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Debug, Serialize, Deserialize, ToSchema)]
|
|
||||||
pub struct ModelInfo {
|
pub struct ModelInfo {
|
||||||
pub name: String,
|
pub name: String,
|
||||||
pub family: Option<String>,
|
pub family: Option<String>,
|
||||||
pub parameter_size: Option<String>,
|
pub parameter_size: Option<String>,
|
||||||
pub quantization: Option<String>,
|
pub metadata: ModelMetadata,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Clone, Default, Serialize, ToSchema)]
|
||||||
|
pub struct ModelMetadata {
|
||||||
|
pub extra: HashMap<String, String>,
|
||||||
|
}
|
||||||
|
|
||||||
|
// ------ Endpoint: /models ------
|
||||||
|
|
||||||
|
#[derive(Debug, Serialize, ToSchema)]
|
||||||
|
pub struct ModelsResponse {
|
||||||
|
pub models: Vec<ModelInfo>,
|
||||||
|
}
|
||||||
|
|
||||||
|
// ------ Endpoint: /models/{model}/load ------
|
||||||
|
|
||||||
#[derive(Debug, Serialize, Deserialize, ToSchema)]
|
#[derive(Debug, Serialize, Deserialize, ToSchema)]
|
||||||
pub struct LoadModelResponse {
|
pub struct LoadModelResponse {
|
||||||
pub model: String,
|
pub model: String,
|
||||||
@@ -34,27 +35,28 @@ pub struct LoadModelResponse {
|
|||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Debug, Serialize, Deserialize, ToSchema)]
|
#[derive(Debug, Serialize, Deserialize, ToSchema)]
|
||||||
pub struct LoadModelBody {
|
pub struct LoadModelRequest {
|
||||||
pub keep_alive: Option<String>,
|
pub keep_alive: String,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ------ Endpoint: /models/{model}/unload ------
|
||||||
|
|
||||||
#[derive(Debug, Serialize, Deserialize, ToSchema)]
|
#[derive(Debug, Serialize, Deserialize, ToSchema)]
|
||||||
pub struct UnloadModelResponse {
|
pub struct UnloadModelResponse {
|
||||||
pub model: String,
|
pub model: String,
|
||||||
pub status: String,
|
pub status: String,
|
||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Debug, Deserialize, Serialize, ToSchema, Default)]
|
// ------ Completions ------
|
||||||
pub struct BaseLLMRequest {
|
|
||||||
pub model: String,
|
|
||||||
|
|
||||||
|
#[derive(Debug, Deserialize, Serialize, ToSchema, Default)]
|
||||||
|
pub struct LLMOptions {
|
||||||
#[serde(default)]
|
#[serde(default)]
|
||||||
pub stream: bool,
|
pub stream: bool,
|
||||||
|
|
||||||
pub temperature: Option<f32>,
|
pub temperature: Option<f32>,
|
||||||
pub top_p: Option<f32>,
|
pub top_p: Option<f32>,
|
||||||
|
|
||||||
// Ollama-native
|
|
||||||
pub top_k: Option<u32>,
|
pub top_k: Option<u32>,
|
||||||
pub repeat_penalty: Option<f32>,
|
pub repeat_penalty: Option<f32>,
|
||||||
pub seed: Option<i64>,
|
pub seed: Option<i64>,
|
||||||
@@ -69,20 +71,35 @@ pub struct BaseLLMRequest {
|
|||||||
pub context_depth: Option<u32>,
|
pub context_depth: Option<u32>,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ------ Endpoint: /completions ------
|
||||||
|
|
||||||
#[derive(Debug, Deserialize, Serialize, ToSchema)]
|
#[derive(Debug, Deserialize, Serialize, ToSchema)]
|
||||||
pub struct CompletionRequest {
|
pub struct CompletionRequest {
|
||||||
#[serde(flatten)]
|
#[serde(flatten)]
|
||||||
pub base: BaseLLMRequest,
|
pub options: LLMOptions,
|
||||||
|
|
||||||
|
pub model: String,
|
||||||
pub prompt: String,
|
pub prompt: String,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Serialize, Deserialize, ToSchema)]
|
||||||
|
pub struct CompletionResponse {
|
||||||
|
pub id: Uuid,
|
||||||
|
pub object: CompletionObject,
|
||||||
|
pub created: String,
|
||||||
|
pub model: String,
|
||||||
|
pub choices: Vec<Choice>,
|
||||||
|
pub usage: Usage,
|
||||||
|
}
|
||||||
|
|
||||||
#[derive(Debug, Serialize, Deserialize, ToSchema, PartialEq, Eq)]
|
#[derive(Debug, Serialize, Deserialize, ToSchema, PartialEq, Eq)]
|
||||||
#[serde(rename_all = "snake_case")]
|
#[serde(rename_all = "snake_case")]
|
||||||
pub enum CompletionObject {
|
pub enum CompletionObject {
|
||||||
TextCompletion,
|
TextCompletion,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ------ Endpoint: /chat/completions ------
|
||||||
|
|
||||||
#[derive(Debug, Serialize, Deserialize, ToSchema, PartialEq, Eq)]
|
#[derive(Debug, Serialize, Deserialize, ToSchema, PartialEq, Eq)]
|
||||||
#[serde(rename_all = "snake_case")]
|
#[serde(rename_all = "snake_case")]
|
||||||
pub enum FinishReason {
|
pub enum FinishReason {
|
||||||
@@ -93,16 +110,6 @@ pub enum FinishReason {
|
|||||||
Error,
|
Error,
|
||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Debug, Serialize, Deserialize, ToSchema)]
|
|
||||||
pub struct CompletionResponse {
|
|
||||||
pub id: String,
|
|
||||||
pub object: CompletionObject,
|
|
||||||
pub created: u64,
|
|
||||||
pub model: String,
|
|
||||||
pub choices: Vec<Choice>,
|
|
||||||
pub usage: Usage,
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Debug, Serialize, Deserialize, ToSchema)]
|
#[derive(Debug, Serialize, Deserialize, ToSchema)]
|
||||||
pub struct Choice {
|
pub struct Choice {
|
||||||
pub text: String,
|
pub text: String,
|
||||||
@@ -127,13 +134,14 @@ pub struct CompletionChunk {
|
|||||||
#[derive(Debug, Deserialize, Serialize, ToSchema)]
|
#[derive(Debug, Deserialize, Serialize, ToSchema)]
|
||||||
pub struct ChatRequest {
|
pub struct ChatRequest {
|
||||||
#[serde(flatten)]
|
#[serde(flatten)]
|
||||||
pub base: BaseLLMRequest,
|
pub base: LLMOptions,
|
||||||
|
|
||||||
pub messages: Vec<Message>,
|
pub model: String,
|
||||||
|
pub message: Message,
|
||||||
|
|
||||||
// Non standard Open AI
|
// Non standard Open AI
|
||||||
pub conversation_id: Option<Uuid>,
|
pub conversation_id: Option<Uuid>,
|
||||||
pub parent_id: Option<Uuid>,
|
pub parent_id: Option<Uuid>, // used when branching, regenerate, etc
|
||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Debug, Deserialize, Serialize, ToSchema)]
|
#[derive(Debug, Deserialize, Serialize, ToSchema)]
|
||||||
@@ -151,15 +159,15 @@ pub enum Role {
|
|||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Debug, Serialize, Deserialize, ToSchema)]
|
#[derive(Debug, Serialize, Deserialize, ToSchema)]
|
||||||
pub struct ChatCompletionResponse {
|
pub struct ChatResponse {
|
||||||
pub id: String,
|
pub id: String,
|
||||||
pub object: String,
|
pub object: String,
|
||||||
pub created: u64,
|
pub created: u64,
|
||||||
pub model: String,
|
pub model: String,
|
||||||
pub choices: Vec<ChatChoice>,
|
pub choices: Vec<ChatChoice>,
|
||||||
pub usage: Option<Usage>, // optional (Ollama may not always provide)
|
pub usage: Usage,
|
||||||
|
|
||||||
pub conversation_id: Option<Uuid>,
|
pub conversation_id: Uuid,
|
||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Debug, Serialize, Deserialize, ToSchema)]
|
#[derive(Debug, Serialize, Deserialize, ToSchema)]
|
||||||
@@ -190,62 +198,96 @@ pub struct Delta {
|
|||||||
pub role: Option<Role>,
|
pub role: Option<Role>,
|
||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Debug, Serialize, Deserialize, ToSchema)]
|
// #[derive(Debug, Serialize, Deserialize, ToSchema)]
|
||||||
pub struct StartEventData {
|
// pub struct StartEventData {
|
||||||
pub conversation_id: Uuid,
|
// pub conversation_id: Uuid,
|
||||||
pub created: u64,
|
// pub created: u64,
|
||||||
pub id: Uuid,
|
// pub id: Uuid,
|
||||||
}
|
// }
|
||||||
|
|
||||||
#[derive(Debug, Serialize, Deserialize, ToSchema)]
|
// #[derive(Debug, Serialize, Deserialize, ToSchema)]
|
||||||
pub struct EndEventData {
|
// pub struct EndEventData {
|
||||||
pub created: u64,
|
// pub created: u64,
|
||||||
pub id: Uuid,
|
// pub id: Uuid,
|
||||||
pub usage: Usage,
|
// pub usage: Usage,
|
||||||
}
|
// }
|
||||||
|
|
||||||
#[derive(Debug, Serialize, Deserialize, ToSchema)]
|
// #[derive(Debug, Serialize, Deserialize, ToSchema)]
|
||||||
#[serde(tag = "type", content = "data")]
|
// #[serde(tag = "type", content = "data")]
|
||||||
pub enum StreamEvent {
|
// pub enum StreamEvent {
|
||||||
#[serde(rename = "start")]
|
// #[serde(rename = "start")]
|
||||||
Start(StartEventData),
|
// Start(StartEventData),
|
||||||
#[serde(rename = "end")]
|
// #[serde(rename = "end")]
|
||||||
End(EndEventData),
|
// End(EndEventData),
|
||||||
#[serde(rename = "delta")]
|
// #[serde(rename = "delta")]
|
||||||
Delta(ChatCompletionChunk),
|
// Delta(ChatCompletionChunk),
|
||||||
|
// }
|
||||||
|
|
||||||
|
// ------ Api Key ------
|
||||||
|
|
||||||
|
#[derive(Debug, Deserialize)]
|
||||||
|
#[serde(rename_all = "snake_case")]
|
||||||
|
pub enum ApiKeyScope {
|
||||||
|
User,
|
||||||
|
Admin,
|
||||||
}
|
}
|
||||||
|
|
||||||
#[derive(serde::Deserialize)]
|
#[derive(serde::Deserialize)]
|
||||||
pub struct CreateApiKeyRequest {
|
pub struct CreateApiKeyRequest {
|
||||||
pub name: String,
|
pub name: String,
|
||||||
pub scopes: Vec<String>,
|
pub scopes: Vec<ApiKeyScope>,
|
||||||
}
|
}
|
||||||
|
|
||||||
#[derive(serde::Serialize)]
|
#[derive(serde::Serialize)]
|
||||||
pub struct CreateApiKeyResponse {
|
pub struct CreateApiKeyResponse {
|
||||||
pub api_key: String, // ONLY returned once
|
pub api_key: String,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ------ Fetch database ------
|
||||||
|
// --- Shared ---
|
||||||
#[derive(Debug, Deserialize)]
|
#[derive(Debug, Deserialize)]
|
||||||
pub struct ConversationQuery {
|
pub struct CursorPage {
|
||||||
pub limit: Option<i64>,
|
pub limit: Option<u32>,
|
||||||
pub before: Option<chrono::DateTime<chrono::Utc>>, // cursor
|
pub before: Option<chrono::DateTime<chrono::Utc>>,
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- Conversation ---
|
||||||
|
|
||||||
|
#[derive(Debug, Serialize)]
|
||||||
|
pub struct ConversationSummary {
|
||||||
|
pub id: Uuid,
|
||||||
|
pub title: String,
|
||||||
|
pub created_at: chrono::DateTime<chrono::Utc>,
|
||||||
|
pub updated_at: chrono::DateTime<chrono::Utc>,
|
||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Debug, Serialize)]
|
#[derive(Debug, Serialize)]
|
||||||
pub struct ConversationListResponse {
|
pub struct ConversationListResponse {
|
||||||
pub conversations: Vec<crate::databases::postgres::chat::types::ConversationSummary>,
|
pub conversations: Vec<ConversationSummary>,
|
||||||
pub has_more: bool,
|
pub has_more: bool,
|
||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Debug, Deserialize)]
|
// --- Message ---
|
||||||
pub struct MessageQuery {
|
|
||||||
pub limit: Option<i64>,
|
#[derive(Debug, Serialize)]
|
||||||
pub before: Option<chrono::DateTime<chrono::Utc>>,
|
pub enum ApiChatRole {
|
||||||
|
System,
|
||||||
|
User,
|
||||||
|
Assistant,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Serialize)]
|
||||||
|
pub struct MessageSummary {
|
||||||
|
pub id: Uuid,
|
||||||
|
pub parent_id: Option<Uuid>,
|
||||||
|
pub role: ApiChatRole,
|
||||||
|
pub content: String,
|
||||||
|
pub created_at: chrono::DateTime<chrono::Utc>,
|
||||||
|
pub tokens: Option<i32>,
|
||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Debug, Serialize)]
|
#[derive(Debug, Serialize)]
|
||||||
pub struct MessageListResponse {
|
pub struct MessageListResponse {
|
||||||
pub messages: Vec<crate::databases::postgres::chat::types::MessageSummary>,
|
pub messages: Vec<MessageSummary>,
|
||||||
pub has_more: bool,
|
pub has_more: bool,
|
||||||
}
|
}
|
||||||
@@ -0,0 +1,27 @@
|
|||||||
|
use uuid::Uuid;
|
||||||
|
|
||||||
|
#[derive(Debug)]
|
||||||
|
pub struct CreateApiKeyRequest {
|
||||||
|
pub user_id: Uuid,
|
||||||
|
pub name: String,
|
||||||
|
pub roles: Vec<KeyRole>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||||
|
pub enum KeyRole {
|
||||||
|
User,
|
||||||
|
Admin,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Clone)]
|
||||||
|
pub struct AuthContext {
|
||||||
|
pub user_id: Uuid,
|
||||||
|
pub _api_key_id: Uuid,
|
||||||
|
pub roles: Vec<KeyRole>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl AuthContext {
|
||||||
|
pub fn has_role(&self, role: &KeyRole) -> bool {
|
||||||
|
self.roles.iter().any(|r| r == role)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,25 @@
|
|||||||
|
// src/core/auth/jwt.rs
|
||||||
|
|
||||||
|
use std::collections::HashMap;
|
||||||
|
use uuid::Uuid;
|
||||||
|
|
||||||
|
#[derive(Debug, Clone)]
|
||||||
|
pub struct JwtClaims {
|
||||||
|
pub user_id: Uuid,
|
||||||
|
pub _username: Option<String>,
|
||||||
|
pub _exp: usize,
|
||||||
|
pub _issuer: String,
|
||||||
|
|
||||||
|
pub realm_roles: Vec<String>,
|
||||||
|
pub client_roles: HashMap<String, Vec<String>>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl JwtClaims {
|
||||||
|
pub fn has_role(&self, role: &str) -> bool {
|
||||||
|
self.realm_roles.iter().any(|r| r == role)
|
||||||
|
|| self
|
||||||
|
.client_roles
|
||||||
|
.get("chat-api")
|
||||||
|
.is_some_and(|roles| roles.iter().any(|r| r == role))
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,33 @@
|
|||||||
|
use uuid::Uuid;
|
||||||
|
|
||||||
|
pub mod api_key;
|
||||||
|
pub mod jwt;
|
||||||
|
|
||||||
|
#[derive(Debug, Clone)]
|
||||||
|
pub enum Auth {
|
||||||
|
Jwt(jwt::JwtClaims),
|
||||||
|
ApiKey(api_key::AuthContext),
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Auth {
|
||||||
|
pub fn user_id(&self) -> Uuid {
|
||||||
|
match self {
|
||||||
|
Auth::Jwt(c) => c.user_id,
|
||||||
|
Auth::ApiKey(c) => c.user_id,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn has_jwt_role(&self, role: &str) -> bool {
|
||||||
|
match self {
|
||||||
|
Auth::Jwt(c) => c.has_role(role),
|
||||||
|
Auth::ApiKey(_) => false,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn has_apikey_role(&self, role: &crate::core::auth::api_key::KeyRole) -> bool {
|
||||||
|
match self {
|
||||||
|
Auth::Jwt(_) => false,
|
||||||
|
Auth::ApiKey(c) => c.has_role(role),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,44 @@
|
|||||||
|
use crate::core::llm::ChatRole;
|
||||||
|
|
||||||
|
use uuid::Uuid;
|
||||||
|
|
||||||
|
#[derive(Debug)]
|
||||||
|
pub struct ConversationSummary {
|
||||||
|
pub id: Uuid,
|
||||||
|
pub title: String,
|
||||||
|
pub created_at: chrono::DateTime<chrono::Utc>,
|
||||||
|
pub updated_at: chrono::DateTime<chrono::Utc>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug)]
|
||||||
|
pub struct ConversationList {
|
||||||
|
pub conversations: Vec<ConversationSummary>,
|
||||||
|
pub has_more: bool,
|
||||||
|
}
|
||||||
|
|
||||||
|
pub struct CursorPage {
|
||||||
|
pub limit: u32,
|
||||||
|
pub before: Option<chrono::DateTime<chrono::Utc>>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug)]
|
||||||
|
pub struct MessageSummary {
|
||||||
|
pub id: Uuid,
|
||||||
|
pub parent_id: Option<Uuid>,
|
||||||
|
pub role: ChatRole,
|
||||||
|
pub content: String,
|
||||||
|
pub created_at: chrono::DateTime<chrono::Utc>,
|
||||||
|
pub tokens: Option<i32>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug)]
|
||||||
|
pub struct MessageList {
|
||||||
|
pub messages: Vec<MessageSummary>,
|
||||||
|
pub has_more: bool,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug)]
|
||||||
|
pub enum ConversationResult {
|
||||||
|
Existing(Uuid),
|
||||||
|
Created(Uuid),
|
||||||
|
}
|
||||||
@@ -0,0 +1 @@
|
|||||||
|
pub mod conversations;
|
||||||
@@ -0,0 +1,79 @@
|
|||||||
|
use super::ChatRole;
|
||||||
|
|
||||||
|
use crate::providers::ollama::errors::LlmError;
|
||||||
|
|
||||||
|
use futures::Stream;
|
||||||
|
use serde::Serialize;
|
||||||
|
use std::pin::Pin;
|
||||||
|
|
||||||
|
#[derive(Debug, Clone, Default)]
|
||||||
|
pub struct ChatCompletionOptions {
|
||||||
|
pub seed: Option<i64>,
|
||||||
|
pub temperature: Option<f32>,
|
||||||
|
pub top_p: Option<f32>,
|
||||||
|
pub top_k: Option<u32>,
|
||||||
|
pub stop: Option<Vec<String>>,
|
||||||
|
pub num_ctx: Option<u32>,
|
||||||
|
pub num_predict: Option<u32>,
|
||||||
|
|
||||||
|
pub keep_alive: Option<String>,
|
||||||
|
pub stream: bool,
|
||||||
|
pub context_depth: u32,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Clone, Serialize)]
|
||||||
|
pub struct Message {
|
||||||
|
pub role: ChatRole,
|
||||||
|
pub content: String,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Clone)]
|
||||||
|
pub struct ChatCompletionRequest {
|
||||||
|
pub model: String,
|
||||||
|
pub message: Message,
|
||||||
|
|
||||||
|
pub options: ChatCompletionOptions,
|
||||||
|
|
||||||
|
pub conversation_id: Option<uuid::Uuid>,
|
||||||
|
pub parent_id: Option<uuid::Uuid>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Clone, Serialize)]
|
||||||
|
pub struct ChatCompletionResultNoStream {
|
||||||
|
pub model: String,
|
||||||
|
pub message: Message,
|
||||||
|
|
||||||
|
pub created_at: String,
|
||||||
|
pub id: uuid::Uuid,
|
||||||
|
|
||||||
|
pub prompt_tokens: u32,
|
||||||
|
pub completion_tokens: u32,
|
||||||
|
|
||||||
|
pub done_reason: Option<String>,
|
||||||
|
|
||||||
|
pub total_duration: Option<u64>,
|
||||||
|
pub load_duration: Option<u64>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Serialize)]
|
||||||
|
pub enum ChatCompletionStreamEvent {
|
||||||
|
Token(String),
|
||||||
|
Final(ChatCompletionResultNoStream),
|
||||||
|
}
|
||||||
|
|
||||||
|
pub type ChatCompletionStream =
|
||||||
|
Pin<Box<dyn Stream<Item = Result<ChatCompletionStreamEvent, LlmError>> + Send>>;
|
||||||
|
|
||||||
|
pub enum ChatCompletionResult {
|
||||||
|
Stream(ChatCompletionStream),
|
||||||
|
NoStream(ChatCompletionResultNoStream),
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Message {
|
||||||
|
pub fn from_summary(summary: crate::core::databases::conversations::MessageSummary) -> Self {
|
||||||
|
Self {
|
||||||
|
role: summary.role,
|
||||||
|
content: summary.content,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,58 @@
|
|||||||
|
use crate::providers::ollama::errors::LlmError;
|
||||||
|
|
||||||
|
use futures::Stream;
|
||||||
|
use serde::Serialize;
|
||||||
|
use std::pin::Pin;
|
||||||
|
|
||||||
|
#[derive(Debug, Clone, Default)]
|
||||||
|
pub struct CompletionOptions {
|
||||||
|
pub seed: Option<i64>,
|
||||||
|
pub temperature: Option<f32>,
|
||||||
|
pub top_p: Option<f32>,
|
||||||
|
pub top_k: Option<u32>,
|
||||||
|
pub stop: Option<Vec<String>>,
|
||||||
|
pub num_ctx: Option<u32>,
|
||||||
|
pub num_predict: Option<u32>,
|
||||||
|
|
||||||
|
pub keep_alive: Option<String>,
|
||||||
|
pub stream: bool,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Clone)]
|
||||||
|
pub struct CompletionRequest {
|
||||||
|
pub model: String,
|
||||||
|
pub prompt: String,
|
||||||
|
|
||||||
|
pub options: CompletionOptions,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Clone, Serialize)]
|
||||||
|
pub struct CompletionResultNoStream {
|
||||||
|
pub model: String,
|
||||||
|
pub text: String,
|
||||||
|
|
||||||
|
pub created_at: String,
|
||||||
|
pub id: uuid::Uuid,
|
||||||
|
|
||||||
|
pub prompt_tokens: u32,
|
||||||
|
pub completion_tokens: u32,
|
||||||
|
|
||||||
|
pub done_reason: Option<String>,
|
||||||
|
|
||||||
|
pub total_duration: Option<u64>,
|
||||||
|
pub load_duration: Option<u64>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Serialize)]
|
||||||
|
pub enum CompletionStreamEvent {
|
||||||
|
Token(String),
|
||||||
|
Final(CompletionResultNoStream),
|
||||||
|
}
|
||||||
|
|
||||||
|
pub type CompletionStream =
|
||||||
|
Pin<Box<dyn Stream<Item = Result<CompletionStreamEvent, LlmError>> + Send>>;
|
||||||
|
|
||||||
|
pub enum CompletionResult {
|
||||||
|
Stream(CompletionStream),
|
||||||
|
NoStream(CompletionResultNoStream),
|
||||||
|
}
|
||||||
@@ -0,0 +1,6 @@
|
|||||||
|
pub mod chat;
|
||||||
|
pub mod completions;
|
||||||
|
pub mod models;
|
||||||
|
pub mod role;
|
||||||
|
|
||||||
|
pub use role::ChatRole;
|
||||||
@@ -0,0 +1,36 @@
|
|||||||
|
use std::collections::HashMap;
|
||||||
|
|
||||||
|
#[derive(Debug, Clone)]
|
||||||
|
pub struct Models {
|
||||||
|
pub models: Vec<Model>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Clone)]
|
||||||
|
pub struct Model {
|
||||||
|
pub name: String,
|
||||||
|
|
||||||
|
/// Optional grouping (OpenAI = "gpt", Ollama = "llama", etc.)
|
||||||
|
pub family: Option<String>,
|
||||||
|
|
||||||
|
/// Human-readable size like "7B", "13B", "gpt-4"
|
||||||
|
pub size: Option<String>,
|
||||||
|
|
||||||
|
/// Optional metadata, provider-specific info normalized into a string map
|
||||||
|
pub metadata: ModelMetadata,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Clone, Default)]
|
||||||
|
pub struct ModelMetadata {
|
||||||
|
pub extra: HashMap<String, String>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Clone)]
|
||||||
|
pub struct LoadModelRequest {
|
||||||
|
pub model: String,
|
||||||
|
pub keep_alive: String,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Clone)]
|
||||||
|
pub struct LoadModelResponse {
|
||||||
|
pub model: String,
|
||||||
|
}
|
||||||
@@ -0,0 +1,8 @@
|
|||||||
|
use serde::Serialize;
|
||||||
|
|
||||||
|
#[derive(Debug, Clone, Serialize)]
|
||||||
|
pub enum ChatRole {
|
||||||
|
System,
|
||||||
|
User,
|
||||||
|
Assistant,
|
||||||
|
}
|
||||||
@@ -0,0 +1,3 @@
|
|||||||
|
pub mod auth;
|
||||||
|
pub mod databases;
|
||||||
|
pub mod llm;
|
||||||
@@ -8,6 +8,9 @@ pub enum DbError {
|
|||||||
#[error("database timeout")]
|
#[error("database timeout")]
|
||||||
Timeout,
|
Timeout,
|
||||||
|
|
||||||
|
#[error("not authorized")]
|
||||||
|
Unauthorized,
|
||||||
|
|
||||||
#[error("not found")]
|
#[error("not found")]
|
||||||
NotFound,
|
NotFound,
|
||||||
}
|
}
|
||||||
@@ -1 +1,3 @@
|
|||||||
pub mod postgres;
|
pub mod postgres;
|
||||||
|
|
||||||
|
pub mod errors;
|
||||||
|
|||||||
@@ -1 +1,2 @@
|
|||||||
pub mod queries;
|
pub mod queries;
|
||||||
|
pub mod types;
|
||||||
|
|||||||
@@ -1,9 +1,11 @@
|
|||||||
use crate::databases::postgres::errors::DbError;
|
use super::types;
|
||||||
|
use super::types::Role;
|
||||||
|
use crate::databases::errors::DbError;
|
||||||
|
|
||||||
use sqlx::PgPool;
|
use sqlx::PgPool;
|
||||||
use uuid::Uuid;
|
use uuid::Uuid;
|
||||||
|
|
||||||
pub async fn update_last_access(pool: &PgPool, api_key_id: Uuid) -> Result<(), DbError> {
|
pub async fn update_last_access(pool: &PgPool, api_key_id: &Uuid) -> Result<(), DbError> {
|
||||||
sqlx::query!(
|
sqlx::query!(
|
||||||
r#"
|
r#"
|
||||||
UPDATE auth.api_key
|
UPDATE auth.api_key
|
||||||
@@ -17,3 +19,41 @@ pub async fn update_last_access(pool: &PgPool, api_key_id: Uuid) -> Result<(), D
|
|||||||
|
|
||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
|
|
||||||
|
pub async fn create(pool: &PgPool, key: super::types::CreateApiKey) -> Result<(), DbError> {
|
||||||
|
sqlx::query!(
|
||||||
|
r#"
|
||||||
|
INSERT INTO auth.api_key (key_hash, name, created_by, scopes)
|
||||||
|
VALUES ($1, $2, $3, $4::auth.role[])
|
||||||
|
"#,
|
||||||
|
key.key_hash,
|
||||||
|
key.name,
|
||||||
|
key.user_id,
|
||||||
|
&key.roles as _
|
||||||
|
)
|
||||||
|
.execute(pool)
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
pub async fn validate(pool: &PgPool, key_hash: &str) -> Result<types::AuthContext, DbError> {
|
||||||
|
let row = sqlx::query_as!(
|
||||||
|
types::AuthContext,
|
||||||
|
r#"
|
||||||
|
SELECT
|
||||||
|
u.id AS user_id,
|
||||||
|
ak.id AS api_key_id,
|
||||||
|
ak.scopes AS "roles!: Vec<Role>"
|
||||||
|
FROM auth.api_key ak
|
||||||
|
JOIN auth.app_user u ON u.id = ak.created_by
|
||||||
|
WHERE ak.key_hash = $1
|
||||||
|
AND ak.revoked_at IS NULL
|
||||||
|
"#,
|
||||||
|
key_hash
|
||||||
|
)
|
||||||
|
.fetch_optional(pool)
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
row.ok_or(DbError::Unauthorized)
|
||||||
|
}
|
||||||
|
|||||||
@@ -0,0 +1,23 @@
|
|||||||
|
use sqlx::Type;
|
||||||
|
|
||||||
|
#[derive(Debug)]
|
||||||
|
pub struct CreateApiKey {
|
||||||
|
pub key_hash: String,
|
||||||
|
pub name: String,
|
||||||
|
pub user_id: uuid::Uuid,
|
||||||
|
pub roles: Vec<Role>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Clone, Copy, PartialEq, Eq, Type)]
|
||||||
|
#[sqlx(type_name = "auth.role", rename_all = "lowercase")]
|
||||||
|
pub enum Role {
|
||||||
|
User,
|
||||||
|
Admin,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, sqlx::FromRow)]
|
||||||
|
pub struct AuthContext {
|
||||||
|
pub user_id: uuid::Uuid,
|
||||||
|
pub api_key_id: uuid::Uuid,
|
||||||
|
pub roles: Vec<Role>,
|
||||||
|
}
|
||||||
@@ -1,10 +1,10 @@
|
|||||||
use crate::databases::postgres::chat::types;
|
use crate::databases::errors::DbError;
|
||||||
use crate::databases::postgres::errors::DbError;
|
use crate::databases::postgres::chat::{types, types::MessageRole};
|
||||||
|
|
||||||
use sqlx::{Acquire, PgPool};
|
use sqlx::{Acquire, PgPool};
|
||||||
use uuid::Uuid;
|
use uuid::Uuid;
|
||||||
|
|
||||||
// ---- Creation ----
|
// ---- Helpers ----
|
||||||
|
|
||||||
async fn validate_conversation<'e, E>(executor: E, conversation_id: Uuid) -> Result<(), DbError>
|
async fn validate_conversation<'e, E>(executor: E, conversation_id: Uuid) -> Result<(), DbError>
|
||||||
where
|
where
|
||||||
@@ -40,14 +40,14 @@ where
|
|||||||
Ok(rec.id)
|
Ok(rec.id)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ------ Creation ------
|
||||||
|
|
||||||
pub async fn get_or_create_conversation(
|
pub async fn get_or_create_conversation(
|
||||||
pool: &PgPool,
|
pool: &PgPool,
|
||||||
conversation_id: Option<Uuid>,
|
conversation_id: Option<Uuid>,
|
||||||
user_id: Uuid,
|
user_id: Uuid,
|
||||||
) -> Result<types::ConversationState, DbError> {
|
) -> Result<types::ConversationState, DbError> {
|
||||||
tracing::debug!("Testing conversation");
|
let mut tx: sqlx::Transaction<'_, sqlx::Postgres> = pool.begin().await?;
|
||||||
|
|
||||||
let mut tx = pool.begin().await?;
|
|
||||||
|
|
||||||
let result = {
|
let result = {
|
||||||
let conn = tx.acquire().await?;
|
let conn = tx.acquire().await?;
|
||||||
@@ -165,7 +165,7 @@ pub async fn update_message_tokens(
|
|||||||
pub async fn get_conversations_entries(
|
pub async fn get_conversations_entries(
|
||||||
pool: &PgPool,
|
pool: &PgPool,
|
||||||
user_id: Uuid,
|
user_id: Uuid,
|
||||||
limit: i64,
|
limit: u32,
|
||||||
before: Option<chrono::DateTime<chrono::Utc>>,
|
before: Option<chrono::DateTime<chrono::Utc>>,
|
||||||
) -> Result<Vec<types::ConversationSummary>, DbError> {
|
) -> Result<Vec<types::ConversationSummary>, DbError> {
|
||||||
let mut tx = pool.begin().await?;
|
let mut tx = pool.begin().await?;
|
||||||
@@ -187,7 +187,7 @@ pub async fn get_conversations_entries(
|
|||||||
"#,
|
"#,
|
||||||
user_id,
|
user_id,
|
||||||
before,
|
before,
|
||||||
limit
|
limit as i64
|
||||||
)
|
)
|
||||||
.fetch_all(&mut *tx)
|
.fetch_all(&mut *tx)
|
||||||
.await?;
|
.await?;
|
||||||
@@ -200,7 +200,7 @@ pub async fn get_conversation_messages(
|
|||||||
pool: &PgPool,
|
pool: &PgPool,
|
||||||
user_id: Uuid,
|
user_id: Uuid,
|
||||||
conversation_id: Uuid,
|
conversation_id: Uuid,
|
||||||
limit: i64,
|
limit: u32,
|
||||||
before: Option<chrono::DateTime<chrono::Utc>>,
|
before: Option<chrono::DateTime<chrono::Utc>>,
|
||||||
) -> Result<Vec<types::MessageSummary>, DbError> {
|
) -> Result<Vec<types::MessageSummary>, DbError> {
|
||||||
let mut conn = pool.acquire().await?;
|
let mut conn = pool.acquire().await?;
|
||||||
@@ -213,16 +213,16 @@ pub async fn get_conversation_messages(
|
|||||||
let rows = sqlx::query_as!(
|
let rows = sqlx::query_as!(
|
||||||
types::MessageSummary,
|
types::MessageSummary,
|
||||||
r#"
|
r#"
|
||||||
SELECT id, parent_id, role, content, created_at, tokens
|
SELECT id, parent_id, role as "role: MessageRole", content, created_at, tokens
|
||||||
FROM chat.message
|
FROM chat.message
|
||||||
WHERE conversation_id = $1
|
WHERE conversation_id = $1
|
||||||
AND ($2::timestamptz IS NULL OR created_at < $2)
|
AND ($2::timestamptz IS NULL OR created_at < $2)
|
||||||
ORDER BY created_at ASC
|
ORDER BY created_at DESC
|
||||||
LIMIT $3
|
LIMIT $3
|
||||||
"#,
|
"#,
|
||||||
conversation_id,
|
conversation_id,
|
||||||
before,
|
before,
|
||||||
limit
|
limit as i64
|
||||||
)
|
)
|
||||||
.fetch_all(&mut *conn)
|
.fetch_all(&mut *conn)
|
||||||
.await?;
|
.await?;
|
||||||
|
|||||||
@@ -1,8 +1,7 @@
|
|||||||
use serde::Serialize;
|
|
||||||
use uuid::Uuid;
|
use uuid::Uuid;
|
||||||
|
|
||||||
#[derive(Debug, Clone, sqlx::Type)]
|
#[derive(Debug, Clone, sqlx::Type)]
|
||||||
#[sqlx(type_name = "text")]
|
#[sqlx(type_name = "chat.role")]
|
||||||
#[sqlx(rename_all = "lowercase")]
|
#[sqlx(rename_all = "lowercase")]
|
||||||
pub enum MessageRole {
|
pub enum MessageRole {
|
||||||
User,
|
User,
|
||||||
@@ -16,7 +15,7 @@ pub enum ConversationState {
|
|||||||
Created(Uuid),
|
Created(Uuid),
|
||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Debug, sqlx::FromRow, Serialize)]
|
#[derive(Debug, sqlx::FromRow)]
|
||||||
pub struct ConversationSummary {
|
pub struct ConversationSummary {
|
||||||
pub id: Uuid,
|
pub id: Uuid,
|
||||||
pub title: Option<String>,
|
pub title: Option<String>,
|
||||||
@@ -24,11 +23,11 @@ pub struct ConversationSummary {
|
|||||||
pub updated_at: chrono::DateTime<chrono::Utc>,
|
pub updated_at: chrono::DateTime<chrono::Utc>,
|
||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Debug, sqlx::FromRow, Serialize)]
|
#[derive(Debug, sqlx::FromRow)]
|
||||||
pub struct MessageSummary {
|
pub struct MessageSummary {
|
||||||
pub id: Uuid,
|
pub id: Uuid,
|
||||||
pub parent_id: Option<Uuid>,
|
pub parent_id: Option<Uuid>,
|
||||||
pub role: String,
|
pub role: MessageRole,
|
||||||
pub content: String,
|
pub content: String,
|
||||||
pub created_at: chrono::DateTime<chrono::Utc>,
|
pub created_at: chrono::DateTime<chrono::Utc>,
|
||||||
pub tokens: Option<i32>,
|
pub tokens: Option<i32>,
|
||||||
|
|||||||
@@ -1,6 +1,5 @@
|
|||||||
pub mod errors;
|
|
||||||
pub mod pool;
|
pub mod pool;
|
||||||
|
|
||||||
pub mod api_key;
|
pub mod api_key;
|
||||||
pub mod chat;
|
pub mod chat;
|
||||||
pub mod user;
|
pub mod user_activity;
|
||||||
|
|||||||
@@ -1,4 +1,4 @@
|
|||||||
use super::errors::DbError;
|
use crate::databases::errors::DbError;
|
||||||
use sqlx::{PgPool, postgres::PgPoolOptions};
|
use sqlx::{PgPool, postgres::PgPoolOptions};
|
||||||
use std::time::Duration;
|
use std::time::Duration;
|
||||||
|
|
||||||
|
|||||||
@@ -1,20 +0,0 @@
|
|||||||
use crate::databases::postgres::errors::DbError;
|
|
||||||
|
|
||||||
use sqlx::PgPool;
|
|
||||||
use uuid::Uuid;
|
|
||||||
|
|
||||||
pub async fn ensure_user_exists(pool: &PgPool, user_id: Uuid) -> Result<(), DbError> {
|
|
||||||
sqlx::query!(
|
|
||||||
r#"
|
|
||||||
INSERT INTO auth.app_user (id)
|
|
||||||
VALUES ($1)
|
|
||||||
ON CONFLICT (id) DO NOTHING
|
|
||||||
"#,
|
|
||||||
user_id
|
|
||||||
)
|
|
||||||
.execute(pool)
|
|
||||||
.await
|
|
||||||
.map_err(DbError::Connection)?;
|
|
||||||
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
@@ -0,0 +1,21 @@
|
|||||||
|
use crate::databases::errors::DbError;
|
||||||
|
|
||||||
|
use sqlx::PgPool;
|
||||||
|
use uuid::Uuid;
|
||||||
|
|
||||||
|
pub async fn upsert_user_activity(pool: &PgPool, user_id: &Uuid) -> Result<(), DbError> {
|
||||||
|
sqlx::query!(
|
||||||
|
r#"
|
||||||
|
INSERT INTO auth.app_user (id, last_seen_at)
|
||||||
|
VALUES ($1, now())
|
||||||
|
ON CONFLICT (id)
|
||||||
|
DO UPDATE SET last_seen_at = now()
|
||||||
|
"#,
|
||||||
|
user_id
|
||||||
|
)
|
||||||
|
.execute(pool)
|
||||||
|
.await
|
||||||
|
.map_err(DbError::Connection)?;
|
||||||
|
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
-55
@@ -1,55 +0,0 @@
|
|||||||
use utoipa::OpenApi;
|
|
||||||
|
|
||||||
use crate::dto::api;
|
|
||||||
use crate::routes;
|
|
||||||
|
|
||||||
#[derive(OpenApi)]
|
|
||||||
#[openapi(
|
|
||||||
info(
|
|
||||||
title = "Ollama Proxy",
|
|
||||||
description = "OpenAI-compatible proxy for local Ollama models",
|
|
||||||
version = "0.1.0",
|
|
||||||
license(
|
|
||||||
name = "MIT",
|
|
||||||
url = "https://opensource.org/licenses/MIT"
|
|
||||||
),
|
|
||||||
),
|
|
||||||
paths(
|
|
||||||
routes::v1::chat::completions,
|
|
||||||
routes::v1::chat::chat_completions,
|
|
||||||
routes::v1::models::list_models,
|
|
||||||
routes::v1::models::load_model,
|
|
||||||
routes::v1::models::unload_model,
|
|
||||||
),
|
|
||||||
components(
|
|
||||||
schemas(
|
|
||||||
api::ErrorResponse,
|
|
||||||
api::ModelsResponse,
|
|
||||||
api::ModelInfo,
|
|
||||||
api::LoadModelResponse,
|
|
||||||
api::LoadModelBody,
|
|
||||||
api::UnloadModelResponse,
|
|
||||||
api::BaseLLMRequest,
|
|
||||||
api::CompletionRequest,
|
|
||||||
api::CompletionObject,
|
|
||||||
api::FinishReason,
|
|
||||||
api::CompletionResponse,
|
|
||||||
api::Choice,
|
|
||||||
api::Usage,
|
|
||||||
api::CompletionChunk,
|
|
||||||
api::ChatRequest,
|
|
||||||
api::Message,
|
|
||||||
api::Role,
|
|
||||||
api::ChatCompletionResponse,
|
|
||||||
api::ChatChoice,
|
|
||||||
api::ChatCompletionChunk,
|
|
||||||
api::ChatChunkChoice,
|
|
||||||
api::Delta,
|
|
||||||
)
|
|
||||||
),
|
|
||||||
tags(
|
|
||||||
(name = "chat", description = "Chat & completions"),
|
|
||||||
(name = "models", description = "Model management")
|
|
||||||
)
|
|
||||||
)]
|
|
||||||
pub struct ApiDoc;
|
|
||||||
@@ -1,2 +0,0 @@
|
|||||||
pub mod api;
|
|
||||||
pub mod ollama;
|
|
||||||
@@ -1,80 +0,0 @@
|
|||||||
use serde::{Deserialize, Serialize};
|
|
||||||
|
|
||||||
use crate::dto::api;
|
|
||||||
|
|
||||||
#[derive(Debug, Serialize, Deserialize)]
|
|
||||||
pub struct OllamaModels {
|
|
||||||
pub models: Vec<OllamaModel>,
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Debug, Serialize, Deserialize)]
|
|
||||||
pub struct OllamaModel {
|
|
||||||
pub name: String,
|
|
||||||
|
|
||||||
pub details: Option<OllamaModelDetails>,
|
|
||||||
|
|
||||||
pub size: Option<u64>,
|
|
||||||
pub digest: Option<String>,
|
|
||||||
pub modified_at: Option<String>,
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Debug, Serialize, Deserialize)]
|
|
||||||
pub struct OllamaModelDetails {
|
|
||||||
pub family: Option<String>,
|
|
||||||
pub parameter_size: Option<String>,
|
|
||||||
pub quantization_level: Option<String>,
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Debug, Serialize, Deserialize)]
|
|
||||||
pub struct OllamaOptions {
|
|
||||||
pub temperature: Option<f32>,
|
|
||||||
pub top_p: Option<f32>,
|
|
||||||
pub top_k: Option<u32>,
|
|
||||||
pub repeat_penalty: Option<f32>,
|
|
||||||
pub seed: Option<i64>,
|
|
||||||
|
|
||||||
pub num_ctx: Option<u32>,
|
|
||||||
pub num_predict: Option<u32>,
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Debug, Serialize)]
|
|
||||||
pub struct OllamaGenerateRequest<'a> {
|
|
||||||
pub model: &'a str,
|
|
||||||
pub prompt: &'a str,
|
|
||||||
pub stream: bool,
|
|
||||||
pub options: OllamaOptions,
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Debug, Serialize, Deserialize)]
|
|
||||||
pub struct OllamaGenerateResponse {
|
|
||||||
pub model: String,
|
|
||||||
pub created_at: Option<String>,
|
|
||||||
pub response: String,
|
|
||||||
pub done: bool,
|
|
||||||
|
|
||||||
#[serde(default)]
|
|
||||||
pub context: Option<Vec<u64>>,
|
|
||||||
|
|
||||||
pub total_duration: Option<u64>,
|
|
||||||
pub load_duration: Option<u64>,
|
|
||||||
pub prompt_eval_count: Option<u32>,
|
|
||||||
pub eval_count: Option<u32>,
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Debug, Serialize)]
|
|
||||||
pub struct OllamaChatRequest<'a> {
|
|
||||||
pub model: &'a str,
|
|
||||||
pub messages: &'a [api::Message],
|
|
||||||
pub stream: bool,
|
|
||||||
pub options: OllamaOptions,
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Debug, Deserialize)]
|
|
||||||
pub struct OllamaChatResponse {
|
|
||||||
pub model: String,
|
|
||||||
pub message: api::Message,
|
|
||||||
pub done: bool,
|
|
||||||
|
|
||||||
pub prompt_eval_count: Option<u32>,
|
|
||||||
pub eval_count: Option<u32>,
|
|
||||||
}
|
|
||||||
+3
-4
@@ -1,7 +1,6 @@
|
|||||||
pub mod api;
|
pub mod api;
|
||||||
|
pub mod core;
|
||||||
pub mod databases;
|
pub mod databases;
|
||||||
pub mod dto;
|
pub mod mappers;
|
||||||
pub mod middlewares;
|
|
||||||
pub mod providers;
|
pub mod providers;
|
||||||
pub mod state;
|
pub mod services;
|
||||||
pub mod utils;
|
|
||||||
|
|||||||
+6
-64
@@ -1,79 +1,21 @@
|
|||||||
mod api;
|
mod api;
|
||||||
|
mod core;
|
||||||
mod databases;
|
mod databases;
|
||||||
mod docs;
|
mod mappers;
|
||||||
mod dto;
|
|
||||||
mod middlewares;
|
|
||||||
mod providers;
|
mod providers;
|
||||||
mod routes;
|
mod services;
|
||||||
mod state;
|
|
||||||
mod utils;
|
|
||||||
|
|
||||||
use crate::databases::postgres;
|
use api::app::build_app;
|
||||||
use crate::providers::ollama::client::OllamaProvider;
|
|
||||||
use crate::state::app_state::AppState;
|
|
||||||
|
|
||||||
use axum::Router;
|
|
||||||
use axum::http::{HeaderName, HeaderValue, Method, header};
|
|
||||||
use once_cell::sync::Lazy;
|
|
||||||
use std::env;
|
|
||||||
use std::net::SocketAddr;
|
use std::net::SocketAddr;
|
||||||
use std::sync::Arc;
|
|
||||||
use tower_http::cors::CorsLayer;
|
|
||||||
use tracing_subscriber::{EnvFilter, fmt};
|
|
||||||
|
|
||||||
static OLLAMA_URL: Lazy<String> = Lazy::new(|| env::var("OLLAMA_URL").expect("OLLAMA_URL not set"));
|
|
||||||
|
|
||||||
pub fn init_tracing() {
|
|
||||||
let filter = env::var("RUST_LOG").unwrap_or_else(|_| "info".to_string());
|
|
||||||
|
|
||||||
fmt().with_env_filter(EnvFilter::new(filter)).init();
|
|
||||||
}
|
|
||||||
|
|
||||||
#[tokio::main]
|
#[tokio::main]
|
||||||
async fn main() {
|
async fn main() {
|
||||||
#[cfg(debug_assertions)]
|
#[cfg(debug_assertions)]
|
||||||
{
|
|
||||||
dotenvy::dotenv().ok();
|
dotenvy::dotenv().ok();
|
||||||
}
|
|
||||||
|
|
||||||
init_tracing();
|
tracing_subscriber::fmt::init();
|
||||||
|
|
||||||
// DB Connection
|
let app = build_app().await;
|
||||||
let database_url = env::var("DATABASE_URL").expect("DATABASE_URL must be set");
|
|
||||||
|
|
||||||
let pool = postgres::pool::create_pool(&database_url)
|
|
||||||
.await
|
|
||||||
.expect("Fatal error");
|
|
||||||
|
|
||||||
let state = AppState {
|
|
||||||
ollama: Arc::new(OllamaProvider::new(OLLAMA_URL.as_str())),
|
|
||||||
postgres: pool,
|
|
||||||
};
|
|
||||||
|
|
||||||
let cors_origin =
|
|
||||||
env::var("CORS_ORIGIN").unwrap_or_else(|_| "http://localhost:3000".to_string());
|
|
||||||
|
|
||||||
let cors = CorsLayer::new()
|
|
||||||
.allow_origin(cors_origin.parse::<HeaderValue>().unwrap())
|
|
||||||
.allow_methods([
|
|
||||||
Method::GET,
|
|
||||||
Method::POST,
|
|
||||||
Method::PUT,
|
|
||||||
Method::DELETE,
|
|
||||||
Method::OPTIONS,
|
|
||||||
])
|
|
||||||
.allow_headers([
|
|
||||||
header::CONTENT_TYPE,
|
|
||||||
header::AUTHORIZATION,
|
|
||||||
header::ACCEPT,
|
|
||||||
HeaderName::from_static("x-api-key"),
|
|
||||||
])
|
|
||||||
.allow_credentials(true);
|
|
||||||
|
|
||||||
let app = Router::new()
|
|
||||||
.nest("/v1", routes::v1::router(state.clone()))
|
|
||||||
.layer(cors)
|
|
||||||
.with_state(state);
|
|
||||||
|
|
||||||
let addr = SocketAddr::from(([0, 0, 0, 0], 3001));
|
let addr = SocketAddr::from(([0, 0, 0, 0], 3001));
|
||||||
tracing::debug!("Server running on {}", addr);
|
tracing::debug!("Server running on {}", addr);
|
||||||
|
|||||||
@@ -0,0 +1,66 @@
|
|||||||
|
use crate::{api, core};
|
||||||
|
|
||||||
|
impl From<api::types::CompletionRequest> for core::llm::completions::CompletionRequest {
|
||||||
|
fn from(m: api::types::CompletionRequest) -> Self {
|
||||||
|
Self {
|
||||||
|
model: m.model,
|
||||||
|
prompt: m.prompt,
|
||||||
|
options: core::llm::completions::CompletionOptions {
|
||||||
|
temperature: m.options.temperature,
|
||||||
|
keep_alive: m.options.keep_alive,
|
||||||
|
num_ctx: m.options.num_ctx,
|
||||||
|
top_p: m.options.top_p,
|
||||||
|
top_k: m.options.top_k,
|
||||||
|
seed: m.options.seed,
|
||||||
|
num_predict: m.options.num_predict,
|
||||||
|
stop: m.options.stop,
|
||||||
|
stream: m.options.stream,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<api::types::ChatRequest> for core::llm::chat::ChatCompletionRequest {
|
||||||
|
fn from(m: api::types::ChatRequest) -> Self {
|
||||||
|
Self {
|
||||||
|
model: m.model,
|
||||||
|
message: m.message.into(),
|
||||||
|
options: core::llm::chat::ChatCompletionOptions {
|
||||||
|
temperature: m.base.temperature,
|
||||||
|
keep_alive: m.base.keep_alive,
|
||||||
|
num_ctx: m.base.num_ctx,
|
||||||
|
top_p: m.base.top_p,
|
||||||
|
top_k: m.base.top_k,
|
||||||
|
seed: m.base.seed,
|
||||||
|
num_predict: m.base.num_predict,
|
||||||
|
stop: m.base.stop,
|
||||||
|
stream: m.base.stream,
|
||||||
|
context_depth: m.base.context_depth.unwrap_or(10),
|
||||||
|
},
|
||||||
|
conversation_id: m.conversation_id,
|
||||||
|
parent_id: m.parent_id,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<api::types::Message> for core::llm::chat::Message {
|
||||||
|
fn from(m: api::types::Message) -> Self {
|
||||||
|
Self {
|
||||||
|
role: match m.role {
|
||||||
|
api::types::Role::System => core::llm::ChatRole::System,
|
||||||
|
api::types::Role::User => core::llm::ChatRole::User,
|
||||||
|
api::types::Role::Assistant => core::llm::ChatRole::Assistant,
|
||||||
|
},
|
||||||
|
content: m.content,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<api::types::ApiKeyScope> for core::auth::api_key::KeyRole {
|
||||||
|
fn from(role: api::types::ApiKeyScope) -> Self {
|
||||||
|
match role {
|
||||||
|
api::types::ApiKeyScope::User => core::auth::api_key::KeyRole::User,
|
||||||
|
api::types::ApiKeyScope::Admin => core::auth::api_key::KeyRole::Admin,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,101 @@
|
|||||||
|
use crate::{api, core};
|
||||||
|
|
||||||
|
impl From<core::llm::models::Models> for api::types::ModelsResponse {
|
||||||
|
fn from(m: core::llm::models::Models) -> Self {
|
||||||
|
Self {
|
||||||
|
models: m.models.into_iter().map(Into::into).collect(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<core::llm::models::ModelMetadata> for api::types::ModelMetadata {
|
||||||
|
fn from(m: core::llm::models::ModelMetadata) -> Self {
|
||||||
|
Self {
|
||||||
|
extra: m
|
||||||
|
.extra
|
||||||
|
.iter()
|
||||||
|
.map(|(k, v)| (k.clone(), v.clone()))
|
||||||
|
.collect(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<core::llm::models::Model> for api::types::ModelInfo {
|
||||||
|
fn from(m: core::llm::models::Model) -> Self {
|
||||||
|
Self {
|
||||||
|
name: m.name,
|
||||||
|
family: m.family,
|
||||||
|
parameter_size: m.size,
|
||||||
|
metadata: m.metadata.into(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<core::databases::conversations::ConversationSummary> for api::types::ConversationSummary {
|
||||||
|
fn from(c: core::databases::conversations::ConversationSummary) -> Self {
|
||||||
|
Self {
|
||||||
|
id: c.id,
|
||||||
|
title: c.title,
|
||||||
|
created_at: c.created_at,
|
||||||
|
updated_at: c.updated_at,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<core::databases::conversations::ConversationList>
|
||||||
|
for api::types::ConversationListResponse
|
||||||
|
{
|
||||||
|
fn from(c: core::databases::conversations::ConversationList) -> Self {
|
||||||
|
Self {
|
||||||
|
conversations: c.conversations.into_iter().map(Into::into).collect(),
|
||||||
|
has_more: c.has_more,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<core::databases::conversations::MessageSummary> for api::types::MessageSummary {
|
||||||
|
fn from(c: core::databases::conversations::MessageSummary) -> Self {
|
||||||
|
Self {
|
||||||
|
id: c.id,
|
||||||
|
parent_id: c.parent_id,
|
||||||
|
role: match c.role {
|
||||||
|
core::llm::ChatRole::User => api::types::ApiChatRole::User,
|
||||||
|
core::llm::ChatRole::Assistant => api::types::ApiChatRole::Assistant,
|
||||||
|
core::llm::ChatRole::System => api::types::ApiChatRole::System,
|
||||||
|
},
|
||||||
|
content: c.content,
|
||||||
|
created_at: c.created_at,
|
||||||
|
tokens: c.tokens,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<core::databases::conversations::MessageList> for api::types::MessageListResponse {
|
||||||
|
fn from(c: core::databases::conversations::MessageList) -> Self {
|
||||||
|
Self {
|
||||||
|
messages: c.messages.into_iter().map(Into::into).collect(),
|
||||||
|
has_more: c.has_more,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<core::llm::completions::CompletionResultNoStream> for api::types::CompletionResponse {
|
||||||
|
fn from(m: core::llm::completions::CompletionResultNoStream) -> Self {
|
||||||
|
Self {
|
||||||
|
id: m.id,
|
||||||
|
object: api::types::CompletionObject::TextCompletion,
|
||||||
|
created: m.created_at,
|
||||||
|
model: m.model,
|
||||||
|
choices: vec![api::types::Choice {
|
||||||
|
finish_reason: api::types::FinishReason::Stop,
|
||||||
|
text: m.text,
|
||||||
|
index: 0,
|
||||||
|
}],
|
||||||
|
usage: api::types::Usage {
|
||||||
|
completion_tokens: m.completion_tokens,
|
||||||
|
prompt_tokens: m.prompt_tokens,
|
||||||
|
total_tokens: m.completion_tokens + m.prompt_tokens,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,30 @@
|
|||||||
|
impl From<crate::core::llm::ChatRole> for crate::databases::postgres::chat::types::MessageRole {
|
||||||
|
fn from(m: crate::core::llm::ChatRole) -> Self {
|
||||||
|
match m {
|
||||||
|
crate::core::llm::ChatRole::System => {
|
||||||
|
crate::databases::postgres::chat::types::MessageRole::System
|
||||||
|
}
|
||||||
|
crate::core::llm::ChatRole::Assistant => {
|
||||||
|
crate::databases::postgres::chat::types::MessageRole::Assistant
|
||||||
|
}
|
||||||
|
crate::core::llm::ChatRole::User => {
|
||||||
|
crate::databases::postgres::chat::types::MessageRole::User
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<crate::core::auth::api_key::KeyRole>
|
||||||
|
for crate::databases::postgres::api_key::types::Role
|
||||||
|
{
|
||||||
|
fn from(role: crate::core::auth::api_key::KeyRole) -> Self {
|
||||||
|
match role {
|
||||||
|
crate::core::auth::api_key::KeyRole::User => {
|
||||||
|
crate::databases::postgres::api_key::types::Role::User
|
||||||
|
}
|
||||||
|
crate::core::auth::api_key::KeyRole::Admin => {
|
||||||
|
crate::databases::postgres::api_key::types::Role::Admin
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,67 @@
|
|||||||
|
use crate::core;
|
||||||
|
use crate::providers::ollama;
|
||||||
|
|
||||||
|
impl From<core::llm::completions::CompletionOptions> for ollama::types::OllamaOptions {
|
||||||
|
fn from(opts: core::llm::completions::CompletionOptions) -> Self {
|
||||||
|
Self {
|
||||||
|
seed: opts.seed,
|
||||||
|
temperature: opts.temperature,
|
||||||
|
top_p: opts.top_p,
|
||||||
|
top_k: opts.top_k,
|
||||||
|
stop: opts.stop,
|
||||||
|
num_ctx: opts.num_ctx,
|
||||||
|
num_predict: opts.num_predict,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<core::llm::chat::ChatCompletionOptions> for ollama::types::OllamaOptions {
|
||||||
|
fn from(opts: core::llm::chat::ChatCompletionOptions) -> Self {
|
||||||
|
Self {
|
||||||
|
seed: opts.seed,
|
||||||
|
temperature: opts.temperature,
|
||||||
|
top_p: opts.top_p,
|
||||||
|
top_k: opts.top_k,
|
||||||
|
stop: opts.stop,
|
||||||
|
num_ctx: opts.num_ctx,
|
||||||
|
num_predict: opts.num_predict,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<core::llm::chat::Message> for ollama::types::OllamaMessage {
|
||||||
|
fn from(m: core::llm::chat::Message) -> Self {
|
||||||
|
Self {
|
||||||
|
role: match m.role {
|
||||||
|
core::llm::ChatRole::System => ollama::types::OllamaRole::System,
|
||||||
|
core::llm::ChatRole::User => ollama::types::OllamaRole::User,
|
||||||
|
core::llm::ChatRole::Assistant => ollama::types::OllamaRole::Assistant,
|
||||||
|
},
|
||||||
|
content: m.content,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<core::llm::completions::CompletionRequest> for ollama::types::OllamaGenerateRequest {
|
||||||
|
fn from(r: core::llm::completions::CompletionRequest) -> Self {
|
||||||
|
Self {
|
||||||
|
model: r.model,
|
||||||
|
prompt: r.prompt,
|
||||||
|
stream: r.options.stream,
|
||||||
|
keep_alive: r.options.keep_alive.clone().unwrap_or("5m".to_string()),
|
||||||
|
options: Some(r.options.clone().into()),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<core::llm::chat::ChatCompletionRequest> for ollama::types::OllamaChatRequest {
|
||||||
|
fn from(r: core::llm::chat::ChatCompletionRequest) -> Self {
|
||||||
|
Self {
|
||||||
|
model: r.model,
|
||||||
|
messages: Vec::new(),
|
||||||
|
stream: r.options.stream,
|
||||||
|
keep_alive: r.options.keep_alive.clone().unwrap_or("5m".to_string()),
|
||||||
|
options: Some(r.options.clone().into()),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,69 @@
|
|||||||
|
use crate::core;
|
||||||
|
use crate::databases::postgres;
|
||||||
|
|
||||||
|
impl From<postgres::chat::types::ConversationSummary>
|
||||||
|
for core::databases::conversations::ConversationSummary
|
||||||
|
{
|
||||||
|
fn from(c: postgres::chat::types::ConversationSummary) -> Self {
|
||||||
|
Self {
|
||||||
|
id: c.id,
|
||||||
|
title: c.title.unwrap_or("No title Generated".to_string()),
|
||||||
|
created_at: c.created_at,
|
||||||
|
updated_at: c.updated_at,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<postgres::chat::types::MessageSummary>
|
||||||
|
for core::databases::conversations::MessageSummary
|
||||||
|
{
|
||||||
|
fn from(m: postgres::chat::types::MessageSummary) -> Self {
|
||||||
|
Self {
|
||||||
|
id: m.id,
|
||||||
|
parent_id: m.parent_id,
|
||||||
|
role: match m.role {
|
||||||
|
postgres::chat::types::MessageRole::User => core::llm::ChatRole::User,
|
||||||
|
postgres::chat::types::MessageRole::Assistant => core::llm::ChatRole::Assistant,
|
||||||
|
postgres::chat::types::MessageRole::System => core::llm::ChatRole::System,
|
||||||
|
},
|
||||||
|
content: m.content,
|
||||||
|
created_at: m.created_at,
|
||||||
|
tokens: m.tokens,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<postgres::api_key::types::Role> for core::auth::api_key::KeyRole {
|
||||||
|
fn from(role: postgres::api_key::types::Role) -> Self {
|
||||||
|
match role {
|
||||||
|
postgres::api_key::types::Role::User => core::auth::api_key::KeyRole::User,
|
||||||
|
postgres::api_key::types::Role::Admin => core::auth::api_key::KeyRole::Admin,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<postgres::api_key::types::AuthContext> for core::auth::api_key::AuthContext {
|
||||||
|
fn from(m: postgres::api_key::types::AuthContext) -> Self {
|
||||||
|
Self {
|
||||||
|
user_id: m.user_id,
|
||||||
|
_api_key_id: m.api_key_id,
|
||||||
|
roles: m.roles.into_iter().map(Into::into).collect(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<postgres::chat::types::ConversationState>
|
||||||
|
for core::databases::conversations::ConversationResult
|
||||||
|
{
|
||||||
|
fn from(m: postgres::chat::types::ConversationState) -> Self {
|
||||||
|
match m {
|
||||||
|
postgres::chat::types::ConversationState::Existing(id) => {
|
||||||
|
core::databases::conversations::ConversationResult::Existing(id)
|
||||||
|
}
|
||||||
|
|
||||||
|
postgres::chat::types::ConversationState::Created(id) => {
|
||||||
|
core::databases::conversations::ConversationResult::Created(id)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,30 @@
|
|||||||
|
use crate::core::auth::jwt::JwtClaims;
|
||||||
|
use crate::providers::keycloak::claims::KeycloakClaims;
|
||||||
|
|
||||||
|
use std::collections::HashMap;
|
||||||
|
use uuid::Uuid;
|
||||||
|
|
||||||
|
impl From<KeycloakClaims> for JwtClaims {
|
||||||
|
fn from(c: KeycloakClaims) -> Self {
|
||||||
|
let realm_roles = c
|
||||||
|
.realm_access
|
||||||
|
.as_ref()
|
||||||
|
.map(|r| r.roles.clone())
|
||||||
|
.unwrap_or_default();
|
||||||
|
|
||||||
|
let client_roles = c
|
||||||
|
.resource_access
|
||||||
|
.into_iter()
|
||||||
|
.map(|(k, v)| (k, v.roles))
|
||||||
|
.collect::<HashMap<_, _>>();
|
||||||
|
|
||||||
|
Self {
|
||||||
|
user_id: c.sub.parse().unwrap_or_else(|_| Uuid::nil()),
|
||||||
|
_username: c.preferred_username,
|
||||||
|
_exp: c.exp,
|
||||||
|
_issuer: c.iss,
|
||||||
|
realm_roles,
|
||||||
|
client_roles,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,7 @@
|
|||||||
|
pub mod api_to_core;
|
||||||
|
pub mod core_to_api;
|
||||||
|
pub mod core_to_database;
|
||||||
|
pub mod core_to_ollama;
|
||||||
|
pub mod database_to_core;
|
||||||
|
pub mod keycloak_to_core;
|
||||||
|
pub mod ollama_to_core;
|
||||||
@@ -0,0 +1,50 @@
|
|||||||
|
use crate::core;
|
||||||
|
use crate::providers::ollama;
|
||||||
|
|
||||||
|
impl From<ollama::types::OllamaModels> for core::llm::models::Models {
|
||||||
|
fn from(m: ollama::types::OllamaModels) -> Self {
|
||||||
|
Self {
|
||||||
|
models: m.models.into_iter().map(Into::into).collect(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<ollama::types::OllamaModel> for core::llm::models::Model {
|
||||||
|
fn from(m: ollama::types::OllamaModel) -> Self {
|
||||||
|
Self {
|
||||||
|
name: m.name,
|
||||||
|
family: m.details.as_ref().and_then(|d| d.family.clone()),
|
||||||
|
size: m.details.as_ref().and_then(|d| d.parameter_size.clone()),
|
||||||
|
metadata: core::llm::models::ModelMetadata {
|
||||||
|
extra: [
|
||||||
|
("digest", m.digest.unwrap_or_default()),
|
||||||
|
("size_bytes", m.size.unwrap_or_default().to_string()),
|
||||||
|
("modified_at", m.modified_at.unwrap_or_default().to_string()),
|
||||||
|
(
|
||||||
|
"quantization_level",
|
||||||
|
m.details
|
||||||
|
.as_ref()
|
||||||
|
.and_then(|d| d.quantization_level.clone())
|
||||||
|
.unwrap_or_default(),
|
||||||
|
),
|
||||||
|
]
|
||||||
|
.into_iter()
|
||||||
|
.map(|(k, v)| (k.to_string(), v))
|
||||||
|
.collect(),
|
||||||
|
},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<ollama::types::OllamaMessage> for core::llm::chat::Message {
|
||||||
|
fn from(m: ollama::types::OllamaMessage) -> Self {
|
||||||
|
Self {
|
||||||
|
role: match m.role {
|
||||||
|
ollama::types::OllamaRole::System => core::llm::ChatRole::System,
|
||||||
|
ollama::types::OllamaRole::User => core::llm::ChatRole::User,
|
||||||
|
ollama::types::OllamaRole::Assistant => core::llm::ChatRole::Assistant,
|
||||||
|
},
|
||||||
|
content: m.content,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -1,15 +0,0 @@
|
|||||||
use uuid::Uuid;
|
|
||||||
|
|
||||||
#[derive(PartialEq, Eq, Clone, Debug)]
|
|
||||||
pub enum ApiKeyClaimsRoles {
|
|
||||||
Admin,
|
|
||||||
Read,
|
|
||||||
Write,
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Clone, Debug)]
|
|
||||||
pub struct ApiKeyClaims {
|
|
||||||
pub sub: Uuid,
|
|
||||||
pub api_key_id: Uuid,
|
|
||||||
pub roles: Vec<ApiKeyClaimsRoles>,
|
|
||||||
}
|
|
||||||
@@ -1,148 +0,0 @@
|
|||||||
use jsonwebtoken::{Algorithm, DecodingKey, Validation, decode, decode_header};
|
|
||||||
use once_cell::sync::Lazy;
|
|
||||||
use serde::{Deserialize, Serialize};
|
|
||||||
use serde_json::Value;
|
|
||||||
use std::collections::HashMap;
|
|
||||||
use std::env;
|
|
||||||
use std::sync::Arc;
|
|
||||||
use std::time::{Duration, Instant};
|
|
||||||
use tokio::sync::RwLock;
|
|
||||||
|
|
||||||
// ------ JWKS ------
|
|
||||||
|
|
||||||
#[derive(Clone)]
|
|
||||||
struct JwksCache {
|
|
||||||
jwks: Value,
|
|
||||||
last_fetched: Instant,
|
|
||||||
}
|
|
||||||
|
|
||||||
static JWK_CACHE: Lazy<Arc<RwLock<Option<JwksCache>>>> = Lazy::new(|| Arc::new(RwLock::new(None)));
|
|
||||||
|
|
||||||
static JWKS_URL: Lazy<String> = Lazy::new(|| env::var("JWKS_URL").expect("JWKS_URL not set"));
|
|
||||||
|
|
||||||
async fn fetch_jwks() -> Result<Value, reqwest::Error> {
|
|
||||||
let jwks = reqwest::get(JWKS_URL.as_str())
|
|
||||||
.await?
|
|
||||||
.json::<Value>()
|
|
||||||
.await?;
|
|
||||||
|
|
||||||
Ok(jwks)
|
|
||||||
}
|
|
||||||
|
|
||||||
pub async fn refresh_jwks() -> Result<Value, reqwest::Error> {
|
|
||||||
let jwks = fetch_jwks().await?;
|
|
||||||
|
|
||||||
let mut write = JWK_CACHE.write().await;
|
|
||||||
|
|
||||||
*write = Some(JwksCache {
|
|
||||||
jwks: jwks.clone(),
|
|
||||||
last_fetched: Instant::now(),
|
|
||||||
});
|
|
||||||
|
|
||||||
Ok(jwks)
|
|
||||||
}
|
|
||||||
|
|
||||||
pub async fn get_jwks() -> Result<Value, reqwest::Error> {
|
|
||||||
let ttl = Duration::from_secs(3600); // 1 hour
|
|
||||||
|
|
||||||
{
|
|
||||||
// Read lock first (fast path)
|
|
||||||
let read = JWK_CACHE.read().await;
|
|
||||||
|
|
||||||
if let Some(cache) = read
|
|
||||||
.as_ref()
|
|
||||||
.filter(|cache| cache.last_fetched.elapsed() < ttl)
|
|
||||||
{
|
|
||||||
return Ok(cache.jwks.clone());
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// Expired or empty → refresh
|
|
||||||
refresh_jwks().await
|
|
||||||
}
|
|
||||||
|
|
||||||
// ------ Claims ------
|
|
||||||
|
|
||||||
#[derive(Debug, Deserialize, Serialize, Clone, Default)]
|
|
||||||
pub struct KeycloakClaims {
|
|
||||||
pub sub: String,
|
|
||||||
pub preferred_username: Option<String>,
|
|
||||||
pub exp: usize,
|
|
||||||
pub iss: String,
|
|
||||||
pub aud: Option<Vec<String>>,
|
|
||||||
pub realm_access: Option<RealmAccess>,
|
|
||||||
#[serde(default)]
|
|
||||||
pub resource_access: HashMap<String, ResourceAccess>,
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Debug, Deserialize, Serialize, Clone)]
|
|
||||||
pub struct RealmAccess {
|
|
||||||
pub roles: Vec<String>,
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Debug, Deserialize, Serialize, Clone)]
|
|
||||||
pub struct ResourceAccess {
|
|
||||||
pub roles: Vec<String>,
|
|
||||||
}
|
|
||||||
|
|
||||||
impl KeycloakClaims {
|
|
||||||
pub fn realm_roles(&self) -> &[String] {
|
|
||||||
self.realm_access
|
|
||||||
.as_ref()
|
|
||||||
.map_or(&[], |r| r.roles.as_slice())
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn has_realm_role(&self, role: &str) -> bool {
|
|
||||||
self.realm_roles().iter().any(|r| r == role)
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn client_roles(&self, client: &str) -> &[String] {
|
|
||||||
self.resource_access
|
|
||||||
.get(client)
|
|
||||||
.map_or(&[], |r| r.roles.as_slice())
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn has_client_role(&self, client: &str, role: &str) -> bool {
|
|
||||||
self.client_roles(client).iter().any(|r| r == role)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// ------ Validation ------
|
|
||||||
|
|
||||||
static ISSUER: Lazy<String> = Lazy::new(|| env::var("ISSUER").expect("ISSUER not set"));
|
|
||||||
|
|
||||||
pub fn validate_token(token: &str, jwks: &Value) -> Result<KeycloakClaims, String> {
|
|
||||||
// 1. Decode header
|
|
||||||
let header = decode_header(token).map_err(|_| "Invalid header")?;
|
|
||||||
|
|
||||||
let kid = header.kid.ok_or("Missing kid")?;
|
|
||||||
|
|
||||||
// 2. Find matching key
|
|
||||||
let keys = jwks["keys"].as_array().ok_or("Invalid JWKS")?;
|
|
||||||
|
|
||||||
let key = keys
|
|
||||||
.iter()
|
|
||||||
.find(|k| k["kid"] == kid)
|
|
||||||
.ok_or("Matching key not found")?;
|
|
||||||
|
|
||||||
// 3. Extract RSA components
|
|
||||||
let n = key["n"].as_str().ok_or("Missing n")?;
|
|
||||||
let e = key["e"].as_str().ok_or("Missing e")?;
|
|
||||||
|
|
||||||
let decoding_key =
|
|
||||||
DecodingKey::from_rsa_components(n, e).map_err(|_| "Invalid decoding key")?;
|
|
||||||
|
|
||||||
// 4. Setup validation rules
|
|
||||||
let mut validation = Validation::new(Algorithm::RS256);
|
|
||||||
|
|
||||||
validation.set_issuer(&[ISSUER.as_str()]);
|
|
||||||
|
|
||||||
validation.validate_exp = true;
|
|
||||||
validation.validate_aud = false;
|
|
||||||
|
|
||||||
// 5. Decode & verify
|
|
||||||
let token_data = decode::<KeycloakClaims>(token, &decoding_key, &validation)
|
|
||||||
.map_err(|_| "Token validation failed")?;
|
|
||||||
|
|
||||||
Ok(token_data.claims)
|
|
||||||
}
|
|
||||||
@@ -1,227 +0,0 @@
|
|||||||
use axum::{
|
|
||||||
extract::{Request, State},
|
|
||||||
http::StatusCode,
|
|
||||||
middleware::Next,
|
|
||||||
response::{IntoResponse, Response},
|
|
||||||
};
|
|
||||||
|
|
||||||
use crate::databases::postgres::{
|
|
||||||
api_key::queries::update_last_access, user::queries::ensure_user_exists,
|
|
||||||
};
|
|
||||||
use crate::middlewares::auth::apikey::{ApiKeyClaims, ApiKeyClaimsRoles};
|
|
||||||
use crate::middlewares::auth::keycloak::{KeycloakClaims, get_jwks, refresh_jwks, validate_token};
|
|
||||||
use crate::state::app_state::AppState;
|
|
||||||
use crate::utils::crypto::hash_key;
|
|
||||||
use uuid::Uuid;
|
|
||||||
|
|
||||||
#[derive(Clone, Debug)]
|
|
||||||
pub enum Auth {
|
|
||||||
Jwt(KeycloakClaims),
|
|
||||||
ApiKey(ApiKeyClaims),
|
|
||||||
}
|
|
||||||
|
|
||||||
impl Auth {
|
|
||||||
pub fn user_id(&self) -> Uuid {
|
|
||||||
match self {
|
|
||||||
Auth::Jwt(c) => c.sub.parse().expect("sub is a valid UUID"),
|
|
||||||
Auth::ApiKey(c) => c.sub,
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn has_realm_role(&self, role: &str) -> bool {
|
|
||||||
match self {
|
|
||||||
Auth::Jwt(c) => c.has_realm_role(role),
|
|
||||||
Auth::ApiKey(_) => false, // API keys carry no roles
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn has_client_role(&self, client: &str, role: &str) -> bool {
|
|
||||||
match self {
|
|
||||||
Auth::Jwt(c) => c.has_client_role(client, role),
|
|
||||||
Auth::ApiKey(_) => false,
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
pub async fn auth_middleware(
|
|
||||||
State(state): State<AppState>,
|
|
||||||
request: Request,
|
|
||||||
next: Next,
|
|
||||||
) -> Result<Response, StatusCode> {
|
|
||||||
match try_jwt(&state, request, next).await {
|
|
||||||
Ok(response) => Ok(response),
|
|
||||||
Err((request, next)) => try_api_key(&state, request, next).await,
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Returns Ok(Response) if JWT was valid and request handled.
|
|
||||||
/// Returns Err((request, next)) if no JWT was present (caller should try next method).
|
|
||||||
/// Returns a 401/500 response directly if JWT was present but invalid.
|
|
||||||
async fn try_jwt(
|
|
||||||
state: &AppState,
|
|
||||||
request: Request,
|
|
||||||
next: Next,
|
|
||||||
) -> Result<Response, (Request, Next)> {
|
|
||||||
let token = request
|
|
||||||
.headers()
|
|
||||||
.get("authorization")
|
|
||||||
.and_then(|v| v.to_str().ok())
|
|
||||||
.and_then(|v| v.strip_prefix("Bearer "))
|
|
||||||
.map(str::to_owned);
|
|
||||||
|
|
||||||
let Some(token) = token else {
|
|
||||||
// No Authorization header at all → let API key branch try
|
|
||||||
return Err((request, next));
|
|
||||||
};
|
|
||||||
|
|
||||||
let jwks = match get_jwks().await {
|
|
||||||
Ok(j) => j,
|
|
||||||
Err(e) => {
|
|
||||||
tracing::error!("Failed to fetch JWKS: {e}");
|
|
||||||
// Token was present but we can't validate → hard 500
|
|
||||||
return Ok(StatusCode::INTERNAL_SERVER_ERROR.into_response());
|
|
||||||
}
|
|
||||||
};
|
|
||||||
|
|
||||||
let claims = match validate_token(&token, &jwks) {
|
|
||||||
Ok(c) => c,
|
|
||||||
Err(_) => {
|
|
||||||
// Try refreshing JWKS once
|
|
||||||
match refresh_jwks().await {
|
|
||||||
Ok(fresh_jwks) => match validate_token(&token, &fresh_jwks) {
|
|
||||||
Ok(c) => c,
|
|
||||||
Err(_) => {
|
|
||||||
tracing::warn!("JWT validation failed after JWKS refresh");
|
|
||||||
return Ok(StatusCode::UNAUTHORIZED.into_response());
|
|
||||||
}
|
|
||||||
},
|
|
||||||
Err(e) => {
|
|
||||||
tracing::error!("Failed to refresh JWKS: {e}");
|
|
||||||
return Ok(StatusCode::INTERNAL_SERVER_ERROR.into_response());
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
};
|
|
||||||
|
|
||||||
tracing::debug!("JWT valid, sub={}", claims.sub);
|
|
||||||
handle_auth(state, request, next, Auth::Jwt(claims))
|
|
||||||
.await
|
|
||||||
.map_err(|_| unreachable!())
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Returns Ok(Response) if API key was valid.
|
|
||||||
/// Returns Err(StatusCode) otherwise (UNAUTHORIZED or INTERNAL_SERVER_ERROR).
|
|
||||||
async fn try_api_key(
|
|
||||||
state: &AppState,
|
|
||||||
request: Request,
|
|
||||||
next: Next,
|
|
||||||
) -> Result<Response, StatusCode> {
|
|
||||||
let key = request
|
|
||||||
.headers()
|
|
||||||
.get("x-api-key")
|
|
||||||
.and_then(|v| v.to_str().ok())
|
|
||||||
.ok_or(StatusCode::UNAUTHORIZED)?
|
|
||||||
.to_owned();
|
|
||||||
|
|
||||||
let key_hash = hash_key(&key);
|
|
||||||
|
|
||||||
let row = sqlx::query!(
|
|
||||||
r#"
|
|
||||||
SELECT u.id AS user_id, ak.id as key_id, ak.scopes as roles
|
|
||||||
FROM auth.api_key ak
|
|
||||||
JOIN auth.app_user u ON u.id = ak.created_by
|
|
||||||
WHERE ak.key_hash = $1
|
|
||||||
AND ak.revoked_at IS NULL
|
|
||||||
"#,
|
|
||||||
key_hash
|
|
||||||
)
|
|
||||||
.fetch_optional(&state.postgres)
|
|
||||||
.await
|
|
||||||
.map_err(|e| {
|
|
||||||
tracing::error!("DB error during API key lookup: {e}");
|
|
||||||
StatusCode::INTERNAL_SERVER_ERROR
|
|
||||||
})?
|
|
||||||
.ok_or(StatusCode::UNAUTHORIZED)?;
|
|
||||||
|
|
||||||
tracing::debug!("API key valid, user_id={}", row.user_id);
|
|
||||||
|
|
||||||
let roles = row
|
|
||||||
.roles
|
|
||||||
.into_iter()
|
|
||||||
.filter_map(|r| match r.as_str() {
|
|
||||||
"admin" => Some(ApiKeyClaimsRoles::Admin),
|
|
||||||
"read" => Some(ApiKeyClaimsRoles::Read),
|
|
||||||
"write" => Some(ApiKeyClaimsRoles::Write),
|
|
||||||
_ => None,
|
|
||||||
})
|
|
||||||
.collect();
|
|
||||||
|
|
||||||
handle_auth(
|
|
||||||
state,
|
|
||||||
request,
|
|
||||||
next,
|
|
||||||
Auth::ApiKey(ApiKeyClaims {
|
|
||||||
sub: row.user_id,
|
|
||||||
api_key_id: row.key_id,
|
|
||||||
roles,
|
|
||||||
}),
|
|
||||||
)
|
|
||||||
.await
|
|
||||||
}
|
|
||||||
|
|
||||||
// ── Shared post-auth logic ────────────────────────────────────────────────────
|
|
||||||
|
|
||||||
/// Ensures the user exists in the DB, inserts `Auth` into extensions, runs the next handler.
|
|
||||||
async fn handle_auth(
|
|
||||||
state: &AppState,
|
|
||||||
mut request: Request,
|
|
||||||
next: Next,
|
|
||||||
auth: Auth,
|
|
||||||
) -> Result<Response, StatusCode> {
|
|
||||||
match &auth {
|
|
||||||
Auth::Jwt(_) => {
|
|
||||||
ensure_user_exists(&state.postgres, auth.user_id())
|
|
||||||
.await
|
|
||||||
.map_err(|e| {
|
|
||||||
tracing::error!("ensure_user_exists failed: {e}");
|
|
||||||
StatusCode::INTERNAL_SERVER_ERROR
|
|
||||||
})?;
|
|
||||||
}
|
|
||||||
Auth::ApiKey(key) => {
|
|
||||||
update_last_access(&state.postgres, key.api_key_id)
|
|
||||||
.await
|
|
||||||
.map_err(|e| {
|
|
||||||
tracing::error!("update_last_access failed: {e}");
|
|
||||||
StatusCode::INTERNAL_SERVER_ERROR
|
|
||||||
})?;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
request.extensions_mut().insert(auth);
|
|
||||||
Ok(next.run(request).await)
|
|
||||||
}
|
|
||||||
|
|
||||||
// ── Role guard ───────────────────────────────────────────────────────────────
|
|
||||||
|
|
||||||
/// Layer-level middleware that checks roles *after* `auth_middleware` has run.
|
|
||||||
pub async fn require_roles(
|
|
||||||
request: Request,
|
|
||||||
next: Next,
|
|
||||||
realm_role: Option<&'static str>,
|
|
||||||
client_role: Option<&'static str>,
|
|
||||||
) -> Result<Response, StatusCode> {
|
|
||||||
let auth = request
|
|
||||||
.extensions()
|
|
||||||
.get::<Auth>()
|
|
||||||
.ok_or(StatusCode::UNAUTHORIZED)?;
|
|
||||||
|
|
||||||
if realm_role.is_some_and(|role| !auth.has_realm_role(role)) {
|
|
||||||
return Err(StatusCode::FORBIDDEN);
|
|
||||||
}
|
|
||||||
|
|
||||||
if client_role.is_some_and(|role| !auth.has_client_role("chat-api", role)) {
|
|
||||||
return Err(StatusCode::FORBIDDEN);
|
|
||||||
}
|
|
||||||
|
|
||||||
Ok(next.run(request).await)
|
|
||||||
}
|
|
||||||
@@ -1,5 +0,0 @@
|
|||||||
pub mod apikey;
|
|
||||||
pub mod keycloak;
|
|
||||||
pub mod middleware;
|
|
||||||
|
|
||||||
pub use middleware::auth_middleware;
|
|
||||||
@@ -0,0 +1,23 @@
|
|||||||
|
use serde::Deserialize;
|
||||||
|
use std::collections::HashMap;
|
||||||
|
|
||||||
|
#[derive(Debug, Clone, Default, Deserialize)]
|
||||||
|
pub struct KeycloakClaims {
|
||||||
|
pub sub: String,
|
||||||
|
pub preferred_username: Option<String>,
|
||||||
|
pub exp: usize,
|
||||||
|
pub iss: String,
|
||||||
|
pub _aud: Option<Vec<String>>,
|
||||||
|
pub realm_access: Option<RealmAccess>,
|
||||||
|
pub resource_access: HashMap<String, ResourceAccess>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Clone, Deserialize)]
|
||||||
|
pub struct RealmAccess {
|
||||||
|
pub roles: Vec<String>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Clone, Deserialize)]
|
||||||
|
pub struct ResourceAccess {
|
||||||
|
pub roles: Vec<String>,
|
||||||
|
}
|
||||||
@@ -0,0 +1,40 @@
|
|||||||
|
use thiserror::Error;
|
||||||
|
|
||||||
|
#[derive(Debug, Error)]
|
||||||
|
pub enum AuthError {
|
||||||
|
#[error("invalid authorization header")]
|
||||||
|
InvalidHeader,
|
||||||
|
|
||||||
|
#[error("missing 'kid' in token header")]
|
||||||
|
MissingKid,
|
||||||
|
|
||||||
|
#[error("invalid JWKS structure")]
|
||||||
|
InvalidJwks,
|
||||||
|
|
||||||
|
#[error("no matching key found for kid")]
|
||||||
|
JwkNotFound,
|
||||||
|
|
||||||
|
#[error("missing RSA modulus")]
|
||||||
|
MissingModulus,
|
||||||
|
|
||||||
|
#[error("missing RSA exponent")]
|
||||||
|
MissingExponent,
|
||||||
|
|
||||||
|
#[error("invalid decoding key")]
|
||||||
|
InvalidDecodingKey,
|
||||||
|
|
||||||
|
#[error("token validation failed")]
|
||||||
|
TokenValidationFailed,
|
||||||
|
|
||||||
|
#[error("invalid or expired token")]
|
||||||
|
InvalidToken,
|
||||||
|
|
||||||
|
#[error("failed to fetch JWKS")]
|
||||||
|
JwksFetchFailed,
|
||||||
|
|
||||||
|
#[error("failed to refresh JWKS")]
|
||||||
|
JwksRefreshFailed,
|
||||||
|
|
||||||
|
#[error(transparent)]
|
||||||
|
Reqwest(#[from] reqwest::Error),
|
||||||
|
}
|
||||||
@@ -0,0 +1,59 @@
|
|||||||
|
use super::errors::AuthError;
|
||||||
|
|
||||||
|
use once_cell::sync::Lazy;
|
||||||
|
use serde_json::Value;
|
||||||
|
use std::env;
|
||||||
|
use std::sync::Arc;
|
||||||
|
use std::time::{Duration, Instant};
|
||||||
|
use tokio::sync::RwLock;
|
||||||
|
|
||||||
|
#[derive(Clone)]
|
||||||
|
struct JwksCache {
|
||||||
|
jwks: Value,
|
||||||
|
last_fetched: Instant,
|
||||||
|
}
|
||||||
|
|
||||||
|
static JWK_CACHE: Lazy<Arc<RwLock<Option<JwksCache>>>> = Lazy::new(|| Arc::new(RwLock::new(None)));
|
||||||
|
|
||||||
|
static JWKS_URL: Lazy<String> = Lazy::new(|| env::var("JWKS_URL").expect("JWKS_URL not set"));
|
||||||
|
|
||||||
|
async fn fetch_jwks() -> Result<Value, AuthError> {
|
||||||
|
let jwks = reqwest::get(JWKS_URL.as_str())
|
||||||
|
.await?
|
||||||
|
.json::<Value>()
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
Ok(jwks)
|
||||||
|
}
|
||||||
|
|
||||||
|
pub async fn refresh_jwks() -> Result<Value, AuthError> {
|
||||||
|
let jwks = fetch_jwks().await?;
|
||||||
|
|
||||||
|
let mut write = JWK_CACHE.write().await;
|
||||||
|
|
||||||
|
*write = Some(JwksCache {
|
||||||
|
jwks: jwks.clone(),
|
||||||
|
last_fetched: Instant::now(),
|
||||||
|
});
|
||||||
|
|
||||||
|
Ok(jwks)
|
||||||
|
}
|
||||||
|
|
||||||
|
pub async fn get_jwks() -> Result<Value, AuthError> {
|
||||||
|
let ttl = Duration::from_secs(3600); // 1 hour
|
||||||
|
|
||||||
|
{
|
||||||
|
// Read lock first (fast path)
|
||||||
|
let read = JWK_CACHE.read().await;
|
||||||
|
|
||||||
|
if let Some(cache) = read
|
||||||
|
.as_ref()
|
||||||
|
.filter(|cache| cache.last_fetched.elapsed() < ttl)
|
||||||
|
{
|
||||||
|
return Ok(cache.jwks.clone());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Expired or empty → refresh
|
||||||
|
refresh_jwks().await
|
||||||
|
}
|
||||||
@@ -0,0 +1,4 @@
|
|||||||
|
pub mod claims;
|
||||||
|
pub mod errors;
|
||||||
|
pub mod jwks;
|
||||||
|
pub mod validator;
|
||||||
@@ -0,0 +1,64 @@
|
|||||||
|
use super::claims::KeycloakClaims;
|
||||||
|
use super::errors::AuthError;
|
||||||
|
|
||||||
|
use jsonwebtoken::{Algorithm, DecodingKey, Validation, decode, decode_header};
|
||||||
|
use once_cell::sync::Lazy;
|
||||||
|
use serde_json::Value;
|
||||||
|
use std::env;
|
||||||
|
|
||||||
|
static ISSUER: Lazy<String> = Lazy::new(|| env::var("ISSUER").expect("ISSUER not set"));
|
||||||
|
|
||||||
|
fn validate_token(token: &str, jwks: &Value) -> Result<KeycloakClaims, AuthError> {
|
||||||
|
// 1. Decode header
|
||||||
|
let header = decode_header(token).map_err(|_| AuthError::InvalidHeader)?;
|
||||||
|
|
||||||
|
let kid = header.kid.ok_or(AuthError::MissingKid)?;
|
||||||
|
|
||||||
|
// 2. Find matching key
|
||||||
|
let keys = jwks["keys"].as_array().ok_or(AuthError::InvalidJwks)?;
|
||||||
|
|
||||||
|
let key = keys
|
||||||
|
.iter()
|
||||||
|
.find(|k| k["kid"] == kid)
|
||||||
|
.ok_or(AuthError::InvalidDecodingKey)?;
|
||||||
|
|
||||||
|
// 3. Extract RSA components
|
||||||
|
let n = key["n"].as_str().ok_or(AuthError::MissingModulus)?;
|
||||||
|
let e = key["e"].as_str().ok_or(AuthError::MissingExponent)?;
|
||||||
|
|
||||||
|
let decoding_key =
|
||||||
|
DecodingKey::from_rsa_components(n, e).map_err(|_| AuthError::JwkNotFound)?;
|
||||||
|
|
||||||
|
// 4. Setup validation rules
|
||||||
|
let mut validation = Validation::new(Algorithm::RS256);
|
||||||
|
|
||||||
|
validation.set_issuer(&[ISSUER.as_str()]);
|
||||||
|
|
||||||
|
validation.validate_exp = true;
|
||||||
|
validation.validate_aud = false;
|
||||||
|
|
||||||
|
// 5. Decode & verify
|
||||||
|
let token_data = decode::<KeycloakClaims>(token, &decoding_key, &validation)
|
||||||
|
.map_err(|_| AuthError::TokenValidationFailed)?;
|
||||||
|
|
||||||
|
Ok(token_data.claims)
|
||||||
|
}
|
||||||
|
|
||||||
|
pub async fn authenticate_jwt(token: &str) -> Result<KeycloakClaims, AuthError> {
|
||||||
|
let jwks = super::jwks::get_jwks()
|
||||||
|
.await
|
||||||
|
.map_err(|_| AuthError::JwksFetchFailed)?;
|
||||||
|
|
||||||
|
match validate_token(token, &jwks) {
|
||||||
|
Ok(claims) => Ok(claims),
|
||||||
|
|
||||||
|
Err(_) => {
|
||||||
|
// one retry with refresh
|
||||||
|
let fresh = super::jwks::refresh_jwks()
|
||||||
|
.await
|
||||||
|
.map_err(|_| AuthError::JwksRefreshFailed)?;
|
||||||
|
|
||||||
|
validate_token(token, &fresh).map_err(|_| AuthError::InvalidToken)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -1 +1,2 @@
|
|||||||
|
pub mod keycloak;
|
||||||
pub mod ollama;
|
pub mod ollama;
|
||||||
|
|||||||
+140
-256
@@ -1,10 +1,8 @@
|
|||||||
use super::errors::OllamaError;
|
use crate::providers::ollama;
|
||||||
use crate::dto::{api, ollama};
|
use crate::providers::ollama::errors::LlmError;
|
||||||
use axum::response::sse::Event;
|
|
||||||
use futures::StreamExt;
|
use futures::StreamExt;
|
||||||
use reqwest::Client;
|
use reqwest::Client;
|
||||||
use serde_json::json;
|
|
||||||
use tokio_stream::wrappers::ReceiverStream;
|
|
||||||
|
|
||||||
#[derive(Clone)]
|
#[derive(Clone)]
|
||||||
pub struct OllamaProvider {
|
pub struct OllamaProvider {
|
||||||
@@ -22,7 +20,7 @@ impl OllamaProvider {
|
|||||||
|
|
||||||
// ── private helpers ──────────────────────────────────────────────────────
|
// ── private helpers ──────────────────────────────────────────────────────
|
||||||
|
|
||||||
async fn model_exists(&self, model: &str) -> Result<bool, OllamaError> {
|
async fn model_exists(&self, model: &str) -> Result<bool, LlmError> {
|
||||||
let url = format!("{}/api/tags", self.base_url);
|
let url = format!("{}/api/tags", self.base_url);
|
||||||
|
|
||||||
let res = self
|
let res = self
|
||||||
@@ -30,61 +28,43 @@ impl OllamaProvider {
|
|||||||
.get(url)
|
.get(url)
|
||||||
.send()
|
.send()
|
||||||
.await?
|
.await?
|
||||||
.json::<ollama::OllamaModels>()
|
.json::<ollama::types::OllamaModels>()
|
||||||
.await?;
|
.await?;
|
||||||
|
|
||||||
Ok(res.models.iter().any(|m| m.name == model))
|
Ok(res.models.iter().any(|m| m.name == model))
|
||||||
}
|
}
|
||||||
|
|
||||||
fn has_user_message(&self, messages: &[api::Message]) -> bool {
|
fn has_user_message(&self, messages: &[ollama::types::OllamaMessage]) -> bool {
|
||||||
messages.iter().any(|m| matches!(m.role, api::Role::User))
|
messages
|
||||||
|
.iter()
|
||||||
|
.any(|m| matches!(m.role, ollama::types::OllamaRole::User))
|
||||||
}
|
}
|
||||||
|
|
||||||
fn extract_completion_params<'a>(
|
// pub fn validate_keep_alive(&self, s: &str) -> Result<(), LlmError> {
|
||||||
&self,
|
// let s = s.trim();
|
||||||
body: &'a api::CompletionRequest,
|
|
||||||
) -> Result<(&'a str, &'a str), OllamaError> {
|
|
||||||
let prompt = body.prompt.trim();
|
|
||||||
|
|
||||||
let model = body.base.model.as_str();
|
// if s == "-1" || s.parse::<u64>().is_ok() {
|
||||||
|
// return Ok(());
|
||||||
|
// }
|
||||||
|
|
||||||
Ok((prompt, model))
|
// let split = s
|
||||||
}
|
// .find(|c: char| c.is_alphabetic())
|
||||||
|
// .ok_or_else(|| LlmError::InvalidKeepAlive(s.to_string()))?;
|
||||||
|
|
||||||
fn extract_chat_params<'a>(
|
// let (num, unit) = s.split_at(split);
|
||||||
&self,
|
|
||||||
body: &'a api::ChatRequest,
|
|
||||||
) -> Result<(&'a [api::Message], &'a str), OllamaError> {
|
|
||||||
let model = body.base.model.as_str();
|
|
||||||
|
|
||||||
Ok((&body.messages, model))
|
// num.parse::<u64>()
|
||||||
}
|
// .map_err(|_| LlmError::InvalidKeepAlive(s.to_string()))?;
|
||||||
|
|
||||||
pub fn parse_keep_alive(&self, s: &str) -> Result<(), OllamaError> {
|
// match unit {
|
||||||
let s = s.trim();
|
// "s" | "m" | "h" => Ok(()),
|
||||||
|
// _ => Err(LlmError::InvalidKeepAlive(s.to_string())),
|
||||||
if s == "-1" || s.parse::<u64>().is_ok() {
|
// }
|
||||||
return Ok(());
|
// }
|
||||||
}
|
|
||||||
|
|
||||||
let split = s
|
|
||||||
.find(|c: char| c.is_alphabetic())
|
|
||||||
.ok_or_else(|| OllamaError::InvalidKeepAlive(s.to_string()))?;
|
|
||||||
|
|
||||||
let (num, unit) = s.split_at(split);
|
|
||||||
|
|
||||||
num.parse::<u64>()
|
|
||||||
.map_err(|_| OllamaError::InvalidKeepAlive(s.to_string()))?;
|
|
||||||
|
|
||||||
match unit {
|
|
||||||
"s" | "m" | "h" => Ok(()),
|
|
||||||
_ => Err(OllamaError::InvalidKeepAlive(s.to_string())),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// // ── public endpoints ─────────────────────────────────────────────────────
|
// // ── public endpoints ─────────────────────────────────────────────────────
|
||||||
|
|
||||||
pub async fn list_models(&self) -> Result<api::ModelsResponse, OllamaError> {
|
pub async fn list_models(&self) -> Result<super::types::OllamaModels, LlmError> {
|
||||||
let url = format!("{}/api/tags", self.base_url);
|
let url = format!("{}/api/tags", self.base_url);
|
||||||
|
|
||||||
let res = self
|
let res = self
|
||||||
@@ -92,156 +72,83 @@ impl OllamaProvider {
|
|||||||
.get(url)
|
.get(url)
|
||||||
.send()
|
.send()
|
||||||
.await?
|
.await?
|
||||||
.json::<ollama::OllamaModels>()
|
.json::<ollama::types::OllamaModels>()
|
||||||
.await?;
|
.await?;
|
||||||
|
|
||||||
let models = res.models.into_iter().map(api::ModelInfo::from).collect();
|
Ok(res)
|
||||||
|
|
||||||
Ok(api::ModelsResponse { models })
|
|
||||||
}
|
|
||||||
|
|
||||||
pub async fn load_model(
|
|
||||||
&self,
|
|
||||||
model: &str,
|
|
||||||
keep_alive: Option<&str>,
|
|
||||||
) -> Result<api::LoadModelResponse, OllamaError> {
|
|
||||||
let url = format!("{}/api/generate", self.base_url);
|
|
||||||
|
|
||||||
let keep_alive = keep_alive.ok_or(OllamaError::MissingKeepAlive)?;
|
|
||||||
|
|
||||||
self.parse_keep_alive(keep_alive)?;
|
|
||||||
|
|
||||||
let exists = self.model_exists(model).await?;
|
|
||||||
if !exists {
|
|
||||||
return Err(OllamaError::ModelNotFound(model.to_string()));
|
|
||||||
}
|
|
||||||
|
|
||||||
let payload = json!({
|
|
||||||
"model": model,
|
|
||||||
"prompt": "",
|
|
||||||
"keep_alive": keep_alive,
|
|
||||||
"stream": false,
|
|
||||||
});
|
|
||||||
|
|
||||||
let _res = self
|
|
||||||
.client
|
|
||||||
.post(url)
|
|
||||||
.json(&payload)
|
|
||||||
.send()
|
|
||||||
.await?
|
|
||||||
.text()
|
|
||||||
.await?;
|
|
||||||
|
|
||||||
Ok(api::LoadModelResponse {
|
|
||||||
model: model.to_string(),
|
|
||||||
status: "loaded".to_string(),
|
|
||||||
keep_alive: keep_alive.to_string(),
|
|
||||||
})
|
|
||||||
}
|
|
||||||
|
|
||||||
pub async fn unload_model(&self, model: &str) -> Result<api::UnloadModelResponse, OllamaError> {
|
|
||||||
let url = format!("{}/api/generate", self.base_url);
|
|
||||||
|
|
||||||
let exists = self.model_exists(model).await?;
|
|
||||||
if !exists {
|
|
||||||
return Err(OllamaError::ModelNotFound(model.to_string()));
|
|
||||||
}
|
|
||||||
|
|
||||||
let payload = json!({
|
|
||||||
"model": model,
|
|
||||||
"prompt": "",
|
|
||||||
"keep_alive": "0",
|
|
||||||
"stream": false,
|
|
||||||
});
|
|
||||||
|
|
||||||
let _res = self
|
|
||||||
.client
|
|
||||||
.post(url)
|
|
||||||
.json(&payload)
|
|
||||||
.send()
|
|
||||||
.await?
|
|
||||||
.text()
|
|
||||||
.await?;
|
|
||||||
|
|
||||||
Ok(api::UnloadModelResponse {
|
|
||||||
model: model.to_string(),
|
|
||||||
status: "unloaded".to_string(),
|
|
||||||
})
|
|
||||||
}
|
}
|
||||||
|
|
||||||
pub async fn completions(
|
pub async fn completions(
|
||||||
&self,
|
&self,
|
||||||
body: &api::CompletionRequest,
|
body: &super::types::OllamaGenerateRequest,
|
||||||
) -> Result<api::CompletionResponse, OllamaError> {
|
) -> Result<super::types::OllamaGenerateResponse, LlmError> {
|
||||||
let url = format!("{}/api/generate", self.base_url);
|
let url = format!("{}/api/generate", self.base_url);
|
||||||
|
|
||||||
let (prompt, model) = self.extract_completion_params(body)?;
|
if body.prompt.is_empty() {
|
||||||
|
return Err(LlmError::MissingPrompt);
|
||||||
if prompt.is_empty() {
|
|
||||||
return Err(OllamaError::MissingPrompt);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
if model.is_empty() {
|
if body.model.is_empty() {
|
||||||
return Err(OllamaError::MissingModel);
|
return Err(LlmError::MissingModel);
|
||||||
}
|
}
|
||||||
|
|
||||||
let exists = self.model_exists(model).await?;
|
let exists = self.model_exists(&body.model).await?;
|
||||||
if !exists {
|
if !exists {
|
||||||
return Err(OllamaError::ModelNotFound(model.to_string()));
|
return Err(LlmError::ModelNotFound(body.model.clone()));
|
||||||
}
|
}
|
||||||
|
|
||||||
let options = ollama::OllamaOptions::from(&body.base);
|
let options = body.options.clone();
|
||||||
|
|
||||||
let payload = ollama::OllamaGenerateRequest {
|
let payload = ollama::types::OllamaGenerateRequest {
|
||||||
model,
|
model: body.model.clone(),
|
||||||
prompt,
|
prompt: body.prompt.clone(),
|
||||||
stream: false,
|
stream: false,
|
||||||
|
keep_alive: body.keep_alive.clone(),
|
||||||
options,
|
options,
|
||||||
};
|
};
|
||||||
|
|
||||||
|
dbg!(&payload);
|
||||||
|
|
||||||
let res = self
|
let res = self
|
||||||
.client
|
.client
|
||||||
.post(url)
|
.post(url)
|
||||||
.json(&payload)
|
.json(&payload)
|
||||||
.send()
|
.send()
|
||||||
.await?
|
.await?
|
||||||
.json::<ollama::OllamaGenerateResponse>()
|
.json::<ollama::types::OllamaGenerateResponse>()
|
||||||
.await?;
|
.await?;
|
||||||
|
|
||||||
Ok(api::CompletionResponse::from(res))
|
Ok(res)
|
||||||
}
|
}
|
||||||
|
|
||||||
pub async fn completions_stream(
|
pub async fn completions_stream(
|
||||||
&self,
|
&self,
|
||||||
body: &api::CompletionRequest,
|
body: &super::types::OllamaGenerateRequest,
|
||||||
) -> Result<ReceiverStream<Result<Event, OllamaError>>, OllamaError> {
|
) -> Result<ollama::types::OllamaGenerateResponseStream, LlmError> {
|
||||||
let url = format!("{}/api/generate", self.base_url);
|
let url = format!("{}/api/generate", self.base_url);
|
||||||
|
|
||||||
let (prompt, model) = self.extract_completion_params(body)?;
|
if body.prompt.is_empty() {
|
||||||
|
return Err(LlmError::MissingPrompt);
|
||||||
if prompt.is_empty() {
|
|
||||||
return Err(OllamaError::MissingPrompt);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
if model.is_empty() {
|
if body.model.is_empty() {
|
||||||
return Err(OllamaError::MissingModel);
|
return Err(LlmError::MissingModel);
|
||||||
}
|
}
|
||||||
|
|
||||||
let exists = self.model_exists(model).await?;
|
let exists = self.model_exists(&body.model.to_string()).await?;
|
||||||
if !exists {
|
if !exists {
|
||||||
return Err(OllamaError::ModelNotFound(model.to_string()));
|
return Err(LlmError::ModelNotFound(body.model.to_string()));
|
||||||
}
|
}
|
||||||
|
|
||||||
let options = ollama::OllamaOptions::from(&body.base);
|
let payload = ollama::types::OllamaGenerateRequest {
|
||||||
|
model: body.model.clone(),
|
||||||
let payload = ollama::OllamaGenerateRequest {
|
prompt: body.prompt.clone(),
|
||||||
model,
|
|
||||||
prompt,
|
|
||||||
stream: true,
|
stream: true,
|
||||||
options,
|
keep_alive: body.keep_alive.clone(),
|
||||||
|
options: body.options.clone(),
|
||||||
};
|
};
|
||||||
|
|
||||||
let mut byte_stream = self
|
let byte_stream = self
|
||||||
.client
|
.client
|
||||||
.post(url)
|
.post(url)
|
||||||
.json(&payload)
|
.json(&payload)
|
||||||
@@ -249,85 +156,80 @@ impl OllamaProvider {
|
|||||||
.await?
|
.await?
|
||||||
.bytes_stream();
|
.bytes_stream();
|
||||||
|
|
||||||
let (tx, rx) = tokio::sync::mpsc::channel(32);
|
let stream = byte_stream.flat_map(|chunk_result| {
|
||||||
|
let mut out: Vec<Result<super::types::OllamaGenerateStreamEvent, LlmError>> =
|
||||||
|
Vec::new();
|
||||||
|
|
||||||
tokio::spawn(async move {
|
let chunk = match chunk_result {
|
||||||
while let Some(chunk) = byte_stream.next().await {
|
|
||||||
let chunk = match chunk {
|
|
||||||
Ok(b) => b,
|
Ok(b) => b,
|
||||||
Err(e) => {
|
Err(e) => {
|
||||||
let _ = tx.send(Err(OllamaError::Http(e))).await;
|
out.push(Err(LlmError::Http(e)));
|
||||||
break;
|
return futures::stream::iter(out);
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
// 🔥 IMPORTANT: typed deserialization
|
for line in chunk.split(|&b| b == b'\n') {
|
||||||
let parsed: ollama::OllamaGenerateResponse = match serde_json::from_slice(&chunk) {
|
if line.is_empty() {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
let parsed: ollama::types::OllamaGenerateResponse =
|
||||||
|
match serde_json::from_slice(line) {
|
||||||
Ok(v) => v,
|
Ok(v) => v,
|
||||||
Err(_) => continue,
|
Err(_) => continue,
|
||||||
};
|
};
|
||||||
|
|
||||||
// map → OpenAI chunk
|
if !parsed.response.is_empty() && !parsed.done {
|
||||||
let event_data = serde_json::to_string(&api::CompletionChunk {
|
out.push(Ok(super::types::OllamaGenerateStreamEvent::Token(
|
||||||
id: "cmpl-ollama".to_string(),
|
parsed.response.clone(),
|
||||||
object: "text_completion".to_string(),
|
)));
|
||||||
choices: vec![api::Choice {
|
}
|
||||||
text: parsed.response,
|
|
||||||
index: 0,
|
|
||||||
finish_reason: if parsed.done {
|
|
||||||
api::FinishReason::Stop
|
|
||||||
} else {
|
|
||||||
api::FinishReason::Length
|
|
||||||
},
|
|
||||||
}],
|
|
||||||
})
|
|
||||||
.unwrap_or_default();
|
|
||||||
|
|
||||||
let _ = tx.send(Ok(Event::default().data(event_data))).await;
|
|
||||||
|
|
||||||
if parsed.done {
|
if parsed.done {
|
||||||
let _ = tx.send(Ok(Event::default().data("[DONE]"))).await;
|
out.push(Ok(super::types::OllamaGenerateStreamEvent::Final(parsed)));
|
||||||
break;
|
return futures::stream::iter(out);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
futures::stream::iter(out)
|
||||||
});
|
});
|
||||||
|
|
||||||
Ok(ReceiverStream::new(rx))
|
Ok(Box::pin(stream))
|
||||||
}
|
}
|
||||||
|
|
||||||
pub async fn chat_completions(
|
pub async fn chat_completions(
|
||||||
&self,
|
&self,
|
||||||
body: &api::ChatRequest,
|
body: &super::types::OllamaChatRequest,
|
||||||
) -> Result<api::ChatCompletionResponse, OllamaError> {
|
) -> Result<super::types::OllamaChatResponse, LlmError> {
|
||||||
let url = format!("{}/api/chat", self.base_url);
|
let url = format!("{}/api/chat", self.base_url);
|
||||||
|
|
||||||
let (messages, model) = self.extract_chat_params(body)?;
|
|
||||||
|
|
||||||
if body.messages.is_empty() {
|
if body.messages.is_empty() {
|
||||||
return Err(OllamaError::MissingMessages);
|
return Err(LlmError::MissingMessages);
|
||||||
}
|
}
|
||||||
|
|
||||||
if !self.has_user_message(&body.messages) {
|
let ollama_messages: Vec<ollama::types::OllamaMessage> = body.messages.clone();
|
||||||
return Err(OllamaError::MissingMessages);
|
|
||||||
|
if !self.has_user_message(&ollama_messages) {
|
||||||
|
return Err(LlmError::MissingMessages);
|
||||||
}
|
}
|
||||||
|
|
||||||
if model.is_empty() {
|
if body.model.is_empty() {
|
||||||
return Err(OllamaError::MissingModel);
|
return Err(LlmError::MissingModel);
|
||||||
}
|
}
|
||||||
|
|
||||||
let exists = self.model_exists(model).await?;
|
let exists = self.model_exists(&body.model).await?;
|
||||||
if !exists {
|
if !exists {
|
||||||
return Err(OllamaError::ModelNotFound(model.to_string()));
|
return Err(LlmError::ModelNotFound(body.model.clone()));
|
||||||
}
|
}
|
||||||
|
|
||||||
// let options = ollama::OllamaOptions::from(body);
|
let options = body.options.clone();
|
||||||
let options = ollama::OllamaOptions::from(&body.base);
|
|
||||||
|
|
||||||
let payload = ollama::OllamaChatRequest {
|
let payload = ollama::types::OllamaChatRequest {
|
||||||
model,
|
model: body.model.clone(),
|
||||||
messages,
|
messages: ollama_messages,
|
||||||
stream: false,
|
stream: false,
|
||||||
options,
|
options,
|
||||||
|
keep_alive: body.keep_alive.clone(),
|
||||||
};
|
};
|
||||||
|
|
||||||
let res = self
|
let res = self
|
||||||
@@ -336,43 +238,46 @@ impl OllamaProvider {
|
|||||||
.json(&payload)
|
.json(&payload)
|
||||||
.send()
|
.send()
|
||||||
.await?
|
.await?
|
||||||
.json::<ollama::OllamaChatResponse>()
|
.json::<ollama::types::OllamaChatResponse>()
|
||||||
.await?;
|
.await?;
|
||||||
|
|
||||||
Ok(api::ChatCompletionResponse::from(res))
|
Ok(res)
|
||||||
}
|
}
|
||||||
|
|
||||||
pub async fn chat_completions_stream(
|
pub async fn chat_completions_stream(
|
||||||
&self,
|
&self,
|
||||||
body: &api::ChatRequest,
|
body: &super::types::OllamaChatRequest,
|
||||||
) -> Result<ReceiverStream<Result<api::ChatCompletionChunk, OllamaError>>, OllamaError> {
|
) -> Result<ollama::types::OllamaChatResponseStream, LlmError> {
|
||||||
let url = format!("{}/api/chat", self.base_url);
|
let url = format!("{}/api/chat", self.base_url);
|
||||||
|
|
||||||
let (messages, model) = self.extract_chat_params(body)?;
|
if body.messages.is_empty() {
|
||||||
|
return Err(LlmError::MissingMessages);
|
||||||
if messages.is_empty() {
|
|
||||||
return Err(OllamaError::MissingMessages);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
if model.is_empty() {
|
if body.model.is_empty() {
|
||||||
return Err(OllamaError::MissingModel);
|
return Err(LlmError::MissingModel);
|
||||||
}
|
}
|
||||||
|
|
||||||
let exists = self.model_exists(model).await?;
|
let exists = self.model_exists(&body.model).await?;
|
||||||
if !exists {
|
if !exists {
|
||||||
return Err(OllamaError::ModelNotFound(model.to_string()));
|
return Err(LlmError::ModelNotFound(body.model.clone()));
|
||||||
}
|
}
|
||||||
|
|
||||||
let options = ollama::OllamaOptions::from(&body.base);
|
let ollama_messages: Vec<ollama::types::OllamaMessage> = body.messages.clone();
|
||||||
|
|
||||||
let payload = ollama::OllamaChatRequest {
|
if !self.has_user_message(&ollama_messages) {
|
||||||
model,
|
return Err(LlmError::MissingMessages);
|
||||||
messages,
|
}
|
||||||
|
|
||||||
|
let payload = ollama::types::OllamaChatRequest {
|
||||||
|
model: body.model.clone(),
|
||||||
|
messages: body.messages.clone().into_iter().collect(),
|
||||||
stream: true,
|
stream: true,
|
||||||
options,
|
options: body.options.clone(),
|
||||||
|
keep_alive: body.keep_alive.clone(),
|
||||||
};
|
};
|
||||||
|
|
||||||
let mut byte_stream = self
|
let byte_stream = self
|
||||||
.client
|
.client
|
||||||
.post(url)
|
.post(url)
|
||||||
.json(&payload)
|
.json(&payload)
|
||||||
@@ -380,63 +285,42 @@ impl OllamaProvider {
|
|||||||
.await?
|
.await?
|
||||||
.bytes_stream();
|
.bytes_stream();
|
||||||
|
|
||||||
let (tx, rx) = tokio::sync::mpsc::channel(32);
|
let stream = byte_stream.flat_map(|chunk_result| {
|
||||||
|
let mut out: Vec<Result<super::types::OllamaChatStreamEvent, LlmError>> = Vec::new();
|
||||||
|
|
||||||
tokio::spawn(async move {
|
let chunk = match chunk_result {
|
||||||
let stream_id = format!("chatcmpl-{}", uuid::Uuid::new_v4());
|
|
||||||
|
|
||||||
while let Some(chunk) = byte_stream.next().await {
|
|
||||||
let chunk = match chunk {
|
|
||||||
Ok(b) => b,
|
Ok(b) => b,
|
||||||
Err(e) => {
|
Err(e) => {
|
||||||
let _ = tx.send(Err(OllamaError::Http(e))).await;
|
out.push(Err(LlmError::Http(e)));
|
||||||
break;
|
return futures::stream::iter(out);
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
let parsed: ollama::OllamaChatResponse = match serde_json::from_slice(&chunk) {
|
for line in chunk.split(|&b| b == b'\n') {
|
||||||
|
if line.is_empty() {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
let parsed: ollama::types::OllamaChatResponse = match serde_json::from_slice(line) {
|
||||||
Ok(v) => v,
|
Ok(v) => v,
|
||||||
Err(_) => continue,
|
Err(_) => continue,
|
||||||
};
|
};
|
||||||
|
|
||||||
let usage = if parsed.done {
|
if !parsed.message.content.is_empty() && !parsed.done {
|
||||||
Some(api::Usage {
|
out.push(Ok(super::types::OllamaChatStreamEvent::Token(
|
||||||
prompt_tokens: parsed.prompt_eval_count.unwrap_or(0) as u32,
|
parsed.message.content.clone(),
|
||||||
completion_tokens: parsed.eval_count.unwrap_or(0) as u32,
|
)));
|
||||||
total_tokens: (parsed.prompt_eval_count.unwrap_or(0)
|
}
|
||||||
+ parsed.eval_count.unwrap_or(0))
|
|
||||||
as u32,
|
|
||||||
})
|
|
||||||
} else {
|
|
||||||
None
|
|
||||||
};
|
|
||||||
|
|
||||||
let event = api::ChatCompletionChunk {
|
|
||||||
id: stream_id.clone(),
|
|
||||||
object: "chat.completion.chunk".to_string(),
|
|
||||||
choices: vec![api::ChatChunkChoice {
|
|
||||||
index: 0,
|
|
||||||
delta: api::Delta {
|
|
||||||
role: Some(parsed.message.role),
|
|
||||||
content: Some(parsed.message.content),
|
|
||||||
},
|
|
||||||
finish_reason: if parsed.done {
|
|
||||||
Some(api::FinishReason::Stop)
|
|
||||||
} else {
|
|
||||||
None
|
|
||||||
},
|
|
||||||
}],
|
|
||||||
usage,
|
|
||||||
};
|
|
||||||
|
|
||||||
let _ = tx.send(Ok(event)).await;
|
|
||||||
|
|
||||||
if parsed.done {
|
if parsed.done {
|
||||||
break;
|
out.push(Ok(super::types::OllamaChatStreamEvent::Final(parsed)));
|
||||||
|
return futures::stream::iter(out);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
futures::stream::iter(out)
|
||||||
});
|
});
|
||||||
|
|
||||||
Ok(ReceiverStream::new(rx))
|
Ok(Box::pin(stream))
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,8 +1,7 @@
|
|||||||
// errors.rs
|
|
||||||
use thiserror::Error;
|
use thiserror::Error;
|
||||||
|
|
||||||
#[derive(Debug, Error)]
|
#[derive(Debug, Error)]
|
||||||
pub enum OllamaError {
|
pub enum LlmError {
|
||||||
#[error("prompt is required and cannot be empty")]
|
#[error("prompt is required and cannot be empty")]
|
||||||
MissingPrompt,
|
MissingPrompt,
|
||||||
|
|
||||||
@@ -15,44 +14,10 @@ pub enum OllamaError {
|
|||||||
#[error("model '{0}' is not available — run `ollama pull {0}` first")]
|
#[error("model '{0}' is not available — run `ollama pull {0}` first")]
|
||||||
ModelNotFound(String),
|
ModelNotFound(String),
|
||||||
|
|
||||||
#[error("keep_alive is required and cannot be empty")]
|
// #[error(
|
||||||
MissingKeepAlive,
|
// "invalid keep_alive format '{0}' — expected <number><unit> (e.g. 30s, 10m, 2h), a plain integer (seconds), or -1"
|
||||||
|
// )]
|
||||||
#[error(
|
// InvalidKeepAlive(String),
|
||||||
"invalid keep_alive format '{0}' — expected <number><unit> (e.g. 30s, 10m, 2h), a plain integer (seconds), or -1"
|
|
||||||
)]
|
|
||||||
InvalidKeepAlive(String),
|
|
||||||
|
|
||||||
#[error(transparent)]
|
#[error(transparent)]
|
||||||
Http(#[from] reqwest::Error),
|
Http(#[from] reqwest::Error),
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn into_http_response(e: OllamaError) -> (axum::http::StatusCode, String) {
|
|
||||||
match e {
|
|
||||||
OllamaError::MissingPrompt => (
|
|
||||||
axum::http::StatusCode::BAD_REQUEST,
|
|
||||||
"prompt is required and cannot be empty".to_string(),
|
|
||||||
),
|
|
||||||
OllamaError::MissingModel => (
|
|
||||||
axum::http::StatusCode::BAD_REQUEST,
|
|
||||||
"model is required and cannot be empty".to_string(),
|
|
||||||
),
|
|
||||||
OllamaError::ModelNotFound(m) => (
|
|
||||||
axum::http::StatusCode::UNPROCESSABLE_ENTITY,
|
|
||||||
format!("model '{m}' is not available — run `ollama pull {m}` first"),
|
|
||||||
),
|
|
||||||
OllamaError::MissingKeepAlive => (
|
|
||||||
axum::http::StatusCode::BAD_REQUEST,
|
|
||||||
"keep alive is required and cannot be empty".to_string(),
|
|
||||||
),
|
|
||||||
OllamaError::InvalidKeepAlive(v) => (
|
|
||||||
axum::http::StatusCode::BAD_REQUEST,
|
|
||||||
format!("invalid keep_alive '{v}'"),
|
|
||||||
),
|
|
||||||
OllamaError::MissingMessages => (
|
|
||||||
axum::http::StatusCode::BAD_REQUEST,
|
|
||||||
"messages array with at least one user message is required".to_string(),
|
|
||||||
),
|
|
||||||
OllamaError::Http(e) => (axum::http::StatusCode::INTERNAL_SERVER_ERROR, e.to_string()),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|||||||
@@ -1,88 +0,0 @@
|
|||||||
use crate::dto::{api, ollama};
|
|
||||||
|
|
||||||
use chrono::Utc;
|
|
||||||
use uuid::Uuid;
|
|
||||||
|
|
||||||
impl From<ollama::OllamaModel> for api::ModelInfo {
|
|
||||||
fn from(m: ollama::OllamaModel) -> Self {
|
|
||||||
Self {
|
|
||||||
name: m.name,
|
|
||||||
|
|
||||||
family: m.details.as_ref().and_then(|d| d.family.clone()),
|
|
||||||
parameter_size: m.details.as_ref().and_then(|d| d.parameter_size.clone()),
|
|
||||||
quantization: m
|
|
||||||
.details
|
|
||||||
.as_ref()
|
|
||||||
.and_then(|d| d.quantization_level.clone()),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
impl From<ollama::OllamaGenerateResponse> for api::CompletionResponse {
|
|
||||||
fn from(res: ollama::OllamaGenerateResponse) -> Self {
|
|
||||||
Self {
|
|
||||||
id: Uuid::new_v4().to_string(),
|
|
||||||
object: api::CompletionObject::TextCompletion,
|
|
||||||
model: res.model,
|
|
||||||
created: Utc::now().timestamp() as u64,
|
|
||||||
|
|
||||||
choices: vec![api::Choice {
|
|
||||||
text: res.response,
|
|
||||||
index: 0,
|
|
||||||
finish_reason: api::FinishReason::Stop,
|
|
||||||
}],
|
|
||||||
|
|
||||||
usage: api::Usage {
|
|
||||||
prompt_tokens: res.prompt_eval_count.unwrap_or(0),
|
|
||||||
completion_tokens: res.eval_count.unwrap_or(0),
|
|
||||||
total_tokens: res.prompt_eval_count.unwrap_or(0) + res.eval_count.unwrap_or(0),
|
|
||||||
},
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
impl From<&api::BaseLLMRequest> for ollama::OllamaOptions {
|
|
||||||
fn from(base: &api::BaseLLMRequest) -> Self {
|
|
||||||
Self {
|
|
||||||
temperature: base.temperature,
|
|
||||||
top_p: base.top_p,
|
|
||||||
top_k: base.top_k,
|
|
||||||
repeat_penalty: base.repeat_penalty,
|
|
||||||
seed: base.seed,
|
|
||||||
num_ctx: base.num_ctx,
|
|
||||||
num_predict: base.num_predict,
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
impl From<ollama::OllamaChatResponse> for api::ChatCompletionResponse {
|
|
||||||
fn from(res: ollama::OllamaChatResponse) -> Self {
|
|
||||||
let prompt_tokens = res.prompt_eval_count.unwrap_or(0);
|
|
||||||
let completion_tokens = res.eval_count.unwrap_or(0);
|
|
||||||
|
|
||||||
Self {
|
|
||||||
id: Uuid::new_v4().to_string(),
|
|
||||||
object: "chat.completion".to_string(),
|
|
||||||
created: Utc::now().timestamp() as u64,
|
|
||||||
model: res.model,
|
|
||||||
|
|
||||||
choices: vec![api::ChatChoice {
|
|
||||||
index: 0,
|
|
||||||
message: res.message,
|
|
||||||
finish_reason: if res.done {
|
|
||||||
api::FinishReason::Stop
|
|
||||||
} else {
|
|
||||||
api::FinishReason::Length
|
|
||||||
},
|
|
||||||
}],
|
|
||||||
|
|
||||||
usage: Some(api::Usage {
|
|
||||||
prompt_tokens,
|
|
||||||
completion_tokens,
|
|
||||||
total_tokens: prompt_tokens + completion_tokens,
|
|
||||||
}),
|
|
||||||
|
|
||||||
conversation_id: None,
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -1,3 +1,3 @@
|
|||||||
pub mod client;
|
pub mod client;
|
||||||
pub mod errors;
|
pub mod errors;
|
||||||
pub mod mapper;
|
pub mod types;
|
||||||
|
|||||||
@@ -0,0 +1,137 @@
|
|||||||
|
use crate::providers::ollama::errors;
|
||||||
|
|
||||||
|
use futures::Stream;
|
||||||
|
use serde::{Deserialize, Serialize};
|
||||||
|
use std::pin::Pin;
|
||||||
|
|
||||||
|
// ------ Models ------
|
||||||
|
|
||||||
|
#[derive(Debug, Deserialize)]
|
||||||
|
pub struct OllamaModels {
|
||||||
|
pub models: Vec<OllamaModel>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Deserialize)]
|
||||||
|
pub struct OllamaModel {
|
||||||
|
pub name: String,
|
||||||
|
|
||||||
|
pub details: Option<OllamaModelDetails>,
|
||||||
|
|
||||||
|
pub size: Option<u64>,
|
||||||
|
pub digest: Option<String>,
|
||||||
|
pub modified_at: Option<String>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Deserialize)]
|
||||||
|
pub struct OllamaModelDetails {
|
||||||
|
pub family: Option<String>,
|
||||||
|
pub parameter_size: Option<String>,
|
||||||
|
pub quantization_level: Option<String>,
|
||||||
|
}
|
||||||
|
|
||||||
|
// ------ Message ------
|
||||||
|
|
||||||
|
#[derive(Debug, Serialize, Deserialize, Clone)]
|
||||||
|
pub struct OllamaMessage {
|
||||||
|
pub role: OllamaRole,
|
||||||
|
pub content: String,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Serialize, Deserialize, PartialEq, Eq, Clone)]
|
||||||
|
#[serde(rename_all = "lowercase")]
|
||||||
|
pub enum OllamaRole {
|
||||||
|
System,
|
||||||
|
User,
|
||||||
|
Assistant,
|
||||||
|
}
|
||||||
|
|
||||||
|
// ------ Shared ------
|
||||||
|
|
||||||
|
#[derive(Debug, Default, Clone, Serialize)]
|
||||||
|
pub struct OllamaOptions {
|
||||||
|
pub seed: Option<i64>,
|
||||||
|
pub temperature: Option<f32>,
|
||||||
|
pub top_p: Option<f32>,
|
||||||
|
pub top_k: Option<u32>,
|
||||||
|
pub stop: Option<Vec<String>>,
|
||||||
|
pub num_ctx: Option<u32>,
|
||||||
|
pub num_predict: Option<u32>,
|
||||||
|
}
|
||||||
|
|
||||||
|
// ------ Completion ------
|
||||||
|
|
||||||
|
#[derive(Debug, Serialize)]
|
||||||
|
pub struct OllamaGenerateRequest {
|
||||||
|
pub model: String,
|
||||||
|
pub prompt: String,
|
||||||
|
pub stream: bool,
|
||||||
|
pub keep_alive: String,
|
||||||
|
pub options: Option<OllamaOptions>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Deserialize)]
|
||||||
|
pub struct OllamaGenerateResponse {
|
||||||
|
pub model: String,
|
||||||
|
pub created_at: String,
|
||||||
|
|
||||||
|
pub response: String,
|
||||||
|
|
||||||
|
pub done: bool,
|
||||||
|
pub done_reason: Option<String>,
|
||||||
|
|
||||||
|
pub total_duration: Option<u64>,
|
||||||
|
pub load_duration: Option<u64>,
|
||||||
|
|
||||||
|
pub prompt_eval_count: Option<u32>,
|
||||||
|
pub eval_count: Option<u32>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug)]
|
||||||
|
pub enum OllamaGenerateStreamEvent {
|
||||||
|
Token(String),
|
||||||
|
Final(OllamaGenerateResponse),
|
||||||
|
}
|
||||||
|
|
||||||
|
pub type OllamaGenerateResponseStream = Pin<
|
||||||
|
Box<
|
||||||
|
dyn Stream<Item = Result<super::types::OllamaGenerateStreamEvent, errors::LlmError>> + Send,
|
||||||
|
>,
|
||||||
|
>;
|
||||||
|
|
||||||
|
// ------ Chat ------
|
||||||
|
|
||||||
|
#[derive(Debug, Serialize)]
|
||||||
|
pub struct OllamaChatRequest {
|
||||||
|
pub model: String,
|
||||||
|
pub messages: Vec<OllamaMessage>,
|
||||||
|
pub stream: bool,
|
||||||
|
pub keep_alive: String,
|
||||||
|
pub options: Option<OllamaOptions>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Deserialize)]
|
||||||
|
pub struct OllamaChatResponse {
|
||||||
|
pub model: String,
|
||||||
|
pub created_at: String,
|
||||||
|
|
||||||
|
pub message: OllamaMessage,
|
||||||
|
|
||||||
|
pub done: bool,
|
||||||
|
pub done_reason: Option<String>,
|
||||||
|
|
||||||
|
pub total_duration: Option<u64>,
|
||||||
|
pub load_duration: Option<u64>,
|
||||||
|
|
||||||
|
pub prompt_eval_count: Option<u32>,
|
||||||
|
pub eval_count: Option<u32>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug)]
|
||||||
|
pub enum OllamaChatStreamEvent {
|
||||||
|
Token(String),
|
||||||
|
Final(OllamaChatResponse),
|
||||||
|
}
|
||||||
|
|
||||||
|
pub type OllamaChatResponseStream = Pin<
|
||||||
|
Box<dyn Stream<Item = Result<super::types::OllamaChatStreamEvent, errors::LlmError>> + Send>,
|
||||||
|
>;
|
||||||
@@ -1,50 +0,0 @@
|
|||||||
use axum::{
|
|
||||||
Json,
|
|
||||||
extract::{Extension, State},
|
|
||||||
http::StatusCode,
|
|
||||||
};
|
|
||||||
use base64::{Engine as _, engine::general_purpose};
|
|
||||||
use rand::RngCore;
|
|
||||||
use rand::rngs::OsRng;
|
|
||||||
|
|
||||||
use crate::dto::api::{CreateApiKeyRequest, CreateApiKeyResponse};
|
|
||||||
use crate::middlewares::auth::apikey::ApiKeyClaimsRoles;
|
|
||||||
use crate::middlewares::auth::middleware::Auth;
|
|
||||||
use crate::state::app_state::AppState;
|
|
||||||
use crate::utils::crypto::hash_key;
|
|
||||||
|
|
||||||
fn generate_api_key() -> String {
|
|
||||||
let mut bytes = [0u8; 32];
|
|
||||||
OsRng.fill_bytes(&mut bytes);
|
|
||||||
general_purpose::URL_SAFE_NO_PAD.encode(bytes)
|
|
||||||
}
|
|
||||||
|
|
||||||
pub async fn create_api_key(
|
|
||||||
State(state): State<AppState>,
|
|
||||||
Extension(claims): Extension<Auth>,
|
|
||||||
Json(body): Json<CreateApiKeyRequest>,
|
|
||||||
) -> Result<Json<CreateApiKeyResponse>, StatusCode> {
|
|
||||||
if matches!(&claims, Auth::ApiKey(api_key) if !api_key.roles.contains(&ApiKeyClaimsRoles::Admin))
|
|
||||||
{
|
|
||||||
return Err(StatusCode::FORBIDDEN);
|
|
||||||
}
|
|
||||||
|
|
||||||
let raw_key = generate_api_key();
|
|
||||||
let key_hash = hash_key(&raw_key);
|
|
||||||
|
|
||||||
sqlx::query!(
|
|
||||||
r#"
|
|
||||||
INSERT INTO auth.api_key (key_hash, name, created_by, scopes)
|
|
||||||
VALUES ($1, $2, $3, $4)
|
|
||||||
"#,
|
|
||||||
key_hash,
|
|
||||||
body.name,
|
|
||||||
claims.user_id(),
|
|
||||||
&body.scopes
|
|
||||||
)
|
|
||||||
.execute(&state.postgres)
|
|
||||||
.await
|
|
||||||
.map_err(|_| StatusCode::INTERNAL_SERVER_ERROR)?;
|
|
||||||
|
|
||||||
Ok(Json(CreateApiKeyResponse { api_key: raw_key }))
|
|
||||||
}
|
|
||||||
@@ -1,504 +0,0 @@
|
|||||||
use crate::{
|
|
||||||
dto::api::BaseLLMRequest, middlewares::auth::middleware::Auth,
|
|
||||||
providers::ollama::client::OllamaProvider,
|
|
||||||
};
|
|
||||||
use axum::{
|
|
||||||
Json,
|
|
||||||
extract::{Extension, Path, Query, State},
|
|
||||||
response::{
|
|
||||||
IntoResponse, Response,
|
|
||||||
sse::{Event, KeepAlive, Sse},
|
|
||||||
},
|
|
||||||
};
|
|
||||||
use sqlx::PgPool;
|
|
||||||
use tokio_stream::StreamExt;
|
|
||||||
use uuid::Uuid;
|
|
||||||
|
|
||||||
use crate::api::errors::ApiError;
|
|
||||||
use crate::databases::postgres::chat::queries::{
|
|
||||||
get_conversation_messages, get_conversations_entries, get_or_create_conversation,
|
|
||||||
insert_message, set_conversation_title, update_message_tokens,
|
|
||||||
};
|
|
||||||
use crate::databases::postgres::chat::types::{ConversationState, MessageRole};
|
|
||||||
use crate::databases::postgres::errors;
|
|
||||||
use crate::dto::api;
|
|
||||||
use crate::providers::ollama::errors::into_http_response;
|
|
||||||
use crate::state::app_state::AppState;
|
|
||||||
|
|
||||||
#[utoipa::path(
|
|
||||||
post,
|
|
||||||
path = "/completions",
|
|
||||||
tag = "chat",
|
|
||||||
request_body(
|
|
||||||
content = api::CompletionRequest,
|
|
||||||
description = "Text completion request",
|
|
||||||
content_type = "application/json"
|
|
||||||
),
|
|
||||||
responses(
|
|
||||||
(
|
|
||||||
status = 200,
|
|
||||||
description = "Text completion response. If stream=true, response is SSE stream of chunks ending in [DONE].",
|
|
||||||
body = api::CompletionResponse,
|
|
||||||
content_type = "application/json"
|
|
||||||
),
|
|
||||||
(
|
|
||||||
status = 400,
|
|
||||||
description = "Invalid request: missing prompt, model, or invalid format",
|
|
||||||
body = api::ErrorResponse,
|
|
||||||
example = json!({ "error": "prompt is required and cannot be empty" })
|
|
||||||
),
|
|
||||||
(
|
|
||||||
status = 422,
|
|
||||||
description = "Model not found or not available locally",
|
|
||||||
body = api::ErrorResponse,
|
|
||||||
example = json!({ "error": "model 'llama3' is not available — run `ollama pull llama3` first" })
|
|
||||||
),
|
|
||||||
(
|
|
||||||
status = 500,
|
|
||||||
description = "Internal server error (Ollama or network failure)",
|
|
||||||
body = api::ErrorResponse,
|
|
||||||
example = json!({ "error": "connection refused" })
|
|
||||||
)
|
|
||||||
)
|
|
||||||
)]
|
|
||||||
pub async fn completions(
|
|
||||||
State(state): State<AppState>,
|
|
||||||
Json(body): Json<api::CompletionRequest>,
|
|
||||||
) -> Result<Response, (axum::http::StatusCode, Json<api::ErrorResponse>)> {
|
|
||||||
tracing::debug!("Received /completion with body {:?}", body);
|
|
||||||
if body.base.stream {
|
|
||||||
let stream = state.ollama.completions_stream(&body).await.map_err(|e| {
|
|
||||||
let (code, msg) = into_http_response(e);
|
|
||||||
(code, Json(api::ErrorResponse::new(msg)))
|
|
||||||
})?;
|
|
||||||
|
|
||||||
Ok(Sse::new(stream)
|
|
||||||
.keep_alive(KeepAlive::default())
|
|
||||||
.into_response())
|
|
||||||
} else {
|
|
||||||
let response = state.ollama.completions(&body).await.map_err(|e| {
|
|
||||||
let (code, msg) = into_http_response(e);
|
|
||||||
(code, Json(api::ErrorResponse::new(msg)))
|
|
||||||
})?;
|
|
||||||
|
|
||||||
Ok(Json(response).into_response())
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
async fn ensure_conversation(
|
|
||||||
state: &AppState,
|
|
||||||
auth: &Auth,
|
|
||||||
conversation_id: Option<Uuid>,
|
|
||||||
first_message: &str,
|
|
||||||
) -> Result<Uuid, ApiError> {
|
|
||||||
let conversation_state =
|
|
||||||
get_or_create_conversation(&state.postgres, conversation_id, auth.user_id()).await?;
|
|
||||||
|
|
||||||
let id = match conversation_state {
|
|
||||||
ConversationState::Existing(uuid) => uuid,
|
|
||||||
ConversationState::Created(uuid) => {
|
|
||||||
let pool = state.postgres.clone();
|
|
||||||
let ollama = state.ollama.clone();
|
|
||||||
let user_id = auth.user_id();
|
|
||||||
let first_message = first_message.to_string();
|
|
||||||
|
|
||||||
tokio::spawn(async move {
|
|
||||||
let title = generate_conversation_title(&ollama, &first_message).await;
|
|
||||||
if let Err(e) = set_conversation_title(&pool, uuid, user_id, &title).await {
|
|
||||||
tracing::warn!(
|
|
||||||
conversation_id = %uuid,
|
|
||||||
error = %e,
|
|
||||||
"Failed to set conversation title"
|
|
||||||
);
|
|
||||||
}
|
|
||||||
tracing::debug!(conversation_id = %uuid, %title, "generated conversation title");
|
|
||||||
});
|
|
||||||
|
|
||||||
uuid
|
|
||||||
}
|
|
||||||
};
|
|
||||||
|
|
||||||
Ok(id)
|
|
||||||
}
|
|
||||||
|
|
||||||
async fn handle_stream(
|
|
||||||
state: AppState,
|
|
||||||
auth: Auth,
|
|
||||||
body: api::ChatRequest,
|
|
||||||
conversation_id: Option<Uuid>,
|
|
||||||
) -> Result<Response, ApiError> {
|
|
||||||
// Handle anonymous (API key) path early — no DB logging
|
|
||||||
let Some(conv_id) = conversation_id else {
|
|
||||||
let stream = state.ollama.chat_completions_stream(&body).await?;
|
|
||||||
|
|
||||||
let plain_stream = stream.map(
|
|
||||||
|item| -> Result<Event, crate::providers::ollama::errors::OllamaError> {
|
|
||||||
match item {
|
|
||||||
Ok(chunk) => Ok(
|
|
||||||
Event::default().data(serde_json::to_string(&chunk).unwrap_or_default())
|
|
||||||
),
|
|
||||||
Err(e) => Err(e),
|
|
||||||
}
|
|
||||||
},
|
|
||||||
);
|
|
||||||
return Ok(Sse::new(plain_stream)
|
|
||||||
.keep_alive(KeepAlive::default())
|
|
||||||
.into_response());
|
|
||||||
};
|
|
||||||
|
|
||||||
// From here conv_id is a plain Uuid — all variables stay in scope
|
|
||||||
let user_msg_id = log_user_message(
|
|
||||||
&state.postgres,
|
|
||||||
auth.user_id(),
|
|
||||||
conv_id,
|
|
||||||
body.parent_id,
|
|
||||||
body.messages
|
|
||||||
.last()
|
|
||||||
.map(|m| m.content.as_str())
|
|
||||||
.unwrap_or(""),
|
|
||||||
None,
|
|
||||||
)
|
|
||||||
.await?;
|
|
||||||
|
|
||||||
let start_event = api::StreamEvent::Start(api::StartEventData {
|
|
||||||
conversation_id: conv_id,
|
|
||||||
created: chrono::Utc::now().timestamp() as u64,
|
|
||||||
id: user_msg_id,
|
|
||||||
});
|
|
||||||
|
|
||||||
let (tx, rx) = tokio::sync::mpsc::channel::<
|
|
||||||
Result<Event, crate::providers::ollama::errors::OllamaError>,
|
|
||||||
>(32);
|
|
||||||
|
|
||||||
// Send start event immediately, before Ollama is contacted
|
|
||||||
let _ = tx
|
|
||||||
.send(Ok(Event::default()
|
|
||||||
.event("metadata")
|
|
||||||
.data(serde_json::to_string(&start_event).unwrap())))
|
|
||||||
.await;
|
|
||||||
|
|
||||||
let pool = state.postgres.clone();
|
|
||||||
let user_id = auth.user_id();
|
|
||||||
|
|
||||||
tokio::spawn(async move {
|
|
||||||
// Ollama called inside spawn — start event already queued
|
|
||||||
let stream = match state.ollama.chat_completions_stream(&body).await {
|
|
||||||
Ok(s) => s,
|
|
||||||
Err(e) => {
|
|
||||||
let _ = tx.send(Err(e)).await;
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
};
|
|
||||||
|
|
||||||
let mut stream = stream;
|
|
||||||
let mut accumulated = String::new();
|
|
||||||
|
|
||||||
while let Some(item) = futures::StreamExt::next(&mut stream).await {
|
|
||||||
match item {
|
|
||||||
Ok(chunk) => {
|
|
||||||
let is_done = chunk.choices[0].finish_reason == Some(api::FinishReason::Stop);
|
|
||||||
|
|
||||||
if let Some(content) = chunk.choices[0].delta.content.as_ref() {
|
|
||||||
accumulated.push_str(content);
|
|
||||||
}
|
|
||||||
|
|
||||||
if is_done {
|
|
||||||
let prompt_tokens = chunk.usage.as_ref().map(|u| u.prompt_tokens);
|
|
||||||
let completion_tokens = chunk.usage.as_ref().map(|u| u.completion_tokens);
|
|
||||||
|
|
||||||
if let Some(pt) = prompt_tokens {
|
|
||||||
let _ = update_message_tokens(&pool, user_id, user_msg_id, pt).await;
|
|
||||||
}
|
|
||||||
|
|
||||||
let assistant_msg_id = log_assistant_message(
|
|
||||||
&pool,
|
|
||||||
user_id,
|
|
||||||
conv_id,
|
|
||||||
user_msg_id,
|
|
||||||
&accumulated,
|
|
||||||
completion_tokens,
|
|
||||||
)
|
|
||||||
.await;
|
|
||||||
|
|
||||||
if let Ok(msg_id) = assistant_msg_id {
|
|
||||||
let end_event = api::StreamEvent::End(api::EndEventData {
|
|
||||||
usage: api::Usage {
|
|
||||||
prompt_tokens: prompt_tokens.unwrap_or(0),
|
|
||||||
completion_tokens: completion_tokens.unwrap_or(0),
|
|
||||||
total_tokens: chunk
|
|
||||||
.usage
|
|
||||||
.as_ref()
|
|
||||||
.map(|u| u.total_tokens)
|
|
||||||
.unwrap_or(0),
|
|
||||||
},
|
|
||||||
id: msg_id,
|
|
||||||
created: chrono::Utc::now().timestamp() as u64,
|
|
||||||
});
|
|
||||||
|
|
||||||
let _ = tx
|
|
||||||
.send(Ok(Event::default()
|
|
||||||
.event("metadata")
|
|
||||||
.data(serde_json::to_string(&end_event).unwrap())))
|
|
||||||
.await;
|
|
||||||
}
|
|
||||||
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
|
|
||||||
let data = api::StreamEvent::Delta(chunk);
|
|
||||||
let json = serde_json::to_string(&data).unwrap();
|
|
||||||
if tx.send(Ok(Event::default().data(json))).await.is_err() {
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
Err(e) => {
|
|
||||||
let _ = tx.send(Err(e)).await;
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
});
|
|
||||||
|
|
||||||
Ok(Sse::new(tokio_stream::wrappers::ReceiverStream::new(rx))
|
|
||||||
.keep_alive(KeepAlive::default())
|
|
||||||
.into_response())
|
|
||||||
}
|
|
||||||
|
|
||||||
async fn handle_non_stream(
|
|
||||||
state: AppState,
|
|
||||||
auth: Auth,
|
|
||||||
body: api::ChatRequest,
|
|
||||||
conversation_id: Option<Uuid>,
|
|
||||||
) -> Result<Response, ApiError> {
|
|
||||||
let mut response = state.ollama.chat_completions(&body).await?;
|
|
||||||
|
|
||||||
response.conversation_id = conversation_id;
|
|
||||||
|
|
||||||
if let Some(conversation_id) = conversation_id {
|
|
||||||
let user_msg_id = log_user_message(
|
|
||||||
&state.postgres,
|
|
||||||
auth.user_id(),
|
|
||||||
conversation_id,
|
|
||||||
body.parent_id,
|
|
||||||
body.messages
|
|
||||||
.last()
|
|
||||||
.map(|m| m.content.as_str())
|
|
||||||
.unwrap_or(""),
|
|
||||||
response.usage.map(|u| u.prompt_tokens),
|
|
||||||
)
|
|
||||||
.await?;
|
|
||||||
|
|
||||||
log_assistant_message(
|
|
||||||
&state.postgres,
|
|
||||||
auth.user_id(),
|
|
||||||
conversation_id,
|
|
||||||
user_msg_id,
|
|
||||||
&response.choices[0].message.content,
|
|
||||||
response.usage.map(|u| u.completion_tokens),
|
|
||||||
)
|
|
||||||
.await?;
|
|
||||||
}
|
|
||||||
|
|
||||||
Ok(Json(response).into_response())
|
|
||||||
}
|
|
||||||
|
|
||||||
#[utoipa::path(
|
|
||||||
post,
|
|
||||||
path = "/chat/completions",
|
|
||||||
tag = "chat",
|
|
||||||
request_body(
|
|
||||||
content = api::ChatRequest,
|
|
||||||
description = "Chat completion request with message history",
|
|
||||||
content_type = "application/json"
|
|
||||||
),
|
|
||||||
responses(
|
|
||||||
(
|
|
||||||
status = 200,
|
|
||||||
description = "Chat completion response. If stream=false returns JSON. If stream=true returns SSE stream of chunks ending with [DONE].",
|
|
||||||
body = api::ChatCompletionResponse,
|
|
||||||
content_type = "application/json"
|
|
||||||
),
|
|
||||||
(
|
|
||||||
status = 400,
|
|
||||||
description = "Invalid request",
|
|
||||||
body = api::ErrorResponse,
|
|
||||||
example = json!({ "error": "messages array with at least one user message is required" })
|
|
||||||
),
|
|
||||||
(
|
|
||||||
status = 401,
|
|
||||||
description = "Unauthorized",
|
|
||||||
body = api::ErrorResponse,
|
|
||||||
example = json!({ "error": "missing or invalid token" })
|
|
||||||
),
|
|
||||||
(
|
|
||||||
status = 422,
|
|
||||||
description = "Model not found or unavailable",
|
|
||||||
body = api::ErrorResponse,
|
|
||||||
example = json!({ "error": "model 'llama3' is not available — run `ollama pull llama3` first" })
|
|
||||||
),
|
|
||||||
(
|
|
||||||
status = 500,
|
|
||||||
description = "Internal server error",
|
|
||||||
body = api::ErrorResponse,
|
|
||||||
example = json!({ "error": "connection refused" })
|
|
||||||
)
|
|
||||||
)
|
|
||||||
)]
|
|
||||||
pub async fn chat_completions(
|
|
||||||
State(state): State<AppState>,
|
|
||||||
Extension(auth): Extension<Auth>,
|
|
||||||
Json(mut body): Json<api::ChatRequest>,
|
|
||||||
) -> Result<Response, ApiError> {
|
|
||||||
tracing::debug!("Received /chat/completion with body {:?}", body);
|
|
||||||
|
|
||||||
let conversation_id = if matches!(&auth, Auth::Jwt(_)) {
|
|
||||||
let first_message = body.messages[0].content.clone();
|
|
||||||
let conv_id =
|
|
||||||
ensure_conversation(&state, &auth, body.conversation_id, &first_message).await?;
|
|
||||||
|
|
||||||
if let Some(depth) = body.base.context_depth {
|
|
||||||
dbg!({ depth });
|
|
||||||
if depth > 0 {
|
|
||||||
let history = get_conversation_messages(
|
|
||||||
&state.postgres,
|
|
||||||
auth.user_id(),
|
|
||||||
conv_id,
|
|
||||||
depth as i64,
|
|
||||||
None, // no cursor — fetch the most recent N messages
|
|
||||||
)
|
|
||||||
.await?;
|
|
||||||
|
|
||||||
// Map MessageSummary → api::Message and prepend to the outgoing request
|
|
||||||
let history_messages: Vec<api::Message> = history
|
|
||||||
.into_iter()
|
|
||||||
.map(|m| api::Message {
|
|
||||||
role: api::Role::Assistant, // TODO
|
|
||||||
content: m.content,
|
|
||||||
})
|
|
||||||
.collect();
|
|
||||||
|
|
||||||
// body.messages = [history_messages, body.messages].concat();
|
|
||||||
body.messages.splice(0..0, history_messages);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
dbg!("{:?}", &body.messages);
|
|
||||||
|
|
||||||
Some(conv_id)
|
|
||||||
} else {
|
|
||||||
None
|
|
||||||
};
|
|
||||||
|
|
||||||
if body.base.stream {
|
|
||||||
handle_stream(state, auth, body, conversation_id).await
|
|
||||||
} else {
|
|
||||||
handle_non_stream(state, auth, body, conversation_id).await
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
async fn generate_conversation_title(ollama: &OllamaProvider, first_message: &str) -> String {
|
|
||||||
let request = api::CompletionRequest {
|
|
||||||
base: BaseLLMRequest {
|
|
||||||
model: "llama3:latest".to_string(),
|
|
||||||
..Default::default()
|
|
||||||
},
|
|
||||||
prompt: format!(
|
|
||||||
"Generate a short title (max 6 words) for the following chat conversation: {}",
|
|
||||||
first_message
|
|
||||||
),
|
|
||||||
};
|
|
||||||
|
|
||||||
ollama
|
|
||||||
.completions(&request)
|
|
||||||
.await
|
|
||||||
.ok()
|
|
||||||
.and_then(|r| r.choices.first().map(|c| c.text.trim().to_string()))
|
|
||||||
.filter(|t| !t.is_empty())
|
|
||||||
.unwrap_or_else(|| "New Conversation".to_string()) // ← default on any error
|
|
||||||
}
|
|
||||||
|
|
||||||
async fn log_user_message(
|
|
||||||
pool: &PgPool,
|
|
||||||
user_id: Uuid,
|
|
||||||
conversation_id: Uuid,
|
|
||||||
parent_id: Option<Uuid>,
|
|
||||||
content: &str,
|
|
||||||
tokens: Option<u32>,
|
|
||||||
) -> Result<Uuid, errors::DbError> {
|
|
||||||
insert_message(
|
|
||||||
pool,
|
|
||||||
user_id,
|
|
||||||
conversation_id,
|
|
||||||
parent_id,
|
|
||||||
MessageRole::User,
|
|
||||||
content,
|
|
||||||
tokens,
|
|
||||||
)
|
|
||||||
.await
|
|
||||||
}
|
|
||||||
|
|
||||||
async fn log_assistant_message(
|
|
||||||
pool: &PgPool,
|
|
||||||
user_id: Uuid,
|
|
||||||
conversation_id: Uuid,
|
|
||||||
parent_id: Uuid,
|
|
||||||
content: &str,
|
|
||||||
tokens: Option<u32>,
|
|
||||||
) -> Result<Uuid, errors::DbError> {
|
|
||||||
insert_message(
|
|
||||||
pool,
|
|
||||||
user_id,
|
|
||||||
conversation_id,
|
|
||||||
Some(parent_id),
|
|
||||||
MessageRole::Assistant,
|
|
||||||
content,
|
|
||||||
tokens,
|
|
||||||
)
|
|
||||||
.await
|
|
||||||
}
|
|
||||||
|
|
||||||
// Conversation retrieveing
|
|
||||||
pub async fn get_conversations(
|
|
||||||
State(state): State<AppState>,
|
|
||||||
Extension(auth): Extension<Auth>,
|
|
||||||
Query(params): Query<api::ConversationQuery>,
|
|
||||||
) -> Result<Json<api::ConversationListResponse>, ApiError> {
|
|
||||||
tracing::debug!("Conversation hit: {:?}", auth);
|
|
||||||
|
|
||||||
let conversations = get_conversations_entries(
|
|
||||||
&state.postgres,
|
|
||||||
auth.user_id(),
|
|
||||||
params.limit.unwrap_or(20),
|
|
||||||
params.before,
|
|
||||||
)
|
|
||||||
.await?;
|
|
||||||
|
|
||||||
let has_more = conversations.len() == params.limit.unwrap_or(20) as usize;
|
|
||||||
|
|
||||||
Ok(Json(api::ConversationListResponse {
|
|
||||||
conversations,
|
|
||||||
has_more,
|
|
||||||
}))
|
|
||||||
}
|
|
||||||
|
|
||||||
pub async fn get_messages(
|
|
||||||
State(state): State<AppState>,
|
|
||||||
Extension(auth): Extension<Auth>,
|
|
||||||
Path(conversation_id): Path<Uuid>,
|
|
||||||
Query(params): Query<api::MessageQuery>,
|
|
||||||
) -> Result<Json<api::MessageListResponse>, ApiError> {
|
|
||||||
tracing::debug!("Messages hit: {:?}", auth);
|
|
||||||
|
|
||||||
let messages = get_conversation_messages(
|
|
||||||
&state.postgres,
|
|
||||||
auth.user_id(),
|
|
||||||
conversation_id,
|
|
||||||
params.limit.unwrap_or(50),
|
|
||||||
params.before,
|
|
||||||
)
|
|
||||||
.await?;
|
|
||||||
|
|
||||||
let has_more = messages.len() == params.limit.unwrap_or(50) as usize;
|
|
||||||
|
|
||||||
Ok(Json(api::MessageListResponse { messages, has_more }))
|
|
||||||
}
|
|
||||||
@@ -1,134 +0,0 @@
|
|||||||
use axum::{
|
|
||||||
Json,
|
|
||||||
extract::{Path, State},
|
|
||||||
};
|
|
||||||
|
|
||||||
use crate::dto::api;
|
|
||||||
use crate::providers::ollama::errors::into_http_response;
|
|
||||||
use crate::state::app_state::AppState;
|
|
||||||
|
|
||||||
#[utoipa::path(
|
|
||||||
get,
|
|
||||||
path = "/models",
|
|
||||||
tag = "models",
|
|
||||||
responses(
|
|
||||||
(
|
|
||||||
status = 200,
|
|
||||||
description = "List of locally available Ollama models",
|
|
||||||
body = api::ModelsResponse,
|
|
||||||
content_type = "application/json",
|
|
||||||
),
|
|
||||||
(
|
|
||||||
status = 500,
|
|
||||||
description = "Internal server error (Ollama or network failure)",
|
|
||||||
body = api::ErrorResponse,
|
|
||||||
example = json!({ "error": "connection refused" })
|
|
||||||
)
|
|
||||||
)
|
|
||||||
)]
|
|
||||||
pub async fn list_models(
|
|
||||||
State(state): State<AppState>,
|
|
||||||
) -> Result<Json<api::ModelsResponse>, (axum::http::StatusCode, String)> {
|
|
||||||
match state.ollama.list_models().await {
|
|
||||||
Ok(models) => Ok(Json(models)),
|
|
||||||
Err(e) => Err(into_http_response(e)),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
#[utoipa::path(
|
|
||||||
post,
|
|
||||||
path = "/models/{model}/load",
|
|
||||||
tag = "models",
|
|
||||||
params(
|
|
||||||
("model" = String, Path, description = "Name of the model to load into memory (e.g. 'llama3')")
|
|
||||||
),
|
|
||||||
request_body(
|
|
||||||
content = api::LoadModelBody,
|
|
||||||
description = "Load model request",
|
|
||||||
content_type = "application/json",
|
|
||||||
example = json!({ "keep_alive": "10m" })
|
|
||||||
),
|
|
||||||
responses(
|
|
||||||
(
|
|
||||||
status = 200,
|
|
||||||
description = "Model successfully loaded into memory",
|
|
||||||
body = api::LoadModelResponse,
|
|
||||||
content_type = "application/json",
|
|
||||||
),
|
|
||||||
(
|
|
||||||
status = 400,
|
|
||||||
description = "Invalid or missing keep_alive format",
|
|
||||||
body = api::ErrorResponse,
|
|
||||||
examples(
|
|
||||||
("Missing" = (value = json!({ "error": "keep alive is required and cannot be empty" }))),
|
|
||||||
("Invalid" = (value = json!({ "error": "invalid keep_alive '10x' — use 30s / 10m / 2h, a plain integer, or -1" })))
|
|
||||||
)
|
|
||||||
),
|
|
||||||
(
|
|
||||||
status = 404,
|
|
||||||
description = "Model not found locally",
|
|
||||||
body = api::ErrorResponse,
|
|
||||||
example = json!({ "error": "model 'llama3' not found — run `ollama pull llama3`" })
|
|
||||||
),
|
|
||||||
(
|
|
||||||
status = 500,
|
|
||||||
description = "Internal server error (Ollama or network failure)",
|
|
||||||
body = api::ErrorResponse,
|
|
||||||
example = json!({ "error": "connection refused" })
|
|
||||||
)
|
|
||||||
)
|
|
||||||
)]
|
|
||||||
pub async fn load_model(
|
|
||||||
State(state): State<AppState>,
|
|
||||||
Path(model): Path<String>,
|
|
||||||
Json(body): Json<api::LoadModelBody>,
|
|
||||||
) -> Result<Json<api::LoadModelResponse>, (axum::http::StatusCode, String)> {
|
|
||||||
let response = state
|
|
||||||
.ollama
|
|
||||||
.load_model(&model, body.keep_alive.as_deref())
|
|
||||||
.await
|
|
||||||
.map_err(into_http_response)?;
|
|
||||||
|
|
||||||
Ok(Json(response))
|
|
||||||
}
|
|
||||||
|
|
||||||
#[utoipa::path(
|
|
||||||
delete,
|
|
||||||
path = "/models/{model}/load",
|
|
||||||
tag = "models",
|
|
||||||
params(
|
|
||||||
("model" = String, Path, description = "Name of the model to unload from memory (e.g. 'llama3')")
|
|
||||||
),
|
|
||||||
responses(
|
|
||||||
(
|
|
||||||
status = 200,
|
|
||||||
description = "Model successfully unloaded from memory",
|
|
||||||
body = api::UnloadModelResponse,
|
|
||||||
content_type = "application/json",
|
|
||||||
),
|
|
||||||
(
|
|
||||||
status = 404,
|
|
||||||
description = "Model not found locally",
|
|
||||||
body = api::ErrorResponse,
|
|
||||||
example = json!({ "error": "model 'llama3' not found — run `ollama pull llama3`" })
|
|
||||||
),
|
|
||||||
(
|
|
||||||
status = 500,
|
|
||||||
description = "Internal server error (Ollama or network failure)",
|
|
||||||
body = api::ErrorResponse,
|
|
||||||
example = json!({ "error": "connection refused" })
|
|
||||||
)
|
|
||||||
)
|
|
||||||
)]
|
|
||||||
pub async fn unload_model(
|
|
||||||
State(state): State<AppState>,
|
|
||||||
Path(model): Path<String>,
|
|
||||||
) -> Result<Json<api::UnloadModelResponse>, (axum::http::StatusCode, String)> {
|
|
||||||
let response = state
|
|
||||||
.ollama
|
|
||||||
.unload_model(&model)
|
|
||||||
.await
|
|
||||||
.map_err(into_http_response)?;
|
|
||||||
|
|
||||||
Ok(Json(response))
|
|
||||||
}
|
|
||||||
@@ -1,8 +0,0 @@
|
|||||||
// use axum::Json;
|
|
||||||
// use utoipa::OpenApi;
|
|
||||||
|
|
||||||
// use crate::openapi::V1ApiDoc;
|
|
||||||
|
|
||||||
// pub async fn openapi_json() -> Json<utoipa::openapi::OpenApi> {
|
|
||||||
// Json(V1ApiDoc::openapi())
|
|
||||||
// }
|
|
||||||
@@ -0,0 +1,100 @@
|
|||||||
|
use crate::core;
|
||||||
|
use crate::services::errors::ServiceError;
|
||||||
|
|
||||||
|
use base64::{Engine as _, engine::general_purpose};
|
||||||
|
use rand::RngCore;
|
||||||
|
use rand::rngs::OsRng;
|
||||||
|
use sha2::{Digest, Sha256};
|
||||||
|
use sqlx::PgPool;
|
||||||
|
|
||||||
|
#[derive(Clone)]
|
||||||
|
pub struct AuthService {
|
||||||
|
postgres: PgPool,
|
||||||
|
}
|
||||||
|
|
||||||
|
fn generate_api_key() -> String {
|
||||||
|
let mut bytes = [0u8; 32];
|
||||||
|
OsRng.fill_bytes(&mut bytes);
|
||||||
|
general_purpose::URL_SAFE_NO_PAD.encode(bytes)
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn hash_key(key: &str) -> String {
|
||||||
|
let mut hasher = Sha256::new();
|
||||||
|
hasher.update(key.as_bytes());
|
||||||
|
hasher
|
||||||
|
.finalize()
|
||||||
|
.iter()
|
||||||
|
.map(|b| format!("{:02x}", b))
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
impl AuthService {
|
||||||
|
pub fn new(postgres: PgPool) -> Self {
|
||||||
|
Self { postgres }
|
||||||
|
}
|
||||||
|
|
||||||
|
// ------ API Key ------
|
||||||
|
|
||||||
|
pub async fn create_api_key(
|
||||||
|
&self,
|
||||||
|
payload: core::auth::api_key::CreateApiKeyRequest,
|
||||||
|
) -> Result<String, ServiceError> {
|
||||||
|
let raw_key = generate_api_key();
|
||||||
|
let key_hash = hash_key(&raw_key);
|
||||||
|
|
||||||
|
let key = crate::databases::postgres::api_key::types::CreateApiKey {
|
||||||
|
key_hash,
|
||||||
|
name: payload.name,
|
||||||
|
user_id: payload.user_id,
|
||||||
|
roles: payload.roles.into_iter().map(Into::into).collect(),
|
||||||
|
};
|
||||||
|
|
||||||
|
crate::databases::postgres::api_key::queries::create(&self.postgres, key).await?;
|
||||||
|
|
||||||
|
Ok(raw_key)
|
||||||
|
}
|
||||||
|
|
||||||
|
pub async fn validate_api_key(
|
||||||
|
&self,
|
||||||
|
key: &str,
|
||||||
|
) -> Result<core::auth::api_key::AuthContext, ServiceError> {
|
||||||
|
let key_hash: String = hash_key(key);
|
||||||
|
|
||||||
|
let auth =
|
||||||
|
crate::databases::postgres::api_key::queries::validate(&self.postgres, &key_hash)
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
Ok(auth.into())
|
||||||
|
}
|
||||||
|
|
||||||
|
pub async fn update_last_access_api_key(
|
||||||
|
&self,
|
||||||
|
key_id: &uuid::Uuid,
|
||||||
|
) -> Result<(), ServiceError> {
|
||||||
|
crate::databases::postgres::api_key::queries::update_last_access(&self.postgres, key_id)
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
// ------ JWT ------
|
||||||
|
|
||||||
|
pub async fn create_user(&self, user_id: &uuid::Uuid) -> Result<(), ServiceError> {
|
||||||
|
crate::databases::postgres::user_activity::queries::upsert_user_activity(
|
||||||
|
&self.postgres,
|
||||||
|
user_id,
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
pub async fn validate_jwt(
|
||||||
|
&self,
|
||||||
|
token: &str,
|
||||||
|
) -> Result<core::auth::jwt::JwtClaims, ServiceError> {
|
||||||
|
let auth = crate::providers::keycloak::validator::authenticate_jwt(token).await?;
|
||||||
|
|
||||||
|
Ok(auth.into())
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,317 @@
|
|||||||
|
use crate::core;
|
||||||
|
use crate::core::llm;
|
||||||
|
use crate::core::llm::completions::{
|
||||||
|
CompletionResult, CompletionResultNoStream, CompletionStreamEvent,
|
||||||
|
};
|
||||||
|
use crate::providers::ollama::types::OllamaChatStreamEvent;
|
||||||
|
use crate::providers::{ollama::client::OllamaProvider, ollama::types::OllamaGenerateStreamEvent};
|
||||||
|
use crate::services::errors::ServiceError;
|
||||||
|
|
||||||
|
use super::ConversationService;
|
||||||
|
|
||||||
|
use futures::StreamExt;
|
||||||
|
use std::boxed::Box;
|
||||||
|
use uuid::Uuid;
|
||||||
|
|
||||||
|
#[derive(Clone)]
|
||||||
|
pub struct ChatService {
|
||||||
|
ollama: OllamaProvider,
|
||||||
|
conversation: ConversationService,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl ChatService {
|
||||||
|
pub fn new(ollama: OllamaProvider, conversation: ConversationService) -> Self {
|
||||||
|
Self {
|
||||||
|
ollama,
|
||||||
|
conversation,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub async fn list_models(&self) -> Result<llm::models::Models, ServiceError> {
|
||||||
|
let models = self.ollama.list_models().await?;
|
||||||
|
|
||||||
|
Ok(models.into())
|
||||||
|
}
|
||||||
|
|
||||||
|
pub async fn load_model(
|
||||||
|
&self,
|
||||||
|
body: crate::core::llm::models::LoadModelRequest,
|
||||||
|
) -> Result<crate::core::llm::models::LoadModelResponse, ServiceError> {
|
||||||
|
let b = crate::providers::ollama::types::OllamaGenerateRequest {
|
||||||
|
model: body.model.clone(),
|
||||||
|
prompt: "load".to_string(),
|
||||||
|
stream: false,
|
||||||
|
keep_alive: body.keep_alive,
|
||||||
|
options: None,
|
||||||
|
};
|
||||||
|
|
||||||
|
self.ollama.completions(&b).await?;
|
||||||
|
|
||||||
|
Ok(crate::core::llm::models::LoadModelResponse { model: body.model })
|
||||||
|
}
|
||||||
|
|
||||||
|
pub async fn complete(
|
||||||
|
&self,
|
||||||
|
body: core::llm::completions::CompletionRequest,
|
||||||
|
) -> Result<core::llm::completions::CompletionResult, ServiceError> {
|
||||||
|
let stream = body.options.stream;
|
||||||
|
|
||||||
|
let request: crate::providers::ollama::types::OllamaGenerateRequest = body.into();
|
||||||
|
|
||||||
|
if stream {
|
||||||
|
let ollama_stream = self.ollama.completions_stream(&request).await?;
|
||||||
|
|
||||||
|
let mapped = ollama_stream.map(|item| {
|
||||||
|
item.map(|event| match event {
|
||||||
|
OllamaGenerateStreamEvent::Token(tok) => CompletionStreamEvent::Token(tok),
|
||||||
|
OllamaGenerateStreamEvent::Final(resp) => {
|
||||||
|
CompletionStreamEvent::Final(CompletionResultNoStream {
|
||||||
|
id: Uuid::new_v4(),
|
||||||
|
created_at: resp.created_at,
|
||||||
|
model: resp.model,
|
||||||
|
text: resp.response,
|
||||||
|
prompt_tokens: resp.prompt_eval_count.unwrap_or(0),
|
||||||
|
completion_tokens: resp.eval_count.unwrap_or(0),
|
||||||
|
done_reason: resp.done_reason,
|
||||||
|
total_duration: resp.total_duration,
|
||||||
|
load_duration: resp.load_duration,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
})
|
||||||
|
});
|
||||||
|
|
||||||
|
Ok(CompletionResult::Stream(Box::pin(mapped)))
|
||||||
|
} else {
|
||||||
|
let response = self.ollama.completions(&request).await?;
|
||||||
|
|
||||||
|
let enriched = core::llm::completions::CompletionResultNoStream {
|
||||||
|
id: Uuid::new_v4(),
|
||||||
|
created_at: response.created_at,
|
||||||
|
model: response.model,
|
||||||
|
text: response.response,
|
||||||
|
prompt_tokens: response.prompt_eval_count.unwrap_or(0),
|
||||||
|
completion_tokens: response.eval_count.unwrap_or(0),
|
||||||
|
done_reason: response.done_reason,
|
||||||
|
total_duration: response.total_duration,
|
||||||
|
load_duration: response.load_duration,
|
||||||
|
};
|
||||||
|
|
||||||
|
Ok(core::llm::completions::CompletionResult::NoStream(enriched))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub async fn chat_complete(
|
||||||
|
&self,
|
||||||
|
body: core::llm::chat::ChatCompletionRequest,
|
||||||
|
auth: &core::auth::Auth,
|
||||||
|
) -> Result<core::llm::chat::ChatCompletionResult, ServiceError> {
|
||||||
|
let conversation_id = self
|
||||||
|
.resolve_conversation_with_title(auth.user_id(), &body)
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
let user_msg_id = self
|
||||||
|
.log_user_message(
|
||||||
|
auth.user_id(),
|
||||||
|
conversation_id,
|
||||||
|
body.parent_id,
|
||||||
|
&body.message.content,
|
||||||
|
None,
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
let stream = body.options.stream;
|
||||||
|
|
||||||
|
let history = self
|
||||||
|
.build_chat_history(
|
||||||
|
auth.user_id(),
|
||||||
|
conversation_id,
|
||||||
|
body.message.clone(),
|
||||||
|
body.options.context_depth,
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
let mut request: crate::providers::ollama::types::OllamaChatRequest = body.into();
|
||||||
|
request.messages = history.into_iter().map(Into::into).collect();
|
||||||
|
|
||||||
|
if stream {
|
||||||
|
let ollama_stream = self.ollama.chat_completions_stream(&request).await?;
|
||||||
|
|
||||||
|
let mapped = ollama_stream.map(|item| {
|
||||||
|
item.map(|event| match event {
|
||||||
|
OllamaChatStreamEvent::Token(tok) => {
|
||||||
|
core::llm::chat::ChatCompletionStreamEvent::Token(tok)
|
||||||
|
}
|
||||||
|
OllamaChatStreamEvent::Final(resp) => {
|
||||||
|
core::llm::chat::ChatCompletionStreamEvent::Final(
|
||||||
|
core::llm::chat::ChatCompletionResultNoStream {
|
||||||
|
id: Uuid::new_v4(),
|
||||||
|
created_at: resp.created_at,
|
||||||
|
model: resp.model,
|
||||||
|
message: resp.message.into(),
|
||||||
|
prompt_tokens: resp.prompt_eval_count.unwrap_or(0),
|
||||||
|
completion_tokens: resp.eval_count.unwrap_or(0),
|
||||||
|
done_reason: resp.done_reason,
|
||||||
|
total_duration: resp.total_duration,
|
||||||
|
load_duration: resp.load_duration,
|
||||||
|
},
|
||||||
|
)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
});
|
||||||
|
|
||||||
|
Ok(core::llm::chat::ChatCompletionResult::Stream(Box::pin(
|
||||||
|
mapped,
|
||||||
|
)))
|
||||||
|
} else {
|
||||||
|
let response = self.ollama.chat_completions(&request).await?;
|
||||||
|
|
||||||
|
self.log_assistant_message(
|
||||||
|
auth.user_id(),
|
||||||
|
conversation_id,
|
||||||
|
user_msg_id,
|
||||||
|
&response.message.content,
|
||||||
|
response.eval_count,
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
self.conversation
|
||||||
|
.update_message_tokens(
|
||||||
|
auth.user_id(),
|
||||||
|
user_msg_id,
|
||||||
|
response.prompt_eval_count.unwrap_or_default(),
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
let enriched = core::llm::chat::ChatCompletionResultNoStream {
|
||||||
|
id: Uuid::new_v4(),
|
||||||
|
created_at: response.created_at,
|
||||||
|
model: response.model,
|
||||||
|
message: response.message.into(),
|
||||||
|
prompt_tokens: response.prompt_eval_count.unwrap_or(0),
|
||||||
|
completion_tokens: response.eval_count.unwrap_or(0),
|
||||||
|
done_reason: response.done_reason,
|
||||||
|
total_duration: response.total_duration,
|
||||||
|
load_duration: response.load_duration,
|
||||||
|
};
|
||||||
|
|
||||||
|
Ok(core::llm::chat::ChatCompletionResult::NoStream(enriched))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ------ Helpers ------
|
||||||
|
|
||||||
|
async fn resolve_conversation_with_title(
|
||||||
|
&self,
|
||||||
|
user_id: Uuid,
|
||||||
|
body: &core::llm::chat::ChatCompletionRequest,
|
||||||
|
) -> Result<Uuid, ServiceError> {
|
||||||
|
let conversation_id = match self
|
||||||
|
.conversation
|
||||||
|
.get_or_create_conversation(user_id, body.conversation_id)
|
||||||
|
.await?
|
||||||
|
{
|
||||||
|
core::databases::conversations::ConversationResult::Existing(id) => id,
|
||||||
|
|
||||||
|
core::databases::conversations::ConversationResult::Created(id) => {
|
||||||
|
let last_message = body.message.content.as_str();
|
||||||
|
|
||||||
|
let prompt = format!(
|
||||||
|
"Generate a title using next message in maximum 6 words: {}",
|
||||||
|
last_message
|
||||||
|
);
|
||||||
|
|
||||||
|
let title_result = self
|
||||||
|
.complete(core::llm::completions::CompletionRequest {
|
||||||
|
model: body.model.clone(),
|
||||||
|
prompt,
|
||||||
|
options: core::llm::completions::CompletionOptions {
|
||||||
|
stream: false,
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
})
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
if let CompletionResult::NoStream(t) = title_result {
|
||||||
|
self.conversation
|
||||||
|
.set_conversation_title(user_id, id, &t.text)
|
||||||
|
.await?;
|
||||||
|
}
|
||||||
|
|
||||||
|
id
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
Ok(conversation_id)
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn build_chat_history(
|
||||||
|
&self,
|
||||||
|
auth_user_id: Uuid,
|
||||||
|
conversation_id: Uuid,
|
||||||
|
body_messages: crate::core::llm::chat::Message,
|
||||||
|
context_depth: u32,
|
||||||
|
) -> Result<Vec<crate::core::llm::chat::Message>, ServiceError> {
|
||||||
|
let messages = self
|
||||||
|
.conversation
|
||||||
|
.get_messages_entries(
|
||||||
|
auth_user_id,
|
||||||
|
conversation_id,
|
||||||
|
crate::core::databases::conversations::CursorPage {
|
||||||
|
limit: context_depth,
|
||||||
|
before: None,
|
||||||
|
},
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
let mut history: Vec<_> = messages
|
||||||
|
.messages
|
||||||
|
.into_iter()
|
||||||
|
.map(crate::core::llm::chat::Message::from_summary)
|
||||||
|
.collect();
|
||||||
|
|
||||||
|
history.reverse();
|
||||||
|
|
||||||
|
history.push(body_messages);
|
||||||
|
|
||||||
|
Ok(history)
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn log_user_message(
|
||||||
|
&self,
|
||||||
|
user_id: Uuid,
|
||||||
|
conversation_id: Uuid,
|
||||||
|
parent_id: Option<Uuid>,
|
||||||
|
content: &str,
|
||||||
|
tokens: Option<u32>,
|
||||||
|
) -> Result<Uuid, ServiceError> {
|
||||||
|
self.conversation
|
||||||
|
.log_message(
|
||||||
|
user_id,
|
||||||
|
conversation_id,
|
||||||
|
parent_id,
|
||||||
|
llm::ChatRole::User,
|
||||||
|
content,
|
||||||
|
tokens,
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn log_assistant_message(
|
||||||
|
&self,
|
||||||
|
user_id: Uuid,
|
||||||
|
conversation_id: Uuid,
|
||||||
|
parent_id: Uuid,
|
||||||
|
content: &str,
|
||||||
|
tokens: Option<u32>,
|
||||||
|
) -> Result<Uuid, ServiceError> {
|
||||||
|
self.conversation
|
||||||
|
.log_message(
|
||||||
|
user_id,
|
||||||
|
conversation_id,
|
||||||
|
Some(parent_id),
|
||||||
|
llm::ChatRole::Assistant,
|
||||||
|
content,
|
||||||
|
tokens,
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,132 @@
|
|||||||
|
use crate::core;
|
||||||
|
use crate::databases::postgres;
|
||||||
|
use crate::services::errors::ServiceError;
|
||||||
|
|
||||||
|
use sqlx::PgPool;
|
||||||
|
use uuid::Uuid;
|
||||||
|
|
||||||
|
#[derive(Clone)]
|
||||||
|
pub struct ConversationService {
|
||||||
|
postgres: PgPool,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl ConversationService {
|
||||||
|
pub fn new(postgres: PgPool) -> Self {
|
||||||
|
Self { postgres }
|
||||||
|
}
|
||||||
|
|
||||||
|
pub async fn get_conversations_entries(
|
||||||
|
&self,
|
||||||
|
user_id: Uuid,
|
||||||
|
pointer: crate::core::databases::conversations::CursorPage,
|
||||||
|
) -> Result<core::databases::conversations::ConversationList, ServiceError> {
|
||||||
|
let limit = pointer.limit;
|
||||||
|
|
||||||
|
let conversations = postgres::chat::queries::get_conversations_entries(
|
||||||
|
&self.postgres,
|
||||||
|
user_id,
|
||||||
|
limit,
|
||||||
|
pointer.before,
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
let has_more = conversations.len() == limit as usize;
|
||||||
|
|
||||||
|
Ok(crate::core::databases::conversations::ConversationList {
|
||||||
|
conversations: conversations.into_iter().map(Into::into).collect(),
|
||||||
|
has_more,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
pub async fn get_messages_entries(
|
||||||
|
&self,
|
||||||
|
user_id: Uuid,
|
||||||
|
conversation_id: Uuid,
|
||||||
|
pointer: crate::core::databases::conversations::CursorPage,
|
||||||
|
) -> Result<core::databases::conversations::MessageList, ServiceError> {
|
||||||
|
let limit = pointer.limit;
|
||||||
|
|
||||||
|
let messages = postgres::chat::queries::get_conversation_messages(
|
||||||
|
&self.postgres,
|
||||||
|
user_id,
|
||||||
|
conversation_id,
|
||||||
|
limit,
|
||||||
|
pointer.before,
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
let has_more = messages.len() == limit as usize;
|
||||||
|
|
||||||
|
Ok(crate::core::databases::conversations::MessageList {
|
||||||
|
messages: messages.into_iter().map(Into::into).collect(),
|
||||||
|
has_more,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
pub async fn get_or_create_conversation(
|
||||||
|
&self,
|
||||||
|
user_id: Uuid,
|
||||||
|
conversation_id: Option<Uuid>,
|
||||||
|
) -> Result<core::databases::conversations::ConversationResult, ServiceError> {
|
||||||
|
let state = postgres::chat::queries::get_or_create_conversation(
|
||||||
|
&self.postgres,
|
||||||
|
conversation_id,
|
||||||
|
user_id,
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
Ok(state.into())
|
||||||
|
}
|
||||||
|
|
||||||
|
pub async fn set_conversation_title(
|
||||||
|
&self,
|
||||||
|
user_id: Uuid,
|
||||||
|
conversation_id: Uuid,
|
||||||
|
title: &str,
|
||||||
|
) -> Result<(), ServiceError> {
|
||||||
|
postgres::chat::queries::set_conversation_title(
|
||||||
|
&self.postgres,
|
||||||
|
conversation_id,
|
||||||
|
user_id,
|
||||||
|
title,
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
pub async fn log_message(
|
||||||
|
&self,
|
||||||
|
user_id: Uuid,
|
||||||
|
conversation_id: Uuid,
|
||||||
|
parent_id: Option<Uuid>,
|
||||||
|
role: core::llm::ChatRole,
|
||||||
|
content: &str,
|
||||||
|
tokens: Option<u32>,
|
||||||
|
) -> Result<Uuid, ServiceError> {
|
||||||
|
let id = postgres::chat::queries::insert_message(
|
||||||
|
&self.postgres,
|
||||||
|
user_id,
|
||||||
|
conversation_id,
|
||||||
|
parent_id,
|
||||||
|
role.into(),
|
||||||
|
content,
|
||||||
|
tokens,
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
Ok(id)
|
||||||
|
}
|
||||||
|
|
||||||
|
pub async fn update_message_tokens(
|
||||||
|
&self,
|
||||||
|
user_id: Uuid,
|
||||||
|
message_id: Uuid,
|
||||||
|
tokens: u32,
|
||||||
|
) -> Result<(), ServiceError> {
|
||||||
|
postgres::chat::queries::update_message_tokens(&self.postgres, user_id, message_id, tokens)
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,27 @@
|
|||||||
|
use crate::databases::errors::DbError;
|
||||||
|
use crate::providers::keycloak::errors::AuthError;
|
||||||
|
use crate::providers::ollama::errors::LlmError;
|
||||||
|
|
||||||
|
pub enum ServiceError {
|
||||||
|
Db(DbError),
|
||||||
|
Llm(LlmError),
|
||||||
|
Auth(AuthError),
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<DbError> for ServiceError {
|
||||||
|
fn from(e: DbError) -> Self {
|
||||||
|
ServiceError::Db(e)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<LlmError> for ServiceError {
|
||||||
|
fn from(e: LlmError) -> Self {
|
||||||
|
ServiceError::Llm(e)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<AuthError> for ServiceError {
|
||||||
|
fn from(e: AuthError) -> Self {
|
||||||
|
ServiceError::Auth(e)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,8 @@
|
|||||||
|
pub mod auth_service;
|
||||||
|
pub mod chat_service;
|
||||||
|
pub mod conversation_service;
|
||||||
|
pub mod errors;
|
||||||
|
|
||||||
|
pub use auth_service::AuthService;
|
||||||
|
pub use chat_service::ChatService;
|
||||||
|
pub use conversation_service::ConversationService;
|
||||||
@@ -1,9 +0,0 @@
|
|||||||
use crate::providers::ollama::client::OllamaProvider;
|
|
||||||
use sqlx::PgPool;
|
|
||||||
use std::sync::Arc;
|
|
||||||
|
|
||||||
#[derive(Clone)]
|
|
||||||
pub struct AppState {
|
|
||||||
pub ollama: Arc<OllamaProvider>,
|
|
||||||
pub postgres: PgPool,
|
|
||||||
}
|
|
||||||
@@ -1 +0,0 @@
|
|||||||
pub mod app_state;
|
|
||||||
@@ -1,11 +0,0 @@
|
|||||||
use sha2::{Digest, Sha256};
|
|
||||||
|
|
||||||
pub fn hash_key(key: &str) -> String {
|
|
||||||
let mut hasher = Sha256::new();
|
|
||||||
hasher.update(key.as_bytes());
|
|
||||||
hasher
|
|
||||||
.finalize()
|
|
||||||
.iter()
|
|
||||||
.map(|b| format!("{:02x}", b))
|
|
||||||
.collect()
|
|
||||||
}
|
|
||||||
@@ -1 +0,0 @@
|
|||||||
pub mod crypto;
|
|
||||||
+419
-419
@@ -1,419 +1,419 @@
|
|||||||
use serde_json::json;
|
// use serde_json::json;
|
||||||
use wiremock::matchers::{method, path};
|
// use wiremock::matchers::{method, path};
|
||||||
use wiremock::{Mock, MockServer, ResponseTemplate};
|
// use wiremock::{Mock, MockServer, ResponseTemplate};
|
||||||
|
|
||||||
use chat::dto::api;
|
// use chat::dto::api;
|
||||||
use chat::providers::ollama::client::OllamaProvider;
|
// use chat::providers::ollama::client::OllamaProvider;
|
||||||
use chat::providers::ollama::errors::OllamaError;
|
// use chat::providers::ollama::errors::OllamaError;
|
||||||
|
|
||||||
// ── helpers ──────────────────────────────────────────────────────────────────
|
// // ── helpers ──────────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
async fn setup() -> (MockServer, OllamaProvider) {
|
// async fn setup() -> (MockServer, OllamaProvider) {
|
||||||
let server = MockServer::start().await;
|
// let server = MockServer::start().await;
|
||||||
let provider = OllamaProvider::new(server.uri());
|
// let provider = OllamaProvider::new(server.uri());
|
||||||
(server, provider)
|
// (server, provider)
|
||||||
}
|
// }
|
||||||
|
|
||||||
fn models_response(names: &[&str]) -> serde_json::Value {
|
// fn models_response(names: &[&str]) -> serde_json::Value {
|
||||||
json!({
|
// json!({
|
||||||
"models": names.iter().map(|n| json!({ "name": n })).collect::<Vec<_>>()
|
// "models": names.iter().map(|n| json!({ "name": n })).collect::<Vec<_>>()
|
||||||
})
|
// })
|
||||||
}
|
// }
|
||||||
|
|
||||||
// ── list_models ───────────────────────────────────────────────────────────────
|
// // ── list_models ───────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
#[tokio::test]
|
// #[tokio::test]
|
||||||
async fn test_list_models_ok() {
|
// async fn test_list_models_ok() {
|
||||||
let (server, provider) = setup().await;
|
// let (server, provider) = setup().await;
|
||||||
|
|
||||||
Mock::given(method("GET"))
|
// Mock::given(method("GET"))
|
||||||
.and(path("/api/tags"))
|
// .and(path("/api/tags"))
|
||||||
.respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
// .respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
||||||
.mount(&server)
|
// .mount(&server)
|
||||||
.await;
|
// .await;
|
||||||
|
|
||||||
let res = provider.list_models().await.unwrap();
|
// let res = provider.list_models().await.unwrap();
|
||||||
assert_eq!(res.models[0].name, "llama3");
|
// assert_eq!(res.models[0].name, "llama3");
|
||||||
}
|
// }
|
||||||
|
|
||||||
// ── completions ───────────────────────────────────────────────────────────────
|
// // ── completions ───────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
#[tokio::test]
|
// #[tokio::test]
|
||||||
async fn test_completions_ok() {
|
// async fn test_completions_ok() {
|
||||||
let (server, provider) = setup().await;
|
// let (server, provider) = setup().await;
|
||||||
|
|
||||||
Mock::given(method("GET"))
|
// Mock::given(method("GET"))
|
||||||
.and(path("/api/tags"))
|
// .and(path("/api/tags"))
|
||||||
.respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
// .respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
||||||
.mount(&server)
|
// .mount(&server)
|
||||||
.await;
|
// .await;
|
||||||
|
|
||||||
Mock::given(method("POST"))
|
// Mock::given(method("POST"))
|
||||||
.and(path("/api/generate"))
|
// .and(path("/api/generate"))
|
||||||
.respond_with(ResponseTemplate::new(200).set_body_json(json!({
|
// .respond_with(ResponseTemplate::new(200).set_body_json(json!({
|
||||||
"model": "llama3",
|
// "model": "llama3",
|
||||||
"response": "I am a helpful assistant.",
|
// "response": "I am a helpful assistant.",
|
||||||
"done": true,
|
// "done": true,
|
||||||
"prompt_eval_count": 10,
|
// "prompt_eval_count": 10,
|
||||||
"eval_count": 8,
|
// "eval_count": 8,
|
||||||
})))
|
// })))
|
||||||
.mount(&server)
|
// .mount(&server)
|
||||||
.await;
|
// .await;
|
||||||
|
|
||||||
let req = api::CompletionRequest {
|
// let req = api::CompletionRequest {
|
||||||
base: api::BaseLLMRequest {
|
// base: api::BaseLLMRequest {
|
||||||
model: "llama3".to_string(),
|
// model: "llama3".to_string(),
|
||||||
..Default::default()
|
// ..Default::default()
|
||||||
},
|
// },
|
||||||
prompt: "Hello".to_string(),
|
// prompt: "Hello".to_string(),
|
||||||
};
|
// };
|
||||||
|
|
||||||
let res = provider.completions(&req).await.unwrap();
|
// let res = provider.completions(&req).await.unwrap();
|
||||||
|
|
||||||
assert_eq!(res.object, api::CompletionObject::TextCompletion);
|
// assert_eq!(res.object, api::CompletionObject::TextCompletion);
|
||||||
|
|
||||||
assert_eq!(res.choices.len(), 1);
|
// assert_eq!(res.choices.len(), 1);
|
||||||
assert_eq!(res.choices[0].text, "I am a helpful assistant.");
|
// assert_eq!(res.choices[0].text, "I am a helpful assistant.");
|
||||||
assert_eq!(res.choices[0].finish_reason, api::FinishReason::Stop);
|
// assert_eq!(res.choices[0].finish_reason, api::FinishReason::Stop);
|
||||||
|
|
||||||
assert_eq!(res.usage.prompt_tokens, 10);
|
// assert_eq!(res.usage.prompt_tokens, 10);
|
||||||
assert_eq!(res.usage.completion_tokens, 8);
|
// assert_eq!(res.usage.completion_tokens, 8);
|
||||||
assert_eq!(res.usage.total_tokens, 18);
|
// assert_eq!(res.usage.total_tokens, 18);
|
||||||
}
|
// }
|
||||||
|
|
||||||
#[tokio::test]
|
// #[tokio::test]
|
||||||
async fn test_completions_missing_prompt() {
|
// async fn test_completions_missing_prompt() {
|
||||||
let (server, provider) = setup().await;
|
// let (server, provider) = setup().await;
|
||||||
|
|
||||||
Mock::given(method("GET"))
|
// Mock::given(method("GET"))
|
||||||
.and(path("/api/tags"))
|
// .and(path("/api/tags"))
|
||||||
.respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
// .respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
||||||
.mount(&server)
|
// .mount(&server)
|
||||||
.await;
|
// .await;
|
||||||
|
|
||||||
let req = api::CompletionRequest {
|
// let req = api::CompletionRequest {
|
||||||
base: api::BaseLLMRequest {
|
// base: api::BaseLLMRequest {
|
||||||
model: "llama3".to_string(),
|
// model: "llama3".to_string(),
|
||||||
..Default::default()
|
// ..Default::default()
|
||||||
},
|
// },
|
||||||
prompt: "".to_string(),
|
// prompt: "".to_string(),
|
||||||
};
|
// };
|
||||||
|
|
||||||
let err = provider.completions(&req).await.unwrap_err();
|
// let err = provider.completions(&req).await.unwrap_err();
|
||||||
|
|
||||||
assert!(matches!(err, OllamaError::MissingPrompt));
|
// assert!(matches!(err, OllamaError::MissingPrompt));
|
||||||
}
|
// }
|
||||||
|
|
||||||
#[tokio::test]
|
// #[tokio::test]
|
||||||
async fn test_completions_empty_prompt() {
|
// async fn test_completions_empty_prompt() {
|
||||||
let (server, provider) = setup().await;
|
// let (server, provider) = setup().await;
|
||||||
|
|
||||||
Mock::given(method("GET"))
|
// Mock::given(method("GET"))
|
||||||
.and(path("/api/tags"))
|
// .and(path("/api/tags"))
|
||||||
.respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
// .respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
||||||
.mount(&server)
|
// .mount(&server)
|
||||||
.await;
|
// .await;
|
||||||
|
|
||||||
let req = api::CompletionRequest {
|
// let req = api::CompletionRequest {
|
||||||
base: api::BaseLLMRequest {
|
// base: api::BaseLLMRequest {
|
||||||
model: "llama3".to_string(),
|
// model: "llama3".to_string(),
|
||||||
..Default::default()
|
// ..Default::default()
|
||||||
},
|
// },
|
||||||
prompt: " ".to_string(),
|
// prompt: " ".to_string(),
|
||||||
};
|
// };
|
||||||
|
|
||||||
let err = provider.completions(&req).await.unwrap_err();
|
// let err = provider.completions(&req).await.unwrap_err();
|
||||||
|
|
||||||
assert!(matches!(err, OllamaError::MissingPrompt));
|
// assert!(matches!(err, OllamaError::MissingPrompt));
|
||||||
}
|
// }
|
||||||
|
|
||||||
#[tokio::test]
|
// #[tokio::test]
|
||||||
async fn test_completions_model_not_found() {
|
// async fn test_completions_model_not_found() {
|
||||||
let (server, provider) = setup().await;
|
// let (server, provider) = setup().await;
|
||||||
|
|
||||||
Mock::given(method("GET"))
|
// Mock::given(method("GET"))
|
||||||
.and(path("/api/tags"))
|
// .and(path("/api/tags"))
|
||||||
.respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
// .respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
||||||
.mount(&server)
|
// .mount(&server)
|
||||||
.await;
|
// .await;
|
||||||
|
|
||||||
let req = api::CompletionRequest {
|
// let req = api::CompletionRequest {
|
||||||
base: api::BaseLLMRequest {
|
// base: api::BaseLLMRequest {
|
||||||
model: "gpt-4".to_string(),
|
// model: "gpt-4".to_string(),
|
||||||
..Default::default()
|
// ..Default::default()
|
||||||
},
|
// },
|
||||||
prompt: "hello".to_string(),
|
// prompt: "hello".to_string(),
|
||||||
};
|
// };
|
||||||
|
|
||||||
let err = provider.completions(&req).await.unwrap_err();
|
// let err = provider.completions(&req).await.unwrap_err();
|
||||||
|
|
||||||
assert!(matches!(err, OllamaError::ModelNotFound(_)));
|
// assert!(matches!(err, OllamaError::ModelNotFound(_)));
|
||||||
}
|
// }
|
||||||
|
|
||||||
// ── chat_completions ──────────────────────────────────────────────────────────
|
// // ── chat_completions ──────────────────────────────────────────────────────────
|
||||||
|
|
||||||
#[tokio::test]
|
// #[tokio::test]
|
||||||
async fn test_chat_completions_ok() {
|
// async fn test_chat_completions_ok() {
|
||||||
let (server, provider) = setup().await;
|
// let (server, provider) = setup().await;
|
||||||
|
|
||||||
Mock::given(method("GET"))
|
// Mock::given(method("GET"))
|
||||||
.and(path("/api/tags"))
|
// .and(path("/api/tags"))
|
||||||
.respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
// .respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
||||||
.mount(&server)
|
// .mount(&server)
|
||||||
.await;
|
// .await;
|
||||||
|
|
||||||
Mock::given(method("POST"))
|
// Mock::given(method("POST"))
|
||||||
.and(path("/api/chat"))
|
// .and(path("/api/chat"))
|
||||||
.respond_with(ResponseTemplate::new(200).set_body_json(json!({
|
// .respond_with(ResponseTemplate::new(200).set_body_json(json!({
|
||||||
"model": "llama3",
|
// "model": "llama3",
|
||||||
"message": { "role": "assistant", "content": "4." },
|
// "message": { "role": "assistant", "content": "4." },
|
||||||
"done": true,
|
// "done": true,
|
||||||
"prompt_eval_count": 5,
|
// "prompt_eval_count": 5,
|
||||||
"eval_count": 2
|
// "eval_count": 2
|
||||||
})))
|
// })))
|
||||||
.mount(&server)
|
// .mount(&server)
|
||||||
.await;
|
// .await;
|
||||||
|
|
||||||
let req = api::ChatRequest {
|
// let req = api::ChatRequest {
|
||||||
base: api::BaseLLMRequest {
|
// base: api::BaseLLMRequest {
|
||||||
model: "llama3".to_string(),
|
// model: "llama3".to_string(),
|
||||||
..Default::default()
|
// ..Default::default()
|
||||||
},
|
// },
|
||||||
messages: vec![api::Message {
|
// messages: vec![api::Message {
|
||||||
role: api::Role::User,
|
// role: api::Role::User,
|
||||||
content: "What is 2+2?".to_string(),
|
// content: "What is 2+2?".to_string(),
|
||||||
}],
|
// }],
|
||||||
conversation_id: None,
|
// conversation_id: None,
|
||||||
parent_id: None,
|
// parent_id: None,
|
||||||
};
|
// };
|
||||||
|
|
||||||
let res = provider.chat_completions(&req).await.unwrap();
|
// let res = provider.chat_completions(&req).await.unwrap();
|
||||||
|
|
||||||
assert_eq!(res.object, "chat.completion");
|
// assert_eq!(res.object, "chat.completion");
|
||||||
assert_eq!(res.choices.len(), 1);
|
// assert_eq!(res.choices.len(), 1);
|
||||||
|
|
||||||
assert_eq!(res.choices[0].message.role, api::Role::Assistant);
|
// assert_eq!(res.choices[0].message.role, api::Role::Assistant);
|
||||||
assert_eq!(res.choices[0].message.content, "4.");
|
// assert_eq!(res.choices[0].message.content, "4.");
|
||||||
|
|
||||||
assert_eq!(res.choices[0].finish_reason, api::FinishReason::Stop);
|
// assert_eq!(res.choices[0].finish_reason, api::FinishReason::Stop);
|
||||||
|
|
||||||
let usage = res.usage.unwrap();
|
// let usage = res.usage.unwrap();
|
||||||
assert_eq!(usage.prompt_tokens, 5);
|
// assert_eq!(usage.prompt_tokens, 5);
|
||||||
assert_eq!(usage.completion_tokens, 2);
|
// assert_eq!(usage.completion_tokens, 2);
|
||||||
assert_eq!(usage.total_tokens, 7);
|
// assert_eq!(usage.total_tokens, 7);
|
||||||
}
|
// }
|
||||||
|
|
||||||
#[tokio::test]
|
// #[tokio::test]
|
||||||
async fn test_chat_completions_missing_messages() {
|
// async fn test_chat_completions_missing_messages() {
|
||||||
let (server, provider) = setup().await;
|
// let (server, provider) = setup().await;
|
||||||
|
|
||||||
Mock::given(method("GET"))
|
// Mock::given(method("GET"))
|
||||||
.and(path("/api/tags"))
|
// .and(path("/api/tags"))
|
||||||
.respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
// .respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
||||||
.mount(&server)
|
// .mount(&server)
|
||||||
.await;
|
// .await;
|
||||||
|
|
||||||
let req = api::ChatRequest {
|
// let req = api::ChatRequest {
|
||||||
base: api::BaseLLMRequest {
|
// base: api::BaseLLMRequest {
|
||||||
model: "llama3".to_string(),
|
// model: "llama3".to_string(),
|
||||||
..Default::default()
|
// ..Default::default()
|
||||||
},
|
// },
|
||||||
messages: vec![],
|
// messages: vec![],
|
||||||
conversation_id: None,
|
// conversation_id: None,
|
||||||
parent_id: None,
|
// parent_id: None,
|
||||||
};
|
// };
|
||||||
|
|
||||||
let err = provider.chat_completions(&req).await.unwrap_err();
|
// let err = provider.chat_completions(&req).await.unwrap_err();
|
||||||
|
|
||||||
assert!(matches!(err, OllamaError::MissingMessages));
|
// assert!(matches!(err, OllamaError::MissingMessages));
|
||||||
}
|
// }
|
||||||
|
|
||||||
#[tokio::test]
|
// #[tokio::test]
|
||||||
async fn test_chat_completions_no_user_message() {
|
// async fn test_chat_completions_no_user_message() {
|
||||||
let (server, provider) = setup().await;
|
// let (server, provider) = setup().await;
|
||||||
|
|
||||||
Mock::given(method("GET"))
|
// Mock::given(method("GET"))
|
||||||
.and(path("/api/tags"))
|
// .and(path("/api/tags"))
|
||||||
.respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
// .respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
||||||
.mount(&server)
|
// .mount(&server)
|
||||||
.await;
|
// .await;
|
||||||
|
|
||||||
let req = api::ChatRequest {
|
// let req = api::ChatRequest {
|
||||||
base: api::BaseLLMRequest {
|
// base: api::BaseLLMRequest {
|
||||||
model: "llama3".to_string(),
|
// model: "llama3".to_string(),
|
||||||
..Default::default()
|
// ..Default::default()
|
||||||
},
|
// },
|
||||||
messages: vec![api::Message {
|
// messages: vec![api::Message {
|
||||||
role: api::Role::System,
|
// role: api::Role::System,
|
||||||
content: "be helpful".to_string(),
|
// content: "be helpful".to_string(),
|
||||||
}],
|
// }],
|
||||||
conversation_id: None,
|
// conversation_id: None,
|
||||||
parent_id: None,
|
// parent_id: None,
|
||||||
};
|
// };
|
||||||
|
|
||||||
let err = provider.chat_completions(&req).await.unwrap_err();
|
// let err = provider.chat_completions(&req).await.unwrap_err();
|
||||||
|
|
||||||
assert!(matches!(err, OllamaError::MissingMessages));
|
// assert!(matches!(err, OllamaError::MissingMessages));
|
||||||
}
|
// }
|
||||||
|
|
||||||
#[tokio::test]
|
// #[tokio::test]
|
||||||
async fn test_chat_completions_model_not_found() {
|
// async fn test_chat_completions_model_not_found() {
|
||||||
let (server, provider) = setup().await;
|
// let (server, provider) = setup().await;
|
||||||
|
|
||||||
Mock::given(method("GET"))
|
// Mock::given(method("GET"))
|
||||||
.and(path("/api/tags"))
|
// .and(path("/api/tags"))
|
||||||
.respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
// .respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
||||||
.mount(&server)
|
// .mount(&server)
|
||||||
.await;
|
// .await;
|
||||||
|
|
||||||
let req = api::ChatRequest {
|
// let req = api::ChatRequest {
|
||||||
base: api::BaseLLMRequest {
|
// base: api::BaseLLMRequest {
|
||||||
model: "gpt-4".to_string(),
|
// model: "gpt-4".to_string(),
|
||||||
..Default::default()
|
// ..Default::default()
|
||||||
},
|
// },
|
||||||
messages: vec![api::Message {
|
// messages: vec![api::Message {
|
||||||
role: api::Role::User,
|
// role: api::Role::User,
|
||||||
content: "hi".to_string(),
|
// content: "hi".to_string(),
|
||||||
}],
|
// }],
|
||||||
conversation_id: None,
|
// conversation_id: None,
|
||||||
parent_id: None,
|
// parent_id: None,
|
||||||
};
|
// };
|
||||||
|
|
||||||
let err = provider.chat_completions(&req).await.unwrap_err();
|
// let err = provider.chat_completions(&req).await.unwrap_err();
|
||||||
|
|
||||||
assert!(matches!(err, OllamaError::ModelNotFound(_)));
|
// assert!(matches!(err, OllamaError::ModelNotFound(_)));
|
||||||
}
|
// }
|
||||||
|
|
||||||
// // ── load_model ────────────────────────────────────────────────────────────────
|
// // // ── load_model ────────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
#[tokio::test]
|
// #[tokio::test]
|
||||||
async fn test_load_model_ok() {
|
// async fn test_load_model_ok() {
|
||||||
let (server, provider) = setup().await;
|
// let (server, provider) = setup().await;
|
||||||
|
|
||||||
Mock::given(method("GET"))
|
// Mock::given(method("GET"))
|
||||||
.and(path("/api/tags"))
|
// .and(path("/api/tags"))
|
||||||
.respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
// .respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
||||||
.mount(&server)
|
// .mount(&server)
|
||||||
.await;
|
// .await;
|
||||||
|
|
||||||
Mock::given(method("POST"))
|
// Mock::given(method("POST"))
|
||||||
.and(path("/api/generate"))
|
// .and(path("/api/generate"))
|
||||||
.respond_with(ResponseTemplate::new(200).set_body_json(json!({
|
// .respond_with(ResponseTemplate::new(200).set_body_json(json!({
|
||||||
"model": "llama3",
|
// "model": "llama3",
|
||||||
"response": "ok",
|
// "response": "ok",
|
||||||
"done": true,
|
// "done": true,
|
||||||
})))
|
// })))
|
||||||
.mount(&server)
|
// .mount(&server)
|
||||||
.await;
|
// .await;
|
||||||
|
|
||||||
let res = provider.load_model("llama3", Some("10m")).await.unwrap();
|
// let res = provider.load_model("llama3", Some("10m")).await.unwrap();
|
||||||
|
|
||||||
assert_eq!(res.model, "llama3");
|
// assert_eq!(res.model, "llama3");
|
||||||
assert_eq!(res.status, "loaded");
|
// assert_eq!(res.status, "loaded");
|
||||||
assert_eq!(res.keep_alive, "10m");
|
// assert_eq!(res.keep_alive, "10m");
|
||||||
}
|
// }
|
||||||
|
|
||||||
#[tokio::test]
|
// #[tokio::test]
|
||||||
async fn test_load_model_not_found() {
|
// async fn test_load_model_not_found() {
|
||||||
let (server, provider) = setup().await;
|
// let (server, provider) = setup().await;
|
||||||
|
|
||||||
Mock::given(method("GET"))
|
// Mock::given(method("GET"))
|
||||||
.and(path("/api/tags"))
|
// .and(path("/api/tags"))
|
||||||
.respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
// .respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
||||||
.mount(&server)
|
// .mount(&server)
|
||||||
.await;
|
// .await;
|
||||||
|
|
||||||
let err = provider.load_model("gpt-4", Some("10m")).await.unwrap_err();
|
// let err = provider.load_model("gpt-4", Some("10m")).await.unwrap_err();
|
||||||
|
|
||||||
assert!(matches!(err, OllamaError::ModelNotFound(_)));
|
// assert!(matches!(err, OllamaError::ModelNotFound(_)));
|
||||||
}
|
// }
|
||||||
|
|
||||||
#[tokio::test]
|
// #[tokio::test]
|
||||||
async fn test_load_model_invalid_keep_alive() {
|
// async fn test_load_model_invalid_keep_alive() {
|
||||||
let (server, provider) = setup().await;
|
// let (server, provider) = setup().await;
|
||||||
|
|
||||||
Mock::given(method("GET"))
|
// Mock::given(method("GET"))
|
||||||
.and(path("/api/tags"))
|
// .and(path("/api/tags"))
|
||||||
.respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
// .respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
||||||
.mount(&server)
|
// .mount(&server)
|
||||||
.await;
|
// .await;
|
||||||
|
|
||||||
let err = provider
|
// let err = provider
|
||||||
.load_model("llama3", Some("10x"))
|
// .load_model("llama3", Some("10x"))
|
||||||
.await
|
// .await
|
||||||
.unwrap_err();
|
// .unwrap_err();
|
||||||
|
|
||||||
assert!(matches!(err, OllamaError::InvalidKeepAlive(_)));
|
// assert!(matches!(err, OllamaError::InvalidKeepAlive(_)));
|
||||||
}
|
// }
|
||||||
|
|
||||||
#[tokio::test]
|
// #[tokio::test]
|
||||||
async fn test_load_model_keep_alive_plain_integer() {
|
// async fn test_load_model_keep_alive_plain_integer() {
|
||||||
let (server, provider) = setup().await;
|
// let (server, provider) = setup().await;
|
||||||
|
|
||||||
Mock::given(method("GET"))
|
// Mock::given(method("GET"))
|
||||||
.and(path("/api/tags"))
|
// .and(path("/api/tags"))
|
||||||
.respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
// .respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
||||||
.mount(&server)
|
// .mount(&server)
|
||||||
.await;
|
// .await;
|
||||||
|
|
||||||
Mock::given(method("POST"))
|
// Mock::given(method("POST"))
|
||||||
.and(path("/api/generate"))
|
// .and(path("/api/generate"))
|
||||||
.respond_with(ResponseTemplate::new(200).set_body_json(json!({
|
// .respond_with(ResponseTemplate::new(200).set_body_json(json!({
|
||||||
"model": "llama3",
|
// "model": "llama3",
|
||||||
"done": true,
|
// "done": true,
|
||||||
})))
|
// })))
|
||||||
.mount(&server)
|
// .mount(&server)
|
||||||
.await;
|
// .await;
|
||||||
|
|
||||||
let res = provider.load_model("llama3", Some("3600")).await.unwrap();
|
// let res = provider.load_model("llama3", Some("3600")).await.unwrap();
|
||||||
|
|
||||||
assert_eq!(res.status, "loaded");
|
// assert_eq!(res.status, "loaded");
|
||||||
assert_eq!(res.keep_alive, "3600");
|
// assert_eq!(res.keep_alive, "3600");
|
||||||
|
|
||||||
let res = provider.load_model("llama3", Some("-1")).await.unwrap();
|
// let res = provider.load_model("llama3", Some("-1")).await.unwrap();
|
||||||
|
|
||||||
assert_eq!(res.status, "loaded");
|
// assert_eq!(res.status, "loaded");
|
||||||
assert_eq!(res.keep_alive, "-1");
|
// assert_eq!(res.keep_alive, "-1");
|
||||||
}
|
// }
|
||||||
|
|
||||||
// // ── unload_model ──────────────────────────────────────────────────────────────
|
// // // ── unload_model ──────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
#[tokio::test]
|
// #[tokio::test]
|
||||||
async fn test_unload_model_ok() {
|
// async fn test_unload_model_ok() {
|
||||||
let (server, provider) = setup().await;
|
// let (server, provider) = setup().await;
|
||||||
|
|
||||||
Mock::given(method("GET"))
|
// Mock::given(method("GET"))
|
||||||
.and(path("/api/tags"))
|
// .and(path("/api/tags"))
|
||||||
.respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
// .respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
||||||
.mount(&server)
|
// .mount(&server)
|
||||||
.await;
|
// .await;
|
||||||
|
|
||||||
Mock::given(method("POST"))
|
// Mock::given(method("POST"))
|
||||||
.and(path("/api/generate"))
|
// .and(path("/api/generate"))
|
||||||
.respond_with(ResponseTemplate::new(200).set_body_json(json!({
|
// .respond_with(ResponseTemplate::new(200).set_body_json(json!({
|
||||||
"model": "llama3",
|
// "model": "llama3",
|
||||||
"response": "ok",
|
// "response": "ok",
|
||||||
"done": true,
|
// "done": true,
|
||||||
})))
|
// })))
|
||||||
.mount(&server)
|
// .mount(&server)
|
||||||
.await;
|
// .await;
|
||||||
|
|
||||||
let res = provider.unload_model("llama3").await.unwrap();
|
// let res = provider.unload_model("llama3").await.unwrap();
|
||||||
|
|
||||||
assert_eq!(res.model, "llama3");
|
// assert_eq!(res.model, "llama3");
|
||||||
assert_eq!(res.status, "unloaded");
|
// assert_eq!(res.status, "unloaded");
|
||||||
}
|
// }
|
||||||
|
|
||||||
#[tokio::test]
|
// #[tokio::test]
|
||||||
async fn test_unload_model_not_found() {
|
// async fn test_unload_model_not_found() {
|
||||||
let (server, provider) = setup().await;
|
// let (server, provider) = setup().await;
|
||||||
|
|
||||||
Mock::given(method("GET"))
|
// Mock::given(method("GET"))
|
||||||
.and(path("/api/tags"))
|
// .and(path("/api/tags"))
|
||||||
.respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
// .respond_with(ResponseTemplate::new(200).set_body_json(models_response(&["llama3"])))
|
||||||
.mount(&server)
|
// .mount(&server)
|
||||||
.await;
|
// .await;
|
||||||
|
|
||||||
let err = provider.unload_model("gpt-4").await.unwrap_err();
|
// let err = provider.unload_model("gpt-4").await.unwrap_err();
|
||||||
assert!(matches!(err, OllamaError::ModelNotFound(_)));
|
// assert!(matches!(err, OllamaError::ModelNotFound(_)));
|
||||||
}
|
// }
|
||||||
|
|||||||
Reference in New Issue
Block a user