fix: format + tracing
This commit is contained in:
@@ -3,7 +3,7 @@ name: Publish & Deploy
|
|||||||
on:
|
on:
|
||||||
push:
|
push:
|
||||||
tags:
|
tags:
|
||||||
- 'v*'
|
- "v*"
|
||||||
|
|
||||||
permissions:
|
permissions:
|
||||||
contents: write
|
contents: write
|
||||||
@@ -86,4 +86,3 @@ jobs:
|
|||||||
REPO_LOWER=$REPO_LOWER TAG=$TAG docker compose up -d
|
REPO_LOWER=$REPO_LOWER TAG=$TAG docker compose up -d
|
||||||
|
|
||||||
docker image prune -af
|
docker image prune -af
|
||||||
|
|
||||||
|
|||||||
@@ -3,10 +3,8 @@
|
|||||||
- Race condition on jwks token refresh
|
- Race condition on jwks token refresh
|
||||||
- Rate Limiting
|
- Rate Limiting
|
||||||
|
|
||||||
|
|
||||||
git tag -d v1.0.0; git push origin :refs/tags/v1.0.0; git tag -a v1.0.0 -m "Release v1.0.0"; git push origin v1.0.0
|
git tag -d v1.0.0; git push origin :refs/tags/v1.0.0; git tag -a v1.0.0 -m "Release v1.0.0"; git push origin v1.0.0
|
||||||
|
|
||||||
|
|
||||||
curl https://chat.iceberg.black/api/v1/models -H "Authorization: Bearer eyJhbGciOiJSUzI1NiIsInR5cCIgOiAiSldUIiwia2lkIiA6ICJublpLek04TkZHVmpWbGFPRXZpMUtFSTVHQWRwaGlsYjh3RHRLeG5JOENZIn0.eyJleHAiOjE3NzU4MTUyODksImlhdCI6MTc3NTgxNDk4OSwianRpIjoiMTgwYzA2NDUtYzZiMC00MDRmLTgyYjEtMzA3YmY4ZmJlNmJlIiwiaXNzIjoiaHR0cHM6Ly9hdXRoLmljZWJlcmcuYmxhY2svcmVhbG1zL2ljZWJlcmciLCJzdWIiOiJmZGRiN2FjZC1kMmE5LTRmMTctOWIxNi1kZjVlN2EzNDI4YjciLCJ0eXAiOiJCZWFyZXIiLCJhenAiOiJjaGF0LWFwaSIsInNjb3BlIjoiIiwiY2xpZW50SG9zdCI6Ijg2LjIxMi44NC4xOTEiLCJjbGllbnRBZGRyZXNzIjoiODYuMjEyLjg0LjE5MSIsImNsaWVudF9pZCI6ImNoYXQtYXBpIn0.abjHABcjCJNiB6vriRw60nfzabEfD7CXwyRhkahFkC8ATgfy4fn0T8PnFfsaRpsXuhamIhWwNskrA7L9V3mbWWKW-JEOvImDhTc8sIX0E5fDTbk8O5wa_2yzNdLpRxdSjLqgL544rB8I-LZ8bl5SxtdN3gHfrnWr5ef8bbLgPRzZIylT3QUpah0uywDM_cfrrve9SMHHOUUItyzmOHLw0Igit1EzyFNyjbWf6OAU6TMjOF_eFTc5sakyBwsdJnGy0nhj5R-wxLpr1ug3iEd3Y-jDzHO4m6jariXVJ7Vvbz71i7sadDEKdSyOKcCeQbU4T0Tf7IxqW4c2DNnk3BFFuw"
|
curl https://chat.iceberg.black/api/v1/models -H "Authorization: Bearer eyJhbGciOiJSUzI1NiIsInR5cCIgOiAiSldUIiwia2lkIiA6ICJublpLek04TkZHVmpWbGFPRXZpMUtFSTVHQWRwaGlsYjh3RHRLeG5JOENZIn0.eyJleHAiOjE3NzU4MTUyODksImlhdCI6MTc3NTgxNDk4OSwianRpIjoiMTgwYzA2NDUtYzZiMC00MDRmLTgyYjEtMzA3YmY4ZmJlNmJlIiwiaXNzIjoiaHR0cHM6Ly9hdXRoLmljZWJlcmcuYmxhY2svcmVhbG1zL2ljZWJlcmciLCJzdWIiOiJmZGRiN2FjZC1kMmE5LTRmMTctOWIxNi1kZjVlN2EzNDI4YjciLCJ0eXAiOiJCZWFyZXIiLCJhenAiOiJjaGF0LWFwaSIsInNjb3BlIjoiIiwiY2xpZW50SG9zdCI6Ijg2LjIxMi44NC4xOTEiLCJjbGllbnRBZGRyZXNzIjoiODYuMjEyLjg0LjE5MSIsImNsaWVudF9pZCI6ImNoYXQtYXBpIn0.abjHABcjCJNiB6vriRw60nfzabEfD7CXwyRhkahFkC8ATgfy4fn0T8PnFfsaRpsXuhamIhWwNskrA7L9V3mbWWKW-JEOvImDhTc8sIX0E5fDTbk8O5wa_2yzNdLpRxdSjLqgL544rB8I-LZ8bl5SxtdN3gHfrnWr5ef8bbLgPRzZIylT3QUpah0uywDM_cfrrve9SMHHOUUItyzmOHLw0Igit1EzyFNyjbWf6OAU6TMjOF_eFTc5sakyBwsdJnGy0nhj5R-wxLpr1ug3iEd3Y-jDzHO4m6jariXVJ7Vvbz71i7sadDEKdSyOKcCeQbU4T0Tf7IxqW4c2DNnk3BFFuw"
|
||||||
|
|
||||||
curl -X POST "https://auth.iceberg.black/realms/iceberg/protocol/openid-connect/token" -H "Content-Type: application/x-www-form-urlencoded" -d "grant_type=client_credentials" -d "client_id=chat-api" -d "client_secret=5fHUp8Z5GoNM70MVOGuQTfKFkaAE48Za"
|
curl -X POST "https://auth.iceberg.black/realms/iceberg/protocol/openid-connect/token" -H "Content-Type: application/x-www-form-urlencoded" -d "grant_type=client_credentials" -d "client_id=chat-api" -d "client_secret=5fHUp8Z5GoNM70MVOGuQTfKFkaAE48Za"
|
||||||
@@ -29,19 +27,20 @@ A high-performance Rust API wrapper around Ollama, providing an OpenAI-compatibl
|
|||||||
|
|
||||||
# 🚀 Features
|
# 🚀 Features
|
||||||
|
|
||||||
* ✅ OpenAI-compatible API (`/v1/...`)
|
- ✅ OpenAI-compatible API (`/v1/...`)
|
||||||
* ⚡ Streaming (Server-Sent Events)
|
- ⚡ Streaming (Server-Sent Events)
|
||||||
* 🧠 Model lifecycle management (load/unload)
|
- 🧠 Model lifecycle management (load/unload)
|
||||||
* 🔐 API key authentication (optional)
|
- 🔐 API key authentication (optional)
|
||||||
* 📊 Usage tracking & observability
|
- 📊 Usage tracking & observability
|
||||||
* 🔀 Model routing & abstraction
|
- 🔀 Model routing & abstraction
|
||||||
* 🧩 Extensible architecture (multi-provider ready)
|
- 🧩 Extensible architecture (multi-provider ready)
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
# 📡 API Endpoints
|
# 📡 API Endpoints
|
||||||
|
|
||||||
## 1. Core LLM API (OpenAI-compatible)
|
## 1. Core LLM API (OpenAI-compatible)
|
||||||
|
|
||||||
- [x] `POST /v1/chat/completions` + streaming
|
- [x] `POST /v1/chat/completions` + streaming
|
||||||
- [x] `POST /v1/completions` + streaming
|
- [x] `POST /v1/completions` + streaming
|
||||||
- [ ] `POST /v1/embeddings`
|
- [ ] `POST /v1/embeddings`
|
||||||
@@ -126,16 +125,16 @@ GET /v1/usage
|
|||||||
|
|
||||||
Tracks:
|
Tracks:
|
||||||
|
|
||||||
* request count
|
- request count
|
||||||
* latency
|
- latency
|
||||||
* per-model usage
|
- per-model usage
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
## 🚦 Rate Limiting
|
## 🚦 Rate Limiting
|
||||||
|
|
||||||
* Requests per minute
|
- Requests per minute
|
||||||
* Tokens per minute
|
- Tokens per minute
|
||||||
|
|
||||||
Returns:
|
Returns:
|
||||||
|
|
||||||
@@ -158,8 +157,8 @@ Stores conversation history server-side.
|
|||||||
|
|
||||||
## ⚡ Caching
|
## ⚡ Caching
|
||||||
|
|
||||||
* Embeddings
|
- Embeddings
|
||||||
* Deterministic prompts (temperature = 0)
|
- Deterministic prompts (temperature = 0)
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
@@ -196,8 +195,8 @@ POST /v1/runtime/evict
|
|||||||
|
|
||||||
Strategies:
|
Strategies:
|
||||||
|
|
||||||
* LRU
|
- LRU
|
||||||
* memory threshold
|
- memory threshold
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
@@ -223,10 +222,10 @@ Client → Rust API → Ollama → Response
|
|||||||
|
|
||||||
### Layers:
|
### Layers:
|
||||||
|
|
||||||
* HTTP (Axum)
|
- HTTP (Axum)
|
||||||
* Service layer (business logic)
|
- Service layer (business logic)
|
||||||
* Provider abstraction
|
- Provider abstraction
|
||||||
* Ollama client
|
- Ollama client
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
@@ -241,19 +240,19 @@ trait LlmProvider {
|
|||||||
|
|
||||||
Supports:
|
Supports:
|
||||||
|
|
||||||
* Ollama (current)
|
- Ollama (current)
|
||||||
* OpenAI (future)
|
- OpenAI (future)
|
||||||
* Others
|
- Others
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
# 🎯 Roadmap
|
# 🎯 Roadmap
|
||||||
|
|
||||||
* [ ] Full OpenAI compatibility
|
- [ ] Full OpenAI compatibility
|
||||||
* [ ] Multi-node routing
|
- [ ] Multi-node routing
|
||||||
* [ ] GPU-aware scheduling
|
- [ ] GPU-aware scheduling
|
||||||
* [ ] Web UI dashboard
|
- [ ] Web UI dashboard
|
||||||
* [ ] Distributed inference
|
- [ ] Distributed inference
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
|
|||||||
+2
-2
@@ -28,13 +28,13 @@ pub fn init_tracing() {
|
|||||||
|
|
||||||
#[tokio::main]
|
#[tokio::main]
|
||||||
async fn main() {
|
async fn main() {
|
||||||
init_tracing();
|
|
||||||
|
|
||||||
#[cfg(debug_assertions)]
|
#[cfg(debug_assertions)]
|
||||||
{
|
{
|
||||||
dotenvy::dotenv().ok();
|
dotenvy::dotenv().ok();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
init_tracing();
|
||||||
|
|
||||||
let state = AppState {
|
let state = AppState {
|
||||||
ollama: Arc::new(OllamaProvider::new(OLLAMA_URL.as_str())),
|
ollama: Arc::new(OllamaProvider::new(OLLAMA_URL.as_str())),
|
||||||
};
|
};
|
||||||
|
|||||||
@@ -41,7 +41,7 @@ pub async fn auth_middleware(mut request: Request, next: Next) -> Result<Respons
|
|||||||
Ok(next.run(request).await)
|
Ok(next.run(request).await)
|
||||||
}
|
}
|
||||||
Err(e) => {
|
Err(e) => {
|
||||||
eprintln!("JWT validation failed: {:?}", e);
|
tracing::error!("JWT validation failed: {:?}", e);
|
||||||
Err(StatusCode::UNAUTHORIZED)
|
Err(StatusCode::UNAUTHORIZED)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -51,6 +51,7 @@ pub async fn completions(
|
|||||||
State(state): State<AppState>,
|
State(state): State<AppState>,
|
||||||
Json(body): Json<api::CompletionRequest>,
|
Json(body): Json<api::CompletionRequest>,
|
||||||
) -> Result<Response, (axum::http::StatusCode, Json<api::ErrorResponse>)> {
|
) -> Result<Response, (axum::http::StatusCode, Json<api::ErrorResponse>)> {
|
||||||
|
tracing::debug!("Received /completion with body {:?}", body);
|
||||||
if body.base.stream {
|
if body.base.stream {
|
||||||
let stream = state.ollama.completions_stream(&body).await.map_err(|e| {
|
let stream = state.ollama.completions_stream(&body).await.map_err(|e| {
|
||||||
let (code, msg) = ollama_err(e);
|
let (code, msg) = ollama_err(e);
|
||||||
@@ -116,6 +117,7 @@ pub async fn chat_completions(
|
|||||||
State(state): State<AppState>,
|
State(state): State<AppState>,
|
||||||
Json(body): Json<api::ChatRequest>,
|
Json(body): Json<api::ChatRequest>,
|
||||||
) -> Result<Response, (axum::http::StatusCode, String)> {
|
) -> Result<Response, (axum::http::StatusCode, String)> {
|
||||||
|
tracing::debug!("Received /completion with body {:?}", body);
|
||||||
if body.base.stream {
|
if body.base.stream {
|
||||||
let stream = state
|
let stream = state
|
||||||
.ollama
|
.ollama
|
||||||
|
|||||||
Reference in New Issue
Block a user