diff --git a/src/api/app.rs b/src/api/app.rs index 94bcfaa..915ca45 100644 --- a/src/api/app.rs +++ b/src/api/app.rs @@ -19,14 +19,14 @@ pub async fn build_app() -> Router { .await .expect("Fatal error"); let conversation_service = ConversationService::new(pool.clone()); - let api_key_service = AuthService::new(pool.clone()); + let auth_service = AuthService::new(pool.clone()); let ollama = OllamaProvider::new(OLLAMA_URL.as_str()); let chat_service = ChatService::new(ollama, conversation_service.clone()); let state = Arc::new(api::state::AppState { conversation_service, - auth_service: api_key_service, + auth_service, chat_service, }); diff --git a/src/api/docs.rs b/src/api/docs.rs index 2cd135a..415bf70 100644 --- a/src/api/docs.rs +++ b/src/api/docs.rs @@ -17,7 +17,7 @@ use crate::api::routes; paths( // routes::v1::chat::completions, // routes::v1::chat::chat_completions, - routes::v1::models::list_models, + routes::v1::llm::list_models, // routes::v1::models::load_model, // routes::v1::models::unload_model, ), diff --git a/src/api/routes/v1/chat.rs b/src/api/routes/v1/llm.rs similarity index 67% rename from src/api/routes/v1/chat.rs rename to src/api/routes/v1/llm.rs index e3a5a5e..a008326 100644 --- a/src/api/routes/v1/chat.rs +++ b/src/api/routes/v1/llm.rs @@ -15,6 +15,138 @@ use axum::{ use futures::StreamExt; use uuid::Uuid; +#[utoipa::path( + get, + path = "/models", + tag = "models", + responses( + ( + status = 200, + description = "List of locally available Ollama models", + body = api::types::ApiModelsResponse, + content_type = "application/json", + ), + ( + status = 500, + description = "Internal server error (Ollama or network failure)", + body = api::errors::ErrorResponse, + example = json!({ "error": "connection refused" }) + ) + ) +)] +// #[axum::debug_handler] +pub async fn list_models( + State(state): State, +) -> Result, api::errors::ApiError> { + let models = state.chat_service.list_models().await?; + + Ok(Json(models.into())) +} + +#[utoipa::path( + post, + path = "/models/{model}/load", + tag = "models", + params( + ("model" = String, Path, description = "Name of the model to load into memory (e.g. 'llama3')") + ), + request_body( + content = api::types::ApiLoadModelRequest, + description = "Load model request", + content_type = "application/json", + example = json!({ "keep_alive": "10m" }) + ), + responses( + ( + status = 200, + description = "Model successfully loaded into memory", + body = api::types::ApiLoadModelResponse, + content_type = "application/json", + ), + ( + status = 400, + description = "Invalid or missing keep_alive format", + body = api::errors::ErrorResponse, + examples( + ("Missing" = (value = json!({ "error": "keep alive is required and cannot be empty" }))), + ("Invalid" = (value = json!({ "error": "invalid keep_alive '10x' — use 30s / 10m / 2h, a plain integer, or -1" }))) + ) + ), + ( + status = 404, + description = "Model not found locally", + body = api::errors::ErrorResponse, + example = json!({ "error": "model 'llama3' not found — run `ollama pull llama3`" }) + ), + ( + status = 500, + description = "Internal server error (Ollama or network failure)", + body = api::errors::ErrorResponse, + example = json!({ "error": "connection refused" }) + ) + ) +)] +pub async fn load_model( + State(state): State, + Path(model): Path, + Json(body): Json, +) -> Result, api::errors::ApiError> { + let response = state + .chat_service + .load_model(crate::core::llm::models::LoadModelRequest { + model, + keep_alive: body.keep_alive.clone(), + }) + .await?; + + Ok(Json(api::types::ApiLoadModelResponse { + model: response.model, + keep_alive: body.keep_alive, + status: "loaded".to_string(), + })) +} + +// #[utoipa::path( +// delete, +// path = "/models/{model}/load", +// tag = "models", +// params( +// ("model" = String, Path, description = "Name of the model to unload from memory (e.g. 'llama3')") +// ), +// responses( +// ( +// status = 200, +// description = "Model successfully unloaded from memory", +// body = api::types::UnloadModelResponse, +// content_type = "application/json", +// ), +// ( +// status = 404, +// description = "Model not found locally", +// body = api::errors::ErrorResponse, +// example = json!({ "error": "model 'llama3' not found — run `ollama pull llama3`" }) +// ), +// ( +// status = 500, +// description = "Internal server error (Ollama or network failure)", +// body = api::errors::ErrorResponse, +// example = json!({ "error": "connection refused" }) +// ) +// ) +// )] +// pub async fn unload_model( +// State(state): State, +// Path(model): Path, +// ) -> Result, (axum::http::StatusCode, String)> { +// let response = state +// .ollama +// .unload_model(&model) +// .await +// .map_err(into_http_response)?; + +// Ok(Json(response)) +// } + #[utoipa::path( post, path = "/completions", diff --git a/src/api/routes/v1/mod.rs b/src/api/routes/v1/mod.rs index dce662a..e72836c 100644 --- a/src/api/routes/v1/mod.rs +++ b/src/api/routes/v1/mod.rs @@ -1,6 +1,5 @@ pub mod apikey; -pub mod chat; -pub mod models; +pub mod llm; use crate::api::docs::ApiDoc; use crate::api::middlewares::auth::auth_middleware; @@ -23,20 +22,20 @@ fn public_router() -> Router { pub fn protected_router() -> Router { Router::new() - .route("/models", get(models::list_models)) - .route("/completions", post(chat::completions)) - .route("/chat/completions", post(chat::chat_completions)) - .route("/models/{model}/load", post(models::load_model)) + .route("/models", get(llm::list_models)) + .route("/completions", post(llm::completions)) + .route("/chat/completions", post(llm::chat_completions)) + .route("/models/{model}/load", post(llm::load_model)) // .route("/models/{model}/unload", post(models::unload_model)) .route( "/keys/generate", post(apikey::create_api_key) // Usage .route_layer(role_guard!(Some("admin"), None)), ) - .route("/conversations", get(chat::get_conversations)) + .route("/conversations", get(llm::get_conversations)) .route( "/conversations/{conversation_id}/messages", - get(chat::get_messages), + get(llm::get_messages), ) } diff --git a/src/api/routes/v1/models.rs b/src/api/routes/v1/models.rs deleted file mode 100644 index 63e4347..0000000 --- a/src/api/routes/v1/models.rs +++ /dev/null @@ -1,138 +0,0 @@ -use crate::api::{self, state::SharedState}; - -use axum::{ - Json, - extract::{Path, State}, -}; - -#[utoipa::path( - get, - path = "/models", - tag = "models", - responses( - ( - status = 200, - description = "List of locally available Ollama models", - body = api::types::ApiModelsResponse, - content_type = "application/json", - ), - ( - status = 500, - description = "Internal server error (Ollama or network failure)", - body = api::errors::ErrorResponse, - example = json!({ "error": "connection refused" }) - ) - ) -)] -// #[axum::debug_handler] -pub async fn list_models( - State(state): State, -) -> Result, api::errors::ApiError> { - let models = state.chat_service.list_models().await?; - - Ok(Json(models.into())) -} - -#[utoipa::path( - post, - path = "/models/{model}/load", - tag = "models", - params( - ("model" = String, Path, description = "Name of the model to load into memory (e.g. 'llama3')") - ), - request_body( - content = api::types::ApiLoadModelRequest, - description = "Load model request", - content_type = "application/json", - example = json!({ "keep_alive": "10m" }) - ), - responses( - ( - status = 200, - description = "Model successfully loaded into memory", - body = api::types::ApiLoadModelResponse, - content_type = "application/json", - ), - ( - status = 400, - description = "Invalid or missing keep_alive format", - body = api::errors::ErrorResponse, - examples( - ("Missing" = (value = json!({ "error": "keep alive is required and cannot be empty" }))), - ("Invalid" = (value = json!({ "error": "invalid keep_alive '10x' — use 30s / 10m / 2h, a plain integer, or -1" }))) - ) - ), - ( - status = 404, - description = "Model not found locally", - body = api::errors::ErrorResponse, - example = json!({ "error": "model 'llama3' not found — run `ollama pull llama3`" }) - ), - ( - status = 500, - description = "Internal server error (Ollama or network failure)", - body = api::errors::ErrorResponse, - example = json!({ "error": "connection refused" }) - ) - ) -)] -pub async fn load_model( - State(state): State, - Path(model): Path, - Json(body): Json, -) -> Result, api::errors::ApiError> { - let response = state - .chat_service - .load_model(crate::core::llm::models::LoadModelRequest { - model, - keep_alive: body.keep_alive.clone(), - }) - .await?; - - Ok(Json(api::types::ApiLoadModelResponse { - model: response.model, - keep_alive: body.keep_alive, - status: "loaded".to_string(), - })) -} - -// #[utoipa::path( -// delete, -// path = "/models/{model}/load", -// tag = "models", -// params( -// ("model" = String, Path, description = "Name of the model to unload from memory (e.g. 'llama3')") -// ), -// responses( -// ( -// status = 200, -// description = "Model successfully unloaded from memory", -// body = api::types::UnloadModelResponse, -// content_type = "application/json", -// ), -// ( -// status = 404, -// description = "Model not found locally", -// body = api::errors::ErrorResponse, -// example = json!({ "error": "model 'llama3' not found — run `ollama pull llama3`" }) -// ), -// ( -// status = 500, -// description = "Internal server error (Ollama or network failure)", -// body = api::errors::ErrorResponse, -// example = json!({ "error": "connection refused" }) -// ) -// ) -// )] -// pub async fn unload_model( -// State(state): State, -// Path(model): Path, -// ) -> Result, (axum::http::StatusCode, String)> { -// let response = state -// .ollama -// .unload_model(&model) -// .await -// .map_err(into_http_response)?; - -// Ok(Json(response)) -// }