use crate::api::{self, state::SharedState}; use axum::{ Json, extract::{Path, State}, }; #[utoipa::path( get, path = "/models", tag = "models", responses( ( status = 200, description = "List of locally available Ollama models", body = api::types::ApiModelsResponse, content_type = "application/json", ), ( status = 500, description = "Internal server error (Ollama or network failure)", body = api::errors::ErrorResponse, example = json!({ "error": "connection refused" }) ) ) )] // #[axum::debug_handler] pub async fn list_models( State(state): State, ) -> Result, api::errors::ApiError> { let models = state.chat_service.list_models().await?; Ok(Json(models.into())) } #[utoipa::path( post, path = "/models/{model}/load", tag = "models", params( ("model" = String, Path, description = "Name of the model to load into memory (e.g. 'llama3')") ), request_body( content = api::types::ApiLoadModelRequest, description = "Load model request", content_type = "application/json", example = json!({ "keep_alive": "10m" }) ), responses( ( status = 200, description = "Model successfully loaded into memory", body = api::types::ApiLoadModelResponse, content_type = "application/json", ), ( status = 400, description = "Invalid or missing keep_alive format", body = api::errors::ErrorResponse, examples( ("Missing" = (value = json!({ "error": "keep alive is required and cannot be empty" }))), ("Invalid" = (value = json!({ "error": "invalid keep_alive '10x' — use 30s / 10m / 2h, a plain integer, or -1" }))) ) ), ( status = 404, description = "Model not found locally", body = api::errors::ErrorResponse, example = json!({ "error": "model 'llama3' not found — run `ollama pull llama3`" }) ), ( status = 500, description = "Internal server error (Ollama or network failure)", body = api::errors::ErrorResponse, example = json!({ "error": "connection refused" }) ) ) )] pub async fn load_model( State(state): State, Path(model): Path, Json(body): Json, ) -> Result, api::errors::ApiError> { let response = state .chat_service .load_model(crate::core::llm::models::LoadModelRequest { model, keep_alive: body.keep_alive.clone(), }) .await?; Ok(Json(api::types::ApiLoadModelResponse { model: response.model, keep_alive: body.keep_alive, status: "loaded".to_string(), })) } // #[utoipa::path( // delete, // path = "/models/{model}/load", // tag = "models", // params( // ("model" = String, Path, description = "Name of the model to unload from memory (e.g. 'llama3')") // ), // responses( // ( // status = 200, // description = "Model successfully unloaded from memory", // body = api::types::UnloadModelResponse, // content_type = "application/json", // ), // ( // status = 404, // description = "Model not found locally", // body = api::errors::ErrorResponse, // example = json!({ "error": "model 'llama3' not found — run `ollama pull llama3`" }) // ), // ( // status = 500, // description = "Internal server error (Ollama or network failure)", // body = api::errors::ErrorResponse, // example = json!({ "error": "connection refused" }) // ) // ) // )] // pub async fn unload_model( // State(state): State, // Path(model): Path, // ) -> Result, (axum::http::StatusCode, String)> { // let response = state // .ollama // .unload_model(&model) // .await // .map_err(into_http_response)?; // Ok(Json(response)) // }