feat: add unload llm route

This commit is contained in:
2026-07-14 12:16:42 +02:00
parent 67312cb7a4
commit 387e0a0cfb
5 changed files with 79 additions and 39 deletions
+45 -39
View File
@@ -106,46 +106,52 @@ pub async fn load_model(
}))
}
// #[utoipa::path(
// delete,
// path = "/models/{model}/load",
// tag = "models",
// params(
// ("model" = String, Path, description = "Name of the model to unload from memory (e.g. 'llama3')")
// ),
// responses(
// (
// status = 200,
// description = "Model successfully unloaded from memory",
// body = api::types::UnloadModelResponse,
// content_type = "application/json",
// ),
// (
// status = 404,
// description = "Model not found locally",
// body = api::errors::ErrorResponse,
// example = json!({ "error": "model 'llama3' not found — run `ollama pull llama3`" })
// ),
// (
// status = 500,
// description = "Internal server error (Ollama or network failure)",
// body = api::errors::ErrorResponse,
// example = json!({ "error": "connection refused" })
// )
// )
// )]
// pub async fn unload_model(
// State(state): State<AppState>,
// Path(model): Path<String>,
// ) -> Result<Json<api::types::UnloadModelResponse>, (axum::http::StatusCode, String)> {
// let response = state
// .ollama
// .unload_model(&model)
// .await
// .map_err(into_http_response)?;
#[utoipa::path(
post,
path = "/models/{model}/unload",
tag = "models",
params(
("model" = String, Path, description = "Name of the model to unload from memory (e.g. 'llama3')")
),
responses(
(
status = 200,
description = "Model successfully unloaded from memory",
body = api::types::ApiUnloadModelResponse,
content_type = "application/json",
),
(
status = 404,
description = "Model not found locally",
body = api::errors::ErrorResponse,
example = json!({ "error": "model 'llama3' not found — run `ollama pull llama3`" })
),
(
status = 500,
description = "Internal server error (Ollama or network failure)",
body = api::errors::ErrorResponse,
example = json!({ "error": "connection refused" })
)
)
)]
pub async fn unload_model(
State(state): State<SharedState>,
Path(model): Path<String>,
Json(body): Json<api::types::ApiUnloadModelRequest>,
) -> Result<Json<api::types::ApiUnloadModelResponse>, api::errors::ApiError> {
let response = state
.chat_service
.unload_model(crate::core::llm::models::UnloadModelRequest {
model,
keep_alive: body.keep_alive.clone(),
})
.await?;
// Ok(Json(response))
// }
Ok(Json(api::types::ApiUnloadModelResponse {
model: response.model,
status: "unloaded".to_string(),
}))
}
#[utoipa::path(
post,
+1
View File
@@ -22,6 +22,7 @@ fn llm_router() -> Router<SharedState> {
.route("/completions", post(llm::completions))
.route("/chat/completions", post(llm::chat_completions))
.route("/models/{model}/load", post(llm::load_model))
.route("/models/{model}/unload", post(llm::unload_model))
}
fn keys_router() -> Router<SharedState> {
+5
View File
@@ -47,6 +47,11 @@ pub struct ApiUnloadModelResponse {
pub status: String,
}
#[derive(Debug, Serialize, Deserialize, ToSchema)]
pub struct ApiUnloadModelRequest {
pub keep_alive: String,
}
// ------ Completions ------
#[derive(Debug, Deserialize, Serialize, ToSchema, Default)]
+11
View File
@@ -34,3 +34,14 @@ pub struct LoadModelRequest {
pub struct LoadModelResponse {
pub model: String,
}
#[derive(Debug, Clone)]
pub struct UnloadModelRequest {
pub model: String,
pub keep_alive: String,
}
#[derive(Debug, Clone)]
pub struct UnloadModelResponse {
pub model: String,
}
+17
View File
@@ -50,6 +50,23 @@ impl ChatService {
Ok(crate::core::llm::models::LoadModelResponse { model: body.model })
}
pub async fn unload_model(
&self,
body: crate::core::llm::models::UnloadModelRequest,
) -> Result<crate::core::llm::models::UnloadModelResponse, ServiceError> {
let b = crate::providers::ollama::types::OllamaGenerateRequest {
model: body.model.clone(),
prompt: "unload".to_string(),
stream: false,
keep_alive: body.keep_alive,
options: None,
};
self.ollama.completions(&b).await?;
Ok(crate::core::llm::models::UnloadModelResponse { model: body.model })
}
pub async fn complete(
&self,
body: core::llm::completions::CompletionRequest,