feat: add load endpoint

This commit is contained in:
2026-04-10 12:29:34 +02:00
parent db14d816cb
commit 1a06577fa3
4 changed files with 114 additions and 6 deletions
+8
View File
@@ -5,10 +5,18 @@ use thiserror::Error;
pub enum OllamaError { pub enum OllamaError {
#[error("prompt is required and cannot be empty")] #[error("prompt is required and cannot be empty")]
MissingPrompt, MissingPrompt,
#[error("messages must be a non-empty array containing at least one user message")] #[error("messages must be a non-empty array containing at least one user message")]
MissingMessages, MissingMessages,
#[error("model '{0}' is not available — run `ollama pull {0}` first")] #[error("model '{0}' is not available — run `ollama pull {0}` first")]
ModelNotFound(String), ModelNotFound(String),
#[error(
"invalid keep_alive format '{0}' — expected <number><unit> (e.g. 30s, 10m, 2h), a plain integer (seconds), or -1"
)]
InvalidKeepAlive(String),
#[error(transparent)] #[error(transparent)]
Http(#[from] reqwest::Error), Http(#[from] reqwest::Error),
} }
+56
View File
@@ -43,6 +43,29 @@ impl OllamaProvider {
Ok(()) Ok(())
} }
fn parse_keep_alive(s: &str) -> Result<(), OllamaError> {
let s = s.trim();
// Ollama also accepts plain integers (seconds) or "-1" (load forever)
if s == "-1" || s.parse::<u64>().is_ok() {
return Ok(());
}
// Otherwise expect: <number><unit> e.g. "10m", "2h", "30s"
let (num, unit) = s
.find(|c: char| c.is_alphabetic())
.map(|i| s.split_at(i))
.ok_or_else(|| OllamaError::InvalidKeepAlive(s.to_string()))?;
num.parse::<u64>()
.map_err(|_| OllamaError::InvalidKeepAlive(s.to_string()))?;
match unit {
"s" | "m" | "h" => Ok(()),
_ => Err(OllamaError::InvalidKeepAlive(s.to_string())),
}
}
// ── public endpoints ───────────────────────────────────────────────────── // ── public endpoints ─────────────────────────────────────────────────────
pub async fn list_models(&self) -> Result<Value, OllamaError> { pub async fn list_models(&self) -> Result<Value, OllamaError> {
@@ -51,6 +74,39 @@ impl OllamaProvider {
Ok(res) Ok(res)
} }
pub async fn load_model(
&self,
model: &str,
keep_alive: Option<&str>,
) -> Result<Value, OllamaError> {
self.validate_model(model).await?;
let keep_alive = keep_alive.unwrap_or("5m");
Self::parse_keep_alive(keep_alive)?; // ← validated before any network call
let payload = json!({
"model": model,
"prompt": "",
"keep_alive": keep_alive,
"stream": false,
});
let res = self
.client
.post(format!("{}/api/generate", self.base_url))
.json(&payload)
.send()
.await?
.json::<Value>()
.await?;
Ok(json!({
"model": res.get("model"),
"status": "loaded",
"keep_alive": keep_alive,
}))
}
pub async fn completions(&self, body: Value) -> Result<Value, OllamaError> { pub async fn completions(&self, body: Value) -> Result<Value, OllamaError> {
let prompt = body let prompt = body
.get("prompt") .get("prompt")
+1
View File
@@ -10,5 +10,6 @@ pub fn router() -> Router<AppState> {
.route("/models", get(models::list_models)) .route("/models", get(models::list_models))
.route("/completions", post(chat::completions)) .route("/completions", post(chat::completions))
.route("/chat/completions", post(chat::chat_completions)) .route("/chat/completions", post(chat::chat_completions))
.route("/models/{model}/load", post(models::load_model))
.layer(middleware::from_fn(auth_middleware)) .layer(middleware::from_fn(auth_middleware))
} }
+49 -6
View File
@@ -1,16 +1,59 @@
use axum::{Json, extract::State}; use axum::{
Json,
extract::{Path, State},
};
use serde::Deserialize;
use serde_json::Value; use serde_json::Value;
use crate::state::app_state::AppState; use crate::{errors::OllamaError, state::app_state::AppState};
#[derive(Deserialize)]
pub struct LoadModelBody {
pub keep_alive: Option<String>,
}
pub async fn list_models( pub async fn list_models(
State(state): State<AppState>, State(state): State<AppState>,
) -> Result<Json<Value>, (axum::http::StatusCode, String)> { ) -> Result<Json<Value>, (axum::http::StatusCode, String)> {
match state.ollama.list_models().await { match state.ollama.list_models().await {
Ok(models) => Ok(Json(models)), Ok(models) => Ok(Json(models)),
Err(err) => Err(( Err(e) => Err(ollama_err(e)),
axum::http::StatusCode::INTERNAL_SERVER_ERROR, }
err.to_string(), }
)),
pub async fn load_model(
State(state): State<AppState>,
Path(model): Path<String>,
Json(body): Json<LoadModelBody>,
) -> Result<Json<Value>, (axum::http::StatusCode, String)> {
match state
.ollama
.load_model(&model, body.keep_alive.as_deref())
.await
{
Ok(response) => Ok(Json(response)),
Err(e) => Err(ollama_err(e)),
}
}
fn ollama_err(e: OllamaError) -> (axum::http::StatusCode, String) {
match e {
OllamaError::ModelNotFound(m) => (
axum::http::StatusCode::NOT_FOUND,
format!("model '{m}' not found — run `ollama pull {m}`"),
),
OllamaError::MissingPrompt => (
axum::http::StatusCode::BAD_REQUEST,
"prompt is required".to_string(),
),
OllamaError::MissingMessages => (
axum::http::StatusCode::BAD_REQUEST,
"messages array with at least one user message is required".to_string(),
),
OllamaError::InvalidKeepAlive(v) => (
axum::http::StatusCode::BAD_REQUEST,
format!("invalid keep_alive '{v}' — use 30s / 10m / 2h, a plain integer, or -1"),
),
OllamaError::Http(e) => (axum::http::StatusCode::INTERNAL_SERVER_ERROR, e.to_string()),
} }
} }