feat: prepare tts
This commit is contained in:
+2
-2
@@ -19,14 +19,14 @@ pub async fn build_app() -> Router {
|
||||
.await
|
||||
.expect("Fatal error");
|
||||
let conversation_service = ConversationService::new(pool.clone());
|
||||
let api_key_service = AuthService::new(pool.clone());
|
||||
let auth_service = AuthService::new(pool.clone());
|
||||
|
||||
let ollama = OllamaProvider::new(OLLAMA_URL.as_str());
|
||||
let chat_service = ChatService::new(ollama, conversation_service.clone());
|
||||
|
||||
let state = Arc::new(api::state::AppState {
|
||||
conversation_service,
|
||||
auth_service: api_key_service,
|
||||
auth_service,
|
||||
chat_service,
|
||||
});
|
||||
|
||||
|
||||
+1
-1
@@ -17,7 +17,7 @@ use crate::api::routes;
|
||||
paths(
|
||||
// routes::v1::chat::completions,
|
||||
// routes::v1::chat::chat_completions,
|
||||
routes::v1::models::list_models,
|
||||
routes::v1::llm::list_models,
|
||||
// routes::v1::models::load_model,
|
||||
// routes::v1::models::unload_model,
|
||||
),
|
||||
|
||||
@@ -15,6 +15,138 @@ use axum::{
|
||||
use futures::StreamExt;
|
||||
use uuid::Uuid;
|
||||
|
||||
#[utoipa::path(
|
||||
get,
|
||||
path = "/models",
|
||||
tag = "models",
|
||||
responses(
|
||||
(
|
||||
status = 200,
|
||||
description = "List of locally available Ollama models",
|
||||
body = api::types::ApiModelsResponse,
|
||||
content_type = "application/json",
|
||||
),
|
||||
(
|
||||
status = 500,
|
||||
description = "Internal server error (Ollama or network failure)",
|
||||
body = api::errors::ErrorResponse,
|
||||
example = json!({ "error": "connection refused" })
|
||||
)
|
||||
)
|
||||
)]
|
||||
// #[axum::debug_handler]
|
||||
pub async fn list_models(
|
||||
State(state): State<SharedState>,
|
||||
) -> Result<Json<api::types::ApiModelsResponse>, api::errors::ApiError> {
|
||||
let models = state.chat_service.list_models().await?;
|
||||
|
||||
Ok(Json(models.into()))
|
||||
}
|
||||
|
||||
#[utoipa::path(
|
||||
post,
|
||||
path = "/models/{model}/load",
|
||||
tag = "models",
|
||||
params(
|
||||
("model" = String, Path, description = "Name of the model to load into memory (e.g. 'llama3')")
|
||||
),
|
||||
request_body(
|
||||
content = api::types::ApiLoadModelRequest,
|
||||
description = "Load model request",
|
||||
content_type = "application/json",
|
||||
example = json!({ "keep_alive": "10m" })
|
||||
),
|
||||
responses(
|
||||
(
|
||||
status = 200,
|
||||
description = "Model successfully loaded into memory",
|
||||
body = api::types::ApiLoadModelResponse,
|
||||
content_type = "application/json",
|
||||
),
|
||||
(
|
||||
status = 400,
|
||||
description = "Invalid or missing keep_alive format",
|
||||
body = api::errors::ErrorResponse,
|
||||
examples(
|
||||
("Missing" = (value = json!({ "error": "keep alive is required and cannot be empty" }))),
|
||||
("Invalid" = (value = json!({ "error": "invalid keep_alive '10x' — use 30s / 10m / 2h, a plain integer, or -1" })))
|
||||
)
|
||||
),
|
||||
(
|
||||
status = 404,
|
||||
description = "Model not found locally",
|
||||
body = api::errors::ErrorResponse,
|
||||
example = json!({ "error": "model 'llama3' not found — run `ollama pull llama3`" })
|
||||
),
|
||||
(
|
||||
status = 500,
|
||||
description = "Internal server error (Ollama or network failure)",
|
||||
body = api::errors::ErrorResponse,
|
||||
example = json!({ "error": "connection refused" })
|
||||
)
|
||||
)
|
||||
)]
|
||||
pub async fn load_model(
|
||||
State(state): State<SharedState>,
|
||||
Path(model): Path<String>,
|
||||
Json(body): Json<api::types::ApiLoadModelRequest>,
|
||||
) -> Result<Json<api::types::ApiLoadModelResponse>, api::errors::ApiError> {
|
||||
let response = state
|
||||
.chat_service
|
||||
.load_model(crate::core::llm::models::LoadModelRequest {
|
||||
model,
|
||||
keep_alive: body.keep_alive.clone(),
|
||||
})
|
||||
.await?;
|
||||
|
||||
Ok(Json(api::types::ApiLoadModelResponse {
|
||||
model: response.model,
|
||||
keep_alive: body.keep_alive,
|
||||
status: "loaded".to_string(),
|
||||
}))
|
||||
}
|
||||
|
||||
// #[utoipa::path(
|
||||
// delete,
|
||||
// path = "/models/{model}/load",
|
||||
// tag = "models",
|
||||
// params(
|
||||
// ("model" = String, Path, description = "Name of the model to unload from memory (e.g. 'llama3')")
|
||||
// ),
|
||||
// responses(
|
||||
// (
|
||||
// status = 200,
|
||||
// description = "Model successfully unloaded from memory",
|
||||
// body = api::types::UnloadModelResponse,
|
||||
// content_type = "application/json",
|
||||
// ),
|
||||
// (
|
||||
// status = 404,
|
||||
// description = "Model not found locally",
|
||||
// body = api::errors::ErrorResponse,
|
||||
// example = json!({ "error": "model 'llama3' not found — run `ollama pull llama3`" })
|
||||
// ),
|
||||
// (
|
||||
// status = 500,
|
||||
// description = "Internal server error (Ollama or network failure)",
|
||||
// body = api::errors::ErrorResponse,
|
||||
// example = json!({ "error": "connection refused" })
|
||||
// )
|
||||
// )
|
||||
// )]
|
||||
// pub async fn unload_model(
|
||||
// State(state): State<AppState>,
|
||||
// Path(model): Path<String>,
|
||||
// ) -> Result<Json<api::types::UnloadModelResponse>, (axum::http::StatusCode, String)> {
|
||||
// let response = state
|
||||
// .ollama
|
||||
// .unload_model(&model)
|
||||
// .await
|
||||
// .map_err(into_http_response)?;
|
||||
|
||||
// Ok(Json(response))
|
||||
// }
|
||||
|
||||
#[utoipa::path(
|
||||
post,
|
||||
path = "/completions",
|
||||
@@ -1,6 +1,5 @@
|
||||
pub mod apikey;
|
||||
pub mod chat;
|
||||
pub mod models;
|
||||
pub mod llm;
|
||||
|
||||
use crate::api::docs::ApiDoc;
|
||||
use crate::api::middlewares::auth::auth_middleware;
|
||||
@@ -23,20 +22,20 @@ fn public_router() -> Router<SharedState> {
|
||||
|
||||
pub fn protected_router() -> Router<SharedState> {
|
||||
Router::new()
|
||||
.route("/models", get(models::list_models))
|
||||
.route("/completions", post(chat::completions))
|
||||
.route("/chat/completions", post(chat::chat_completions))
|
||||
.route("/models/{model}/load", post(models::load_model))
|
||||
.route("/models", get(llm::list_models))
|
||||
.route("/completions", post(llm::completions))
|
||||
.route("/chat/completions", post(llm::chat_completions))
|
||||
.route("/models/{model}/load", post(llm::load_model))
|
||||
// .route("/models/{model}/unload", post(models::unload_model))
|
||||
.route(
|
||||
"/keys/generate",
|
||||
post(apikey::create_api_key) // Usage
|
||||
.route_layer(role_guard!(Some("admin"), None)),
|
||||
)
|
||||
.route("/conversations", get(chat::get_conversations))
|
||||
.route("/conversations", get(llm::get_conversations))
|
||||
.route(
|
||||
"/conversations/{conversation_id}/messages",
|
||||
get(chat::get_messages),
|
||||
get(llm::get_messages),
|
||||
)
|
||||
}
|
||||
|
||||
|
||||
@@ -1,138 +0,0 @@
|
||||
use crate::api::{self, state::SharedState};
|
||||
|
||||
use axum::{
|
||||
Json,
|
||||
extract::{Path, State},
|
||||
};
|
||||
|
||||
#[utoipa::path(
|
||||
get,
|
||||
path = "/models",
|
||||
tag = "models",
|
||||
responses(
|
||||
(
|
||||
status = 200,
|
||||
description = "List of locally available Ollama models",
|
||||
body = api::types::ApiModelsResponse,
|
||||
content_type = "application/json",
|
||||
),
|
||||
(
|
||||
status = 500,
|
||||
description = "Internal server error (Ollama or network failure)",
|
||||
body = api::errors::ErrorResponse,
|
||||
example = json!({ "error": "connection refused" })
|
||||
)
|
||||
)
|
||||
)]
|
||||
// #[axum::debug_handler]
|
||||
pub async fn list_models(
|
||||
State(state): State<SharedState>,
|
||||
) -> Result<Json<api::types::ApiModelsResponse>, api::errors::ApiError> {
|
||||
let models = state.chat_service.list_models().await?;
|
||||
|
||||
Ok(Json(models.into()))
|
||||
}
|
||||
|
||||
#[utoipa::path(
|
||||
post,
|
||||
path = "/models/{model}/load",
|
||||
tag = "models",
|
||||
params(
|
||||
("model" = String, Path, description = "Name of the model to load into memory (e.g. 'llama3')")
|
||||
),
|
||||
request_body(
|
||||
content = api::types::ApiLoadModelRequest,
|
||||
description = "Load model request",
|
||||
content_type = "application/json",
|
||||
example = json!({ "keep_alive": "10m" })
|
||||
),
|
||||
responses(
|
||||
(
|
||||
status = 200,
|
||||
description = "Model successfully loaded into memory",
|
||||
body = api::types::ApiLoadModelResponse,
|
||||
content_type = "application/json",
|
||||
),
|
||||
(
|
||||
status = 400,
|
||||
description = "Invalid or missing keep_alive format",
|
||||
body = api::errors::ErrorResponse,
|
||||
examples(
|
||||
("Missing" = (value = json!({ "error": "keep alive is required and cannot be empty" }))),
|
||||
("Invalid" = (value = json!({ "error": "invalid keep_alive '10x' — use 30s / 10m / 2h, a plain integer, or -1" })))
|
||||
)
|
||||
),
|
||||
(
|
||||
status = 404,
|
||||
description = "Model not found locally",
|
||||
body = api::errors::ErrorResponse,
|
||||
example = json!({ "error": "model 'llama3' not found — run `ollama pull llama3`" })
|
||||
),
|
||||
(
|
||||
status = 500,
|
||||
description = "Internal server error (Ollama or network failure)",
|
||||
body = api::errors::ErrorResponse,
|
||||
example = json!({ "error": "connection refused" })
|
||||
)
|
||||
)
|
||||
)]
|
||||
pub async fn load_model(
|
||||
State(state): State<SharedState>,
|
||||
Path(model): Path<String>,
|
||||
Json(body): Json<api::types::ApiLoadModelRequest>,
|
||||
) -> Result<Json<api::types::ApiLoadModelResponse>, api::errors::ApiError> {
|
||||
let response = state
|
||||
.chat_service
|
||||
.load_model(crate::core::llm::models::LoadModelRequest {
|
||||
model,
|
||||
keep_alive: body.keep_alive.clone(),
|
||||
})
|
||||
.await?;
|
||||
|
||||
Ok(Json(api::types::ApiLoadModelResponse {
|
||||
model: response.model,
|
||||
keep_alive: body.keep_alive,
|
||||
status: "loaded".to_string(),
|
||||
}))
|
||||
}
|
||||
|
||||
// #[utoipa::path(
|
||||
// delete,
|
||||
// path = "/models/{model}/load",
|
||||
// tag = "models",
|
||||
// params(
|
||||
// ("model" = String, Path, description = "Name of the model to unload from memory (e.g. 'llama3')")
|
||||
// ),
|
||||
// responses(
|
||||
// (
|
||||
// status = 200,
|
||||
// description = "Model successfully unloaded from memory",
|
||||
// body = api::types::UnloadModelResponse,
|
||||
// content_type = "application/json",
|
||||
// ),
|
||||
// (
|
||||
// status = 404,
|
||||
// description = "Model not found locally",
|
||||
// body = api::errors::ErrorResponse,
|
||||
// example = json!({ "error": "model 'llama3' not found — run `ollama pull llama3`" })
|
||||
// ),
|
||||
// (
|
||||
// status = 500,
|
||||
// description = "Internal server error (Ollama or network failure)",
|
||||
// body = api::errors::ErrorResponse,
|
||||
// example = json!({ "error": "connection refused" })
|
||||
// )
|
||||
// )
|
||||
// )]
|
||||
// pub async fn unload_model(
|
||||
// State(state): State<AppState>,
|
||||
// Path(model): Path<String>,
|
||||
// ) -> Result<Json<api::types::UnloadModelResponse>, (axum::http::StatusCode, String)> {
|
||||
// let response = state
|
||||
// .ollama
|
||||
// .unload_model(&model)
|
||||
// .await
|
||||
// .map_err(into_http_response)?;
|
||||
|
||||
// Ok(Json(response))
|
||||
// }
|
||||
Reference in New Issue
Block a user