diff --git a/src/api/docs.rs b/src/api/docs.rs index 415bf70..54866b7 100644 --- a/src/api/docs.rs +++ b/src/api/docs.rs @@ -15,33 +15,39 @@ use crate::api::routes; ), ), paths( - // routes::v1::chat::completions, - // routes::v1::chat::chat_completions, routes::v1::llm::list_models, - // routes::v1::models::load_model, - // routes::v1::models::unload_model, + routes::v1::llm::completions, + routes::v1::llm::chat_completions, + routes::v1::llm::load_model, + routes::v1::llm::unload_model, ), components( schemas( api::errors::ErrorResponse, + api::types::ApiModelsResponse, api::types::ApiModelInfo, + api::types::ApiModelMetadata, + api::types::ApiLoadModelResponse, api::types::ApiLoadModelRequest, api::types::ApiUnloadModelResponse, + api::types::ApiLlmOptions, + api::types::ApiCompletionRequest, - api::types::ApiCompletionObject, - api::types::ApiFinishReason, api::types::ApiCompletionResponse, + api::types::ApiCompletionObject, api::types::Choice, api::types::Usage, - api::types::CompletionChunk, + api::types::ApiFinishReason, + api::types::ApiChatRequest, api::types::ApiMessage, api::types::ApiRole, api::types::ApiChatResponse, api::types::ApiChatChoice, + api::types::ChatCompletionChunk, api::types::ChatChunkChoice, api::types::Delta, diff --git a/src/api/routes/v1/llm.rs b/src/api/routes/v1/llm.rs index fb7688e..d456fb0 100644 --- a/src/api/routes/v1/llm.rs +++ b/src/api/routes/v1/llm.rs @@ -17,7 +17,7 @@ use uuid::Uuid; #[utoipa::path( get, - path = "/models", + path = "/llm/models", tag = "models", responses( ( @@ -34,7 +34,6 @@ use uuid::Uuid; ) ) )] -// #[axum::debug_handler] pub async fn list_models( State(state): State, ) -> Result, api::errors::ApiError> { @@ -45,7 +44,7 @@ pub async fn list_models( #[utoipa::path( post, - path = "/models/{model}/load", + path = "/llm/models/{model}/load", tag = "models", params( ("model" = String, Path, description = "Name of the model to load into memory (e.g. 'llama3')") @@ -108,7 +107,7 @@ pub async fn load_model( #[utoipa::path( post, - path = "/models/{model}/unload", + path = "/llm/models/{model}/unload", tag = "models", params( ("model" = String, Path, description = "Name of the model to unload from memory (e.g. 'llama3')") @@ -155,7 +154,7 @@ pub async fn unload_model( #[utoipa::path( post, - path = "/completions", + path = "/llm/completions", tag = "chat", request_body( content = api::types::ApiCompletionRequest, @@ -219,7 +218,7 @@ pub async fn completions( #[utoipa::path( post, - path = "/chat/completions", + path = "/llm/chat/completions", tag = "chat", request_body( content = api::types::ApiChatRequest, diff --git a/src/api/types.rs b/src/api/types.rs index 843906d..7a9d1fb 100644 --- a/src/api/types.rs +++ b/src/api/types.rs @@ -129,12 +129,12 @@ pub struct Usage { pub total_tokens: u32, } -#[derive(Debug, Serialize, Deserialize, ToSchema)] -pub struct CompletionChunk { - pub id: String, - pub object: String, - pub choices: Vec, -} +// #[derive(Debug, Serialize, Deserialize, ToSchema)] +// pub struct CompletionChunk { +// pub id: String, +// pub object: String, +// pub choices: Vec, +// } #[derive(Debug, Deserialize, Serialize, ToSchema)] pub struct ApiChatRequest {