From 4d6acab6a138cc6d48447d43f1f5f7107a138187 Mon Sep 17 00:00:00 2001 From: flemming-it Date: Thu, 20 Aug 2026 17:13:42 +0200 Subject: [PATCH] feat: api accepts 'vllm' as a first-class value MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 'openai' reads as the cloud company — an operator running self-hosted vLLM should be able to write what they mean. The new value speaks the identical OpenAI wire (a guard test asserts the emitted request stays byte-identical to api: openai, so the alias can never drift into a dialect) but yields vLLM-specific error hints (vllm serve, port 8000) instead of generic OpenAI prose. Manifest + docs name the value in DE and EN. Signed-off-by: flemming-it --- Cargo.lock | 2 +- MODULE.de.md | 2 +- MODULE.md | 2 +- module.yaml | 19 +++++++++++++------ src/lib.rs | 15 ++++++++++----- src/llm.rs | 40 +++++++++++++++++++++++++++++++++++----- 6 files changed, 61 insertions(+), 19 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index fec05ad..cae6529 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -265,7 +265,7 @@ dependencies = [ [[package]] name = "text_translate" -version = "0.1.0" +version = "0.2.0" dependencies = [ "chain-module-sdk", "serde", diff --git a/MODULE.de.md b/MODULE.de.md index 144c399..aad0605 100644 --- a/MODULE.de.md +++ b/MODULE.de.md @@ -17,7 +17,7 @@ audit-taugliche Modell-Herkunfts-Felder. | `target_language` | text | Zielsprache in einfachem Englisch (z.B. `German`, `French`, `ja-JP`). | | `source_language` | text | Optionaler Hinweis auf die Ausgangssprache. Leer = Modell erkennt selbst. | | `endpoint` | text | Chat-Endpunkt passend zum gewählten `api` (Ollama `/api/chat`, OpenAI-kompatibel `/v1/chat/completions` — vLLM u. a., Anthropic `/v1/messages`). | -| `api` | text | Optionales Wire-Format: `ollama` (Default), `openai`, `anthropic`. | +| `api` | text | Optionales Wire-Format: `ollama` (Default), `vllm`, `openai`, `anthropic` (`vllm` = OpenAI-Wire mit vLLM-spezifischen Hinweisen). | | `model` | text | Modell-ID am Endpunkt (z.B. `qwen2.5:14b`). | | `api_key` | text | Optionaler Bearer-Token für Cloud-Endpunkte. | diff --git a/MODULE.md b/MODULE.md index a914b7c..5382031 100644 --- a/MODULE.md +++ b/MODULE.md @@ -17,7 +17,7 @@ model-provenance fields. | `target_language` | text | Target language in plain English (e.g. `German`, `French`, `ja-JP`). | | `source_language` | text | Optional source-language hint. Empty = let the model auto-detect. | | `endpoint` | text | Chat endpoint URL matching the selected `api` (Ollama `/api/chat`, OpenAI-compatible `/v1/chat/completions` — vLLM etc., Anthropic `/v1/messages`). | -| `api` | text | Optional wire format: `ollama` (default), `openai`, `anthropic`. | +| `api` | text | Optional wire format: `ollama` (default), `vllm`, `openai`, `anthropic` (`vllm` = the OpenAI wire with vLLM-specific hints). | | `model` | text | Model identifier the endpoint serves (e.g. `qwen2.5:14b`). | | `api_key` | text | Optional bearer token for cloud-hosted endpoints. | diff --git a/module.yaml b/module.yaml index 7bf3611..c47fa8a 100644 --- a/module.yaml +++ b/module.yaml @@ -49,13 +49,20 @@ inputs: type: text description: en: | - Optional wire format: "ollama" (default), "openai" - (OpenAI-compatible servers such as vLLM or LM Studio), - or "anthropic" (Messages API). Empty = ollama. + Optional wire format: "ollama" (default), "vllm" + (self-hosted vLLM), "openai" (OpenAI or other + OpenAI-compatible servers such as LM Studio), or + "anthropic" (Messages API). vllm and openai speak the + same wire; the separate value exists so you can write + what you mean and get vLLM-specific hints. Empty = ollama. de: | - Optionales Wire-Format: "ollama" (Default), "openai" - (OpenAI-kompatible Server wie vLLM oder LM Studio) - oder "anthropic" (Messages API). Leer = ollama. + Optionales Wire-Format: "ollama" (Default), "vllm" + (selbst gehostetes vLLM), "openai" (OpenAI oder andere + OpenAI-kompatible Server wie LM Studio) oder "anthropic" + (Messages API). vllm und openai sprechen dieselbe Wire; + der eigene Wert existiert, damit Du schreibst, was Du + meinst, und vLLM-spezifische Hinweise bekommst. + Leer = ollama. model: type: text description: diff --git a/src/lib.rs b/src/lib.rs index ddea369..c0f6b02 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -6,9 +6,9 @@ //! fields, same as `llm.chat`. //! //! The wire format is selected by the optional `api` input: -//! `ollama` (default, unchanged v0.1.x behavior), `openai` -//! (OpenAI-compatible `/v1/chat/completions` — vLLM, LM Studio, -//! LiteLLM, cloud OpenAI), or `anthropic` (Messages API). The +//! `ollama` (default, unchanged v0.1.x behavior), `vllm` +//! (self-hosted vLLM), `openai` (any other OpenAI-compatible +//! server), or `anthropic` (Messages API). The //! client logic in `llm.rs` is kept in lockstep with `llm.chat`. mod llm; @@ -83,7 +83,12 @@ fn llm_error_to_module_error( )), (Api::Openai, LlmError::Http(detail)) => ModuleError::internal(format!( "LLM endpoint {endpoint} not reachable ({detail}). Expected an \ - OpenAI-compatible server (OpenAI, vLLM, ...) at a /v1/chat/completions URL." + OpenAI-compatible server at a /v1/chat/completions URL." + )), + (Api::Vllm, LlmError::Http(detail)) => ModuleError::internal(format!( + "LLM endpoint {endpoint} not reachable ({detail}). Is vLLM running? \ + Start it with `vllm serve {model}`, then point `endpoint` at \ + http://:8000/v1/chat/completions." )), (Api::Anthropic, LlmError::Http(detail)) => ModuleError::internal(format!( "LLM endpoint {endpoint} not reachable ({detail}). Expected the \ @@ -107,7 +112,7 @@ fn llm_error_to_module_error( ModuleError::invalid_input(format!("missing required input '{name}'")) } (_, LlmError::UnsupportedApi(raw)) => ModuleError::invalid_input(format!( - "unsupported api '{raw}' (expected: ollama, openai, anthropic)" + "unsupported api '{raw}' (expected: ollama, openai, vllm, anthropic)" )), } } diff --git a/src/llm.rs b/src/llm.rs index 90565f5..12a890c 100644 --- a/src/llm.rs +++ b/src/llm.rs @@ -33,7 +33,7 @@ const ANTHROPIC_VERSION: &str = "2023-06-01"; pub enum LlmError { #[error("missing required input '{0}'")] MissingInput(&'static str), - #[error("unsupported api '{0}' (expected: ollama, openai, anthropic)")] + #[error("unsupported api '{0}' (expected: ollama, openai, vllm, anthropic)")] UnsupportedApi(String), #[error("http error: {0}")] Http(String), @@ -48,6 +48,13 @@ pub enum LlmError { pub enum Api { Ollama, Openai, + /// vLLM speaks the OpenAI wire format verbatim; it is a + /// separate variant ONLY so operators can write what they + /// mean (`api: vllm`) and get vLLM-specific error hints — + /// "openai" would read as the cloud company, not a + /// self-hosted server. Same rationale as the hub's + /// SystemLlmProvider vocabulary. + Vllm, Anthropic, } @@ -58,6 +65,7 @@ impl Api { match raw.trim().to_ascii_lowercase().as_str() { "" | "ollama" => Ok(Api::Ollama), "openai" => Ok(Api::Openai), + "vllm" => Ok(Api::Vllm), "anthropic" => Ok(Api::Anthropic), other => Err(LlmError::UnsupportedApi(other.to_string())), } @@ -92,7 +100,7 @@ pub struct ChatParams<'a> { pub fn build_headers(p: &ChatParams) -> Vec<(&'static str, String)> { let mut headers = Vec::with_capacity(2); match p.api { - Api::Ollama | Api::Openai => { + Api::Ollama | Api::Openai | Api::Vllm => { if !p.api_key.is_empty() { headers.push(("Authorization", format!("Bearer {}", p.api_key))); } @@ -261,19 +269,19 @@ pub fn chat_with_identity( } let body = match p.api { Api::Ollama => build_ollama_body(p), - Api::Openai => build_openai_body(p), + Api::Openai | Api::Vllm => build_openai_body(p), Api::Anthropic => build_anthropic_body(p), }; let headers = build_headers(p); let response_body = client.post_json(p.endpoint, &body, &headers)?; let response = match p.api { Api::Ollama => extract_ollama_content(&response_body)?, - Api::Openai => extract_openai_content(&response_body)?, + Api::Openai | Api::Vllm => extract_openai_content(&response_body)?, Api::Anthropic => extract_anthropic_content(&response_body)?, }; let model_digest = match p.api { Api::Ollama => probe_model_digest(client, p), - Api::Openai | Api::Anthropic => None, + Api::Openai | Api::Vllm | Api::Anthropic => None, }; Ok(ChatWithIdentity { response, @@ -717,4 +725,26 @@ mod tests { Err(LlmError::MissingInput("prompt")) )); } + + /// Guard: `api: vllm` exists ONLY for operator clarity — the + /// request it emits (URL, body, headers) must stay identical to + /// `api: openai`. If the two ever diverge, that is a new wire + /// dialect and needs its own deliberate design, not drift. + #[test] + fn vllm_is_the_openai_wire_verbatim() { + assert_eq!(Api::parse("vllm").unwrap(), Api::Vllm); + let openai_reply = r#"{"choices":[{"message":{"role":"assistant","content":"x"}}]}"#; + let recorded: Vec = [Api::Vllm, Api::Openai] + .into_iter() + .map(|api| { + let client = MockClient::new(vec![Ok(openai_reply.to_string())]); + chat_with_identity(&client, ¶ms(api)).unwrap(); + client.calls.into_inner().pop().unwrap() + }) + .collect(); + assert_eq!( + recorded[0], recorded[1], + "vllm request drifted from openai" + ); + } }