From fcbe736e1ae64c7ac79477594e02e4a27a064759 Mon Sep 17 00:00:00 2001 From: jhqxxx <18280426169@163.com> Date: Sat, 7 Feb 2026 13:28:13 +0800 Subject: [PATCH] update docs --- README.md | 10 +++--- README.zh-CN.md | 12 +++---- docs/api.md | 65 +++++++++++++++++++-------------- docs/api.zh-CN.md | 67 +++++++++++++++++++++-------------- docs/concepts.md | 12 +++---- docs/concepts.zh-CN.md | 10 +++--- docs/development.md | 30 +++------------- docs/development.zh-CN.md | 30 +++------------- docs/getting-started.md | 45 ++++++++++++++--------- docs/getting-started.zh-CN.md | 52 +++++++++++++++++---------- docs/installation.md | 10 +++--- docs/installation.zh-CN.md | 10 +++--- src/models/mod.rs | 2 +- 13 files changed, 180 insertions(+), 175 deletions(-) diff --git a/README.md b/README.md index ce1ed65..8c65f49 100644 --- a/README.md +++ b/README.md @@ -68,13 +68,10 @@ cargo build --release --features cuda cargo build --release --features metal # Flash Attention (faster inference) -cargo build --release --features flash-attn +cargo build --release --features cuda,flash-attn # FFmpeg (multimedia processing) cargo build --release --features ffmpeg - -# Combine multiple features -cargo build --release --features "cuda,flash-attn" ``` ### CLI Quick Reference @@ -111,7 +108,8 @@ curl http://localhost:10100/chat/completions \ -H "Content-Type: application/json" \ -d '{ "model": "qwen3-0.6b", - "messages": [{"role": "user", "content": "Hello!"}] + "messages": [{"role": "user", "content": "Hello!"}], + "stream": false } ' ``` @@ -154,7 +152,7 @@ curl http://localhost:10100/chat/completions \ ### Using aha as a Library > cargo add aha -```bash +```rust # VoxCPM example use aha::models::voxcpm::generate::VoxCPMGenerate; use aha::utils::audio_utils::save_wav; diff --git a/README.zh-CN.md b/README.zh-CN.md index 2718100..a3c72c2 100644 --- a/README.zh-CN.md +++ b/README.zh-CN.md @@ -22,7 +22,7 @@ **轻量 AI 推理引擎 —— 文本、视觉、语音与 OCR 一站式解决方案** -aha 是一款基于 Rust 和 Candle 框架构建的高性能跨平台 AI 推理引擎。将最先进的 AI 模型带到您的本地机器——无需 API 密钥,无需云依赖,纯粹、快速的 AI 直接在您的硬件上运行。 +aha 是一款基于 Rust 和 Candle 框架构建的高性能跨平台 AI 推理引擎。将最先进的 AI 模型带到您的本地机器——无需 API 密钥,无需云依赖,纯粹、快速的 AI,直接在您的硬件上运行。 ## 更新日志 @@ -68,13 +68,10 @@ cargo build --release --features cuda cargo build --release --features metal # Flash Attention (更快推理) -cargo build --release --features flash-attn +cargo build --release --features cuda,flash-attn # FFmpeg (多媒体处理) cargo build --release --features ffmpeg - -# 组合多个特性 -cargo build --release --features "cuda,flash-attn" ``` ### CLI 快速参考 @@ -111,7 +108,8 @@ curl http://localhost:10100/chat/completions \ -H "Content-Type: application/json" \ -d '{ "model": "qwen3-0.6b", - "messages": [{"role": "user", "content": "你好!"}] + "messages": [{"role": "user", "content": "你好!"}], + "stream": false }' ``` @@ -154,7 +152,7 @@ curl http://localhost:10100/chat/completions \ ### aha 作为库使用 > cargo add aha -```bash +```rust # VoxCPM示例 use aha::models::voxcpm::generate::VoxCPMGenerate; use aha::utils::audio_utils::save_wav; diff --git a/docs/api.md b/docs/api.md index 2bf3260..65c4d44 100644 --- a/docs/api.md +++ b/docs/api.md @@ -39,8 +39,6 @@ Success responses follow this structure: "data": { ... }, "model": "model-name", "usage": { - "prompt_tokens": 10, - "completion_tokens": 20, "total_tokens": 30 } } @@ -95,15 +93,15 @@ For vision/audio models, content can be an array: "role": "user", "content": [ {"type": "text", "text": "Describe this image"}, - {"type": "image_url", "image_url": {"url": "file:///path/to/image.jpg"}} + {"type": "image", "image_url": {"url": "file:///path/to/image.jpg"}} ] } ``` Supported content types: - `text` - Text content -- `image_url` - Image file (file://, base64://, or http://) -- `audio_url` - Audio file (file:// or base64://) +- `image_url` - Image file (file://, base64://, https:// or http://) +- `audio_url` - Audio file (file://, base64://, https:// or http://) #### Examples @@ -148,7 +146,7 @@ curl http://127.0.0.1:10100/chat/completions \ "role": "user", "content": [ {"type": "text", "text": "What is in this image?"}, - {"type": "image_url", "image_url": {"url": "file:///path/to/image.jpg"}} + {"type": "image", "image_url": {"url": "file:///path/to/image.jpg"}} ] } ] @@ -167,7 +165,7 @@ curl http://127.0.0.1:10100/chat/completions \ "role": "user", "content": [ {"type": "text", "text": "Extract all text"}, - {"type": "image_url", "image_url": {"url": "file:///path/to/document.png"}} + {"type": "image", "image_url": {"url": "file:///path/to/document.png"}} ] } ] @@ -186,7 +184,7 @@ curl http://127.0.0.1:10100/chat/completions \ "role": "user", "content": [ {"type": "text", "text": "Transcribe this audio"}, - {"type": "audio_url", "audio_url": {"url": "file:///path/to/audio.wav"}} + {"type": "audio", "audio_url": {"url": "file:///path/to/audio.wav"}} ] } ] @@ -237,8 +235,6 @@ data: [DONE] } ], "usage": { - "prompt_tokens": 10, - "completion_tokens": 9, "total_tokens": 19 } } @@ -265,8 +261,7 @@ POST /audio/speech | Parameter | Type | Required | Description | |-----------|------|----------|-------------| | `model` | string | Yes | Model identifier (e.g., "voxcpm1.5") | -| `input` | string | Yes | Text to convert to speech | -| `voice` | string | No | Voice selection (default: "default") | +| `messages` | array | Yes | Array of message objects | #### Example @@ -275,15 +270,21 @@ curl http://127.0.0.1:10100/audio/speech \ -H "Content-Type: application/json" \ -d '{ "model": "voxcpm1.5", - "input": "你好,世界!", - "voice": "default" - }' \ - --output speech.wav + "messages": [ + { + "role": "user", + "content": [ + {"type": "text", "text": "Hello, this is AHA speaking."}, + {"type": "audio", "audio_url": {"url": "https://package-release.coderbox.cn/aiway/test/other/%E5%93%AA%E5%90%92.wav"}} + ] + } + ] + }' ``` #### Response -Returns audio data in WAV format. +Returns audio data in base64 WAV format. #### Supported Models @@ -303,7 +304,7 @@ POST /images/remove_background | Parameter | Type | Required | Description | |-----------|------|----------|-------------| | `model` | string | Yes | Model identifier (e.g., "rmbg2.0") | -| `image` | string | Yes | Image file path (file://) or base64 data | +| `messages` | array | Yes | Array of message objects | #### Example @@ -314,9 +315,15 @@ curl http://127.0.0.1:10100/images/remove_background \ -H "Content-Type: application/json" \ -d '{ "model": "rmbg2.0", - "image": "file:///path/to/photo.png" - }' \ - --output no-background.png + "messages": [ + { + "role": "user", + "content": [ + {"type": "image", "image_url": {"url": "file:///path/to/document.jpg"}} + ] + } + ] + }' ``` **From Base64:** @@ -326,14 +333,20 @@ curl http://127.0.0.1:10100/images/remove_background \ -H "Content-Type: application/json" \ -d '{ "model": "rmbg2.0", - "image": "base64://$(base64 -w 0 photo.png)" - }' \ - --output no-background.png + "messages": [ + { + "role": "user", + "content": [ + {"type": "image", "image_url": {"url": "base64://$(base64 -w 0 photo.png)"}} + ] + } + ] + }' ``` #### Response -Returns the processed image in PNG format. +Returns the processed image in base64 PNG format. #### Supported Models @@ -373,7 +386,7 @@ Currently, AHA does not implement rate limiting. The server can handle concurren ## OpenAI Compatibility -AHA's API is designed to be compatible with OpenAI's API format. This means you can use existing OpenAI client libraries with minimal changes: +AHA's text generation API is designed to be compatible with OpenAI's API format. Multimodal APIs are derived from the text generation API with minimal changes: ### Python Example diff --git a/docs/api.zh-CN.md b/docs/api.zh-CN.md index 226d7a5..cb51888 100644 --- a/docs/api.zh-CN.md +++ b/docs/api.zh-CN.md @@ -39,8 +39,6 @@ Content-Type: application/json "data": { ... }, "model": "model-name", "usage": { - "prompt_tokens": 10, - "completion_tokens": 20, "total_tokens": 30 } } @@ -77,7 +75,7 @@ POST /chat/completions | `temperature` | number | 否 | 采样温度(0-2,默认:1) | | `top_p` | number | 否 | 核采样(0-1,默认:1) | | `max_tokens` | integer | 否 | 要生成的最大令牌数 | -| `stream` | boolean | 否 | 启用流式传输(默认:false) | +| `stream` | boolean | 否 | 启用流式传输(默认:true) | #### 消息对象 @@ -95,15 +93,15 @@ POST /chat/completions "role": "user", "content": [ {"type": "text", "text": "描述这张图片"}, - {"type": "image_url", "image_url": {"url": "file:///path/to/image.jpg"}} + {"type": "image", "image_url": {"url": "file:///path/to/image.jpg"}} ] } ``` 支持的内容类型: - `text` - 文本内容 -- `image_url` - 图像文件(file://、base64:// 或 http://) -- `audio_url` - 音频文件(file:// 或 base64://) +- `image` - 图像文件(file://,base64://, https:// 或 http://) +- `audio` - 音频文件(file://,base64://, https:// 或 http://) #### 示例 @@ -148,7 +146,7 @@ curl http://127.0.0.1:10100/chat/completions \ "role": "user", "content": [ {"type": "text", "text": "这张图片里有什么?"}, - {"type": "image_url", "image_url": {"url": "file:///path/to/image.jpg"}} + {"type": "image", "image_url": {"url": "file:///path/to/image.jpg"}} ] } ] @@ -167,7 +165,7 @@ curl http://127.0.0.1:10100/chat/completions \ "role": "user", "content": [ {"type": "text", "text": "提取所有文本"}, - {"type": "image_url", "image_url": {"url": "file:///path/to/document.png"}} + {"type": "image", "image_url": {"url": "file:///path/to/document.png"}} ] } ] @@ -186,7 +184,7 @@ curl http://127.0.0.1:10100/chat/completions \ "role": "user", "content": [ {"type": "text", "text": "转写这段音频"}, - {"type": "audio_url", "audio_url": {"url": "file:///path/to/audio.wav"}} + {"type": "audio", "audio_url": {"url": "file:///path/to/audio.wav"}} ] } ] @@ -237,8 +235,6 @@ data: [DONE] } ], "usage": { - "prompt_tokens": 10, - "completion_tokens": 9, "total_tokens": 19 } } @@ -265,8 +261,7 @@ POST /audio/speech | 参数 | 类型 | 必需 | 描述 | |------|------|------|------| | `model` | string | 是 | 模型标识符(如 "voxcpm1.5") | -| `input` | string | 是 | 要转换为语音的文本 | -| `voice` | string | 否 | 语音选择(默认:"default") | +| `messages` | array | 是 | 消息对象数组 | #### 示例 @@ -275,15 +270,21 @@ curl http://127.0.0.1:10100/audio/speech \ -H "Content-Type: application/json" \ -d '{ "model": "voxcpm1.5", - "input": "你好,世界!", - "voice": "default" - }' \ - --output speech.wav + "messages": [ + { + "role": "user", + "content": [ + {"type": "text", "text": "你好,这是 AHA 在说话。"}, + {"type": "audio", "audio_url": {"url": "https://package-release.coderbox.cn/aiway/test/other/%E5%93%AA%E5%90%92.wav"}} + ] + } + ] + }' ``` #### 响应 -以 WAV 格式返回音频数据。 +以 base64 WAV 格式返回音频数据。 #### 支持的模型 @@ -303,7 +304,7 @@ POST /images/remove_background | 参数 | 类型 | 必需 | 描述 | |------|------|------|------| | `model` | string | 是 | 模型标识符(如 "rmbg2.0") | -| `image` | string | 是 | 图像文件路径(file://)或 base64 数据 | +| `messages` | array | 是 | 消息对象数组 | #### 示例 @@ -314,9 +315,15 @@ curl http://127.0.0.1:10100/images/remove_background \ -H "Content-Type: application/json" \ -d '{ "model": "rmbg2.0", - "image": "file:///path/to/photo.png" - }' \ - --output no-background.png + "messages": [ + { + "role": "user", + "content": [ + {"type": "image", "image_url": {"url": "file:///path/to/document.jpg"}} + ] + } + ] + }' ``` **从 Base64:** @@ -326,14 +333,20 @@ curl http://127.0.0.1:10100/images/remove_background \ -H "Content-Type: application/json" \ -d '{ "model": "rmbg2.0", - "image": "base64://$(base64 -w 0 photo.png)" - }' \ - --output no-background.png + "messages": [ + { + "role": "user", + "content": [ + {"type": "image", "image_url": {"url": "base64://$(base64 -w 0 photo.png)"}} + ] + } + ] + }' ``` #### 响应 -以 PNG 格式返回处理后的图像。 +以base64 PNG 格式返回处理后的图像。 #### 支持的模型 @@ -373,7 +386,7 @@ curl http://127.0.0.1:10100/images/remove_background \ ## OpenAI 兼容性 -AHA 的 API 设计为与 OpenAI 的 API 格式兼容。这意味着您可以使用现有的 OpenAI 客户端库,只需最少的更改: +AHA 的 文本生成 API 设计与 OpenAI 的 API 格式兼容。多模态 API 源自文本生成 API,仅做了少量改动: ### Python 示例 diff --git a/docs/concepts.md b/docs/concepts.md index db4c784..3beca33 100644 --- a/docs/concepts.md +++ b/docs/concepts.md @@ -126,9 +126,7 @@ All models implement the `GenerateModel` trait: ```rust pub trait GenerateModel { - // Initialize model with weights - fn init(model_path: &str, config: Option) -> Result; - + // Generate response fn generate(&mut self, prompt: &str, params: GenerationParams) -> Result; @@ -167,8 +165,8 @@ This provides: - `audio_utils.rs` - Audio processing (WAV, MP3) - `image_utils.rs` - Image processing (resize, encode/decode) -- `download.rs` - Model downloading with retry logic -- `common.rs` - Common utilities and constants +- `tensor_utils.rs` - Tensor utility methods +- `mod.rs` - Common utilities and constants ## Design Patterns @@ -269,7 +267,7 @@ Flash Attention optimizes long-sequence processing: Enable with: ```bash -cargo build --features "cuda,flash-attn" +cargo build --features cuda,flash-attn ``` ### Memory-Mapped Tensors @@ -294,7 +292,7 @@ Dynamic precision selection based on hardware: - No external API calls after model download - No telemetry or data collection -- Full control over your data +- Data remains entirely on the local system ### Memory Safety diff --git a/docs/concepts.zh-CN.md b/docs/concepts.zh-CN.md index 2e1020e..1b3596c 100644 --- a/docs/concepts.zh-CN.md +++ b/docs/concepts.zh-CN.md @@ -126,8 +126,6 @@ HTTP API 层提供 REST 端点: ```rust pub trait GenerateModel { - // 使用权重初始化模型 - fn init(model_path: &str, config: Option) -> Result; // 生成响应 fn generate(&mut self, prompt: &str, params: GenerationParams) -> Result; @@ -167,8 +165,8 @@ pub trait GenerateModel { - `audio_utils.rs` - 音频处理(WAV、MP3) - `image_utils.rs` - 图像处理(调整大小、编码/解码) -- `download.rs` - 带重试逻辑的模型下载 -- `common.rs` - 通用工具和常量 +- `tensor_utils.rs` - Tensor常用方法 +- `mod.rs` - 通用工具和常量 ## 设计模式 @@ -269,7 +267,7 @@ Flash Attention 优化长序列处理: 启用方式: ```bash -cargo build --features "cuda,flash-attn" +cargo build --features cuda,flash-attn ``` ### 内存映射张量 @@ -294,7 +292,7 @@ cargo build --features "cuda,flash-attn" - 模型下载后无外部 API 调用 - 无遥测或数据收集 -- 完全控制您的数据 +- 数据完全保留在本地 ### 内存安全 diff --git a/docs/development.md b/docs/development.md index 7a4a15d..6f1fe2e 100644 --- a/docs/development.md +++ b/docs/development.md @@ -34,9 +34,9 @@ cargo build cargo build --release # Build with features -cargo build --features "cuda,flash-attn" -cargo build --features "metal" -cargo build --features "ffmpeg" +cargo build --features cuda,flash-attn +cargo build --features metal +cargo build --features ffmpeg ``` ### Development Workflow @@ -98,13 +98,12 @@ aha/ │ └── utils/ # Utility functions │ ├── audio_utils.rs │ ├── image_utils.rs -│ ├── download.rs -│ └── common.rs +│ ├── tensor_utils.rs +│ └── mod.rs ├── tests/ # Integration tests │ ├── test_qwen2_5vl.rs │ ├── test_qwen3vl.rs │ └── ... -├── examples/ # Example code └── docs/ # Documentation ``` @@ -140,16 +139,6 @@ pub struct NewModelConfig { // ... other config fields } -impl Default for NewModelConfig { - fn default() -> Self { - Self { - vocab_size: 32000, - hidden_size: 2048, - num_hidden_layers: 24, - num_attention_heads: 32, - } - } -} #[derive(Debug, Clone, Serialize, Deserialize)] pub struct GenerationConfig { @@ -158,15 +147,6 @@ pub struct GenerationConfig { pub max_tokens: usize, } -impl Default for GenerationConfig { - fn default() -> Self { - Self { - temperature: 1.0, - top_p: 1.0, - max_tokens: 100, - } - } -} ``` #### model.rs diff --git a/docs/development.zh-CN.md b/docs/development.zh-CN.md index 6170b47..c7a900d 100644 --- a/docs/development.zh-CN.md +++ b/docs/development.zh-CN.md @@ -34,9 +34,9 @@ cargo build cargo build --release # 使用功能构建 -cargo build --features "cuda,flash-attn" -cargo build --features "metal" -cargo build --features "ffmpeg" +cargo build --features cuda,flash-attn +cargo build --features metal +cargo build --features ffmpeg ``` ### 开发工作流程 @@ -98,13 +98,12 @@ aha/ │ └── utils/ # 工具函数 │ ├── audio_utils.rs │ ├── image_utils.rs -│ ├── download.rs -│ └── common.rs +│ ├── tensor_utils.rs +│ └── mod.rs ├── tests/ # 集成测试 │ ├── test_qwen2_5vl.rs │ ├── test_qwen3vl.rs │ └── ... -├── examples/ # 示例代码 └── docs/ # 文档 ``` @@ -140,16 +139,6 @@ pub struct NewModelConfig { // ... 其他配置字段 } -impl Default for NewModelConfig { - fn default() -> Self { - Self { - vocab_size: 32000, - hidden_size: 2048, - num_hidden_layers: 24, - num_attention_heads: 32, - } - } -} #[derive(Debug, Clone, Serialize, Deserialize)] pub struct GenerationConfig { @@ -158,15 +147,6 @@ pub struct GenerationConfig { pub max_tokens: usize, } -impl Default for GenerationConfig { - fn default() -> Self { - Self { - temperature: 1.0, - top_p: 1.0, - max_tokens: 100, - } - } -} ``` #### model.rs diff --git a/docs/getting-started.md b/docs/getting-started.md index 921752d..204d099 100644 --- a/docs/getting-started.md +++ b/docs/getting-started.md @@ -110,10 +110,11 @@ curl http://127.0.0.1:10100/chat/completions \ "role": "user", "content": [ {"type": "text", "text": "Describe this image in detail."}, - {"type": "image_url", "image_url": {"url": "file:///path/to/image.jpg"}} + {"type": "image", "image_url": {"url": "file:///path/to/image.jpg"}} ] } - ] + ], + "stream": false }' ``` @@ -133,7 +134,7 @@ curl http://127.0.0.1:10100/chat/completions \ "role": "user", "content": [ {"type": "text", "text": "Extract all text from this image."}, - {"type": "image_url", "image_url": {"url": "file:///path/to/document.jpg"}} + {"type": "image", "image_url": {"url": "file:///path/to/document.jpg"}} ] } ] @@ -156,7 +157,7 @@ curl http://127.0.0.1:10100/chat/completions \ "role": "user", "content": [ {"type": "text", "text": "Transcribe this audio."}, - {"type": "audio_url", "audio_url": {"url": "file:///path/to/audio.wav"}} + {"type": "audio", "audio_url": {"url": "file:///path/to/audio.wav"}} ] } ] @@ -174,10 +175,16 @@ curl http://127.0.0.1:10100/audio/speech \ -H "Content-Type: application/json" \ -d '{ "model": "voxcpm1.5", - "input": "Hello, this is AHA speaking.", - "voice": "default" - }' \ - --output speech.wav + "messages": [ + { + "role": "user", + "content": [ + {"type": "text", "text": "Hello, this is AHA speaking."}, + {"type": "audio", "audio_url": {"url": "https://package-release.coderbox.cn/aiway/test/other/%E5%93%AA%E5%90%92.wav"}} + ] + } + ] + }' ``` ### Background Removal @@ -191,9 +198,15 @@ curl http://127.0.0.1:10100/images/remove_background \ -H "Content-Type: application/json" \ -d '{ "model": "rmbg2.0", - "image": "file:///path/to/photo.png" - }' \ - --output no-background.png + "messages": [ + { + "role": "user", + "content": [ + {"type": "image", "image_url": {"url": "file:///path/to/document.jpg"}} + ] + } + ] + }' ``` ### Direct Inference (Without Server) @@ -202,7 +215,7 @@ curl http://127.0.0.1:10100/images/remove_background \ # Run inference directly without starting HTTP server aha run -m qwen3-0.6b \ -i "Write a haiku about AI" \ - --weight-path ~/.aha/qwen3-0.6b + --weight-path ~/.aha/Qwen/Qwen3-0.6B ``` ## Configuration Options @@ -239,7 +252,7 @@ aha download -m qwen3vl-2b -s /data/models ## Streaming Responses -For real-time responses, use streaming: +For chat/completions, using no "stream" field or "stream": true enables streaming, while "stream": false is for non-streaming responses: ```bash curl http://127.0.0.1:10100/chat/completions \ @@ -249,7 +262,7 @@ curl http://127.0.0.1:10100/chat/completions \ "messages": [ {"role": "user", "content": "Tell me a story"} ], - "stream": true + "stream": false }' ``` @@ -308,7 +321,7 @@ aha download -m qwen3vl-2b Then use them later without internet: ```bash -aha serv -m qwen3vl-2b --weight-path ~/.aha/qwen3vl-2b +aha serv -m qwen3vl-2b --weight-path ~/.aha/Qwen/Qwen3-VL-2B-Instruct ``` ### 4. Manage Disk Space @@ -366,7 +379,7 @@ aha cli -m qwen3-0.6b ## Examples Repository -For more examples, check out the [examples](../examples/) directory in the repository. +For more examples, check out the [tests](../tests/) directory in the repository. ## See Also diff --git a/docs/getting-started.zh-CN.md b/docs/getting-started.zh-CN.md index ce9856e..625882d 100644 --- a/docs/getting-started.zh-CN.md +++ b/docs/getting-started.zh-CN.md @@ -13,7 +13,7 @@ aha list ### 2. 下载第一个模型 ```bash -# 下载一个小型文本模型开始 +# 从下载一个小型文本模型开始 aha download -m qwen3-0.6b ``` @@ -37,7 +37,8 @@ curl http://127.0.0.1:10100/chat/completions \ "model": "qwen3-0.6b", "messages": [ {"role": "user", "content": "你好,AHA!"} - ] + ], + "stream": false }' ``` @@ -110,10 +111,11 @@ curl http://127.0.0.1:10100/chat/completions \ "role": "user", "content": [ {"type": "text", "text": "详细描述这张图片。"}, - {"type": "image_url", "image_url": {"url": "file:///path/to/image.jpg"}} + {"type": "image", "image_url": {"url": "file:///path/to/image.jpg"}} ] } - ] + ], + "stream": false }' ``` @@ -133,7 +135,7 @@ curl http://127.0.0.1:10100/chat/completions \ "role": "user", "content": [ {"type": "text", "text": "提取这张图片中的所有文本。"}, - {"type": "image_url", "image_url": {"url": "file:///path/to/document.jpg"}} + {"type": "image", "image_url": {"url": "file:///path/to/document.jpg"}} ] } ] @@ -156,7 +158,7 @@ curl http://127.0.0.1:10100/chat/completions \ "role": "user", "content": [ {"type": "text", "text": "转写这段音频。"}, - {"type": "audio_url", "audio_url": {"url": "file:///path/to/audio.wav"}} + {"type": "audio", "audio_url": {"url": "file:///path/to/audio.wav"}} ] } ] @@ -174,10 +176,16 @@ curl http://127.0.0.1:10100/audio/speech \ -H "Content-Type: application/json" \ -d '{ "model": "voxcpm1.5", - "input": "你好,这是 AHA 在说话。", - "voice": "default" - }' \ - --output speech.wav + "messages": [ + { + "role": "user", + "content": [ + {"type": "text", "text": "你好,这是 AHA 在说话。"}, + {"type": "audio", "audio_url": {"url": "https://package-release.coderbox.cn/aiway/test/other/%E5%93%AA%E5%90%92.wav"}} + ] + } + ] + }' ``` ### 背景移除 @@ -191,9 +199,15 @@ curl http://127.0.0.1:10100/images/remove_background \ -H "Content-Type: application/json" \ -d '{ "model": "rmbg2.0", - "image": "file:///path/to/photo.png" - }' \ - --output no-background.png + "messages": [ + { + "role": "user", + "content": [ + {"type": "image", "image_url": {"url": "file:///path/to/document.jpg"}} + ] + } + ] + }' ``` ### 直接推理(无需服务器) @@ -201,8 +215,8 @@ curl http://127.0.0.1:10100/images/remove_background \ ```bash # 直接运行推理,无需启动 HTTP 服务器 aha run -m qwen3-0.6b \ - -i "写一首关于 AI 的俳句" \ - --weight-path ~/.aha/qwen3-0.6b + -i "写一首关于AI的俳句" \ + --weight-path ~/.aha/Qwen/Qwen3-0.6B ``` ## 配置选项 @@ -239,7 +253,7 @@ aha download -m qwen3vl-2b -s /data/models ## 流式响应 -对于实时响应,使用流式传输: +对于chat/completions,无"stream"字段或"stream": true时使用流式传输,"stream": false为非流式: ```bash curl http://127.0.0.1:10100/chat/completions \ @@ -249,7 +263,7 @@ curl http://127.0.0.1:10100/chat/completions \ "messages": [ {"role": "user", "content": "给我讲个故事"} ], - "stream": true + "stream": false }' ``` @@ -308,7 +322,7 @@ aha download -m qwen3vl-2b 稍后在没有网络的情况下使用: ```bash -aha serv -m qwen3vl-2b --weight-path ~/.aha/qwen3vl-2b +aha serv -m qwen3vl-2b --weight-path ~/.aha/Qwen/Qwen3-VL-2B-Instruct ``` ### 4. 管理磁盘空间 @@ -366,7 +380,7 @@ aha cli -m qwen3-0.6b ## 示例仓库 -更多示例,请查看仓库中的 [examples](../examples/) 目录。 +更多示例,请查看仓库中的 [tests](../tests/) 目录。 ## 另见 diff --git a/docs/installation.md b/docs/installation.md index 3aff0cf..bf74ccb 100644 --- a/docs/installation.md +++ b/docs/installation.md @@ -70,7 +70,7 @@ cargo build --release --features cuda cargo build --release --features metal # With Flash Attention -cargo build --release --features "cuda,flash-attn" +cargo build --release --features cuda,flash-attn # With FFmpeg support cargo build --release --features ffmpeg @@ -213,7 +213,7 @@ cargo build --release --features metal Enables Flash Attention for optimized long-sequence processing. ```bash -cargo build --release --features "cuda,flash-attn" +cargo build --release --features cuda,flash-attn ``` **Requirements**: @@ -250,13 +250,13 @@ You can combine multiple features: ```bash # Maximum performance on NVIDIA GPU -cargo build --release --features "cuda,flash-attn" +cargo build --release --features cuda,flash-attn # Apple Silicon with audio support -cargo build --release --features "metal,ffmpeg" +cargo build --release --features metal,ffmpeg # Everything enabled -cargo build --release --features "cuda,flash-attn,ffmpeg" +cargo build --release --features cuda,flash-attn,ffmpeg ``` ## Verification diff --git a/docs/installation.zh-CN.md b/docs/installation.zh-CN.md index 03ae3a2..3a68bb6 100644 --- a/docs/installation.zh-CN.md +++ b/docs/installation.zh-CN.md @@ -70,7 +70,7 @@ cargo build --release --features cuda cargo build --release --features metal # 启用 Flash Attention -cargo build --release --features "cuda,flash-attn" +cargo build --release --features cuda,flash-attn # 启用 FFmpeg 支持 cargo build --release --features ffmpeg @@ -212,7 +212,7 @@ cargo build --release --features metal 启用 Flash Attention 以优化长序列处理。 ```bash -cargo build --release --features "cuda,flash-attn" +cargo build --release --features cuda,flash-attn ``` **要求**: @@ -249,13 +249,13 @@ cargo build --release --features ffmpeg ```bash # NVIDIA GPU 上的最佳性能 -cargo build --release --features "cuda,flash-attn" +cargo build --release --features cuda,flash-attn # 带音频支持的 Apple Silicon -cargo build --release --features "metal,ffmpeg" +cargo build --release --features metal,ffmpeg # 启用所有功能 -cargo build --release --features "cuda,flash-attn,ffmpeg" +cargo build --release --features cuda,flash-attn,ffmpeg ``` ## 验证安装 diff --git a/src/models/mod.rs b/src/models/mod.rs index e0fa3d2..3ad06af 100644 --- a/src/models/mod.rs +++ b/src/models/mod.rs @@ -48,7 +48,7 @@ pub enum WhichModel { Qwen3ASR0_6B, #[value(name = "qwen3asr-1.7b", hide = true)] Qwen3ASR1_7B, - #[value(name = "qwen3vl-4b", hide = true)] + #[value(name = "qwen3vl-2b", hide = true)] Qwen3vl2B, #[value(name = "qwen3vl-4b", hide = true)] Qwen3vl4B,