feat: switch to MiniCPM5-1B-Claude-Opus-Fable5-V2-Thinking-Q8_0 GGUF

- Updated model path + model ID for new 1B thinking model
- Updated prompt builder to use MiniCPM5 native /think trigger
- Updated clean_text to strip MiniCPM5 special tokens
- Bumped n_ctx from 2048 to 8192
This commit is contained in:
Asep Haryana
2026-07-26 15:38:05 +07:00
parent 14032db705
commit d8425ea3b3
3 changed files with 13 additions and 8 deletions
+9 -4
View File
@@ -89,8 +89,8 @@ pub fn build_prompt(messages: &[ChatMessage], tools: &Option<Vec<ToolDef>>) -> S
}
}
// Generation prompt: non-thinking mode
prompt.push_str("<|im_start|>assistant\n<think>\n\n</think>\n\n");
// Generation prompt: thinking mode (MiniCPM5 native)
prompt.push_str("<|im_start|>assistant\n/think\n\n");
prompt
}
@@ -182,8 +182,13 @@ pub fn build_sampler(params: &SamplerParams) -> LlamaSampler {
pub fn clean_text(text: &str) -> String {
text.replace("<|im_end|>", "")
.replace("<|im_start|>", "")
.replace("<think>", "")
.replace("</think>", "")
.replace("<|thought_begin|>", "")
.replace("<|thought_end|>", "")
.replace("<|tool_call|>", "")
.replace("<|execute_start|>", "")
.replace("<|execute_end|>", "")
.replace("/think", "")
.replace("/no_think", "")
.trim()
.to_string()
}
+3 -3
View File
@@ -4,8 +4,8 @@
use std::sync::LazyLock;
const DEFAULT_MODEL_PATH: &str = "/models/MiniCPM-V-4.6-Q4_K_M.gguf";
pub const MODEL_ID: &str = "minicpm-v-4.6";
const DEFAULT_MODEL_PATH: &str = "/models/MiniCPM5-1B-Claude-Opus-Fable5-V2-Thinking-Q8_0.gguf";
pub const MODEL_ID: &str = "minicpm5-1b-fable5-v2-thinking";
/// Application configuration loaded at startup from environment variables.
#[derive(Debug, Clone)]
@@ -44,7 +44,7 @@ impl AppConfig {
.and_then(|v| v.parse().ok())
.unwrap_or(8080),
log_level: std::env::var("RUST_LOG").unwrap_or_else(|_| "info".to_string()),
n_ctx: 2048,
n_ctx: 8192,
n_batch: 512,
n_threads: 4,
}
+1 -1
View File
@@ -8,6 +8,6 @@ use crate::domain::entity::HealthResponse;
pub async fn health_check() -> Json<HealthResponse> {
Json(HealthResponse {
status: "ok".into(),
model: format!("{MODEL_ID}-q4_k_m"),
model: format!("{MODEL_ID}-q8_0"),
})
}