Expand Ollama context windows (num_ctx) and output limits

This commit is contained in:
Riz Ashraf committed 2026-10-08 22:53:24 +01:00
1 parent 7eae2fa0aa
commit 4005f566cf
1 file changed
+17 -2
+17 -2
View File
@@ -23,6 +23,8 @@ struct GenerateRequest<'a> {
stream: bool,
#[serde(skip_serializing_if = "Option::is_none")]
images: Option<Vec<&'a str>>,
#[serde(skip_serializing_if = "Option::is_none")]
options: Option<serde_json::Value>,
}
#[derive(Deserialize)]
@@ -50,12 +52,17 @@ impl OllamaClient {
let reasoning_model =
env::var("OLLAMA_REASONING_MODEL").unwrap_or_else(|_| "deepseek-r1:1.5b".to_string());
let vision_model =
env::var("OLLAMA_VISION_MODEL").unwrap_or_else(|_| "qwen3-vl:2b".to_string());
env::var("OLLAMA_VISION_MODEL").unwrap_or_else(|_| "moondream".to_string());
let embed_model = env::var("OLLAMA_EMBED_MODEL")
.unwrap_or_else(|_| "nomic-embed-text:latest".to_string());
let timeout_sec = env::var("OLLAMA_TIMEOUT_SEC")
.unwrap_or_else(|_| "60".to_string())
.parse::<u64>()
.unwrap_or(60);
let client = reqwest::Client::builder()
.timeout(Duration::from_secs(60))
.timeout(Duration::from_secs(timeout_sec))
.build()
.unwrap_or_default();
@@ -112,6 +119,10 @@ impl OllamaClient {
system,
stream: false,
images: None,
options: Some(serde_json::json!({
"num_ctx": 32768, // Massive context window win
"num_predict": 4096 // Give reasoning models plenty of output room
})),
};
let res = self
@@ -151,6 +162,10 @@ impl OllamaClient {
),
stream: false,
images: Some(vec![image_base64]),
options: Some(serde_json::json!({
"num_ctx": 8192,
"num_predict": 1024
})),
};
let res = self