diff --git a/server/src/ollama.rs b/server/src/ollama.rs index 2a70a9b..2ee7b70 100644 --- a/server/src/ollama.rs +++ b/server/src/ollama.rs @@ -23,6 +23,8 @@ struct GenerateRequest<'a> { stream: bool, #[serde(skip_serializing_if = "Option::is_none")] images: Option>, + #[serde(skip_serializing_if = "Option::is_none")] + options: Option, } #[derive(Deserialize)] @@ -50,12 +52,17 @@ impl OllamaClient { let reasoning_model = env::var("OLLAMA_REASONING_MODEL").unwrap_or_else(|_| "deepseek-r1:1.5b".to_string()); let vision_model = - env::var("OLLAMA_VISION_MODEL").unwrap_or_else(|_| "qwen3-vl:2b".to_string()); + env::var("OLLAMA_VISION_MODEL").unwrap_or_else(|_| "moondream".to_string()); let embed_model = env::var("OLLAMA_EMBED_MODEL") .unwrap_or_else(|_| "nomic-embed-text:latest".to_string()); + let timeout_sec = env::var("OLLAMA_TIMEOUT_SEC") + .unwrap_or_else(|_| "60".to_string()) + .parse::() + .unwrap_or(60); + let client = reqwest::Client::builder() - .timeout(Duration::from_secs(60)) + .timeout(Duration::from_secs(timeout_sec)) .build() .unwrap_or_default(); @@ -112,6 +119,10 @@ impl OllamaClient { system, stream: false, images: None, + options: Some(serde_json::json!({ + "num_ctx": 32768, // Massive context window win + "num_predict": 4096 // Give reasoning models plenty of output room + })), }; let res = self @@ -151,6 +162,10 @@ impl OllamaClient { ), stream: false, images: Some(vec![image_base64]), + options: Some(serde_json::json!({ + "num_ctx": 8192, + "num_predict": 1024 + })), }; let res = self