feat(ollama): keep models in VRAM for 1h to prevent cold starts
This commit is contained in:
1 parent
4005f566cf
commit
8952bd5399
1 file changed
+7
@@ -25,6 +25,8 @@ struct GenerateRequest<'a> {
|
||||
images: Option<Vec<&'a str>>,
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
options: Option<serde_json::Value>,
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
keep_alive: Option<&'a str>,
|
||||
}
|
||||
|
||||
#[derive(Deserialize)]
|
||||
@@ -36,6 +38,8 @@ struct GenerateResponse {
|
||||
struct EmbeddingRequest<'a> {
|
||||
model: &'a str,
|
||||
prompt: &'a str,
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
keep_alive: Option<&'a str>,
|
||||
}
|
||||
|
||||
#[derive(Deserialize)]
|
||||
@@ -123,6 +127,7 @@ impl OllamaClient {
|
||||
"num_ctx": 32768, // Massive context window win
|
||||
"num_predict": 4096 // Give reasoning models plenty of output room
|
||||
})),
|
||||
keep_alive: Some("1h"),
|
||||
};
|
||||
|
||||
let res = self
|
||||
@@ -166,6 +171,7 @@ impl OllamaClient {
|
||||
"num_ctx": 8192,
|
||||
"num_predict": 1024
|
||||
})),
|
||||
keep_alive: Some("1h"),
|
||||
};
|
||||
|
||||
let res = self
|
||||
@@ -196,6 +202,7 @@ impl OllamaClient {
|
||||
let body = EmbeddingRequest {
|
||||
model: &self.embed_model,
|
||||
prompt: text,
|
||||
keep_alive: Some("1h"),
|
||||
};
|
||||
|
||||
let res = self
|
||||
|
||||
Reference in new issue
Block a user