supports small llama and gemma models

Refactor inference dedicated crates for llama and gemma inferencing, not integrated
2025-09-08 22:46:44 +00:00 · 2025-08-29 18:15:29 -04:00
parent d06b16bb12
commit 315ef17605
26 changed files with 2136 additions and 1402 deletions
--- a/crates/llama-runner/src/lib.rs
+++ b/crates/llama-runner/src/lib.rs
@@ -0,0 +1,8 @@
+pub mod llama_api;
+
+use clap::ValueEnum;
+pub use llama_api::{run_llama_inference, LlamaInferenceConfig, WhichModel};
+
+// Re-export constants and types that might be needed
+pub const EOS_TOKEN: &str = "</s>";
+