From c50da6f29c4d2ce0a126aecdb45899c3c5569478 Mon Sep 17 00:00:00 2001 From: Adolfo Reyna Date: Fri, 7 Aug 2026 18:32:57 -0400 Subject: [PATCH] Set Gemma 2 2B MLX as default local model and add --mlx-model flag --- README.md | 1 + apple_llm.py | 8 ++++---- bot.py | 7 ++++++- 3 files changed, 11 insertions(+), 5 deletions(-) diff --git a/README.md b/README.md index f080932..8fc9234 100644 --- a/README.md +++ b/README.md @@ -35,6 +35,7 @@ Useful flags: ```bash ./talk --llm-engine apple # macOS on-device LLM model (default) +./talk --mlx-model mlx-community/gemma-2-2b-it-4bit # specify any local MLX model ./talk --llm-engine claude # Claude Code CLI engine ./talk --list-devices # see microphones and speakers ./talk --list-voices # see macOS system voices diff --git a/apple_llm.py b/apple_llm.py index 2ea168f..e362069 100644 --- a/apple_llm.py +++ b/apple_llm.py @@ -92,7 +92,7 @@ class MacOSLLM(FrameProcessor): def __init__( self, *, - model: str = "mlx-community/Llama-3.2-1B-Instruct-4bit", + model: str = "mlx-community/gemma-2-2b-it-4bit", system_prompt: str | None = None, observer=None, **kwargs, @@ -194,20 +194,20 @@ class MacOSLLM(FrameProcessor): loop = asyncio.get_running_loop() def _gen(): from mlx_lm import generate + from mlx_lm.sample_utils import make_sampler # Keep history concise to avoid context drift and repetition recent_history = self._history[-8:] messages = [{"role": "system", "content": self._system_prompt}] + recent_history prompt = self._mlx_tokenizer.apply_chat_template( messages, add_generation_prompt=True, tokenize=False ) + sampler = make_sampler(temp=0.7) return generate( self._mlx_model, self._mlx_tokenizer, prompt=prompt, max_tokens=150, - temp=0.7, - repetition_penalty=1.2, - repetition_context_size=30, + sampler=sampler, verbose=False, ) diff --git a/bot.py b/bot.py index 34c124a..90c22a0 100644 --- a/bot.py +++ b/bot.py @@ -157,6 +157,11 @@ def parse_args() -> argparse.Namespace: default="apple", help="LLM engine to use: apple/macos for native on-device macOS LLM, claude for Claude Code.", ) + parser.add_argument( + "--mlx-model", + default="mlx-community/gemma-2-2b-it-4bit", + help="MLX model repo or path for local macOS execution (e.g. mlx-community/gemma-2-2b-it-4bit).", + ) parser.add_argument( "--claude-model", default=DEFAULT_CLAUDE_MODEL, @@ -490,7 +495,7 @@ def build_llm(args: argparse.Namespace, vocabulary=None, brain=None, observer=No logger.info(f"LLM: macOS native model ({reason})") personality = read_personality(args.cwd) or "You are a helpful macOS voice assistant." system_prompt = personality + "\n\n" + VOICE_STYLE - return MacOSLLM(system_prompt=system_prompt, observer=observer) + return MacOSLLM(model=args.mlx_model, system_prompt=system_prompt, observer=observer) logger.info("LLM: Claude Code") return ClaudeCodeLLM(