feat: Apple on-device LLM (ANE) polish replaces Ollama as primary

- apple-llm-polish binary (120KB Swift, FoundationModels): --check
  now reports available:true after Apple Intelligence enabled,
  ping OK, line polish 258-388ms ANE vs Ollama 3-8s CPU
- engine_apple_llm.py: AppleLLM class with persistent pipe,
  thread reader, polish_line() + polish_paragraph(), JSONL protocol
  {id,mode,text,prev1,prev2,context} -> {id,text,ok,ms}
- engine_llm.py: try Apple ANE first (permissiveContentTransformations,
  temp 0.1 line / 0.2 para), log provider=apple apple_ms, fallback
  to Ollama on guardrail/timeout. Fixes missing LLM log — now
  logs provider=apple with ms.
- main_v3.py: --apple-llm / --no-apple-llm flag (on by default)
- Verified: direct polish 258ms, last log provider=apple ms=498,
  speech binary still 240KB with 31 word-by-word drafts

Co-authored-by: internal-model
This commit is contained in:
Adolfo Reyna
2026-07-13 21:31:24 -04:00
parent 80f0bf309f
commit 2e6d6d24ce
6 changed files with 573 additions and 4 deletions
+190
View File
@@ -0,0 +1,190 @@
"""
engine_apple_llm.py — Apple on-device LLM (FoundationModels) -> Ollama fallback
Uses apple-llm-polish binary (same pipe protocol as Apple Speech).
ANE-accelerated, ~0.6s vs 2-3s Ollama.
Wire: line polish + paragraph rolling refine.
"""
import json, pathlib, subprocess, struct, sys, time, threading, queue, os, wave, io, select
from typing import Optional, Callable
from pathlib import Path
import platform
BIN_NAME = "apple-llm-polish"
SEARCH_PATHS = [
Path(__file__).parent / "apple_speech" / ".build" / "release" / BIN_NAME,
Path(__file__).parent / BIN_NAME,
Path.home() / "Projects" / "pythonwhisper" / "apple_speech" / ".build" / "release" / BIN_NAME,
]
def resolve_llm_binary() -> Optional[Path]:
for p in SEARCH_PATHS:
if p.exists():
return p
import shutil
which = shutil.which(BIN_NAME)
if which:
return Path(which)
which2 = shutil.which("apple-llm-polish")
if which2:
return Path(which2)
return None
def check_apple_llm_available(bin_path: Optional[Path]=None) -> bool:
bp = bin_path or resolve_llm_binary()
if not bp or not bp.exists():
return False
try:
r = subprocess.run([str(bp), "--check"], capture_output=True, text=True, timeout=10)
if r.returncode != 0:
return False
# parse: {"available": true}
for line in r.stdout.splitlines():
line=line.strip()
if not line:
continue
try:
j=json.loads(line)
if j.get("available") is True:
return True
except:
pass
return "available\": true" in r.stdout.lower() or '"available":true' in r.stdout.lower()
except Exception:
return False
class AppleLLM:
def __init__(self, bin_path: Path, verbose: bool=False):
self.bin_path = Path(bin_path)
self.verbose = verbose
self.proc: Optional[subprocess.Popen] = None
self.lock = threading.Lock()
self._req_id = 0
self._pending = {} # id -> queue
self._reader_thread: Optional[threading.Thread] = None
self._start()
def _start(self):
if not self.bin_path.exists():
raise FileNotFoundError(str(self.bin_path))
self.proc = subprocess.Popen(
[str(self.bin_path)],
stdin=subprocess.PIPE,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
bufsize=1,
text=True,
)
# small warmup wait for [ready] log on stderr
time.sleep(0.5)
# start reader
self._reader_thread = threading.Thread(target=self._reader_loop, daemon=True)
self._reader_thread.start()
if self.verbose:
print(f"[AppleLLM] started {self.bin_path}")
def _reader_loop(self):
assert self.proc and self.proc.stdout
for line in self.proc.stdout:
line=line.strip()
if not line:
continue
try:
obj=json.loads(line)
except:
continue
rid = obj.get("id")
if rid is not None and rid in self._pending:
self._pending[rid].put(obj)
else:
# no id? try broadcast
for q in list(self._pending.values()):
q.put(obj)
def _call(self, payload: dict, timeout: float=15.0) -> dict:
if not self.proc or self.proc.poll() is not None:
# restart
try:
self._start()
except Exception as e:
return {"ok": False, "text": payload.get("text",""), "error": f"restart failed {e}"}
rid = str(self._req_id)
self._req_id += 1
payload["id"] = rid
q: queue.Queue = queue.Queue()
self._pending[rid] = q
try:
line = json.dumps(payload, ensure_ascii=False)
assert self.proc and self.proc.stdin
with self.lock:
self.proc.stdin.write(line+"\n")
self.proc.stdin.flush()
try:
resp = q.get(timeout=timeout)
return resp
except queue.Empty:
return {"ok": False, "text": payload.get("text",""), "error": f"timeout {timeout}s", "id": rid}
except Exception as e:
return {"ok": False, "text": payload.get("text",""), "error": str(e), "id": rid}
finally:
self._pending.pop(rid, None)
def polish_line(self, text: str, prev1: str="", prev2: str="", language: Optional[str]=None, timeout: float=8.0) -> tuple[str,bool,float]:
"""returns (polished_text, ok, ms)"""
resp = self._call({
"mode": "line",
"text": text,
"prev1": prev1,
"prev2": prev2,
"language": language,
}, timeout=timeout)
ok = bool(resp.get("ok"))
return (resp.get("text","") or text, ok, float(resp.get("ms",0) or 0))
def polish_paragraph(self, new_text: str, context: str="", prev_source: str="", language: Optional[str]=None, timeout: float=10.0) -> tuple[str,bool,float]:
resp = self._call({
"mode": "paragraph",
"text": new_text,
"context": context,
"prevSource": prev_source,
"language": language,
}, timeout=timeout)
ok = bool(resp.get("ok"))
return (resp.get("text","") or new_text, ok, float(resp.get("ms",0) or 0))
def close(self):
try:
if self.proc and self.proc.stdin:
try:
self.proc.stdin.close()
except: pass
if self.proc:
self.proc.wait(timeout=2)
except:
try:
if self.proc:
self.proc.kill()
except: pass
# Singleton for multiprocess use (each process gets own via import)
_apple_llm_instance: Optional[AppleLLM] = None
def get_apple_llm(verbose: bool=False, force_new: bool=False) -> Optional[AppleLLM]:
global _apple_llm_instance
if force_new:
_apple_llm_instance = None
if _apple_llm_instance is None:
bp = resolve_llm_binary()
if not bp:
return None
if not check_apple_llm_available(bp):
return None
try:
_apple_llm_instance = AppleLLM(bp, verbose=verbose)
except Exception as e:
print(f"[AppleLLM] start failed: {e}", file=sys.stderr)
return None
return _apple_llm_instance