Add LLM logging and prompt experiment harness
This commit is contained in:
@@ -29,6 +29,32 @@ A real-time, low-latency audio transcription and translation tool utilizing Open
|
|||||||
|
|
||||||
Run the script using `python3 transcribe.py` with optional flags.
|
Run the script using `python3 transcribe.py` with optional flags.
|
||||||
|
|
||||||
|
### LLM Prompt Testing
|
||||||
|
You can now send the line-correction or paragraph-refinement prompts directly without starting live transcription. Every LLM request is appended as JSON Lines to `logs/llm_requests.jsonl` by default.
|
||||||
|
|
||||||
|
- **Test the line-correction prompt:**
|
||||||
|
```bash
|
||||||
|
python3 main_v2.py --post-correct-model qwen3.5:0.8b --llm-test-line "so um we should probably ship it tomorrow" --llm-test-prev1 "We finished the staging deploy." --llm-test-prev2 "QA signed off this morning."
|
||||||
|
```
|
||||||
|
- **Test the paragraph-refinement prompt:**
|
||||||
|
```bash
|
||||||
|
python3 main_v2.py --post-correct-model qwen3.5:0.8b --llm-test-context "We reviewed the launch checklist." --llm-test-segments "we confirmed monitoring we confirmed rollback and then talked about the release window"
|
||||||
|
```
|
||||||
|
- **Change where requests are logged:**
|
||||||
|
```bash
|
||||||
|
python3 main_v2.py --llm-request-log-path tmp/my_llm_requests.jsonl --llm-test-line "example text"
|
||||||
|
```
|
||||||
|
- **Replay the most recent logged request:**
|
||||||
|
```bash
|
||||||
|
python3 main_v2.py --llm-test-from-log -1
|
||||||
|
```
|
||||||
|
- **Replay a logged request using the original model from the log entry:**
|
||||||
|
```bash
|
||||||
|
python3 main_v2.py --llm-test-from-log 12 --llm-test-use-logged-model
|
||||||
|
```
|
||||||
|
|
||||||
|
Each log entry includes the timestamp, provider, model, mode, temperature, and full `messages` payload that was sent to the LLM.
|
||||||
|
|
||||||
### Common Commands
|
### Common Commands
|
||||||
- **List available audio devices:**
|
- **List available audio devices:**
|
||||||
```bash
|
```bash
|
||||||
|
|||||||
@@ -25,6 +25,7 @@
|
|||||||
"post_correct_min_overlap": 0.45,
|
"post_correct_min_overlap": 0.45,
|
||||||
"post_correct_debug": false,
|
"post_correct_debug": false,
|
||||||
"llm_paragraph": false,
|
"llm_paragraph": false,
|
||||||
|
"llm_request_log_path": "logs/llm_requests.jsonl",
|
||||||
"only_translate_llm": false,
|
"only_translate_llm": false,
|
||||||
"temperature_fallback": [0.0, 0.2, 0.4, 0.6, 0.8, 1.0],
|
"temperature_fallback": [0.0, 0.2, 0.4, 0.6, 0.8, 1.0],
|
||||||
"logprob_threshold": -0.8,
|
"logprob_threshold": -0.8,
|
||||||
|
|||||||
+264
-44
@@ -19,6 +19,244 @@ import re
|
|||||||
import json
|
import json
|
||||||
import requests
|
import requests
|
||||||
from collections import deque
|
from collections import deque
|
||||||
|
from datetime import datetime, timezone
|
||||||
|
|
||||||
|
|
||||||
|
def build_paragraph_messages(previous_refined_context, previous_source_text, new_source_text):
|
||||||
|
system_prompt = """You are a careful live transcript editor working from noisy translated source text.
|
||||||
|
|
||||||
|
Goal:
|
||||||
|
Produce the most faithful readable English update.
|
||||||
|
|
||||||
|
Decision rules:
|
||||||
|
1. NEW SOURCE TEXT is the primary evidence and should dominate the output.
|
||||||
|
2. Use PREVIOUS SOURCE TEXT and PREVIOUS REFINED CONTEXT only when they clearly help resolve or continue the new material.
|
||||||
|
3. If the new material starts a fresh thought, output only the new material.
|
||||||
|
4. Remove exact or near-exact repetition unless the repetition is clearly intentional rhetoric.
|
||||||
|
5. Do not add explanations, disclaimers, or meta commentary.
|
||||||
|
6. Prefer conservative wording over guessed meaning.
|
||||||
|
7. Return only the revised transcript text."""
|
||||||
|
|
||||||
|
prompt = f"""Edit this transcript update.
|
||||||
|
|
||||||
|
[PREVIOUS REFINED CONTEXT]
|
||||||
|
{previous_refined_context}
|
||||||
|
|
||||||
|
[PREVIOUS SOURCE TEXT]
|
||||||
|
{previous_source_text}
|
||||||
|
|
||||||
|
[NEW SOURCE TEXT]
|
||||||
|
{new_source_text}
|
||||||
|
"""
|
||||||
|
return [
|
||||||
|
{"role": "system", "content": system_prompt},
|
||||||
|
{"role": "user", "content": prompt},
|
||||||
|
], prompt
|
||||||
|
|
||||||
|
|
||||||
|
def build_line_messages(prev1, prev2, corrected):
|
||||||
|
system_prompt = """You are a real-time English caption corrector for live speech.
|
||||||
|
|
||||||
|
Task:
|
||||||
|
Clean only the current caption line.
|
||||||
|
|
||||||
|
Hard rules:
|
||||||
|
1. Preserve meaning exactly. Never replace current content with prior context.
|
||||||
|
2. Remove disfluencies and false starts.
|
||||||
|
3. Fix punctuation, casing, and obvious STT typos.
|
||||||
|
4. If uncertain, return the original line unchanged.
|
||||||
|
5. Output only one corrected English line."""
|
||||||
|
|
||||||
|
prompt = (
|
||||||
|
"Context (reference only):\n"
|
||||||
|
f"Previous line 1: {prev1}\n"
|
||||||
|
f"Previous line 2: {prev2}\n"
|
||||||
|
"Current line to correct:\n"
|
||||||
|
f"{corrected}\n"
|
||||||
|
"Corrected:"
|
||||||
|
)
|
||||||
|
return [
|
||||||
|
{"role": "system", "content": system_prompt},
|
||||||
|
{"role": "user", "content": prompt},
|
||||||
|
], prompt
|
||||||
|
|
||||||
|
|
||||||
|
def append_llm_request_log(log_path, entry):
|
||||||
|
if not log_path:
|
||||||
|
return
|
||||||
|
log_dir = os.path.dirname(log_path)
|
||||||
|
if log_dir:
|
||||||
|
os.makedirs(log_dir, exist_ok=True)
|
||||||
|
with open(log_path, "a", encoding="utf-8") as handle:
|
||||||
|
json.dump(entry, handle, ensure_ascii=False)
|
||||||
|
handle.write("\n")
|
||||||
|
|
||||||
|
|
||||||
|
def load_llm_request_log_entry(log_path, entry_index):
|
||||||
|
if not log_path or not os.path.exists(log_path):
|
||||||
|
raise FileNotFoundError(f"LLM request log not found: {log_path}")
|
||||||
|
|
||||||
|
with open(log_path, "r", encoding="utf-8") as handle:
|
||||||
|
entries = [json.loads(line) for line in handle if line.strip()]
|
||||||
|
|
||||||
|
if not entries:
|
||||||
|
raise ValueError(f"LLM request log is empty: {log_path}")
|
||||||
|
|
||||||
|
if entry_index is None or entry_index == -1:
|
||||||
|
return entries[-1]
|
||||||
|
|
||||||
|
if entry_index < 0:
|
||||||
|
entry_index = len(entries) + entry_index
|
||||||
|
|
||||||
|
if entry_index < 0 or entry_index >= len(entries):
|
||||||
|
raise IndexError(f"LLM request log index {entry_index} is out of range for {len(entries)} entries")
|
||||||
|
|
||||||
|
return entries[entry_index]
|
||||||
|
|
||||||
|
|
||||||
|
def run_llm_prompt_test(args):
|
||||||
|
from dotenv import load_dotenv
|
||||||
|
load_dotenv()
|
||||||
|
|
||||||
|
api_key = os.environ.get("OPENAI_API_KEY")
|
||||||
|
log_path = getattr(args, "llm_request_log_path", "logs/llm_requests.jsonl")
|
||||||
|
|
||||||
|
def call_openai(model, messages):
|
||||||
|
response = requests.post(
|
||||||
|
"https://api.openai.com/v1/chat/completions",
|
||||||
|
json={"model": model, "messages": messages},
|
||||||
|
headers={"Authorization": f"Bearer {api_key}", "Content-Type": "application/json"},
|
||||||
|
timeout=getattr(args, "post_correct_llm_timeout", 8.0),
|
||||||
|
)
|
||||||
|
if response.status_code >= 400:
|
||||||
|
try:
|
||||||
|
payload = response.json()
|
||||||
|
detail = json.dumps(payload, ensure_ascii=True)
|
||||||
|
except Exception:
|
||||||
|
detail = response.text.strip()
|
||||||
|
raise RuntimeError(f"OpenAI API error {response.status_code}: {detail}")
|
||||||
|
return response.json()["choices"][0]["message"]["content"].strip()
|
||||||
|
|
||||||
|
def call_ollama(model, prompt, temperature, system=None):
|
||||||
|
response = requests.post(
|
||||||
|
getattr(args, "post_correct_ollama_url", "http://127.0.0.1:11434/api/generate"),
|
||||||
|
json={
|
||||||
|
"model": model,
|
||||||
|
"prompt": prompt,
|
||||||
|
"system": system or "",
|
||||||
|
"stream": False,
|
||||||
|
"options": {"temperature": temperature},
|
||||||
|
"keep_alive": getattr(args, "post_correct_keep_alive", "30m"),
|
||||||
|
},
|
||||||
|
timeout=getattr(args, "post_correct_llm_timeout", 8.0),
|
||||||
|
)
|
||||||
|
response.raise_for_status()
|
||||||
|
return response.json().get("response", "").strip()
|
||||||
|
|
||||||
|
def send_test_request(mode, messages, prompt, temperature, metadata, model=None):
|
||||||
|
request_model = model or args.post_correct_model
|
||||||
|
is_openai = request_model.startswith("gpt-")
|
||||||
|
entry = {
|
||||||
|
"timestamp": datetime.now(timezone.utc).isoformat(),
|
||||||
|
"mode": mode,
|
||||||
|
"provider": "openai" if is_openai else "ollama",
|
||||||
|
"model": request_model,
|
||||||
|
"temperature": temperature,
|
||||||
|
"messages": messages,
|
||||||
|
"metadata": metadata,
|
||||||
|
}
|
||||||
|
started_at = time.time()
|
||||||
|
if is_openai:
|
||||||
|
if not api_key:
|
||||||
|
raise RuntimeError("OPENAI_API_KEY is not set")
|
||||||
|
try:
|
||||||
|
response_text = call_openai(request_model, messages)
|
||||||
|
entry["response"] = {"status": "ok", "content": response_text}
|
||||||
|
return response_text
|
||||||
|
except Exception as exc:
|
||||||
|
entry["response"] = {"status": "error", "error": str(exc)}
|
||||||
|
raise
|
||||||
|
finally:
|
||||||
|
entry["duration_ms"] = round((time.time() - started_at) * 1000, 2)
|
||||||
|
append_llm_request_log(log_path, entry)
|
||||||
|
system = ""
|
||||||
|
for message in messages:
|
||||||
|
if message.get("role") == "system":
|
||||||
|
system = message.get("content", "")
|
||||||
|
break
|
||||||
|
try:
|
||||||
|
response_text = call_ollama(request_model, prompt, temperature, system=system)
|
||||||
|
entry["response"] = {"status": "ok", "content": response_text}
|
||||||
|
return response_text
|
||||||
|
except Exception as exc:
|
||||||
|
entry["response"] = {"status": "error", "error": str(exc)}
|
||||||
|
raise
|
||||||
|
finally:
|
||||||
|
entry["duration_ms"] = round((time.time() - started_at) * 1000, 2)
|
||||||
|
append_llm_request_log(log_path, entry)
|
||||||
|
|
||||||
|
if getattr(args, "llm_test_from_log", None) is not None:
|
||||||
|
logged_entry = load_llm_request_log_entry(log_path, args.llm_test_from_log)
|
||||||
|
logged_messages = logged_entry.get("messages")
|
||||||
|
if not logged_messages:
|
||||||
|
raise ValueError("Selected log entry does not contain messages")
|
||||||
|
logged_prompt = next((m.get("content", "") for m in logged_messages if m.get("role") == "user"), "")
|
||||||
|
logged_model = logged_entry.get("model", args.post_correct_model)
|
||||||
|
replay_model = logged_model if getattr(args, "llm_test_use_logged_model", False) else args.post_correct_model
|
||||||
|
response = send_test_request(
|
||||||
|
logged_entry.get("mode", "replay"),
|
||||||
|
logged_messages,
|
||||||
|
logged_prompt,
|
||||||
|
logged_entry.get("temperature", 0.1),
|
||||||
|
{
|
||||||
|
"replay_source_log_path": log_path,
|
||||||
|
"replay_source_index": args.llm_test_from_log,
|
||||||
|
"replay_source_timestamp": logged_entry.get("timestamp"),
|
||||||
|
"replay_source_model": logged_model,
|
||||||
|
},
|
||||||
|
model=replay_model,
|
||||||
|
)
|
||||||
|
print("[LLM TEST] Replayed request from log:")
|
||||||
|
print(f"source_index={args.llm_test_from_log} source_model={logged_model} replay_model={replay_model}")
|
||||||
|
print(response)
|
||||||
|
|
||||||
|
if getattr(args, "llm_test_line", None):
|
||||||
|
messages, prompt = build_line_messages(
|
||||||
|
getattr(args, "llm_test_prev1", ""),
|
||||||
|
getattr(args, "llm_test_prev2", ""),
|
||||||
|
args.llm_test_line,
|
||||||
|
)
|
||||||
|
response = send_test_request(
|
||||||
|
"line",
|
||||||
|
messages,
|
||||||
|
prompt,
|
||||||
|
0.1,
|
||||||
|
{
|
||||||
|
"prev1": getattr(args, "llm_test_prev1", ""),
|
||||||
|
"prev2": getattr(args, "llm_test_prev2", ""),
|
||||||
|
"input": args.llm_test_line,
|
||||||
|
},
|
||||||
|
)
|
||||||
|
print("[LLM TEST] Line response:")
|
||||||
|
print(response)
|
||||||
|
|
||||||
|
if getattr(args, "llm_test_segments", None):
|
||||||
|
messages, prompt = build_paragraph_messages(
|
||||||
|
getattr(args, "llm_test_context", ""),
|
||||||
|
args.llm_test_segments,
|
||||||
|
)
|
||||||
|
response = send_test_request(
|
||||||
|
"paragraph",
|
||||||
|
messages,
|
||||||
|
prompt,
|
||||||
|
0.1,
|
||||||
|
{
|
||||||
|
"context": getattr(args, "llm_test_context", ""),
|
||||||
|
"segments": args.llm_test_segments,
|
||||||
|
},
|
||||||
|
)
|
||||||
|
print("[LLM TEST] Paragraph response:")
|
||||||
|
print(response)
|
||||||
|
|
||||||
def run_llm_processor(in_queue, out_queue, args):
|
def run_llm_processor(in_queue, out_queue, args):
|
||||||
from dotenv import load_dotenv
|
from dotenv import load_dotenv
|
||||||
@@ -29,6 +267,7 @@ def run_llm_processor(in_queue, out_queue, args):
|
|||||||
api_key = os.environ.get("OPENAI_API_KEY")
|
api_key = os.environ.get("OPENAI_API_KEY")
|
||||||
is_openai = args.post_correct_model.startswith("gpt-")
|
is_openai = args.post_correct_model.startswith("gpt-")
|
||||||
llm_state = {"line_warned": False, "paragraph_warned": False}
|
llm_state = {"line_warned": False, "paragraph_warned": False}
|
||||||
|
log_path = getattr(args, "llm_request_log_path", "logs/llm_requests.jsonl")
|
||||||
|
|
||||||
def check_ollama_health():
|
def check_ollama_health():
|
||||||
try:
|
try:
|
||||||
@@ -130,49 +369,49 @@ def run_llm_processor(in_queue, out_queue, args):
|
|||||||
return response.json().get("response", "").strip()
|
return response.json().get("response", "").strip()
|
||||||
|
|
||||||
def call_llm(messages, prompt, temperature, warn_key):
|
def call_llm(messages, prompt, temperature, warn_key):
|
||||||
|
entry = {
|
||||||
|
"timestamp": datetime.now(timezone.utc).isoformat(),
|
||||||
|
"mode": "paragraph" if warn_key == "paragraph_warned" else "line",
|
||||||
|
"provider": "openai" if is_openai else "ollama",
|
||||||
|
"model": args.post_correct_model,
|
||||||
|
"temperature": temperature,
|
||||||
|
"messages": messages,
|
||||||
|
}
|
||||||
|
started_at = time.time()
|
||||||
try:
|
try:
|
||||||
if is_openai:
|
if is_openai:
|
||||||
if not api_key:
|
if not api_key:
|
||||||
raise RuntimeError("OPENAI_API_KEY is not set")
|
raise RuntimeError("OPENAI_API_KEY is not set")
|
||||||
return call_openai(messages)
|
response_text = call_openai(messages)
|
||||||
|
entry["response"] = {"status": "ok", "content": response_text}
|
||||||
|
return response_text
|
||||||
system = ""
|
system = ""
|
||||||
if messages:
|
if messages:
|
||||||
for message in messages:
|
for message in messages:
|
||||||
if message.get("role") == "system":
|
if message.get("role") == "system":
|
||||||
system = message.get("content", "")
|
system = message.get("content", "")
|
||||||
break
|
break
|
||||||
return call_ollama(prompt, temperature, system=system)
|
response_text = call_ollama(prompt, temperature, system=system)
|
||||||
|
entry["response"] = {"status": "ok", "content": response_text}
|
||||||
|
return response_text
|
||||||
except Exception as exc:
|
except Exception as exc:
|
||||||
|
entry["response"] = {"status": "error", "error": str(exc)}
|
||||||
if not llm_state[warn_key]:
|
if not llm_state[warn_key]:
|
||||||
label = "Paragraph LLM" if warn_key == "paragraph_warned" else "Line LLM"
|
label = "Paragraph LLM" if warn_key == "paragraph_warned" else "Line LLM"
|
||||||
print(f"[LLM] {label} unavailable ({exc}). Falling back to deterministic text.")
|
print(f"[LLM] {label} unavailable ({exc}). Falling back to deterministic text.")
|
||||||
llm_state[warn_key] = True
|
llm_state[warn_key] = True
|
||||||
return ""
|
return ""
|
||||||
|
finally:
|
||||||
|
entry["duration_ms"] = round((time.time() - started_at) * 1000, 2)
|
||||||
|
append_llm_request_log(log_path, entry)
|
||||||
|
|
||||||
# State: This is the ONLY text we send to the LLM as context
|
# State: This is the ONLY text we send to the LLM as context
|
||||||
active_context = ""
|
active_context = ""
|
||||||
|
previous_source_window = ""
|
||||||
recent_lines = deque(maxlen=2)
|
recent_lines = deque(maxlen=2)
|
||||||
|
|
||||||
def call_llm_rolling_refine(new_segments_str, context):
|
def call_llm_rolling_refine(new_segments_str, context, previous_source_text):
|
||||||
prompt = f"""Task: Refine the following live transcription stream into clean, professional paragraphs.
|
messages, prompt = build_paragraph_messages(context, previous_source_text, new_segments_str)
|
||||||
|
|
||||||
[PREVIOUS WORKING CONTEXT]
|
|
||||||
"{context}"
|
|
||||||
|
|
||||||
[NEW RAW ASR SEGMENTS]
|
|
||||||
"{new_segments_str}"
|
|
||||||
|
|
||||||
[INSTRUCTIONS]
|
|
||||||
1. INTEGRATE: Polished and merge the new segments into the flow of the 'PREVIOUS WORKING CONTEXT'.
|
|
||||||
2. CONSOLIDATE: Remove redundant repetitions and translator echoes.
|
|
||||||
3. ORGANIZE: Use a double newline (\\n\\n) to start a new paragraph when a topic changes or the current one is complete.
|
|
||||||
4. TARGET LANGUAGE: Output ONLY in English.
|
|
||||||
5. OUTPUT: Provide ONLY the refined, consolidated text. Do not explain anything.
|
|
||||||
"""
|
|
||||||
messages = [
|
|
||||||
{"role": "system", "content": "You are a live transcript editor."},
|
|
||||||
{"role": "user", "content": prompt},
|
|
||||||
]
|
|
||||||
return call_llm(messages, prompt, 0.1, "paragraph_warned")
|
return call_llm(messages, prompt, 0.1, "paragraph_warned")
|
||||||
|
|
||||||
def post_correct_line(text):
|
def post_correct_line(text):
|
||||||
@@ -185,27 +424,7 @@ def run_llm_processor(in_queue, out_queue, args):
|
|||||||
|
|
||||||
prev2 = recent_lines[-2] if len(recent_lines) >= 2 else ""
|
prev2 = recent_lines[-2] if len(recent_lines) >= 2 else ""
|
||||||
prev1 = recent_lines[-1] if len(recent_lines) >= 1 else ""
|
prev1 = recent_lines[-1] if len(recent_lines) >= 1 else ""
|
||||||
prompt = (
|
messages, prompt = build_line_messages(prev1, prev2, corrected)
|
||||||
"You are a real-time English caption corrector for live speech.\n"
|
|
||||||
"Task:\n"
|
|
||||||
"Clean ONLY the current caption line.\n"
|
|
||||||
"Hard rules:\n"
|
|
||||||
"1. Preserve meaning exactly. Never replace current content with prior context.\n"
|
|
||||||
"2. Remove disfluencies and false starts.\n"
|
|
||||||
"3. Fix punctuation, casing, and obvious STT typos.\n"
|
|
||||||
"4. If uncertain, return the original line unchanged.\n"
|
|
||||||
"5. Output ONLY one corrected English line.\n"
|
|
||||||
"Context (reference only):\n"
|
|
||||||
f"Previous line 1: {prev1}\n"
|
|
||||||
f"Previous line 2: {prev2}\n"
|
|
||||||
"Current line to correct:\n"
|
|
||||||
f"{corrected}\n"
|
|
||||||
"Corrected:"
|
|
||||||
)
|
|
||||||
messages = [
|
|
||||||
{"role": "system", "content": "You are a real-time English caption corrector."},
|
|
||||||
{"role": "user", "content": prompt},
|
|
||||||
]
|
|
||||||
candidate = normalize_english_caption(call_llm(messages, prompt, 0.1, "line_warned"))
|
candidate = normalize_english_caption(call_llm(messages, prompt, 0.1, "line_warned"))
|
||||||
if candidate and word_overlap_ratio(corrected, candidate) >= getattr(args, "post_correct_min_overlap", 0.45):
|
if candidate and word_overlap_ratio(corrected, candidate) >= getattr(args, "post_correct_min_overlap", 0.45):
|
||||||
return candidate
|
return candidate
|
||||||
@@ -244,7 +463,7 @@ def run_llm_processor(in_queue, out_queue, args):
|
|||||||
paragraph_from_llm = False
|
paragraph_from_llm = False
|
||||||
if args.llm_paragraph and should_refine:
|
if args.llm_paragraph and should_refine:
|
||||||
new_batch = " ".join(pending_buffer)
|
new_batch = " ".join(pending_buffer)
|
||||||
result = call_llm_rolling_refine(new_batch, active_context)
|
result = call_llm_rolling_refine(new_batch, active_context, previous_source_window)
|
||||||
|
|
||||||
if result:
|
if result:
|
||||||
# Logic to handle paragraph breaks
|
# Logic to handle paragraph breaks
|
||||||
@@ -263,6 +482,7 @@ def run_llm_processor(in_queue, out_queue, args):
|
|||||||
else:
|
else:
|
||||||
structured_paragraph = new_batch
|
structured_paragraph = new_batch
|
||||||
active_context = new_batch
|
active_context = new_batch
|
||||||
|
previous_source_window = new_batch
|
||||||
|
|
||||||
pending_buffer = []
|
pending_buffer = []
|
||||||
last_llm_call_time = time.time()
|
last_llm_call_time = time.time()
|
||||||
|
|||||||
+18
@@ -279,3 +279,21 @@ The project now features a high-performance, Apple Silicon-optimized pipeline th
|
|||||||
- Removed the explicit Chat Completions temperature override after discovering current GPT-5-family models reject non-default temperature values on this endpoint.
|
- Removed the explicit Chat Completions temperature override after discovering current GPT-5-family models reject non-default temperature values on this endpoint.
|
||||||
- Aligned the Ollama path with the OpenAI path by forwarding the same system instructions to local generation requests, including warmup calls.
|
- Aligned the Ollama path with the OpenAI path by forwarding the same system instructions to local generation requests, including warmup calls.
|
||||||
- **Outcome:** GPT-backed runs now fail with useful diagnostics, OpenAI requests are compatible with current GPT-5-mini behavior, and local-vs-remote prompting is more consistent.
|
- **Outcome:** GPT-backed runs now fail with useful diagnostics, OpenAI requests are compatible with current GPT-5-mini behavior, and local-vs-remote prompting is more consistent.
|
||||||
|
|
||||||
|
## Phase 32: LLM Request Logging + Prompt Replay Harness
|
||||||
|
- **Goal:** Make prompt behavior observable and reproducible so local LLM tuning can happen without rerunning live transcription.
|
||||||
|
- **Approach:**
|
||||||
|
- Added JSONL request logging for every LLM call, including full message payloads, timing, and the final response or error outcome.
|
||||||
|
- Added prompt-only test entry points in `main_v2.py` so line-correction and paragraph-refinement prompts can be sent directly without starting the transcription pipeline.
|
||||||
|
- Added replay-from-log support so a captured request can be resent with the current model or the originally logged model.
|
||||||
|
- Documented the new logging and replay workflow in `README.md` and added a default log path in `config.json`.
|
||||||
|
- **Outcome:** We can now inspect exactly what the app sends to the LLM, replay real requests on demand, and compare prompt changes against stored payloads and outputs.
|
||||||
|
|
||||||
|
## Phase 33: Grounded Paragraph Prompt Experiments
|
||||||
|
- **Goal:** Reduce paragraph drift and self-reinforcing hallucinations in the rolling LLM refiner.
|
||||||
|
- **Approach:**
|
||||||
|
- Built `prompt_experiments.py` to reconstruct representative paragraph-refinement samples from `distribute_debug.log` and send multiple prompt variants directly to Ollama.
|
||||||
|
- Compared several context arrangements, including the previous refined paragraph alone, raw-only windows, and mixed refined-plus-raw grounding strategies.
|
||||||
|
- Identified the best-performing variant as one that treats the new source window as primary evidence while using prior refined and prior raw text only when they clearly continue the thought.
|
||||||
|
- Updated `engine_llm.py` so live paragraph refinement now sends `PREVIOUS REFINED CONTEXT`, `PREVIOUS SOURCE TEXT`, and `NEW SOURCE TEXT` with a more conservative, grounding-oriented system prompt.
|
||||||
|
- **Outcome:** Paragraph refinement is now substantially better anchored to current source text, with less stale-context carryover and a much lower rate of unsupported content in direct Ollama tests.
|
||||||
|
|||||||
+13
-1
@@ -10,7 +10,7 @@ load_dotenv()
|
|||||||
|
|
||||||
# Light imports (heavy ones moved inside run functions)
|
# Light imports (heavy ones moved inside run functions)
|
||||||
from engine_transcribe import run_transcription, list_audio_devices
|
from engine_transcribe import run_transcription, list_audio_devices
|
||||||
from engine_llm import run_llm_processor
|
from engine_llm import run_llm_processor, run_llm_prompt_test
|
||||||
from engine_translate import run_translation
|
from engine_translate import run_translation
|
||||||
from engine_distribute import run_distribution
|
from engine_distribute import run_distribution
|
||||||
|
|
||||||
@@ -57,6 +57,14 @@ def main():
|
|||||||
parser.add_argument("--post-correct-min-overlap", type=float, default=defaults.get("post_correct_min_overlap", 0.45))
|
parser.add_argument("--post-correct-min-overlap", type=float, default=defaults.get("post_correct_min_overlap", 0.45))
|
||||||
parser.add_argument("--post-correct-debug", action="store_true", default=defaults.get("post_correct_debug", False))
|
parser.add_argument("--post-correct-debug", action="store_true", default=defaults.get("post_correct_debug", False))
|
||||||
parser.add_argument("--llm-paragraph", action="store_true", default=defaults.get("llm_paragraph", False))
|
parser.add_argument("--llm-paragraph", action="store_true", default=defaults.get("llm_paragraph", False))
|
||||||
|
parser.add_argument("--llm-request-log-path", type=str, default=defaults.get("llm_request_log_path", "logs/llm_requests.jsonl"))
|
||||||
|
parser.add_argument("--llm-test-line", type=str, default=None, help="Send a single line-correction prompt without starting transcription.")
|
||||||
|
parser.add_argument("--llm-test-prev1", type=str, default="", help="Optional previous line 1 context for --llm-test-line.")
|
||||||
|
parser.add_argument("--llm-test-prev2", type=str, default="", help="Optional previous line 2 context for --llm-test-line.")
|
||||||
|
parser.add_argument("--llm-test-segments", type=str, default=None, help="Send a single paragraph-refinement prompt without starting transcription.")
|
||||||
|
parser.add_argument("--llm-test-context", type=str, default="", help="Optional existing paragraph context for --llm-test-segments.")
|
||||||
|
parser.add_argument("--llm-test-from-log", type=int, default=None, help="Replay a logged LLM request by JSONL entry index. Use -1 for the most recent entry.")
|
||||||
|
parser.add_argument("--llm-test-use-logged-model", action="store_true", default=False, help="When replaying from log, use the model stored in the log entry instead of --post-correct-model.")
|
||||||
|
|
||||||
# Translate Args
|
# Translate Args
|
||||||
parser.add_argument("-es", action="store_true", default=defaults.get("es", False))
|
parser.add_argument("-es", action="store_true", default=defaults.get("es", False))
|
||||||
@@ -80,6 +88,10 @@ def main():
|
|||||||
print("[Main] Enabling --llm-paragraph because --only-translate-llm was requested.")
|
print("[Main] Enabling --llm-paragraph because --only-translate-llm was requested.")
|
||||||
args.llm_paragraph = True
|
args.llm_paragraph = True
|
||||||
|
|
||||||
|
if args.llm_test_line or args.llm_test_segments or args.llm_test_from_log is not None:
|
||||||
|
run_llm_prompt_test(args)
|
||||||
|
return
|
||||||
|
|
||||||
# Handle device listing
|
# Handle device listing
|
||||||
if args.list_devices:
|
if args.list_devices:
|
||||||
list_audio_devices()
|
list_audio_devices()
|
||||||
|
|||||||
@@ -0,0 +1,356 @@
|
|||||||
|
import argparse
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import re
|
||||||
|
import statistics
|
||||||
|
from collections import Counter
|
||||||
|
from urllib import request, error
|
||||||
|
|
||||||
|
|
||||||
|
def parse_distribute_debug_log(path):
|
||||||
|
entries = []
|
||||||
|
current = {"final": None, "en_bridge": None, "llm_paragraph": None}
|
||||||
|
|
||||||
|
with open(path, "r", encoding="utf-8") as handle:
|
||||||
|
for raw_line in handle:
|
||||||
|
line = raw_line.rstrip("\n")
|
||||||
|
if "[Final]:" in line:
|
||||||
|
current["final"] = line.split("[Final]:", 1)[1].strip()
|
||||||
|
elif "[EN-BRIDGE]:" in line:
|
||||||
|
current["en_bridge"] = line.split("[EN-BRIDGE]:", 1)[1].strip()
|
||||||
|
elif "[LLM-PARAGRAPH]:" in line:
|
||||||
|
current["llm_paragraph"] = line.split("[LLM-PARAGRAPH]:", 1)[1].strip()
|
||||||
|
entries.append(current)
|
||||||
|
current = {"final": None, "en_bridge": None, "llm_paragraph": None}
|
||||||
|
elif not line.strip():
|
||||||
|
continue
|
||||||
|
|
||||||
|
return entries
|
||||||
|
|
||||||
|
|
||||||
|
def normalize_paragraph(text):
|
||||||
|
return (text or "").strip().strip('"').strip()
|
||||||
|
|
||||||
|
|
||||||
|
def reconstruct_samples(path):
|
||||||
|
samples = []
|
||||||
|
rows = parse_distribute_debug_log(path)
|
||||||
|
previous_llm = ""
|
||||||
|
pending_bridges = []
|
||||||
|
recent_raw_windows = []
|
||||||
|
|
||||||
|
for row in rows:
|
||||||
|
if row.get("en_bridge"):
|
||||||
|
pending_bridges.append(row["en_bridge"])
|
||||||
|
if not row.get("llm_paragraph"):
|
||||||
|
continue
|
||||||
|
|
||||||
|
llm_output = normalize_paragraph(row["llm_paragraph"])
|
||||||
|
new_raw = " ".join(pending_bridges).strip()
|
||||||
|
prev_raw = recent_raw_windows[-1] if recent_raw_windows else ""
|
||||||
|
|
||||||
|
samples.append(
|
||||||
|
{
|
||||||
|
"previous_refined": previous_llm,
|
||||||
|
"previous_raw": prev_raw,
|
||||||
|
"new_raw": new_raw,
|
||||||
|
"observed_output": llm_output,
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
|
previous_llm = llm_output
|
||||||
|
recent_raw_windows.append(new_raw)
|
||||||
|
pending_bridges = []
|
||||||
|
|
||||||
|
return samples
|
||||||
|
|
||||||
|
|
||||||
|
def build_variants(sample):
|
||||||
|
prev_refined = sample["previous_refined"]
|
||||||
|
prev_raw = sample["previous_raw"]
|
||||||
|
new_raw = sample["new_raw"]
|
||||||
|
|
||||||
|
variants = []
|
||||||
|
|
||||||
|
system_a = """You are a live transcript editor.
|
||||||
|
|
||||||
|
Task:
|
||||||
|
Refine live transcription stream into clean, professional paragraphs.
|
||||||
|
|
||||||
|
Rules:
|
||||||
|
1. Integrate new segments into the flow of the previous working context.
|
||||||
|
2. Remove redundant repetitions and translator echoes.
|
||||||
|
3. Use a double newline (\\n\\n) to start a new paragraph when a topic changes or the current one is complete.
|
||||||
|
4. Output only in English.
|
||||||
|
5. Return only the refined, consolidated text with no explanation."""
|
||||||
|
user_a = f"""Refine the following live transcription stream.
|
||||||
|
|
||||||
|
[PREVIOUS WORKING CONTEXT]
|
||||||
|
"{prev_refined}"
|
||||||
|
|
||||||
|
[NEW SOURCE TEXT]
|
||||||
|
"{new_raw}"
|
||||||
|
"""
|
||||||
|
variants.append(("current_refined_plus_new", system_a, user_a))
|
||||||
|
|
||||||
|
system_b = """You are a live transcript editor for noisy speech recognition output.
|
||||||
|
|
||||||
|
Goal:
|
||||||
|
Produce a clean English paragraph update grounded in the source text.
|
||||||
|
|
||||||
|
Grounding rules:
|
||||||
|
1. Treat PREVIOUS REFINED CONTEXT as style and continuity help, not as unquestionable truth.
|
||||||
|
2. Use PREVIOUS SOURCE TEXT and NEW SOURCE TEXT as the factual anchor.
|
||||||
|
3. Do not add details that are not supported by the source text.
|
||||||
|
4. Preserve uncertainty when the source is unclear instead of inventing a cleaner claim.
|
||||||
|
5. Remove obvious repetitions and translator echoes.
|
||||||
|
6. Start a new paragraph with a double newline only when there is a clear topic shift.
|
||||||
|
7. Return only the refined English text with no explanation."""
|
||||||
|
user_b = f"""Update the transcript using the source text below.
|
||||||
|
|
||||||
|
[PREVIOUS REFINED CONTEXT]
|
||||||
|
"{prev_refined}"
|
||||||
|
|
||||||
|
[PREVIOUS SOURCE TEXT]
|
||||||
|
"{prev_raw}"
|
||||||
|
|
||||||
|
[NEW SOURCE TEXT]
|
||||||
|
"{new_raw}"
|
||||||
|
"""
|
||||||
|
variants.append(("refined_plus_prev_raw_plus_new_raw", system_b, user_b))
|
||||||
|
|
||||||
|
system_c = """You are a careful transcript editor.
|
||||||
|
|
||||||
|
Goal:
|
||||||
|
Produce clean English paragraphs directly from noisy source text.
|
||||||
|
|
||||||
|
Rules:
|
||||||
|
1. Use only the supplied source text as evidence.
|
||||||
|
2. Do not infer missing facts.
|
||||||
|
3. Keep wording conservative when the source is noisy.
|
||||||
|
4. Remove repeated fragments and obvious translation artifacts.
|
||||||
|
5. Use a double newline only for a clear paragraph break.
|
||||||
|
6. Return only the refined English text."""
|
||||||
|
user_c = f"""Refine this transcript source.
|
||||||
|
|
||||||
|
[RECENT SOURCE TEXT]
|
||||||
|
"{prev_raw}"
|
||||||
|
|
||||||
|
[NEW SOURCE TEXT]
|
||||||
|
"{new_raw}"
|
||||||
|
"""
|
||||||
|
variants.append(("prev_raw_plus_new_raw_only", system_c, user_c))
|
||||||
|
|
||||||
|
system_d = """You are a live transcript editor for sermon audio translated into English.
|
||||||
|
|
||||||
|
Goal:
|
||||||
|
Create the most faithful readable English paragraph you can from imperfect source text.
|
||||||
|
|
||||||
|
Priority order:
|
||||||
|
1. Faithfulness to source text.
|
||||||
|
2. Remove duplicated phrases, stutters, and translation echoes.
|
||||||
|
3. Maintain continuity with the previous refined context only when it matches the source text.
|
||||||
|
4. Prefer slight awkwardness over hallucination.
|
||||||
|
5. If a phrase is unclear, keep it modestly literal instead of guessing.
|
||||||
|
6. Use a double newline only for a real topic change.
|
||||||
|
7. Return only the revised English transcript.
|
||||||
|
|
||||||
|
Hard constraints:
|
||||||
|
- Do not add people, events, or meanings absent from the source text.
|
||||||
|
- Do not turn rhetorical questions into factual claims unless the source clearly does that.
|
||||||
|
- Do not overwrite NEW SOURCE TEXT with PREVIOUS REFINED CONTEXT."""
|
||||||
|
user_d = f"""Revise the transcript update.
|
||||||
|
|
||||||
|
[PREVIOUS REFINED CONTEXT]
|
||||||
|
"{prev_refined}"
|
||||||
|
|
||||||
|
[RECENT SOURCE WINDOW]
|
||||||
|
"{prev_raw}"
|
||||||
|
|
||||||
|
[CURRENT SOURCE WINDOW]
|
||||||
|
"{new_raw}"
|
||||||
|
"""
|
||||||
|
variants.append(("grounded_sermon_editor", system_d, user_d))
|
||||||
|
|
||||||
|
system_e = """You are a careful live transcript editor working from noisy translated source text.
|
||||||
|
|
||||||
|
Goal:
|
||||||
|
Produce the most faithful readable English update.
|
||||||
|
|
||||||
|
Decision rules:
|
||||||
|
1. NEW SOURCE TEXT is the primary evidence and should dominate the output.
|
||||||
|
2. Use PREVIOUS SOURCE TEXT and PREVIOUS REFINED CONTEXT only when they clearly help resolve or continue the new material.
|
||||||
|
3. If the new material starts a fresh thought, output only the new material.
|
||||||
|
4. Remove exact or near-exact repetition unless the repetition is clearly intentional rhetoric.
|
||||||
|
5. Do not add explanations, disclaimers, or meta commentary.
|
||||||
|
6. Prefer conservative wording over guessed meaning.
|
||||||
|
7. Return only the revised transcript text."""
|
||||||
|
user_e = f"""Edit this transcript update.
|
||||||
|
|
||||||
|
[PREVIOUS REFINED CONTEXT]
|
||||||
|
"{prev_refined}"
|
||||||
|
|
||||||
|
[PREVIOUS SOURCE TEXT]
|
||||||
|
"{prev_raw}"
|
||||||
|
|
||||||
|
[NEW SOURCE TEXT]
|
||||||
|
"{new_raw}"
|
||||||
|
"""
|
||||||
|
variants.append(("grounded_new_dominant", system_e, user_e))
|
||||||
|
|
||||||
|
return variants
|
||||||
|
|
||||||
|
|
||||||
|
def ollama_generate(model, system_prompt, user_prompt, url, timeout, keep_alive):
|
||||||
|
payload = json.dumps(
|
||||||
|
{
|
||||||
|
"model": model,
|
||||||
|
"prompt": user_prompt,
|
||||||
|
"system": system_prompt,
|
||||||
|
"stream": False,
|
||||||
|
"options": {"temperature": 0.1},
|
||||||
|
"keep_alive": keep_alive,
|
||||||
|
}
|
||||||
|
).encode("utf-8")
|
||||||
|
req = request.Request(
|
||||||
|
url,
|
||||||
|
data=payload,
|
||||||
|
headers={"Content-Type": "application/json"},
|
||||||
|
method="POST",
|
||||||
|
)
|
||||||
|
try:
|
||||||
|
with request.urlopen(req, timeout=timeout) as response:
|
||||||
|
body = json.loads(response.read().decode("utf-8"))
|
||||||
|
except error.HTTPError as exc:
|
||||||
|
detail = exc.read().decode("utf-8", errors="replace")
|
||||||
|
raise RuntimeError(f"Ollama HTTP error {exc.code}: {detail}") from exc
|
||||||
|
return body.get("response", "").strip()
|
||||||
|
|
||||||
|
|
||||||
|
def tokenize(text):
|
||||||
|
return re.findall(r"[a-z0-9']+", (text or "").lower())
|
||||||
|
|
||||||
|
|
||||||
|
def repeated_ngram_ratio(text, n=3):
|
||||||
|
tokens = tokenize(text)
|
||||||
|
if len(tokens) < n:
|
||||||
|
return 0.0
|
||||||
|
grams = [tuple(tokens[i:i + n]) for i in range(len(tokens) - n + 1)]
|
||||||
|
counts = Counter(grams)
|
||||||
|
repeated = sum(count - 1 for count in counts.values() if count > 1)
|
||||||
|
return repeated / max(1, len(grams))
|
||||||
|
|
||||||
|
|
||||||
|
def unsupported_token_ratio(output, evidence):
|
||||||
|
evidence_tokens = set(tokenize(evidence))
|
||||||
|
output_tokens = tokenize(output)
|
||||||
|
if not output_tokens:
|
||||||
|
return 0.0
|
||||||
|
unsupported = [tok for tok in output_tokens if tok not in evidence_tokens]
|
||||||
|
return len(unsupported) / len(output_tokens)
|
||||||
|
|
||||||
|
|
||||||
|
def evaluate_output(output, sample):
|
||||||
|
evidence = " ".join(
|
||||||
|
part for part in [sample["previous_raw"], sample["new_raw"]] if part
|
||||||
|
)
|
||||||
|
return {
|
||||||
|
"output_len": len(tokenize(output)),
|
||||||
|
"repeated_ngram_ratio": round(repeated_ngram_ratio(output), 4),
|
||||||
|
"unsupported_token_ratio": round(unsupported_token_ratio(output, evidence), 4),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def run_experiments(args):
|
||||||
|
samples = reconstruct_samples(args.log_file)
|
||||||
|
if args.limit:
|
||||||
|
samples = samples[: args.limit]
|
||||||
|
|
||||||
|
results = []
|
||||||
|
for index, sample in enumerate(samples):
|
||||||
|
print(f"Running sample {index + 1}/{len(samples)}")
|
||||||
|
sample_results = []
|
||||||
|
for variant_name, system_prompt, user_prompt in build_variants(sample):
|
||||||
|
print(f" variant={variant_name}")
|
||||||
|
try:
|
||||||
|
output = ollama_generate(
|
||||||
|
args.model,
|
||||||
|
system_prompt,
|
||||||
|
user_prompt,
|
||||||
|
args.ollama_url,
|
||||||
|
args.timeout,
|
||||||
|
args.keep_alive,
|
||||||
|
)
|
||||||
|
metrics = evaluate_output(output, sample)
|
||||||
|
error_message = None
|
||||||
|
except Exception as exc:
|
||||||
|
output = ""
|
||||||
|
metrics = None
|
||||||
|
error_message = str(exc)
|
||||||
|
sample_results.append(
|
||||||
|
{
|
||||||
|
"variant": variant_name,
|
||||||
|
"output": output,
|
||||||
|
"metrics": metrics,
|
||||||
|
"error": error_message,
|
||||||
|
}
|
||||||
|
)
|
||||||
|
results.append(
|
||||||
|
{
|
||||||
|
"sample_index": index,
|
||||||
|
"sample": sample,
|
||||||
|
"results": sample_results,
|
||||||
|
}
|
||||||
|
)
|
||||||
|
return results
|
||||||
|
|
||||||
|
|
||||||
|
def summarize_results(results):
|
||||||
|
summary = {}
|
||||||
|
for sample_result in results:
|
||||||
|
for variant_result in sample_result["results"]:
|
||||||
|
bucket = summary.setdefault(
|
||||||
|
variant_result["variant"],
|
||||||
|
{"repeated_ngram_ratio": [], "unsupported_token_ratio": [], "output_len": [], "errors": 0},
|
||||||
|
)
|
||||||
|
if variant_result.get("error"):
|
||||||
|
bucket["errors"] += 1
|
||||||
|
continue
|
||||||
|
for key, value in variant_result["metrics"].items():
|
||||||
|
bucket[key].append(value)
|
||||||
|
|
||||||
|
for variant_name, metrics in summary.items():
|
||||||
|
summary[variant_name] = {
|
||||||
|
key: round(statistics.mean(values), 4) if isinstance(values, list) and values else values
|
||||||
|
for key, values in metrics.items()
|
||||||
|
}
|
||||||
|
return summary
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
parser = argparse.ArgumentParser(description="Run prompt experiments against Ollama using distribute_debug.log samples.")
|
||||||
|
parser.add_argument("--log-file", type=str, default="distribute_debug.log")
|
||||||
|
parser.add_argument("--model", type=str, default="qwen2.5:3b-instruct")
|
||||||
|
parser.add_argument("--ollama-url", type=str, default="http://127.0.0.1:11434/api/generate")
|
||||||
|
parser.add_argument("--timeout", type=float, default=40.0)
|
||||||
|
parser.add_argument("--keep-alive", type=str, default="30m")
|
||||||
|
parser.add_argument("--limit", type=int, default=0)
|
||||||
|
parser.add_argument("--output-file", type=str, default="logs/prompt_experiment_results.json")
|
||||||
|
args = parser.parse_args()
|
||||||
|
|
||||||
|
results = run_experiments(args)
|
||||||
|
summary = summarize_results(results)
|
||||||
|
payload = {"model": args.model, "summary": summary, "results": results}
|
||||||
|
|
||||||
|
output_dir = os.path.dirname(args.output_file)
|
||||||
|
if output_dir:
|
||||||
|
os.makedirs(output_dir, exist_ok=True)
|
||||||
|
with open(args.output_file, "w", encoding="utf-8") as handle:
|
||||||
|
json.dump(payload, handle, ensure_ascii=False, indent=2)
|
||||||
|
|
||||||
|
print(json.dumps(summary, ensure_ascii=False, indent=2))
|
||||||
|
print(f"Wrote detailed results to {args.output_file}")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
Reference in New Issue
Block a user