Tune paragraph prompts for multilingual noise
This commit is contained in:
@@ -25,6 +25,9 @@
|
|||||||
"post_correct_min_overlap": 0.45,
|
"post_correct_min_overlap": 0.45,
|
||||||
"post_correct_debug": false,
|
"post_correct_debug": false,
|
||||||
"llm_paragraph": false,
|
"llm_paragraph": false,
|
||||||
|
"llm_paragraph_temperature": 0.2,
|
||||||
|
"llm_paragraph_top_p": 0.8,
|
||||||
|
"llm_paragraph_repeat_penalty": 1.0,
|
||||||
"llm_request_log_path": "logs/llm_requests.jsonl",
|
"llm_request_log_path": "logs/llm_requests.jsonl",
|
||||||
"only_translate_llm": false,
|
"only_translate_llm": false,
|
||||||
"temperature_fallback": [0.0, 0.2, 0.4, 0.6, 0.8, 1.0],
|
"temperature_fallback": [0.0, 0.2, 0.4, 0.6, 0.8, 1.0],
|
||||||
|
|||||||
+36
-14
@@ -31,11 +31,15 @@ Produce the most faithful readable English update.
|
|||||||
Decision rules:
|
Decision rules:
|
||||||
1. NEW SOURCE TEXT is the primary evidence and should dominate the output.
|
1. NEW SOURCE TEXT is the primary evidence and should dominate the output.
|
||||||
2. Use PREVIOUS SOURCE TEXT and PREVIOUS REFINED CONTEXT only when they clearly help resolve or continue the new material.
|
2. Use PREVIOUS SOURCE TEXT and PREVIOUS REFINED CONTEXT only when they clearly help resolve or continue the new material.
|
||||||
3. If the new material starts a fresh thought, output only the new material.
|
3. The final answer must be English only.
|
||||||
4. Remove exact or near-exact repetition unless the repetition is clearly intentional rhetoric.
|
4. Never copy non-English words into the output unless they are proper names.
|
||||||
5. Do not add explanations, disclaimers, or meta commentary.
|
5. If a non-English fragment appears as an isolated clause, trailing fragment, or side comment without a clear English anchor, drop it.
|
||||||
6. Prefer conservative wording over guessed meaning.
|
6. Only translate a non-English fragment when it is clearly central to the same idea and its meaning is reasonably obvious from context.
|
||||||
7. Return only the revised transcript text."""
|
7. If the new material starts a fresh thought, output only the new material.
|
||||||
|
8. Remove exact or near-exact repetition unless the repetition is clearly intentional rhetoric.
|
||||||
|
9. Do not add explanations, disclaimers, or meta commentary.
|
||||||
|
10. Prefer conservative wording over guessed meaning.
|
||||||
|
11. Return only the revised transcript text."""
|
||||||
|
|
||||||
prompt = f"""Edit this transcript update.
|
prompt = f"""Edit this transcript update.
|
||||||
|
|
||||||
@@ -137,7 +141,10 @@ def run_llm_prompt_test(args):
|
|||||||
raise RuntimeError(f"OpenAI API error {response.status_code}: {detail}")
|
raise RuntimeError(f"OpenAI API error {response.status_code}: {detail}")
|
||||||
return response.json()["choices"][0]["message"]["content"].strip()
|
return response.json()["choices"][0]["message"]["content"].strip()
|
||||||
|
|
||||||
def call_ollama(model, prompt, temperature, system=None):
|
def call_ollama(model, prompt, temperature, system=None, extra_options=None):
|
||||||
|
options = {"temperature": temperature}
|
||||||
|
if extra_options:
|
||||||
|
options.update(extra_options)
|
||||||
response = requests.post(
|
response = requests.post(
|
||||||
getattr(args, "post_correct_ollama_url", "http://127.0.0.1:11434/api/generate"),
|
getattr(args, "post_correct_ollama_url", "http://127.0.0.1:11434/api/generate"),
|
||||||
json={
|
json={
|
||||||
@@ -145,7 +152,7 @@ def run_llm_prompt_test(args):
|
|||||||
"prompt": prompt,
|
"prompt": prompt,
|
||||||
"system": system or "",
|
"system": system or "",
|
||||||
"stream": False,
|
"stream": False,
|
||||||
"options": {"temperature": temperature},
|
"options": options,
|
||||||
"keep_alive": getattr(args, "post_correct_keep_alive", "30m"),
|
"keep_alive": getattr(args, "post_correct_keep_alive", "30m"),
|
||||||
},
|
},
|
||||||
timeout=getattr(args, "post_correct_llm_timeout", 8.0),
|
timeout=getattr(args, "post_correct_llm_timeout", 8.0),
|
||||||
@@ -153,7 +160,7 @@ def run_llm_prompt_test(args):
|
|||||||
response.raise_for_status()
|
response.raise_for_status()
|
||||||
return response.json().get("response", "").strip()
|
return response.json().get("response", "").strip()
|
||||||
|
|
||||||
def send_test_request(mode, messages, prompt, temperature, metadata, model=None):
|
def send_test_request(mode, messages, prompt, temperature, metadata, model=None, extra_options=None):
|
||||||
request_model = model or args.post_correct_model
|
request_model = model or args.post_correct_model
|
||||||
is_openai = request_model.startswith("gpt-")
|
is_openai = request_model.startswith("gpt-")
|
||||||
entry = {
|
entry = {
|
||||||
@@ -162,6 +169,7 @@ def run_llm_prompt_test(args):
|
|||||||
"provider": "openai" if is_openai else "ollama",
|
"provider": "openai" if is_openai else "ollama",
|
||||||
"model": request_model,
|
"model": request_model,
|
||||||
"temperature": temperature,
|
"temperature": temperature,
|
||||||
|
"options": extra_options or {},
|
||||||
"messages": messages,
|
"messages": messages,
|
||||||
"metadata": metadata,
|
"metadata": metadata,
|
||||||
}
|
}
|
||||||
@@ -185,7 +193,7 @@ def run_llm_prompt_test(args):
|
|||||||
system = message.get("content", "")
|
system = message.get("content", "")
|
||||||
break
|
break
|
||||||
try:
|
try:
|
||||||
response_text = call_ollama(request_model, prompt, temperature, system=system)
|
response_text = call_ollama(request_model, prompt, temperature, system=system, extra_options=extra_options)
|
||||||
entry["response"] = {"status": "ok", "content": response_text}
|
entry["response"] = {"status": "ok", "content": response_text}
|
||||||
return response_text
|
return response_text
|
||||||
except Exception as exc:
|
except Exception as exc:
|
||||||
@@ -352,7 +360,10 @@ def run_llm_processor(in_queue, out_queue, args):
|
|||||||
raise RuntimeError(f"OpenAI API error {response.status_code}: {detail}")
|
raise RuntimeError(f"OpenAI API error {response.status_code}: {detail}")
|
||||||
return response.json()["choices"][0]["message"]["content"].strip()
|
return response.json()["choices"][0]["message"]["content"].strip()
|
||||||
|
|
||||||
def call_ollama(prompt, temperature, system=None):
|
def call_ollama(prompt, temperature, system=None, extra_options=None):
|
||||||
|
options = {"temperature": temperature}
|
||||||
|
if extra_options:
|
||||||
|
options.update(extra_options)
|
||||||
response = requests.post(
|
response = requests.post(
|
||||||
getattr(args, "post_correct_ollama_url", "http://127.0.0.1:11434/api/generate"),
|
getattr(args, "post_correct_ollama_url", "http://127.0.0.1:11434/api/generate"),
|
||||||
json={
|
json={
|
||||||
@@ -360,7 +371,7 @@ def run_llm_processor(in_queue, out_queue, args):
|
|||||||
"prompt": prompt,
|
"prompt": prompt,
|
||||||
"system": system or "",
|
"system": system or "",
|
||||||
"stream": False,
|
"stream": False,
|
||||||
"options": {"temperature": temperature},
|
"options": options,
|
||||||
"keep_alive": getattr(args, "post_correct_keep_alive", "30m"),
|
"keep_alive": getattr(args, "post_correct_keep_alive", "30m"),
|
||||||
},
|
},
|
||||||
timeout=getattr(args, "post_correct_llm_timeout", 8.0),
|
timeout=getattr(args, "post_correct_llm_timeout", 8.0),
|
||||||
@@ -368,13 +379,14 @@ def run_llm_processor(in_queue, out_queue, args):
|
|||||||
response.raise_for_status()
|
response.raise_for_status()
|
||||||
return response.json().get("response", "").strip()
|
return response.json().get("response", "").strip()
|
||||||
|
|
||||||
def call_llm(messages, prompt, temperature, warn_key):
|
def call_llm(messages, prompt, temperature, warn_key, extra_options=None):
|
||||||
entry = {
|
entry = {
|
||||||
"timestamp": datetime.now(timezone.utc).isoformat(),
|
"timestamp": datetime.now(timezone.utc).isoformat(),
|
||||||
"mode": "paragraph" if warn_key == "paragraph_warned" else "line",
|
"mode": "paragraph" if warn_key == "paragraph_warned" else "line",
|
||||||
"provider": "openai" if is_openai else "ollama",
|
"provider": "openai" if is_openai else "ollama",
|
||||||
"model": args.post_correct_model,
|
"model": args.post_correct_model,
|
||||||
"temperature": temperature,
|
"temperature": temperature,
|
||||||
|
"options": extra_options or {},
|
||||||
"messages": messages,
|
"messages": messages,
|
||||||
}
|
}
|
||||||
started_at = time.time()
|
started_at = time.time()
|
||||||
@@ -391,7 +403,7 @@ def run_llm_processor(in_queue, out_queue, args):
|
|||||||
if message.get("role") == "system":
|
if message.get("role") == "system":
|
||||||
system = message.get("content", "")
|
system = message.get("content", "")
|
||||||
break
|
break
|
||||||
response_text = call_ollama(prompt, temperature, system=system)
|
response_text = call_ollama(prompt, temperature, system=system, extra_options=extra_options)
|
||||||
entry["response"] = {"status": "ok", "content": response_text}
|
entry["response"] = {"status": "ok", "content": response_text}
|
||||||
return response_text
|
return response_text
|
||||||
except Exception as exc:
|
except Exception as exc:
|
||||||
@@ -412,7 +424,17 @@ def run_llm_processor(in_queue, out_queue, args):
|
|||||||
|
|
||||||
def call_llm_rolling_refine(new_segments_str, context, previous_source_text):
|
def call_llm_rolling_refine(new_segments_str, context, previous_source_text):
|
||||||
messages, prompt = build_paragraph_messages(context, previous_source_text, new_segments_str)
|
messages, prompt = build_paragraph_messages(context, previous_source_text, new_segments_str)
|
||||||
return call_llm(messages, prompt, 0.1, "paragraph_warned")
|
paragraph_options = {
|
||||||
|
"top_p": getattr(args, "llm_paragraph_top_p", 0.8),
|
||||||
|
"repeat_penalty": getattr(args, "llm_paragraph_repeat_penalty", 1.0),
|
||||||
|
}
|
||||||
|
return call_llm(
|
||||||
|
messages,
|
||||||
|
prompt,
|
||||||
|
getattr(args, "llm_paragraph_temperature", 0.2),
|
||||||
|
"paragraph_warned",
|
||||||
|
extra_options=paragraph_options,
|
||||||
|
)
|
||||||
|
|
||||||
def post_correct_line(text):
|
def post_correct_line(text):
|
||||||
if not getattr(args, "post_correct", False):
|
if not getattr(args, "post_correct", False):
|
||||||
|
|||||||
+10
@@ -297,3 +297,13 @@ The project now features a high-performance, Apple Silicon-optimized pipeline th
|
|||||||
- Identified the best-performing variant as one that treats the new source window as primary evidence while using prior refined and prior raw text only when they clearly continue the thought.
|
- Identified the best-performing variant as one that treats the new source window as primary evidence while using prior refined and prior raw text only when they clearly continue the thought.
|
||||||
- Updated `engine_llm.py` so live paragraph refinement now sends `PREVIOUS REFINED CONTEXT`, `PREVIOUS SOURCE TEXT`, and `NEW SOURCE TEXT` with a more conservative, grounding-oriented system prompt.
|
- Updated `engine_llm.py` so live paragraph refinement now sends `PREVIOUS REFINED CONTEXT`, `PREVIOUS SOURCE TEXT`, and `NEW SOURCE TEXT` with a more conservative, grounding-oriented system prompt.
|
||||||
- **Outcome:** Paragraph refinement is now substantially better anchored to current source text, with less stale-context carryover and a much lower rate of unsupported content in direct Ollama tests.
|
- **Outcome:** Paragraph refinement is now substantially better anchored to current source text, with less stale-context carryover and a much lower rate of unsupported content in direct Ollama tests.
|
||||||
|
|
||||||
|
## Phase 34: Multilingual Noise Filtering + Paragraph Generation Sweep
|
||||||
|
- **Goal:** Make paragraph refinement more robust when stray Spanish or other non-English fragments leak into the English bridge text.
|
||||||
|
- **Approach:**
|
||||||
|
- Extended `prompt_experiments.py` with multilingual-noise test cases that inject Spanish, Portuguese, and French fragments into reconstructed live samples.
|
||||||
|
- Ran direct-Ollama comparisons across multiple paragraph prompt variants and generation settings, including temperature, `top_p`, and `repeat_penalty`.
|
||||||
|
- Found that the best overall paragraph generation settings for `qwen2.5:3b-instruct` remained conservative: `temperature=0.2`, `top_p=0.8`, `repeat_penalty=1.0`.
|
||||||
|
- Tightened the live paragraph system prompt in `engine_llm.py` so output must stay English-only, non-English fragments should not be copied through, isolated trailing foreign fragments should be dropped, and only clearly central foreign content should be translated into English.
|
||||||
|
- Added paragraph-specific generation controls to `main_v2.py` and `config.json` so paragraph tuning can evolve independently of line-correction behavior.
|
||||||
|
- **Outcome:** Paragraph refinement is now stricter about keeping the final output in English while preserving the more grounded context layout and the best-tested Ollama generation settings.
|
||||||
|
|||||||
@@ -57,6 +57,9 @@ def main():
|
|||||||
parser.add_argument("--post-correct-min-overlap", type=float, default=defaults.get("post_correct_min_overlap", 0.45))
|
parser.add_argument("--post-correct-min-overlap", type=float, default=defaults.get("post_correct_min_overlap", 0.45))
|
||||||
parser.add_argument("--post-correct-debug", action="store_true", default=defaults.get("post_correct_debug", False))
|
parser.add_argument("--post-correct-debug", action="store_true", default=defaults.get("post_correct_debug", False))
|
||||||
parser.add_argument("--llm-paragraph", action="store_true", default=defaults.get("llm_paragraph", False))
|
parser.add_argument("--llm-paragraph", action="store_true", default=defaults.get("llm_paragraph", False))
|
||||||
|
parser.add_argument("--llm-paragraph-temperature", type=float, default=defaults.get("llm_paragraph_temperature", 0.2))
|
||||||
|
parser.add_argument("--llm-paragraph-top-p", type=float, default=defaults.get("llm_paragraph_top_p", 0.8))
|
||||||
|
parser.add_argument("--llm-paragraph-repeat-penalty", type=float, default=defaults.get("llm_paragraph_repeat_penalty", 1.0))
|
||||||
parser.add_argument("--llm-request-log-path", type=str, default=defaults.get("llm_request_log_path", "logs/llm_requests.jsonl"))
|
parser.add_argument("--llm-request-log-path", type=str, default=defaults.get("llm_request_log_path", "logs/llm_requests.jsonl"))
|
||||||
parser.add_argument("--llm-test-line", type=str, default=None, help="Send a single line-correction prompt without starting transcription.")
|
parser.add_argument("--llm-test-line", type=str, default=None, help="Send a single line-correction prompt without starting transcription.")
|
||||||
parser.add_argument("--llm-test-prev1", type=str, default="", help="Optional previous line 1 context for --llm-test-line.")
|
parser.add_argument("--llm-test-prev1", type=str, default="", help="Optional previous line 1 context for --llm-test-line.")
|
||||||
|
|||||||
+230
-6
@@ -7,6 +7,14 @@ from collections import Counter
|
|||||||
from urllib import request, error
|
from urllib import request, error
|
||||||
|
|
||||||
|
|
||||||
|
FOREIGN_INSERTIONS = [
|
||||||
|
"pero en espanol dice que no retrocedas",
|
||||||
|
"y tambien dijo que Dios responde",
|
||||||
|
"mas ele continuou em fe",
|
||||||
|
"et il a dit de rester ferme",
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
def parse_distribute_debug_log(path):
|
def parse_distribute_debug_log(path):
|
||||||
entries = []
|
entries = []
|
||||||
current = {"final": None, "en_bridge": None, "llm_paragraph": None}
|
current = {"final": None, "en_bridge": None, "llm_paragraph": None}
|
||||||
@@ -65,7 +73,39 @@ def reconstruct_samples(path):
|
|||||||
return samples
|
return samples
|
||||||
|
|
||||||
|
|
||||||
def build_variants(sample):
|
def inject_multilingual_noise(text, sample_index, slot_name):
|
||||||
|
if not text:
|
||||||
|
return text
|
||||||
|
inserts = [
|
||||||
|
FOREIGN_INSERTIONS[(sample_index + 0) % len(FOREIGN_INSERTIONS)],
|
||||||
|
FOREIGN_INSERTIONS[(sample_index + 1) % len(FOREIGN_INSERTIONS)],
|
||||||
|
]
|
||||||
|
sentences = [part.strip() for part in re.split(r"(?<=[.!?])\s+", text.strip()) if part.strip()]
|
||||||
|
if not sentences:
|
||||||
|
return text
|
||||||
|
if len(sentences) == 1:
|
||||||
|
return f'{sentences[0]} {inserts[0]}.'
|
||||||
|
head = sentences[0]
|
||||||
|
tail = " ".join(sentences[1:])
|
||||||
|
if slot_name == "previous_raw":
|
||||||
|
return f"{head} {inserts[0]}. {tail}"
|
||||||
|
return f"{head} {tail} {inserts[1]}."
|
||||||
|
|
||||||
|
|
||||||
|
def maybe_augment_samples(samples, mode):
|
||||||
|
if mode != "multilingual_noise":
|
||||||
|
return samples
|
||||||
|
|
||||||
|
augmented = []
|
||||||
|
for idx, sample in enumerate(samples):
|
||||||
|
updated = dict(sample)
|
||||||
|
updated["previous_raw"] = inject_multilingual_noise(sample["previous_raw"], idx, "previous_raw")
|
||||||
|
updated["new_raw"] = inject_multilingual_noise(sample["new_raw"], idx, "new_raw")
|
||||||
|
augmented.append(updated)
|
||||||
|
return augmented
|
||||||
|
|
||||||
|
|
||||||
|
def build_variants(sample, variant_filter=None):
|
||||||
prev_refined = sample["previous_refined"]
|
prev_refined = sample["previous_refined"]
|
||||||
prev_raw = sample["previous_raw"]
|
prev_raw = sample["previous_raw"]
|
||||||
new_raw = sample["new_raw"]
|
new_raw = sample["new_raw"]
|
||||||
@@ -198,17 +238,140 @@ Decision rules:
|
|||||||
"""
|
"""
|
||||||
variants.append(("grounded_new_dominant", system_e, user_e))
|
variants.append(("grounded_new_dominant", system_e, user_e))
|
||||||
|
|
||||||
|
system_g = """You are a careful live transcript editor working from noisy translated source text.
|
||||||
|
|
||||||
|
Goal:
|
||||||
|
Produce the most faithful readable English update.
|
||||||
|
|
||||||
|
Decision rules:
|
||||||
|
1. NEW SOURCE TEXT is the primary evidence and should dominate the output.
|
||||||
|
2. Use PREVIOUS SOURCE TEXT and PREVIOUS REFINED CONTEXT only when they clearly help resolve or continue the new material.
|
||||||
|
3. The final answer must be English only.
|
||||||
|
4. If source text contains Spanish, Portuguese, French, or other non-English fragments, translate their meaning into English when reasonably clear.
|
||||||
|
5. Never copy non-English words into the output unless they are proper names.
|
||||||
|
6. If the new material starts a fresh thought, output only the new material.
|
||||||
|
7. Remove exact or near-exact repetition unless the repetition is clearly intentional rhetoric.
|
||||||
|
8. Do not add explanations, disclaimers, or meta commentary.
|
||||||
|
9. Prefer conservative wording over guessed meaning.
|
||||||
|
10. Return only the revised transcript text."""
|
||||||
|
user_g = f"""Edit this transcript update.
|
||||||
|
|
||||||
|
[PREVIOUS REFINED CONTEXT]
|
||||||
|
"{prev_refined}"
|
||||||
|
|
||||||
|
[PREVIOUS SOURCE TEXT]
|
||||||
|
"{prev_raw}"
|
||||||
|
|
||||||
|
[NEW SOURCE TEXT]
|
||||||
|
"{new_raw}"
|
||||||
|
"""
|
||||||
|
variants.append(("grounded_english_only_translate", system_g, user_g))
|
||||||
|
|
||||||
|
system_h = """You are a careful live transcript editor working from noisy translated source text.
|
||||||
|
|
||||||
|
Goal:
|
||||||
|
Produce the most faithful readable English update.
|
||||||
|
|
||||||
|
Decision rules:
|
||||||
|
1. NEW SOURCE TEXT is the primary evidence and should dominate the output.
|
||||||
|
2. Use PREVIOUS SOURCE TEXT and PREVIOUS REFINED CONTEXT only when they clearly help resolve or continue the new material.
|
||||||
|
3. The final answer must be English only.
|
||||||
|
4. Treat any Spanish, Portuguese, French, or other non-English fragments as contamination from surrounding audio or translation noise unless they clearly repeat the same idea as nearby English.
|
||||||
|
5. When a non-English fragment clearly repeats nearby English content, keep only the English meaning once.
|
||||||
|
6. Never copy non-English words into the output unless they are proper names.
|
||||||
|
7. If the new material starts a fresh thought, output only the new material.
|
||||||
|
8. Remove exact or near-exact repetition unless the repetition is clearly intentional rhetoric.
|
||||||
|
9. Do not add explanations, disclaimers, or meta commentary.
|
||||||
|
10. Prefer conservative wording over guessed meaning.
|
||||||
|
11. Return only the revised transcript text."""
|
||||||
|
user_h = f"""Edit this transcript update.
|
||||||
|
|
||||||
|
[PREVIOUS REFINED CONTEXT]
|
||||||
|
"{prev_refined}"
|
||||||
|
|
||||||
|
[PREVIOUS SOURCE TEXT]
|
||||||
|
"{prev_raw}"
|
||||||
|
|
||||||
|
[NEW SOURCE TEXT]
|
||||||
|
"{new_raw}"
|
||||||
|
"""
|
||||||
|
variants.append(("grounded_english_only_ignore_noise", system_h, user_h))
|
||||||
|
|
||||||
|
system_i = """You are a careful live transcript editor working from noisy translated source text.
|
||||||
|
|
||||||
|
Goal:
|
||||||
|
Produce the most faithful readable English update.
|
||||||
|
|
||||||
|
Decision rules:
|
||||||
|
1. NEW SOURCE TEXT is the primary evidence and should dominate the output.
|
||||||
|
2. Use PREVIOUS SOURCE TEXT and PREVIOUS REFINED CONTEXT only when they clearly help resolve or continue the new material.
|
||||||
|
3. The final answer must be English only.
|
||||||
|
4. Never copy non-English words into the output unless they are proper names.
|
||||||
|
5. If a non-English fragment appears as an isolated clause, trailing fragment, or side comment without a clear English anchor, drop it.
|
||||||
|
6. Only translate a non-English fragment when it is clearly central to the same idea and its meaning is reasonably obvious from context.
|
||||||
|
7. If the new material starts a fresh thought, output only the new material.
|
||||||
|
8. Remove exact or near-exact repetition unless the repetition is clearly intentional rhetoric.
|
||||||
|
9. Do not add explanations, disclaimers, or meta commentary.
|
||||||
|
10. Prefer conservative wording over guessed meaning.
|
||||||
|
11. Return only the revised transcript text."""
|
||||||
|
user_i = f"""Edit this transcript update.
|
||||||
|
|
||||||
|
[PREVIOUS REFINED CONTEXT]
|
||||||
|
"{prev_refined}"
|
||||||
|
|
||||||
|
[PREVIOUS SOURCE TEXT]
|
||||||
|
"{prev_raw}"
|
||||||
|
|
||||||
|
[NEW SOURCE TEXT]
|
||||||
|
"{new_raw}"
|
||||||
|
"""
|
||||||
|
variants.append(("grounded_english_only_drop_isolated", system_i, user_i))
|
||||||
|
|
||||||
|
system_f = """You are a careful live transcript editor working from noisy translated source text.
|
||||||
|
|
||||||
|
Goal:
|
||||||
|
Produce the most faithful readable English update.
|
||||||
|
|
||||||
|
Decision rules:
|
||||||
|
1. NEW SOURCE TEXT is the primary evidence and should dominate the output.
|
||||||
|
2. Use PREVIOUS SOURCE TEXT and PREVIOUS REFINED CONTEXT only when they clearly help resolve or continue the new material.
|
||||||
|
3. If the new material starts a fresh thought, output only the new material.
|
||||||
|
4. Remove exact or near-exact repetition unless the repetition is clearly intentional rhetoric.
|
||||||
|
5. If the source repeats the same short clause three or more times in a row, keep at most two repetitions unless losing more would change the speaker's emphasis.
|
||||||
|
6. Do not add explanations, disclaimers, or meta commentary.
|
||||||
|
7. Prefer conservative wording over guessed meaning.
|
||||||
|
8. Return only the revised transcript text."""
|
||||||
|
user_f = f"""Edit this transcript update.
|
||||||
|
|
||||||
|
[PREVIOUS REFINED CONTEXT]
|
||||||
|
"{prev_refined}"
|
||||||
|
|
||||||
|
[PREVIOUS SOURCE TEXT]
|
||||||
|
"{prev_raw}"
|
||||||
|
|
||||||
|
[NEW SOURCE TEXT]
|
||||||
|
"{new_raw}"
|
||||||
|
"""
|
||||||
|
variants.append(("grounded_new_dominant_dedupe", system_f, user_f))
|
||||||
|
|
||||||
|
if variant_filter:
|
||||||
|
allowed = set(x.strip() for x in variant_filter.split(",") if x.strip())
|
||||||
|
variants = [variant for variant in variants if variant[0] in allowed]
|
||||||
return variants
|
return variants
|
||||||
|
|
||||||
|
|
||||||
def ollama_generate(model, system_prompt, user_prompt, url, timeout, keep_alive):
|
def ollama_generate(model, system_prompt, user_prompt, url, timeout, keep_alive, temperature, top_p, repeat_penalty):
|
||||||
payload = json.dumps(
|
payload = json.dumps(
|
||||||
{
|
{
|
||||||
"model": model,
|
"model": model,
|
||||||
"prompt": user_prompt,
|
"prompt": user_prompt,
|
||||||
"system": system_prompt,
|
"system": system_prompt,
|
||||||
"stream": False,
|
"stream": False,
|
||||||
"options": {"temperature": 0.1},
|
"options": {
|
||||||
|
"temperature": temperature,
|
||||||
|
"top_p": top_p,
|
||||||
|
"repeat_penalty": repeat_penalty,
|
||||||
|
},
|
||||||
"keep_alive": keep_alive,
|
"keep_alive": keep_alive,
|
||||||
}
|
}
|
||||||
).encode("utf-8")
|
).encode("utf-8")
|
||||||
@@ -258,20 +421,65 @@ def evaluate_output(output, sample):
|
|||||||
"output_len": len(tokenize(output)),
|
"output_len": len(tokenize(output)),
|
||||||
"repeated_ngram_ratio": round(repeated_ngram_ratio(output), 4),
|
"repeated_ngram_ratio": round(repeated_ngram_ratio(output), 4),
|
||||||
"unsupported_token_ratio": round(unsupported_token_ratio(output, evidence), 4),
|
"unsupported_token_ratio": round(unsupported_token_ratio(output, evidence), 4),
|
||||||
|
"foreign_leak_ratio": round(foreign_leak_ratio(output), 4),
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def foreign_leak_ratio(output):
|
||||||
|
foreign_markers = {
|
||||||
|
"pero", "espanol", "tambien", "dijo", "dios", "responde",
|
||||||
|
"mas", "ele", "continuou", "fe",
|
||||||
|
"et", "rester", "ferme",
|
||||||
|
}
|
||||||
|
tokens = tokenize(output)
|
||||||
|
if not tokens:
|
||||||
|
return 0.0
|
||||||
|
leaked = [tok for tok in tokens if tok in foreign_markers]
|
||||||
|
return len(leaked) / len(tokens)
|
||||||
|
|
||||||
|
|
||||||
|
def parse_float_grid(raw_value, default_values):
|
||||||
|
if not raw_value:
|
||||||
|
return default_values
|
||||||
|
return [float(x.strip()) for x in raw_value.split(",") if x.strip()]
|
||||||
|
|
||||||
|
|
||||||
|
def build_param_sets(args):
|
||||||
|
temperatures = parse_float_grid(args.temperature_grid, [0.1])
|
||||||
|
top_ps = parse_float_grid(args.top_p_grid, [0.9])
|
||||||
|
repeat_penalties = parse_float_grid(args.repeat_penalty_grid, [1.1])
|
||||||
|
|
||||||
|
param_sets = []
|
||||||
|
for temperature in temperatures:
|
||||||
|
for top_p in top_ps:
|
||||||
|
for repeat_penalty in repeat_penalties:
|
||||||
|
label = f"temp={temperature}_top_p={top_p}_repeat_penalty={repeat_penalty}"
|
||||||
|
param_sets.append(
|
||||||
|
{
|
||||||
|
"label": label,
|
||||||
|
"temperature": temperature,
|
||||||
|
"top_p": top_p,
|
||||||
|
"repeat_penalty": repeat_penalty,
|
||||||
|
}
|
||||||
|
)
|
||||||
|
return param_sets
|
||||||
|
|
||||||
|
|
||||||
def run_experiments(args):
|
def run_experiments(args):
|
||||||
samples = reconstruct_samples(args.log_file)
|
samples = reconstruct_samples(args.log_file)
|
||||||
|
samples = maybe_augment_samples(samples, args.sample_mode)
|
||||||
if args.limit:
|
if args.limit:
|
||||||
samples = samples[: args.limit]
|
samples = samples[: args.limit]
|
||||||
|
|
||||||
|
param_sets = build_param_sets(args)
|
||||||
results = []
|
results = []
|
||||||
for index, sample in enumerate(samples):
|
for index, sample in enumerate(samples):
|
||||||
print(f"Running sample {index + 1}/{len(samples)}")
|
print(f"Running sample {index + 1}/{len(samples)}")
|
||||||
sample_results = []
|
sample_results = []
|
||||||
for variant_name, system_prompt, user_prompt in build_variants(sample):
|
for variant_name, system_prompt, user_prompt in build_variants(sample, args.variant_filter):
|
||||||
print(f" variant={variant_name}")
|
print(f" variant={variant_name}")
|
||||||
|
for params in param_sets:
|
||||||
|
print(f" params={params['label']}")
|
||||||
try:
|
try:
|
||||||
output = ollama_generate(
|
output = ollama_generate(
|
||||||
args.model,
|
args.model,
|
||||||
@@ -280,6 +488,9 @@ def run_experiments(args):
|
|||||||
args.ollama_url,
|
args.ollama_url,
|
||||||
args.timeout,
|
args.timeout,
|
||||||
args.keep_alive,
|
args.keep_alive,
|
||||||
|
params["temperature"],
|
||||||
|
params["top_p"],
|
||||||
|
params["repeat_penalty"],
|
||||||
)
|
)
|
||||||
metrics = evaluate_output(output, sample)
|
metrics = evaluate_output(output, sample)
|
||||||
error_message = None
|
error_message = None
|
||||||
@@ -290,6 +501,7 @@ def run_experiments(args):
|
|||||||
sample_results.append(
|
sample_results.append(
|
||||||
{
|
{
|
||||||
"variant": variant_name,
|
"variant": variant_name,
|
||||||
|
"params": params,
|
||||||
"output": output,
|
"output": output,
|
||||||
"metrics": metrics,
|
"metrics": metrics,
|
||||||
"error": error_message,
|
"error": error_message,
|
||||||
@@ -309,9 +521,16 @@ def summarize_results(results):
|
|||||||
summary = {}
|
summary = {}
|
||||||
for sample_result in results:
|
for sample_result in results:
|
||||||
for variant_result in sample_result["results"]:
|
for variant_result in sample_result["results"]:
|
||||||
|
summary_key = f"{variant_result['variant']}|{variant_result['params']['label']}"
|
||||||
bucket = summary.setdefault(
|
bucket = summary.setdefault(
|
||||||
variant_result["variant"],
|
summary_key,
|
||||||
{"repeated_ngram_ratio": [], "unsupported_token_ratio": [], "output_len": [], "errors": 0},
|
{
|
||||||
|
"repeated_ngram_ratio": [],
|
||||||
|
"unsupported_token_ratio": [],
|
||||||
|
"foreign_leak_ratio": [],
|
||||||
|
"output_len": [],
|
||||||
|
"errors": 0,
|
||||||
|
},
|
||||||
)
|
)
|
||||||
if variant_result.get("error"):
|
if variant_result.get("error"):
|
||||||
bucket["errors"] += 1
|
bucket["errors"] += 1
|
||||||
@@ -336,6 +555,11 @@ def main():
|
|||||||
parser.add_argument("--keep-alive", type=str, default="30m")
|
parser.add_argument("--keep-alive", type=str, default="30m")
|
||||||
parser.add_argument("--limit", type=int, default=0)
|
parser.add_argument("--limit", type=int, default=0)
|
||||||
parser.add_argument("--output-file", type=str, default="logs/prompt_experiment_results.json")
|
parser.add_argument("--output-file", type=str, default="logs/prompt_experiment_results.json")
|
||||||
|
parser.add_argument("--variant-filter", type=str, default="", help="Comma-separated subset of variants to run.")
|
||||||
|
parser.add_argument("--temperature-grid", type=str, default="", help="Comma-separated temperatures to test.")
|
||||||
|
parser.add_argument("--top-p-grid", type=str, default="", help="Comma-separated top_p values to test.")
|
||||||
|
parser.add_argument("--repeat-penalty-grid", type=str, default="", help="Comma-separated repeat_penalty values to test.")
|
||||||
|
parser.add_argument("--sample-mode", type=str, default="base", help="Sample mode: base or multilingual_noise.")
|
||||||
args = parser.parse_args()
|
args = parser.parse_args()
|
||||||
|
|
||||||
results = run_experiments(args)
|
results = run_experiments(args)
|
||||||
|
|||||||
Reference in New Issue
Block a user