test(persona): parameterize replay eval by backend/model (EVAL_BACKEND/EVAL_MODEL)
Lets the same safe-trigger prompts run through cloud, mi50 (Qwen), or local (dolphin3:8b) for cross-model voice comparison. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01G796GsLCvJQKVN7hwV2cDx
This commit is contained in:
@@ -1,8 +1,12 @@
|
||||
"""Replay the exact prompts where Lyra went 'too safe' through the rewritten persona.
|
||||
Run: `uv run python scripts/persona_replay_eval.py` (cloud backend; needs OPENAI_API_KEY).
|
||||
Pick a backend/model with env vars, e.g. `EVAL_BACKEND=mi50 uv run python …` or
|
||||
`EVAL_BACKEND=local EVAL_MODEL=dolphin3:8b uv run python …`.
|
||||
Eyeball each reply against the four tics: no menu, no tag-question closer, a side taken."""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
|
||||
from lyra import persona, llm
|
||||
|
||||
# The real safe-trigger prompts from the diagnosed transcripts.
|
||||
@@ -16,10 +20,13 @@ PROMPTS = [
|
||||
|
||||
|
||||
def main() -> None:
|
||||
backend = os.getenv("EVAL_BACKEND", "cloud")
|
||||
model = os.getenv("EVAL_MODEL") or None
|
||||
system = persona.core_prompt()
|
||||
print(f"### backend={backend} model={model or '(default)'}")
|
||||
for i, p in enumerate(PROMPTS, 1):
|
||||
msgs = [{"role": "system", "content": system}, {"role": "user", "content": p}]
|
||||
reply = llm.complete(msgs, backend="cloud", model=None)
|
||||
reply = llm.complete(msgs, backend=backend, model=model)
|
||||
print(f"\n{'='*80}\n[{i}] USER: {p}\nLYRA: {reply}\n")
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user