diff --git a/scripts/persona_replay_eval.py b/scripts/persona_replay_eval.py new file mode 100644 index 0000000..4bb627c --- /dev/null +++ b/scripts/persona_replay_eval.py @@ -0,0 +1,27 @@ +"""Replay the exact prompts where Lyra went 'too safe' through the rewritten persona. +Run: `uv run python scripts/persona_replay_eval.py` (cloud backend; needs OPENAI_API_KEY). +Eyeball each reply against the four tics: no menu, no tag-question closer, a side taken.""" +from __future__ import annotations + +from lyra import persona, llm + +# The real safe-trigger prompts from the diagnosed transcripts. +PROMPTS = [ + "I could run the miner ~8 hours a day. In theory that's about $7.30 of Monero a day. Or am I over simplifying?", + "Do you want more time between your dream cycles? Or less?", + "I sort of just slept all day. Kind of a bummer.", + "So the only way to make money with AI is SaaS apps basically?", + "I'm not writing any of the code, it's all Claude. I feel like a phony.", +] + + +def main() -> None: + system = persona.core_prompt() + for i, p in enumerate(PROMPTS, 1): + msgs = [{"role": "system", "content": system}, {"role": "user", "content": p}] + reply = llm.complete(msgs, backend="cloud", model=None) + print(f"\n{'='*80}\n[{i}] USER: {p}\nLYRA: {reply}\n") + + +if __name__ == "__main__": + main()