feat: cloud-first consolidation routing + graceful backend fallback
Nail down which backend each LLM path uses, and make the dream cycle resilient to a backend being down (the MI50 outage left profile/era/narrative — pinned to mi50 with no fallback — aborting every dream cycle). - Consolidation (summaries + profile/era/narrative) -> cloud via SUMMARY_BACKEND= cloud (.env, not committed). Matches the documented lesson that the MI50 is too slow/hot for bulk consolidation; nothing background touches the card now. - llm.complete_with_fallback(): try the primary backend, fall back to cloud on error (re-raise if already cloud / no key). Wired into reflect + think so the introspection voice (3090/dolphin) survives the gaming PC being powered off. - dream coherence stage is now fault-isolated: a rebuild failure logs + continues instead of sinking the whole pass (reflection still runs). - .env: removed stale INTROSPECTION_BACKEND=mi50 (live routing is the web-switchable introspection_mode DB setting = dolphin/3090; the var only fed a dead fallback). Verified: forced cycle runs consolidation on cloud, introspection on the 3090, completes with zero MI50 calls. 206 pass, ruff clean. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_015yrEb5qpPGv2FjyxrB7LLk
This commit is contained in:
+12
-5
@@ -124,11 +124,18 @@ def dream_cycle(backend: Backend | None = None, force: bool = False) -> dict:
|
||||
|
||||
# --- coherence: fold gists up into profile / eras / narrative ---
|
||||
if (force or drives["coherence"] >= THRESHOLD) and not _over_budget(deadline):
|
||||
profile.rebuild_profile(backend=backend)
|
||||
era.rebuild_eras(backend=backend)
|
||||
narrative.rebuild_narrative(backend=backend)
|
||||
actions.append("integrated knowledge (profile/eras/narrative)")
|
||||
drives["coherence"] = 0.0
|
||||
# A backend hiccup here must not sink the whole pass (reflection still
|
||||
# deserves to run); log it and move on, leaving coherence unrelieved so a
|
||||
# later cycle retries.
|
||||
try:
|
||||
profile.rebuild_profile(backend=backend)
|
||||
era.rebuild_eras(backend=backend)
|
||||
narrative.rebuild_narrative(backend=backend)
|
||||
actions.append("integrated knowledge (profile/eras/narrative)")
|
||||
drives["coherence"] = 0.0
|
||||
except Exception as exc:
|
||||
logbus.log("error", "coherence stage failed", error=str(exc)[:200])
|
||||
actions.append("coherence stage failed")
|
||||
# Off-hot-path villain identity housekeeping: propose likely same-person
|
||||
# merges for Brian to confirm on the Players page. Never sinks the cycle.
|
||||
try:
|
||||
|
||||
+20
@@ -92,6 +92,26 @@ def complete(messages: list[Message], backend: Backend = "local", model: str | N
|
||||
return out
|
||||
|
||||
|
||||
def complete_with_fallback(messages: list[Message], backend: Backend, model: str | None = None,
|
||||
*, fallback: Backend = "cloud",
|
||||
max_tokens: int | None = None, timeout: float | None = None) -> str:
|
||||
"""`complete()` but if the primary backend errors (e.g. a local GPU that's
|
||||
powered off or down), retry once on `fallback` (cloud) instead of failing.
|
||||
Lets local/GPU-routed work (introspection, consolidation) degrade gracefully.
|
||||
Re-raises if the primary is already the fallback or no cloud key is configured."""
|
||||
try:
|
||||
return complete(messages, backend=backend, model=model,
|
||||
max_tokens=max_tokens, timeout=timeout)
|
||||
except Exception as exc:
|
||||
can_fallback = backend != fallback and (fallback != "cloud" or load().openai_api_key)
|
||||
if not can_fallback:
|
||||
raise
|
||||
logbus.log("info", "llm fell back", primary=backend, to=fallback, error=str(exc)[:80])
|
||||
# Drop the primary's model on fallback — let the fallback pick its own default.
|
||||
return complete(messages, backend=fallback, model=None,
|
||||
max_tokens=max_tokens, timeout=timeout)
|
||||
|
||||
|
||||
def chat_call(
|
||||
messages: list, backend: Backend = "cloud", model: str | None = None,
|
||||
tools: list | None = None,
|
||||
|
||||
+3
-3
@@ -317,7 +317,7 @@ def reflect(backend: Backend | None = None, session_id: str | None = None,
|
||||
)
|
||||
|
||||
# Step 1 — draft a reflection.
|
||||
draft = _safe_json(llm.complete(
|
||||
draft = _safe_json(llm.complete_with_fallback(
|
||||
[{"role": "system", "content": _REFLECT_PROMPT}, {"role": "user", "content": body}],
|
||||
backend=backend, model=model,
|
||||
))
|
||||
@@ -326,7 +326,7 @@ def reflect(backend: Backend | None = None, session_id: str | None = None,
|
||||
update, critique, revised = draft, None, None
|
||||
if draft:
|
||||
examine_body = body + "\n\nYOUR DRAFT REFLECTION:\n" + json.dumps(draft, indent=2)
|
||||
revised = _safe_json(llm.complete(
|
||||
revised = _safe_json(llm.complete_with_fallback(
|
||||
[{"role": "system", "content": _EXAMINE_PROMPT},
|
||||
{"role": "user", "content": examine_body}],
|
||||
backend=backend, model=model,
|
||||
@@ -417,7 +417,7 @@ def _consolidate_self(backend: Backend | None = None, model: str | None = None,
|
||||
body = ("STABLE ANCHOR (who you are — this holds):\n" + IDENTITY_ANCHOR
|
||||
+ "\n\nYOUR RECENT REFLECTIONS (what's actually been on your mind):\n"
|
||||
+ "\n".join(f"- {r}" for r in refs))
|
||||
out = _safe_json(llm.complete(
|
||||
out = _safe_json(llm.complete_with_fallback(
|
||||
[{"role": "system", "content": _CONSOLIDATE_PROMPT}, {"role": "user", "content": body}],
|
||||
backend=backend, model=model,
|
||||
))
|
||||
|
||||
+2
-2
@@ -414,7 +414,7 @@ def _compose_reachout(title: str, content: str, backend, model) -> str:
|
||||
"""Auto-write her a short personal text about a genuinely salient thought she didn't
|
||||
explicitly flag — so the good ones reach Brian, in her voice, not as a thought-dump."""
|
||||
try:
|
||||
out = llm.complete(
|
||||
out = llm.complete_with_fallback(
|
||||
[{"role": "system", "content": _REACHOUT_PROMPT},
|
||||
{"role": "user", "content": f'Thought "{title}": {content}'}],
|
||||
backend=backend, model=model,
|
||||
@@ -612,7 +612,7 @@ def think(backend: Backend | None = None, force_mode: str | None = None,
|
||||
)
|
||||
|
||||
body = f"{time_line}\n\n{inner}{norestate}\n\n{task}"
|
||||
out = _safe_json(llm.complete(
|
||||
out = _safe_json(llm.complete_with_fallback(
|
||||
[{"role": "system", "content": _THINK_PROMPT}, {"role": "user", "content": body}],
|
||||
backend=backend, model=model,
|
||||
))
|
||||
|
||||
Reference in New Issue
Block a user