fix: cap MI50 summary length + fast-fail cloud fallback

The dream cycle's summarize_all ran uncapped against the MI50: no max_tokens
and no timeout, so the OpenAI SDK's 600s x2-retry default meant ~30 min per
call. Combined with summary.py's own retry loop, one unsummarizable session
pegged the GPU for hours (observed 2026-07-04: stuck since 23:02, nothing saved
since 00:56, 7-8k-token runaway generations, all 4 llama.cpp slots busy). Not
context overflow (0 shifts/truncations) - purely unbounded length on a slow
backend timing out and retrying.

- llm.complete(): add optional max_tokens (caps generation; num_predict for
  Ollama) and timeout (bounds the request and sets max_retries=0 so the caller
  owns retry policy). Both default None -> unchanged for every existing caller.
- summary.py: cap gists at 768 tokens, 150s/call fast-fail, 2 MI50 attempts
  then one cloud fallback (when primary isn't already cloud and a key exists).

Known limitation (scoped out per decision): the fallback triggers on
timeouts/exceptions, not on a degraded backend returning garbage as a 200.

Tests: fallback fires after 2 MI50 failures; no fallback when primary is cloud
or no key; cap+timeout threaded into every complete() call; llm bounds tests.
172 pass, ruff clean.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_015yrEb5qpPGv2FjyxrB7LLk
This commit is contained in:
2026-07-04 07:19:31 +00:00
parent 07153fc53d
commit 29a4d59661
5 changed files with 224 additions and 24 deletions
+29 -14
View File
@@ -37,30 +37,45 @@ def _resolved_model(cfg, backend: Backend, model: str | None) -> str:
return model or cfg.local_model
def complete(messages: list[Message], backend: Backend = "local", model: str | None = None) -> str:
def complete(messages: list[Message], backend: Backend = "local", model: str | None = None,
max_tokens: int | None = None, timeout: float | None = None) -> str:
"""Generate a completion. `model` overrides the backend's default model
(used so live chat can run a stronger cloud model than bulk consolidation)."""
(used so live chat can run a stronger cloud model than bulk consolidation).
`max_tokens` caps the generation length (guards a slow local model against
rambling for thousands of tokens). `timeout`, when set, bounds each request
and disables the SDK's own retries so the caller owns retry/fallback policy.
Both default to None → unchanged behavior for every existing caller."""
cfg = load()
mdl = _resolved_model(cfg, backend, model)
logbus.log("info", "llm call", kind="complete", backend=backend, model=mdl, tok=_approx_tok(messages))
t0 = time.monotonic()
if backend == "cloud":
if not cfg.openai_api_key:
raise RuntimeError("OPENAI_API_KEY is not set")
client = OpenAI(api_key=cfg.openai_api_key)
resp = client.chat.completions.create(model=mdl, messages=messages)
out = resp.choices[0].message.content or ""
elif backend == "mi50":
# MI50 box runs an OpenAI-compatible llama.cpp server; key is unused.
client = OpenAI(api_key="not-needed", base_url=cfg.mi50_base_url)
resp = client.chat.completions.create(model=mdl, messages=messages)
if backend in ("cloud", "mi50"):
if backend == "cloud":
if not cfg.openai_api_key:
raise RuntimeError("OPENAI_API_KEY is not set")
client_kwargs: dict = {"api_key": cfg.openai_api_key}
else:
# MI50 box runs an OpenAI-compatible llama.cpp server; key is unused.
client_kwargs = {"api_key": "not-needed", "base_url": cfg.mi50_base_url}
if timeout is not None:
client_kwargs["timeout"] = timeout
client_kwargs["max_retries"] = 0 # caller owns retries (see summary.py)
client = OpenAI(**client_kwargs)
create_kwargs: dict = {"model": mdl, "messages": messages}
if max_tokens is not None:
create_kwargs["max_tokens"] = max_tokens
resp = client.chat.completions.create(**create_kwargs)
out = resp.choices[0].message.content or ""
else:
payload: dict = {"model": mdl, "messages": messages, "stream": False}
if max_tokens is not None:
payload["options"] = {"num_predict": max_tokens}
resp = httpx.post(
f"{cfg.local_base_url}/api/chat",
json={"model": mdl, "messages": messages, "stream": False},
timeout=120,
json=payload,
timeout=timeout or 120,
)
resp.raise_for_status()
out = resp.json()["message"]["content"]
+34 -9
View File
@@ -17,7 +17,16 @@ from concurrent.futures import ThreadPoolExecutor, as_completed
from lyra import config, llm, logbus, memory
from lyra.llm import Backend, Message
_RETRIES = 4
# Consolidation LLM budget. A gist is short (a handful of sentences), so cap the
# generation hard — an uncapped local model will otherwise ramble for thousands
# of tokens and, on a slow GPU, blow the request timeout. 768 is ~3x the longest
# real gist we've stored.
SUMMARY_MAX_TOKENS = 768
# Attempts on the primary backend before falling back to cloud.
MI50_ATTEMPTS = 2
# Per-call timeout (seconds). A capped 768-token gist finishes in ~60-90s on the
# MI50; 150s is headroom but bails a hung call fast so fallback isn't slow.
SUMMARY_TIMEOUT = 150
# Re-summarize a session once it has accumulated this many new raw exchanges.
SUMMARIZE_AFTER = 20
@@ -61,16 +70,32 @@ def _summarize_text(text: str, backend: Backend) -> str:
{"role": "system", "content": _PROMPT},
{"role": "user", "content": text},
]
# Retry transient backend errors (e.g. the GPU server restarting) with backoff.
for attempt in range(_RETRIES):
def _call(be: Backend) -> str:
return llm.complete(messages, backend=be,
max_tokens=SUMMARY_MAX_TOKENS, timeout=SUMMARY_TIMEOUT)
# Try the primary backend a bounded number of times (each call fast-fails via
# SUMMARY_TIMEOUT), with a short backoff for a transient blip / restarting GPU.
last_exc: Exception | None = None
for attempt in range(MI50_ATTEMPTS):
try:
return llm.complete(messages, backend=backend)
return _call(backend)
except Exception as exc:
if attempt == _RETRIES - 1:
raise
logbus.log("debug", "summary retry", attempt=attempt + 1, error=str(exc)[:80])
time.sleep(5 * (attempt + 1))
raise RuntimeError("unreachable")
last_exc = exc
logbus.log("debug", "summary retry", attempt=attempt + 1,
backend=backend, error=str(exc)[:80])
if attempt < MI50_ATTEMPTS - 1:
time.sleep(5 * (attempt + 1))
# Primary exhausted. If it wasn't already cloud and cloud is configured, fall
# back once so a stuck/offline MI50 doesn't sink consolidation for the night.
if backend != "cloud" and config.load().openai_api_key:
logbus.log("info", "summary fell back to cloud", primary=backend,
error=str(last_exc)[:80] if last_exc else None)
return _call("cloud")
raise last_exc if last_exc else RuntimeError("summary failed")
def _summarize_transcript(transcript: str, backend: Backend) -> str: