fix: cap MI50 summary length + fast-fail cloud fallback
The dream cycle's summarize_all ran uncapped against the MI50: no max_tokens and no timeout, so the OpenAI SDK's 600s x2-retry default meant ~30 min per call. Combined with summary.py's own retry loop, one unsummarizable session pegged the GPU for hours (observed 2026-07-04: stuck since 23:02, nothing saved since 00:56, 7-8k-token runaway generations, all 4 llama.cpp slots busy). Not context overflow (0 shifts/truncations) - purely unbounded length on a slow backend timing out and retrying. - llm.complete(): add optional max_tokens (caps generation; num_predict for Ollama) and timeout (bounds the request and sets max_retries=0 so the caller owns retry policy). Both default None -> unchanged for every existing caller. - summary.py: cap gists at 768 tokens, 150s/call fast-fail, 2 MI50 attempts then one cloud fallback (when primary isn't already cloud and a key exists). Known limitation (scoped out per decision): the fallback triggers on timeouts/exceptions, not on a degraded backend returning garbage as a 200. Tests: fallback fires after 2 MI50 failures; no fallback when primary is cloud or no key; cap+timeout threaded into every complete() call; llm bounds tests. 172 pass, ruff clean. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_015yrEb5qpPGv2FjyxrB7LLk
This commit is contained in:
+29
-14
@@ -37,30 +37,45 @@ def _resolved_model(cfg, backend: Backend, model: str | None) -> str:
|
||||
return model or cfg.local_model
|
||||
|
||||
|
||||
def complete(messages: list[Message], backend: Backend = "local", model: str | None = None) -> str:
|
||||
def complete(messages: list[Message], backend: Backend = "local", model: str | None = None,
|
||||
max_tokens: int | None = None, timeout: float | None = None) -> str:
|
||||
"""Generate a completion. `model` overrides the backend's default model
|
||||
(used so live chat can run a stronger cloud model than bulk consolidation)."""
|
||||
(used so live chat can run a stronger cloud model than bulk consolidation).
|
||||
|
||||
`max_tokens` caps the generation length (guards a slow local model against
|
||||
rambling for thousands of tokens). `timeout`, when set, bounds each request
|
||||
and disables the SDK's own retries so the caller owns retry/fallback policy.
|
||||
Both default to None → unchanged behavior for every existing caller."""
|
||||
cfg = load()
|
||||
mdl = _resolved_model(cfg, backend, model)
|
||||
logbus.log("info", "llm call", kind="complete", backend=backend, model=mdl, tok=_approx_tok(messages))
|
||||
t0 = time.monotonic()
|
||||
|
||||
if backend == "cloud":
|
||||
if not cfg.openai_api_key:
|
||||
raise RuntimeError("OPENAI_API_KEY is not set")
|
||||
client = OpenAI(api_key=cfg.openai_api_key)
|
||||
resp = client.chat.completions.create(model=mdl, messages=messages)
|
||||
out = resp.choices[0].message.content or ""
|
||||
elif backend == "mi50":
|
||||
# MI50 box runs an OpenAI-compatible llama.cpp server; key is unused.
|
||||
client = OpenAI(api_key="not-needed", base_url=cfg.mi50_base_url)
|
||||
resp = client.chat.completions.create(model=mdl, messages=messages)
|
||||
if backend in ("cloud", "mi50"):
|
||||
if backend == "cloud":
|
||||
if not cfg.openai_api_key:
|
||||
raise RuntimeError("OPENAI_API_KEY is not set")
|
||||
client_kwargs: dict = {"api_key": cfg.openai_api_key}
|
||||
else:
|
||||
# MI50 box runs an OpenAI-compatible llama.cpp server; key is unused.
|
||||
client_kwargs = {"api_key": "not-needed", "base_url": cfg.mi50_base_url}
|
||||
if timeout is not None:
|
||||
client_kwargs["timeout"] = timeout
|
||||
client_kwargs["max_retries"] = 0 # caller owns retries (see summary.py)
|
||||
client = OpenAI(**client_kwargs)
|
||||
create_kwargs: dict = {"model": mdl, "messages": messages}
|
||||
if max_tokens is not None:
|
||||
create_kwargs["max_tokens"] = max_tokens
|
||||
resp = client.chat.completions.create(**create_kwargs)
|
||||
out = resp.choices[0].message.content or ""
|
||||
else:
|
||||
payload: dict = {"model": mdl, "messages": messages, "stream": False}
|
||||
if max_tokens is not None:
|
||||
payload["options"] = {"num_predict": max_tokens}
|
||||
resp = httpx.post(
|
||||
f"{cfg.local_base_url}/api/chat",
|
||||
json={"model": mdl, "messages": messages, "stream": False},
|
||||
timeout=120,
|
||||
json=payload,
|
||||
timeout=timeout or 120,
|
||||
)
|
||||
resp.raise_for_status()
|
||||
out = resp.json()["message"]["content"]
|
||||
|
||||
+34
-9
@@ -17,7 +17,16 @@ from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||
from lyra import config, llm, logbus, memory
|
||||
from lyra.llm import Backend, Message
|
||||
|
||||
_RETRIES = 4
|
||||
# Consolidation LLM budget. A gist is short (a handful of sentences), so cap the
|
||||
# generation hard — an uncapped local model will otherwise ramble for thousands
|
||||
# of tokens and, on a slow GPU, blow the request timeout. 768 is ~3x the longest
|
||||
# real gist we've stored.
|
||||
SUMMARY_MAX_TOKENS = 768
|
||||
# Attempts on the primary backend before falling back to cloud.
|
||||
MI50_ATTEMPTS = 2
|
||||
# Per-call timeout (seconds). A capped 768-token gist finishes in ~60-90s on the
|
||||
# MI50; 150s is headroom but bails a hung call fast so fallback isn't slow.
|
||||
SUMMARY_TIMEOUT = 150
|
||||
|
||||
# Re-summarize a session once it has accumulated this many new raw exchanges.
|
||||
SUMMARIZE_AFTER = 20
|
||||
@@ -61,16 +70,32 @@ def _summarize_text(text: str, backend: Backend) -> str:
|
||||
{"role": "system", "content": _PROMPT},
|
||||
{"role": "user", "content": text},
|
||||
]
|
||||
# Retry transient backend errors (e.g. the GPU server restarting) with backoff.
|
||||
for attempt in range(_RETRIES):
|
||||
|
||||
def _call(be: Backend) -> str:
|
||||
return llm.complete(messages, backend=be,
|
||||
max_tokens=SUMMARY_MAX_TOKENS, timeout=SUMMARY_TIMEOUT)
|
||||
|
||||
# Try the primary backend a bounded number of times (each call fast-fails via
|
||||
# SUMMARY_TIMEOUT), with a short backoff for a transient blip / restarting GPU.
|
||||
last_exc: Exception | None = None
|
||||
for attempt in range(MI50_ATTEMPTS):
|
||||
try:
|
||||
return llm.complete(messages, backend=backend)
|
||||
return _call(backend)
|
||||
except Exception as exc:
|
||||
if attempt == _RETRIES - 1:
|
||||
raise
|
||||
logbus.log("debug", "summary retry", attempt=attempt + 1, error=str(exc)[:80])
|
||||
time.sleep(5 * (attempt + 1))
|
||||
raise RuntimeError("unreachable")
|
||||
last_exc = exc
|
||||
logbus.log("debug", "summary retry", attempt=attempt + 1,
|
||||
backend=backend, error=str(exc)[:80])
|
||||
if attempt < MI50_ATTEMPTS - 1:
|
||||
time.sleep(5 * (attempt + 1))
|
||||
|
||||
# Primary exhausted. If it wasn't already cloud and cloud is configured, fall
|
||||
# back once so a stuck/offline MI50 doesn't sink consolidation for the night.
|
||||
if backend != "cloud" and config.load().openai_api_key:
|
||||
logbus.log("info", "summary fell back to cloud", primary=backend,
|
||||
error=str(last_exc)[:80] if last_exc else None)
|
||||
return _call("cloud")
|
||||
|
||||
raise last_exc if last_exc else RuntimeError("summary failed")
|
||||
|
||||
|
||||
def _summarize_transcript(transcript: str, backend: Backend) -> str:
|
||||
|
||||
Reference in New Issue
Block a user