fix(projekt-matching): break infinite alert loop (issue 2)
Root cause chain (evidence: 25+ failed traces on nproj/3022414): straight quote in page text ends the guided-decoding JSON string early -> grammar deadlock -> whitespace loop until context/token limit -> chat_json fails 3x (~35 min) -> IMAP conn idles out -> move_to_trash raises -> mail never trashed -> reprocessed forever. - ingest/mailer: MailBox.ensure_alive() reconnect before finalize; move_to_trash failure falls back to flag_checked, run continues - stages: page_text() feeds only div.content minus skill-tag cloud to the LLM and maps quote chars to apostrophes (the actual degeneration trigger, live-verified: extract now OK in 21 s) - llm: requirements maxItems 40; max_tokens 8192 caps degenerate runs at ~100 s instead of ~10 min Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
@@ -119,6 +119,10 @@ def run_ingest(cfg, mailbox=None, dispatch=None, espo=None, resolve=None):
|
||||
or f"unbekannter Status: {result.get('status')!r}")
|
||||
results.append(result)
|
||||
summary[result["status"]] += 1
|
||||
try:
|
||||
box.ensure_alive() # Flow-2-Dispatches können >30 min dauern — Server-Idle-Timeout
|
||||
except Exception: # noqa: BLE001
|
||||
pass # Alert/Trash unten haben eigene Fallbacks
|
||||
failed = [r for r in results if r["status"] == "failed"]
|
||||
if failed:
|
||||
lines = "\n\n".join(
|
||||
@@ -134,7 +138,14 @@ def run_ingest(cfg, mailbox=None, dispatch=None, espo=None, resolve=None):
|
||||
"werden:\n\n" + lines + "\n")
|
||||
except Exception: # noqa: BLE001
|
||||
summary["alertFailed"] = summary.get("alertFailed", 0) + 1
|
||||
box.move_to_trash(uid)
|
||||
try:
|
||||
box.move_to_trash(uid)
|
||||
except Exception: # noqa: BLE001
|
||||
summary["trashFailed"] = summary.get("trashFailed", 0) + 1
|
||||
try:
|
||||
box.flag_checked(uid) # nie erneut verarbeiten — Terminal-Zustand
|
||||
except Exception: # noqa: BLE001
|
||||
pass # nächster Lauf räumt auf
|
||||
return summary
|
||||
finally:
|
||||
box.close()
|
||||
|
||||
@@ -1,7 +1,9 @@
|
||||
"""vLLM (OpenAI-compatible) calls with response_format json_schema + German prompts.
|
||||
|
||||
max_tokens stays UNSET so nothing can truncate the final answer (65k
|
||||
context window). Thinking is DISABLED via chat_template_kwargs: with the
|
||||
max_tokens is capped at 8192: every valid answer fits comfortably, and an
|
||||
uncapped run lets a degenerate generation grind for ~10 min before failing
|
||||
(observed live: 100+-requirement JSON, broken at ~char 1725, 3 retries =
|
||||
~35 min per cycle). Thinking is DISABLED via chat_template_kwargs: with the
|
||||
vLLM qwen3 reasoning parser active, json_schema guided decoding plus
|
||||
thinking degenerates — the run burns the entire context window
|
||||
(finish_reason "length", ~63k completion tokens) and returns empty or
|
||||
@@ -34,6 +36,7 @@ EXTRACT_SCHEMA = {
|
||||
"contactPerson": {"type": ["string", "null"]},
|
||||
"requirements": {
|
||||
"type": "array",
|
||||
"maxItems": 40,
|
||||
"items": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
@@ -139,6 +142,7 @@ z. B. vLLM, TGI, Triton" ab).
|
||||
|
||||
def _post(base, model, messages, schema):
|
||||
body = {"model": model, "messages": messages, "temperature": 0.1,
|
||||
"max_tokens": 8192,
|
||||
"chat_template_kwargs": {"enable_thinking": False},
|
||||
"response_format": {"type": "json_schema", "json_schema": {
|
||||
"name": "out", "schema": schema}}}
|
||||
|
||||
@@ -16,12 +16,31 @@ KEYWORD = "$ProjektChecked"
|
||||
|
||||
class MailBox:
|
||||
def __init__(self, user, password, conn=None):
|
||||
self.conn = conn or imaplib.IMAP4_SSL(IMAP_HOST, IMAP_PORT)
|
||||
if conn is None:
|
||||
self.conn.login(user, password)
|
||||
self.user, self.password = user, password
|
||||
self.conn = conn or self._connect()
|
||||
self.conn.select("INBOX")
|
||||
self._trash = None
|
||||
|
||||
def _connect(self):
|
||||
conn = imaplib.IMAP4_SSL(IMAP_HOST, IMAP_PORT)
|
||||
conn.login(self.user, self.password)
|
||||
return conn
|
||||
|
||||
def ensure_alive(self):
|
||||
"""Reconnect if the server dropped the connection (idle timeout)."""
|
||||
try:
|
||||
typ, _ = self.conn.noop()
|
||||
if typ == "OK":
|
||||
return
|
||||
except Exception: # noqa: BLE001
|
||||
pass
|
||||
try:
|
||||
self.conn.logout()
|
||||
except Exception: # noqa: BLE001
|
||||
pass
|
||||
self.conn = self._connect()
|
||||
self.conn.select("INBOX")
|
||||
|
||||
def unchecked_uids(self):
|
||||
typ, data = self.conn.uid("SEARCH", None, f"UNKEYWORD {KEYWORD}")
|
||||
if typ != "OK" or not data or not data[0]:
|
||||
|
||||
@@ -20,6 +20,7 @@ TEAM_BY_OFFER = {"Projekt": "DesTEngS",
|
||||
"Arbeitnehmer-Angebot": "Arbeitnehmer",
|
||||
"ANÜ": "ANÜ"}
|
||||
MIN_PAGE_CHARS = 200
|
||||
QUOTES = str.maketrans({c: "'" for c in "„“”\"«»‹›"})
|
||||
|
||||
|
||||
def run_stage(name, fn, ctx, cfg, **kw):
|
||||
@@ -32,6 +33,25 @@ def run_stage(name, fn, ctx, cfg, **kw):
|
||||
return ctx
|
||||
|
||||
|
||||
def page_text(html):
|
||||
"""Project text for the LLM: header + description, WITHOUT the skill-tag
|
||||
cloud (div.project-body-badges, ~100 entries on some pages — blows the
|
||||
requirements list up until the model emits broken JSON) and without
|
||||
site chrome. Falls back to whole-page text if div.content is missing."""
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
for tag in soup(["script", "style", "noscript"]):
|
||||
tag.decompose()
|
||||
root = soup.find("div", class_="content") or soup
|
||||
for tag in root.find_all(class_="project-body-badges"):
|
||||
tag.decompose()
|
||||
# Anführungszeichen -> Apostroph: ein wörtlich übernommenes " beendet den
|
||||
# JSON-String vorzeitig und treibt guided decoding in eine Whitespace-
|
||||
# Endlosschleife (finish_reason "length", live verifiziert an nproj/3022414)
|
||||
return "\n".join(line for line in
|
||||
root.get_text("\n", strip=True).splitlines()
|
||||
if line.strip()).translate(QUOTES)
|
||||
|
||||
|
||||
def stage_fetch(ctx, cfg):
|
||||
last = None
|
||||
for attempt in range(3):
|
||||
@@ -39,12 +59,7 @@ def stage_fetch(ctx, cfg):
|
||||
resp = requests.get(ctx["canonical"], timeout=30,
|
||||
headers={"User-Agent": BROWSER_UA})
|
||||
resp.raise_for_status()
|
||||
soup = BeautifulSoup(resp.text, "html.parser")
|
||||
for tag in soup(["script", "style", "noscript"]):
|
||||
tag.decompose()
|
||||
text = "\n".join(line for line in
|
||||
soup.get_text("\n", strip=True).splitlines()
|
||||
if line.strip())
|
||||
text = page_text(resp.text)
|
||||
if len(text) < MIN_PAGE_CHARS:
|
||||
raise ValueError(
|
||||
f"Seitentext zu kurz ({len(text)} Zeichen) — "
|
||||
|
||||
Reference in New Issue
Block a user