"""E-mail parsing: own-mail detection, project splitting, canonical URLs.""" from __future__ import annotations import re from urllib.parse import urlsplit, urlunsplit import requests from bs4 import BeautifulSoup OWN_PREFIXES = ("[Projekt-Match]", "[Projekt-Match-Fehler]") PROJECT_URL_RE = re.compile( r"https?://(?:www\.)?freelancermap\.de/(?:nproj/|projekt/)[^\s\"'<>)\]]+") BROWSER_UA = ("Mozilla/5.0 (X11; Linux x86_64; rv:128.0) " "Gecko/20100101 Firefox/128.0") def is_own_mail(subject) -> bool: return bool(subject) and subject.strip().startswith(OWN_PREFIXES) def bodies(msg): """Return (html, text) of the first text/html and text/plain parts.""" html = text = "" for part in msg.walk(): if part.get_content_maintype() == "multipart": continue try: payload = part.get_content() except Exception: continue ctype = part.get_content_type() if ctype == "text/html" and not html: html = payload elif ctype == "text/plain" and not text: text = payload return html, text def _url_key(url: str) -> str: return urlsplit(url).path def split_projects(html: str, text: str) -> list: """Extract project items {title, url} from a mail body. HTML preferred; per project (URL path) the LONGEST anchor text wins (skips 'Zum Projekt' buttons). Plain-text fallback: URL line + nearest preceding non-empty line.""" best = {} # key -> {"title", "url"} order = [] if html: soup = BeautifulSoup(html, "html.parser") for a in soup.find_all("a", href=PROJECT_URL_RE): url = a["href"] key = _url_key(url) title = " ".join(a.get_text(" ", strip=True).split()) if key not in best: best[key] = {"title": title, "url": url} order.append(key) elif len(title) > len(best[key]["title"]): best[key]["title"] = title if not best and text: last_line = "" for line in text.splitlines(): match = PROJECT_URL_RE.search(line) stripped = line.strip() if match: key = _url_key(match.group(0)) if key not in best: best[key] = {"title": last_line or match.group(0), "url": match.group(0)} order.append(key) elif stripped: last_line = stripped return [best[k] for k in order if best[k]["title"]] def canonical_url(url: str, session=None, timeout=20) -> str: """Follow redirects, strip query string and fragment.""" sess = session or requests.Session() resp = sess.get(url, timeout=timeout, allow_redirects=True, headers={"User-Agent": BROWSER_UA}) resp.raise_for_status() parts = urlsplit(resp.url) return urlunsplit((parts.scheme, parts.netloc, parts.path, "", ""))