diff --git a/projekt-matching/projektmatch/mailparse.py b/projekt-matching/projektmatch/mailparse.py new file mode 100644 index 0000000..b82ba5b --- /dev/null +++ b/projekt-matching/projektmatch/mailparse.py @@ -0,0 +1,83 @@ +"""E-mail parsing: own-mail detection, project splitting, canonical URLs.""" +from __future__ import annotations + +import re +from urllib.parse import urlsplit, urlunsplit + +import requests +from bs4 import BeautifulSoup + +OWN_PREFIXES = ("[Projekt-Match]", "[Projekt-Match-Fehler]") +PROJECT_URL_RE = re.compile( + r"https?://(?:www\.)?freelancermap\.de/(?:nproj/|projekt/)[^\s\"'<>)\]]+") +BROWSER_UA = ("Mozilla/5.0 (X11; Linux x86_64; rv:128.0) " + "Gecko/20100101 Firefox/128.0") + + +def is_own_mail(subject) -> bool: + return bool(subject) and subject.strip().startswith(OWN_PREFIXES) + + +def bodies(msg): + """Return (html, text) of the first text/html and text/plain parts.""" + html = text = "" + for part in msg.walk(): + if part.get_content_maintype() == "multipart": + continue + try: + payload = part.get_content() + except Exception: + continue + ctype = part.get_content_type() + if ctype == "text/html" and not html: + html = payload + elif ctype == "text/plain" and not text: + text = payload + return html, text + + +def _url_key(url: str) -> str: + return urlsplit(url).path + + +def split_projects(html: str, text: str) -> list: + """Extract project items {title, url} from a mail body. HTML preferred; + per project (URL path) the LONGEST anchor text wins (skips 'Zum Projekt' + buttons). Plain-text fallback: URL line + nearest preceding non-empty line.""" + best = {} # key -> {"title", "url"} + order = [] + if html: + soup = BeautifulSoup(html, "html.parser") + for a in soup.find_all("a", href=PROJECT_URL_RE): + url = a["href"] + key = _url_key(url) + title = " ".join(a.get_text(" ", strip=True).split()) + if key not in best: + best[key] = {"title": title, "url": url} + order.append(key) + elif len(title) > len(best[key]["title"]): + best[key]["title"] = title + if not best and text: + last_line = "" + for line in text.splitlines(): + match = PROJECT_URL_RE.search(line) + stripped = line.strip() + if match: + key = _url_key(match.group(0)) + if key not in best: + best[key] = {"title": last_line or match.group(0), + "url": match.group(0)} + order.append(key) + elif stripped: + last_line = stripped + return [best[k] for k in order if best[k]["title"]] + + +def canonical_url(url: str, session=None, timeout=20) -> str: + """Follow redirects, strip query string and fragment.""" + sess = session or requests.Session() + resp = sess.get(url, timeout=timeout, allow_redirects=True, + headers={"User-Agent": BROWSER_UA}) + resp.raise_for_status() + parts = urlsplit(resp.url) + return urlunsplit((parts.scheme, parts.netloc, parts.path, "", "")) diff --git a/projekt-matching/tests/test_mailparse.py b/projekt-matching/tests/test_mailparse.py new file mode 100644 index 0000000..5d3d0af --- /dev/null +++ b/projekt-matching/tests/test_mailparse.py @@ -0,0 +1,69 @@ +import email +from email.policy import default as default_policy +from unittest import mock + +from projektmatch import mailparse + +HTML = """ +
+Beschreibung ...
+Zum Projekt +Hallo
", "Hallo") == [] + + +def test_bodies_multipart(): + msg = email.message.EmailMessage(policy=default_policy) + msg["Subject"] = "x" + msg.set_content("plain body") + msg.add_alternative("html body
", subtype="html") + html, text = mailparse.bodies(msg) + assert "html body" in html and "plain body" in text + + +def test_canonical_url_strips_query(): + session = mock.Mock() + session.get.return_value = mock.Mock( + url="https://www.freelancermap.de/projekt/python-entwickler?ref=1", + raise_for_status=lambda: None) + out = mailparse.canonical_url("https://www.freelancermap.de/nproj/1.html?x=1", + session=session) + assert out == "https://www.freelancermap.de/projekt/python-entwickler"