email_utils.py
| 1 | """Email helpers for RL Agent: clean raw emails into a compact state and a ready-made set of email questions. |
| 2 | |
| 3 | Jev-style models lose accuracy on long, noisy state, and RL Agent reads at most max_len (512) tokens, |
| 4 | so strip quoted replies, signatures and disclaimers in code before asking questions. |
| 5 | """ |
| 6 | import re |
| 7 | |
| 8 | _QUOTE_HEADERS = [ |
| 9 | re.compile(r"^\s*On .{0,300}wrote:\s*$", re.I), |
| 10 | re.compile(r"^\s*-{2,}\s*(Original|Forwarded) Message\s*-{2,}", re.I), |
| 11 | re.compile(r"^\s*_{8,}\s*$"), |
| 12 | re.compile(r"^\s*From:\s.+$", re.I), |
| 13 | ] |
| 14 | _SIGNATURE_MARKERS = [ |
| 15 | re.compile(r"^\s*--\s*$"), |
| 16 | re.compile(r"^\s*(best|kind|warm|many thanks|thanks|thank you|regards|cheers|sincerely)[\w ,!.]*$", re.I), |
| 17 | re.compile(r"^\s*sent from my (iphone|android|mobile|ipad)", re.I), |
| 18 | ] |
| 19 | _DISCLAIMER = re.compile(r"(confidential|intended (solely )?for the (use of the )?(named )?(addressee|recipient)|" |
| 20 | r"if you (have )?received this (e-?mail|message) in error)", re.I) |
| 21 | |
| 22 | |
| 23 | def clean_email_body(body, max_chars=3000): |
| 24 | """Remove quoted history, signature and legal disclaimer; collapse whitespace; truncate.""" |
| 25 | text = (body or "").replace("\r\n", "\n").replace("\r", "\n").replace("\\n", "\n") |
| 26 | lines = [] |
| 27 | for line in text.split("\n"): |
| 28 | if any(p.match(line) for p in _QUOTE_HEADERS) and lines: |
| 29 | break # everything below is the previous thread |
| 30 | if line.lstrip().startswith(">"): |
| 31 | continue |
| 32 | lines.append(line.rstrip()) |
| 33 | # a sign-off only counts near the end (last 40%, or last 8 lines of a short email) and must be a short line |
| 34 | cut = len(lines) |
| 35 | for i in range(max(1, min(int(len(lines) * 0.6), len(lines) - 8)), len(lines)): |
| 36 | if len(lines[i].strip()) <= 40 and any(p.match(lines[i]) for p in _SIGNATURE_MARKERS): |
| 37 | cut = i |
| 38 | break |
| 39 | lines = lines[:cut] |
| 40 | paragraphs = [p for p in re.split(r"\n\s*\n", "\n".join(lines)) if not _DISCLAIMER.search(p)] |
| 41 | text = re.sub(r"[ \t]+", " ", "\n\n".join(p.strip() for p in paragraphs if p.strip())) |
| 42 | return text[:max_chars] |
| 43 | |
| 44 | |
| 45 | def email_state(subject, body, sender=None, clean=True, **extra): |
| 46 | """Build the state dict the email questions refer to (`subject`, `body`, optional `from`).""" |
| 47 | state = {"subject": (subject or "").strip(), "body": clean_email_body(body) if clean else (body or "")} |
| 48 | if sender: |
| 49 | state["from"] = sender |
| 50 | state.update({k: v for k, v in extra.items() if v is not None}) |
| 51 | return state |
| 52 | |
| 53 | |
| 54 | def email_questions(categories=None): |
| 55 | """A default fan-out of email questions. `categories` = {key: description} for your own routing labels.""" |
| 56 | categories = categories or { |
| 57 | "billing": "invoices, payments, refunds", "technical": "bugs, outages, integrations", |
| 58 | "sales": "pricing, demos, new purchases", "account": "login, access, profile changes", |
| 59 | "hr": "hiring, leave, payroll", "other": "none of the above", |
| 60 | } |
| 61 | return { |
| 62 | "category": {"type": "choice", "instructions": "Which team should handle the email in `body`?", "criteria": categories}, |
| 63 | "is_spam": {"type": "noul", "instructions": "Is this email unsolicited spam or bulk marketing?"}, |
| 64 | "is_phishing": {"type": "noul", "instructions": "Is this email a phishing or scam attempt to steal money, credentials, or personal data?", |
| 65 | "criteria": {"true": "phishing, scam, or fraud", "false": "a legitimate email"}}, |
| 66 | "urgency": {"type": "score", "instructions": "How urgent is the issue described in `body`?", |
| 67 | "criteria": ["no time pressure", "needs attention soon", "blocking issue or hard deadline"]}, |
| 68 | "needs_reply": {"type": "noul", "instructions": "Does the sender expect a reply?"}, |
| 69 | "sentiment": {"type": "score", "instructions": "What is the sender's tone in `body`?", |
| 70 | "criteria": ["angry or very negative", "negative", "neutral", "positive"]}, |
| 71 | } |
| 72 | |