email_utils.py
3.8 KB · 72 lines · python Raw
1 """Email helpers for RL Agent: clean raw emails into a compact state and a ready-made set of email questions.
2
3 Jev-style models lose accuracy on long, noisy state, and RL Agent reads at most max_len (512) tokens,
4 so strip quoted replies, signatures and disclaimers in code before asking questions.
5 """
6 import re
7
8 _QUOTE_HEADERS = [
9 re.compile(r"^\s*On .{0,300}wrote:\s*$", re.I),
10 re.compile(r"^\s*-{2,}\s*(Original|Forwarded) Message\s*-{2,}", re.I),
11 re.compile(r"^\s*_{8,}\s*$"),
12 re.compile(r"^\s*From:\s.+$", re.I),
13 ]
14 _SIGNATURE_MARKERS = [
15 re.compile(r"^\s*--\s*$"),
16 re.compile(r"^\s*(best|kind|warm|many thanks|thanks|thank you|regards|cheers|sincerely)[\w ,!.]*$", re.I),
17 re.compile(r"^\s*sent from my (iphone|android|mobile|ipad)", re.I),
18 ]
19 _DISCLAIMER = re.compile(r"(confidential|intended (solely )?for the (use of the )?(named )?(addressee|recipient)|"
20 r"if you (have )?received this (e-?mail|message) in error)", re.I)
21
22
23 def clean_email_body(body, max_chars=3000):
24 """Remove quoted history, signature and legal disclaimer; collapse whitespace; truncate."""
25 text = (body or "").replace("\r\n", "\n").replace("\r", "\n").replace("\\n", "\n")
26 lines = []
27 for line in text.split("\n"):
28 if any(p.match(line) for p in _QUOTE_HEADERS) and lines:
29 break # everything below is the previous thread
30 if line.lstrip().startswith(">"):
31 continue
32 lines.append(line.rstrip())
33 # a sign-off only counts near the end (last 40%, or last 8 lines of a short email) and must be a short line
34 cut = len(lines)
35 for i in range(max(1, min(int(len(lines) * 0.6), len(lines) - 8)), len(lines)):
36 if len(lines[i].strip()) <= 40 and any(p.match(lines[i]) for p in _SIGNATURE_MARKERS):
37 cut = i
38 break
39 lines = lines[:cut]
40 paragraphs = [p for p in re.split(r"\n\s*\n", "\n".join(lines)) if not _DISCLAIMER.search(p)]
41 text = re.sub(r"[ \t]+", " ", "\n\n".join(p.strip() for p in paragraphs if p.strip()))
42 return text[:max_chars]
43
44
45 def email_state(subject, body, sender=None, clean=True, **extra):
46 """Build the state dict the email questions refer to (`subject`, `body`, optional `from`)."""
47 state = {"subject": (subject or "").strip(), "body": clean_email_body(body) if clean else (body or "")}
48 if sender:
49 state["from"] = sender
50 state.update({k: v for k, v in extra.items() if v is not None})
51 return state
52
53
54 def email_questions(categories=None):
55 """A default fan-out of email questions. `categories` = {key: description} for your own routing labels."""
56 categories = categories or {
57 "billing": "invoices, payments, refunds", "technical": "bugs, outages, integrations",
58 "sales": "pricing, demos, new purchases", "account": "login, access, profile changes",
59 "hr": "hiring, leave, payroll", "other": "none of the above",
60 }
61 return {
62 "category": {"type": "choice", "instructions": "Which team should handle the email in `body`?", "criteria": categories},
63 "is_spam": {"type": "noul", "instructions": "Is this email unsolicited spam or bulk marketing?"},
64 "is_phishing": {"type": "noul", "instructions": "Is this email a phishing or scam attempt to steal money, credentials, or personal data?",
65 "criteria": {"true": "phishing, scam, or fraud", "false": "a legitimate email"}},
66 "urgency": {"type": "score", "instructions": "How urgent is the issue described in `body`?",
67 "criteria": ["no time pressure", "needs attention soon", "blocking issue or hard deadline"]},
68 "needs_reply": {"type": "noul", "instructions": "Does the sender expect a reply?"},
69 "sentiment": {"type": "score", "instructions": "What is the sender's tone in `body`?",
70 "criteria": ["angry or very negative", "negative", "neutral", "positive"]},
71 }
72