mirror of
https://github.com/open-webui/open-webui.git
synced 2026-08-29 08:07:34 -05:00
fix: Comprehensive HTML stripping for email indexing
Problem:
1. HTML garbage in keywords (border, padding, cellpadding)
2. DOCTYPE declarations appearing in snippets
3. MSO conditional comments ('[if mso]') not removed
4. Inline styles (style="...") left garbage text
Solution - Both files now strip:
- DOCTYPE and XML declarations (<!DOCTYPE...>, <?xml?>)
- CDATA sections and HTML comments
- MSO Office conditional comments ([if mso]...[endif])
- Script and style blocks with content
- Inline style attributes BEFORE removing tags
- All remaining HTML tags
Defense in depth:
- gmail_processor.py: First-pass HTML stripping
- gmail_indexer_v2.py: Secondary stripping in _deep_clean_for_embeddings
This ensures clean text for embeddings, keywords, and snippets.
This commit is contained in:
@@ -595,6 +595,33 @@ class GmailIndexerV2:
|
||||
text = text.replace("\\\\", "")
|
||||
text = text.replace("\\/", "/")
|
||||
|
||||
# STEP 1.5: Strip HTML completely (defense in depth)
|
||||
# Remove DOCTYPE declarations (<!DOCTYPE...>)
|
||||
text = re.sub(r"<![^>]*>", "", text, flags=re.IGNORECASE)
|
||||
# Remove XML declarations (<?xml...?>)
|
||||
text = re.sub(r"<\?[^>]*\?>", "", text, flags=re.IGNORECASE)
|
||||
# Remove CDATA sections
|
||||
text = re.sub(r"<!\[CDATA\[.*?\]\]>", "", text, flags=re.DOTALL)
|
||||
# Remove HTML comments (<!--...-->)
|
||||
text = re.sub(r"<!--.*?-->", "", text, flags=re.DOTALL)
|
||||
# Remove script and style tags with content
|
||||
text = re.sub(
|
||||
r"<script[^>]*>.*?</script>", "", text, flags=re.DOTALL | re.IGNORECASE
|
||||
)
|
||||
text = re.sub(
|
||||
r"<style[^>]*>.*?</style>", "", text, flags=re.DOTALL | re.IGNORECASE
|
||||
)
|
||||
# Remove inline styles (style="...") before removing tags
|
||||
text = re.sub(
|
||||
r'\s*style\s*=\s*["\'][^"\']*["\']', "", text, flags=re.IGNORECASE
|
||||
)
|
||||
# Remove all HTML tags
|
||||
text = re.sub(r"<[^>]+>", "", text)
|
||||
# Remove MSO (Microsoft Office) conditional comments
|
||||
text = re.sub(
|
||||
r"\[if[^\]]*\].*?\[endif\]", "", text, flags=re.DOTALL | re.IGNORECASE
|
||||
)
|
||||
|
||||
# STEP 2: Remove quoted reply blocks
|
||||
text = re.sub(r"On .{1,100}wrote:.*", "", text, flags=re.DOTALL | re.IGNORECASE)
|
||||
text = re.sub(r"^>.*$", "", text, flags=re.MULTILINE)
|
||||
|
||||
@@ -26,7 +26,7 @@ logger.setLevel(logging.INFO)
|
||||
class GmailProcessor:
|
||||
"""
|
||||
Processes Gmail API responses into structured email data.
|
||||
|
||||
|
||||
This is a pure data transformation class - no external API calls,
|
||||
no database operations, just parsing and cleaning.
|
||||
"""
|
||||
@@ -38,70 +38,75 @@ class GmailProcessor:
|
||||
def parse_email(self, email_data: dict, user_id: str) -> Dict:
|
||||
"""
|
||||
Parse a Gmail API message response into structured data.
|
||||
|
||||
|
||||
Args:
|
||||
email_data: Full Gmail API message response (format='full')
|
||||
user_id: User ID for metadata isolation
|
||||
|
||||
|
||||
Returns:
|
||||
Dict with parsed email data and metadata
|
||||
|
||||
|
||||
Example:
|
||||
>>> processor = GmailProcessor()
|
||||
>>> result = processor.parse_email(gmail_api_response, "user_123")
|
||||
>>> print(result["subject"])
|
||||
"Q4 Budget Discussion"
|
||||
"""
|
||||
|
||||
|
||||
try:
|
||||
# Extract basic identifiers
|
||||
email_id = email_data.get("id", "")
|
||||
thread_id = email_data.get("threadId", "")
|
||||
|
||||
|
||||
if not email_id:
|
||||
raise ValueError("Email data missing 'id' field")
|
||||
|
||||
|
||||
# Extract headers
|
||||
headers_dict = self._extract_headers(email_data)
|
||||
|
||||
|
||||
# Extract core email fields
|
||||
subject = headers_dict.get("Subject", "(No Subject)")
|
||||
from_addr = headers_dict.get("From", "")
|
||||
to_addr = headers_dict.get("To", "")
|
||||
cc_addr = headers_dict.get("Cc", "")
|
||||
date_str = headers_dict.get("Date", "")
|
||||
|
||||
|
||||
# Parse date
|
||||
date_timestamp, date_readable = self._parse_date(
|
||||
date_str,
|
||||
email_data.get("internalDate")
|
||||
date_str, email_data.get("internalDate")
|
||||
)
|
||||
|
||||
|
||||
# Extract email body
|
||||
payload = email_data.get("payload", {})
|
||||
body_raw = self._extract_body(payload)
|
||||
snippet = email_data.get("snippet", "")
|
||||
|
||||
|
||||
# Clean email body with advanced cleaning
|
||||
body_full_clean, body_original_only = EmailCleaner.clean_email_body(body_raw)
|
||||
|
||||
body_full_clean, body_original_only = EmailCleaner.clean_email_body(
|
||||
body_raw
|
||||
)
|
||||
|
||||
# Use original message (no quotes/signatures) for primary content
|
||||
body_clean = body_original_only if body_original_only else body_full_clean
|
||||
|
||||
|
||||
# Extract labels
|
||||
labels = email_data.get("labelIds", [])
|
||||
|
||||
|
||||
# Extract attachment information
|
||||
attachments = self._get_attachments(payload)
|
||||
has_attachments = len(attachments) > 0
|
||||
|
||||
|
||||
# Parse email addresses
|
||||
from_name, from_email = EmailCleaner.parse_email_address(from_addr)
|
||||
to_name, to_email = EmailCleaner.parse_email_address(to_addr)
|
||||
|
||||
|
||||
# Detect if this is a reply
|
||||
is_reply = "RE:" in subject.upper() or "Re:" in subject or any(label == "SENT" for label in labels)
|
||||
|
||||
is_reply = (
|
||||
"RE:" in subject.upper()
|
||||
or "Re:" in subject
|
||||
or any(label == "SENT" for label in labels)
|
||||
)
|
||||
|
||||
# Create document text (what will be embedded) - cleaner format
|
||||
document_text = self._create_document_text(
|
||||
subject=subject,
|
||||
@@ -110,9 +115,9 @@ class GmailProcessor:
|
||||
to_email=to_email,
|
||||
date_readable=date_readable,
|
||||
body=body_clean,
|
||||
snippet=snippet
|
||||
snippet=snippet,
|
||||
)
|
||||
|
||||
|
||||
# Build metadata (improved structure for better search)
|
||||
metadata = {
|
||||
# Core identification
|
||||
@@ -120,7 +125,6 @@ class GmailProcessor:
|
||||
"email_id": email_id,
|
||||
"thread_id": thread_id,
|
||||
"user_id": user_id,
|
||||
|
||||
# Email fields (structured)
|
||||
"subject": subject,
|
||||
"from_name": from_name,
|
||||
@@ -133,7 +137,6 @@ class GmailProcessor:
|
||||
"date": date_readable,
|
||||
"date_timestamp": date_timestamp,
|
||||
"labels": labels, # Keep as array, not comma-separated string
|
||||
|
||||
# Content metadata
|
||||
"has_attachments": has_attachments,
|
||||
"attachment_count": len(attachments),
|
||||
@@ -141,21 +144,16 @@ class GmailProcessor:
|
||||
"word_count": len(body_clean.split()),
|
||||
"body_length": len(body_clean),
|
||||
"original_body_length": len(body_raw),
|
||||
|
||||
# Cleaned text fields
|
||||
"body_full_clean": body_full_clean, # Full email with cleaned quotes
|
||||
"body_original": body_clean, # Just the original message
|
||||
|
||||
# Categorization
|
||||
"source": "gmail",
|
||||
"doc_type": "email",
|
||||
|
||||
# Hash for deduplication
|
||||
"hash": hashlib.sha256(
|
||||
f"{email_id}{user_id}".encode()
|
||||
).hexdigest(),
|
||||
"hash": hashlib.sha256(f"{email_id}{user_id}".encode()).hexdigest(),
|
||||
}
|
||||
|
||||
|
||||
return {
|
||||
"email_id": email_id,
|
||||
"thread_id": thread_id,
|
||||
@@ -166,42 +164,40 @@ class GmailProcessor:
|
||||
"snippet": snippet,
|
||||
"attachments": attachments, # List of attachment metadata
|
||||
}
|
||||
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Error parsing email {email_data.get('id', 'unknown')}: {e}")
|
||||
raise
|
||||
|
||||
|
||||
def _extract_headers(self, email_data: dict) -> Dict[str, str]:
|
||||
"""Extract email headers from Gmail API response"""
|
||||
|
||||
|
||||
payload = email_data.get("payload", {})
|
||||
headers_list = payload.get("headers", [])
|
||||
|
||||
|
||||
# Convert list of {name, value} to dict
|
||||
headers_dict = {}
|
||||
for header in headers_list:
|
||||
name = header.get("name", "")
|
||||
value = header.get("value", "")
|
||||
headers_dict[name] = value
|
||||
|
||||
|
||||
return headers_dict
|
||||
|
||||
|
||||
def _parse_date(
|
||||
self,
|
||||
date_str: str,
|
||||
internal_date: Optional[str] = None
|
||||
self, date_str: str, internal_date: Optional[str] = None
|
||||
) -> Tuple[int, str]:
|
||||
"""
|
||||
Parse email date into timestamp and readable format.
|
||||
|
||||
|
||||
Args:
|
||||
date_str: Date header from email (e.g., "Mon, 15 Nov 2020 14:30:00 +0000")
|
||||
internal_date: Gmail internal date (milliseconds since epoch)
|
||||
|
||||
|
||||
Returns:
|
||||
Tuple of (unix_timestamp, readable_date_string)
|
||||
"""
|
||||
|
||||
|
||||
# Try to parse the Date header
|
||||
try:
|
||||
date_obj = parsedate_to_datetime(date_str)
|
||||
@@ -210,7 +206,7 @@ class GmailProcessor:
|
||||
return timestamp, readable
|
||||
except Exception as e:
|
||||
logger.debug(f"Could not parse date '{date_str}': {e}")
|
||||
|
||||
|
||||
# Fallback to internal date
|
||||
if internal_date:
|
||||
try:
|
||||
@@ -220,164 +216,194 @@ class GmailProcessor:
|
||||
return timestamp, readable
|
||||
except Exception as e:
|
||||
logger.debug(f"Could not parse internal date '{internal_date}': {e}")
|
||||
|
||||
|
||||
# Final fallback: current time
|
||||
now = int(datetime.now().timestamp())
|
||||
return now, "Unknown Date"
|
||||
|
||||
|
||||
def _extract_body(self, payload: dict) -> str:
|
||||
"""
|
||||
Recursively extract email body from Gmail API payload.
|
||||
|
||||
|
||||
Handles:
|
||||
- Plain text emails
|
||||
- HTML emails (strips tags)
|
||||
- Multipart emails (multipart/alternative, multipart/mixed)
|
||||
- Nested MIME structures
|
||||
|
||||
|
||||
Args:
|
||||
payload: Gmail API message payload
|
||||
|
||||
|
||||
Returns:
|
||||
Extracted email body as plain text
|
||||
"""
|
||||
|
||||
|
||||
# Handle multipart emails (most common)
|
||||
if "parts" in payload:
|
||||
return self._extract_from_parts(payload["parts"])
|
||||
|
||||
|
||||
# Handle single-part emails
|
||||
elif "body" in payload and "data" in payload["body"]:
|
||||
mime_type = payload.get("mimeType", "")
|
||||
data = payload["body"]["data"]
|
||||
return self._decode_body_data(data, mime_type)
|
||||
|
||||
|
||||
return ""
|
||||
|
||||
|
||||
def _extract_from_parts(self, parts: list) -> str:
|
||||
"""
|
||||
Extract body from multipart email structure.
|
||||
|
||||
|
||||
Priority:
|
||||
1. text/plain (preferred)
|
||||
2. text/html (strip tags)
|
||||
3. Nested parts (recurse)
|
||||
"""
|
||||
|
||||
|
||||
# First pass: look for text/plain
|
||||
for part in parts:
|
||||
mime_type = part.get("mimeType", "")
|
||||
|
||||
|
||||
if mime_type == "text/plain":
|
||||
data = part.get("body", {}).get("data", "")
|
||||
if data:
|
||||
body = self._decode_body_data(data, mime_type)
|
||||
if body:
|
||||
return body
|
||||
|
||||
|
||||
# Second pass: look for text/html
|
||||
for part in parts:
|
||||
mime_type = part.get("mimeType", "")
|
||||
|
||||
|
||||
if mime_type == "text/html":
|
||||
data = part.get("body", {}).get("data", "")
|
||||
if data:
|
||||
body = self._decode_body_data(data, mime_type)
|
||||
if body:
|
||||
return body
|
||||
|
||||
|
||||
# Third pass: recurse into nested parts
|
||||
for part in parts:
|
||||
if "parts" in part:
|
||||
body = self._extract_from_parts(part["parts"])
|
||||
if body:
|
||||
return body
|
||||
|
||||
|
||||
return ""
|
||||
|
||||
|
||||
def _decode_body_data(self, data: str, mime_type: str) -> str:
|
||||
"""
|
||||
Decode base64-encoded email body data.
|
||||
|
||||
|
||||
Args:
|
||||
data: Base64-encoded body data
|
||||
mime_type: MIME type (text/plain or text/html)
|
||||
|
||||
|
||||
Returns:
|
||||
Decoded and cleaned text
|
||||
"""
|
||||
|
||||
|
||||
if not data:
|
||||
return ""
|
||||
|
||||
|
||||
try:
|
||||
# Gmail API uses URL-safe base64 encoding
|
||||
decoded_bytes = base64.urlsafe_b64decode(data)
|
||||
decoded_text = decoded_bytes.decode("utf-8", errors="ignore")
|
||||
|
||||
|
||||
# Clean based on MIME type
|
||||
if mime_type == "text/html":
|
||||
decoded_text = self._strip_html_tags(decoded_text)
|
||||
|
||||
|
||||
return decoded_text.strip()
|
||||
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Error decoding body data: {e}")
|
||||
return ""
|
||||
|
||||
|
||||
def _strip_html_tags(self, html_text: str) -> str:
|
||||
"""
|
||||
Strip HTML tags and extract plain text.
|
||||
|
||||
Basic implementation - good enough for email bodies.
|
||||
|
||||
Comprehensive implementation handles:
|
||||
- DOCTYPE and XML declarations
|
||||
- Script and style blocks
|
||||
- HTML comments and CDATA
|
||||
- MSO (Microsoft Office) conditional comments
|
||||
- Inline styles
|
||||
- All HTML tags
|
||||
"""
|
||||
|
||||
|
||||
if not html_text:
|
||||
return ""
|
||||
|
||||
# Decode HTML entities first
|
||||
text = unescape(html_text)
|
||||
|
||||
|
||||
text = html_text
|
||||
|
||||
# Remove DOCTYPE declarations (<!DOCTYPE...>)
|
||||
text = re.sub(r"<![^>]*>", "", text, flags=re.IGNORECASE)
|
||||
# Remove XML declarations (<?xml...?>)
|
||||
text = re.sub(r"<\?[^>]*\?>", "", text, flags=re.IGNORECASE)
|
||||
# Remove CDATA sections
|
||||
text = re.sub(r"<!\[CDATA\[.*?\]\]>", "", text, flags=re.DOTALL)
|
||||
# Remove HTML comments (<!--...-->)
|
||||
text = re.sub(r"<!--.*?-->", "", text, flags=re.DOTALL)
|
||||
# Remove MSO (Microsoft Office) conditional comments
|
||||
text = re.sub(
|
||||
r"\[if[^\]]*\].*?\[endif\]", "", text, flags=re.DOTALL | re.IGNORECASE
|
||||
)
|
||||
|
||||
# Remove script and style tags with their content
|
||||
text = re.sub(r'<script[^>]*>.*?</script>', '', text, flags=re.DOTALL | re.IGNORECASE)
|
||||
text = re.sub(r'<style[^>]*>.*?</style>', '', text, flags=re.DOTALL | re.IGNORECASE)
|
||||
|
||||
text = re.sub(
|
||||
r"<script[^>]*>.*?</script>", "", text, flags=re.DOTALL | re.IGNORECASE
|
||||
)
|
||||
text = re.sub(
|
||||
r"<style[^>]*>.*?</style>", "", text, flags=re.DOTALL | re.IGNORECASE
|
||||
)
|
||||
|
||||
# Remove inline styles (before removing tags to prevent garbage)
|
||||
text = re.sub(
|
||||
r'\s*style\s*=\s*["\'][^"\']*["\']', "", text, flags=re.IGNORECASE
|
||||
)
|
||||
|
||||
# Replace common block elements with newlines
|
||||
text = re.sub(r'</(p|div|h[1-6]|li|tr)>', '\n', text, flags=re.IGNORECASE)
|
||||
text = re.sub(r'<br\s*/?>', '\n', text, flags=re.IGNORECASE)
|
||||
|
||||
text = re.sub(r"</(p|div|h[1-6]|li|tr)>", "\n", text, flags=re.IGNORECASE)
|
||||
text = re.sub(r"<br\s*/?>", "\n", text, flags=re.IGNORECASE)
|
||||
|
||||
# Remove all remaining HTML tags
|
||||
text = re.sub(r'<[^>]+>', '', text)
|
||||
|
||||
text = re.sub(r"<[^>]+>", "", text)
|
||||
|
||||
# Decode HTML entities
|
||||
text = unescape(text)
|
||||
|
||||
# Clean up whitespace
|
||||
text = re.sub(r'\n\s*\n', '\n\n', text)
|
||||
text = re.sub(r' +', ' ', text)
|
||||
|
||||
text = re.sub(r"\n\s*\n", "\n\n", text)
|
||||
text = re.sub(r" +", " ", text)
|
||||
|
||||
return text.strip()
|
||||
|
||||
|
||||
def _has_attachments(self, payload: dict) -> bool:
|
||||
"""Check if email has attachments (returns boolean only)."""
|
||||
attachments = self._get_attachments(payload)
|
||||
return len(attachments) > 0
|
||||
|
||||
|
||||
def _get_attachments(self, payload: dict, attachments: list = None) -> list:
|
||||
"""
|
||||
Extract attachment metadata from email payload.
|
||||
|
||||
|
||||
Args:
|
||||
payload: Gmail API message payload
|
||||
attachments: List to accumulate attachments (for recursion)
|
||||
|
||||
|
||||
Returns:
|
||||
List of attachment dicts with: filename, mimeType, size, attachmentId
|
||||
"""
|
||||
if attachments is None:
|
||||
attachments = []
|
||||
|
||||
|
||||
if "parts" in payload:
|
||||
for part in payload["parts"]:
|
||||
filename = part.get("filename", "")
|
||||
|
||||
|
||||
# Part has a filename and body.attachmentId - it's an attachment
|
||||
if filename and part.get("body", {}).get("attachmentId"):
|
||||
attachment_info = {
|
||||
@@ -387,13 +413,13 @@ class GmailProcessor:
|
||||
"attachmentId": part.get("body", {}).get("attachmentId"),
|
||||
}
|
||||
attachments.append(attachment_info)
|
||||
|
||||
|
||||
# Recursively check nested parts
|
||||
if "parts" in part:
|
||||
self._get_attachments(part, attachments)
|
||||
|
||||
|
||||
return attachments
|
||||
|
||||
|
||||
def _create_document_text(
|
||||
self,
|
||||
subject: str,
|
||||
@@ -402,23 +428,23 @@ class GmailProcessor:
|
||||
to_email: str,
|
||||
date_readable: str,
|
||||
body: str,
|
||||
snippet: str
|
||||
snippet: str,
|
||||
) -> str:
|
||||
"""
|
||||
Create clean document text for embedding.
|
||||
|
||||
|
||||
Simple, clean format without headers (better for semantic search).
|
||||
"""
|
||||
|
||||
|
||||
# Use cleaned body if available, otherwise snippet
|
||||
content = body if body else snippet
|
||||
|
||||
|
||||
# Build clean document - just subject and content (no "Email Subject:" labels)
|
||||
if subject and subject != "(No Subject)":
|
||||
document_text = f"{subject}\n\n{content}"
|
||||
else:
|
||||
document_text = content
|
||||
|
||||
|
||||
return document_text.strip()
|
||||
|
||||
|
||||
@@ -430,10 +456,10 @@ class GmailProcessor:
|
||||
def create_sample_gmail_response() -> dict:
|
||||
"""
|
||||
Create a sample Gmail API response for testing.
|
||||
|
||||
|
||||
This is useful for unit tests and development.
|
||||
"""
|
||||
|
||||
|
||||
return {
|
||||
"id": "msg_18c2a3b4d5e6f7g8",
|
||||
"threadId": "thread_12345",
|
||||
@@ -452,20 +478,20 @@ def create_sample_gmail_response() -> dict:
|
||||
"data": base64.urlsafe_b64encode(
|
||||
b"Hi team,\n\nI wanted to discuss our Q4 budget allocation.\n\nBest,\nJohn"
|
||||
).decode()
|
||||
}
|
||||
}
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def test_processor():
|
||||
"""Quick test function to verify processor works"""
|
||||
|
||||
|
||||
processor = GmailProcessor()
|
||||
sample_data = create_sample_gmail_response()
|
||||
|
||||
|
||||
try:
|
||||
result = processor.parse_email(sample_data, "test_user_123")
|
||||
|
||||
|
||||
print("✅ GmailProcessor Test Results:")
|
||||
print(f" Email ID: {result['email_id']}")
|
||||
print(f" Subject: {result['metadata']['subject']}")
|
||||
@@ -475,9 +501,9 @@ def test_processor():
|
||||
print(f" Has attachments: {result['metadata']['has_attachments']}")
|
||||
print(f"\n Document text preview:")
|
||||
print(f" {result['document_text'][:200]}...")
|
||||
|
||||
|
||||
return True
|
||||
|
||||
|
||||
except Exception as e:
|
||||
print(f"❌ Test failed: {e}")
|
||||
return False
|
||||
@@ -486,4 +512,3 @@ def test_processor():
|
||||
if __name__ == "__main__":
|
||||
# Run test when executed directly
|
||||
test_processor()
|
||||
|
||||
|
||||
Reference in New Issue
Block a user