diff --git a/backend/open_webui/utils/gmail_indexer_v2.py b/backend/open_webui/utils/gmail_indexer_v2.py index bb9582359a..a19e3e4e67 100644 --- a/backend/open_webui/utils/gmail_indexer_v2.py +++ b/backend/open_webui/utils/gmail_indexer_v2.py @@ -595,6 +595,33 @@ class GmailIndexerV2: text = text.replace("\\\\", "") text = text.replace("\\/", "/") + # STEP 1.5: Strip HTML completely (defense in depth) + # Remove DOCTYPE declarations () + text = re.sub(r"]*>", "", text, flags=re.IGNORECASE) + # Remove XML declarations () + text = re.sub(r"<\?[^>]*\?>", "", text, flags=re.IGNORECASE) + # Remove CDATA sections + text = re.sub(r"", "", text, flags=re.DOTALL) + # Remove HTML comments () + text = re.sub(r"", "", text, flags=re.DOTALL) + # Remove script and style tags with content + text = re.sub( + r"]*>.*?", "", text, flags=re.DOTALL | re.IGNORECASE + ) + text = re.sub( + r"]*>.*?", "", text, flags=re.DOTALL | re.IGNORECASE + ) + # Remove inline styles (style="...") before removing tags + text = re.sub( + r'\s*style\s*=\s*["\'][^"\']*["\']', "", text, flags=re.IGNORECASE + ) + # Remove all HTML tags + text = re.sub(r"<[^>]+>", "", text) + # Remove MSO (Microsoft Office) conditional comments + text = re.sub( + r"\[if[^\]]*\].*?\[endif\]", "", text, flags=re.DOTALL | re.IGNORECASE + ) + # STEP 2: Remove quoted reply blocks text = re.sub(r"On .{1,100}wrote:.*", "", text, flags=re.DOTALL | re.IGNORECASE) text = re.sub(r"^>.*$", "", text, flags=re.MULTILINE) diff --git a/backend/open_webui/utils/gmail_processor.py b/backend/open_webui/utils/gmail_processor.py index 997e51f1f0..f4a886e29d 100644 --- a/backend/open_webui/utils/gmail_processor.py +++ b/backend/open_webui/utils/gmail_processor.py @@ -26,7 +26,7 @@ logger.setLevel(logging.INFO) class GmailProcessor: """ Processes Gmail API responses into structured email data. - + This is a pure data transformation class - no external API calls, no database operations, just parsing and cleaning. """ @@ -38,70 +38,75 @@ class GmailProcessor: def parse_email(self, email_data: dict, user_id: str) -> Dict: """ Parse a Gmail API message response into structured data. - + Args: email_data: Full Gmail API message response (format='full') user_id: User ID for metadata isolation - + Returns: Dict with parsed email data and metadata - + Example: >>> processor = GmailProcessor() >>> result = processor.parse_email(gmail_api_response, "user_123") >>> print(result["subject"]) "Q4 Budget Discussion" """ - + try: # Extract basic identifiers email_id = email_data.get("id", "") thread_id = email_data.get("threadId", "") - + if not email_id: raise ValueError("Email data missing 'id' field") - + # Extract headers headers_dict = self._extract_headers(email_data) - + # Extract core email fields subject = headers_dict.get("Subject", "(No Subject)") from_addr = headers_dict.get("From", "") to_addr = headers_dict.get("To", "") cc_addr = headers_dict.get("Cc", "") date_str = headers_dict.get("Date", "") - + # Parse date date_timestamp, date_readable = self._parse_date( - date_str, - email_data.get("internalDate") + date_str, email_data.get("internalDate") ) - + # Extract email body payload = email_data.get("payload", {}) body_raw = self._extract_body(payload) snippet = email_data.get("snippet", "") - + # Clean email body with advanced cleaning - body_full_clean, body_original_only = EmailCleaner.clean_email_body(body_raw) - + body_full_clean, body_original_only = EmailCleaner.clean_email_body( + body_raw + ) + # Use original message (no quotes/signatures) for primary content body_clean = body_original_only if body_original_only else body_full_clean - + # Extract labels labels = email_data.get("labelIds", []) - + # Extract attachment information attachments = self._get_attachments(payload) has_attachments = len(attachments) > 0 - + # Parse email addresses from_name, from_email = EmailCleaner.parse_email_address(from_addr) to_name, to_email = EmailCleaner.parse_email_address(to_addr) - + # Detect if this is a reply - is_reply = "RE:" in subject.upper() or "Re:" in subject or any(label == "SENT" for label in labels) - + is_reply = ( + "RE:" in subject.upper() + or "Re:" in subject + or any(label == "SENT" for label in labels) + ) + # Create document text (what will be embedded) - cleaner format document_text = self._create_document_text( subject=subject, @@ -110,9 +115,9 @@ class GmailProcessor: to_email=to_email, date_readable=date_readable, body=body_clean, - snippet=snippet + snippet=snippet, ) - + # Build metadata (improved structure for better search) metadata = { # Core identification @@ -120,7 +125,6 @@ class GmailProcessor: "email_id": email_id, "thread_id": thread_id, "user_id": user_id, - # Email fields (structured) "subject": subject, "from_name": from_name, @@ -133,7 +137,6 @@ class GmailProcessor: "date": date_readable, "date_timestamp": date_timestamp, "labels": labels, # Keep as array, not comma-separated string - # Content metadata "has_attachments": has_attachments, "attachment_count": len(attachments), @@ -141,21 +144,16 @@ class GmailProcessor: "word_count": len(body_clean.split()), "body_length": len(body_clean), "original_body_length": len(body_raw), - # Cleaned text fields "body_full_clean": body_full_clean, # Full email with cleaned quotes "body_original": body_clean, # Just the original message - # Categorization "source": "gmail", "doc_type": "email", - # Hash for deduplication - "hash": hashlib.sha256( - f"{email_id}{user_id}".encode() - ).hexdigest(), + "hash": hashlib.sha256(f"{email_id}{user_id}".encode()).hexdigest(), } - + return { "email_id": email_id, "thread_id": thread_id, @@ -166,42 +164,40 @@ class GmailProcessor: "snippet": snippet, "attachments": attachments, # List of attachment metadata } - + except Exception as e: logger.error(f"Error parsing email {email_data.get('id', 'unknown')}: {e}") raise - + def _extract_headers(self, email_data: dict) -> Dict[str, str]: """Extract email headers from Gmail API response""" - + payload = email_data.get("payload", {}) headers_list = payload.get("headers", []) - + # Convert list of {name, value} to dict headers_dict = {} for header in headers_list: name = header.get("name", "") value = header.get("value", "") headers_dict[name] = value - + return headers_dict - + def _parse_date( - self, - date_str: str, - internal_date: Optional[str] = None + self, date_str: str, internal_date: Optional[str] = None ) -> Tuple[int, str]: """ Parse email date into timestamp and readable format. - + Args: date_str: Date header from email (e.g., "Mon, 15 Nov 2020 14:30:00 +0000") internal_date: Gmail internal date (milliseconds since epoch) - + Returns: Tuple of (unix_timestamp, readable_date_string) """ - + # Try to parse the Date header try: date_obj = parsedate_to_datetime(date_str) @@ -210,7 +206,7 @@ class GmailProcessor: return timestamp, readable except Exception as e: logger.debug(f"Could not parse date '{date_str}': {e}") - + # Fallback to internal date if internal_date: try: @@ -220,164 +216,194 @@ class GmailProcessor: return timestamp, readable except Exception as e: logger.debug(f"Could not parse internal date '{internal_date}': {e}") - + # Final fallback: current time now = int(datetime.now().timestamp()) return now, "Unknown Date" - + def _extract_body(self, payload: dict) -> str: """ Recursively extract email body from Gmail API payload. - + Handles: - Plain text emails - HTML emails (strips tags) - Multipart emails (multipart/alternative, multipart/mixed) - Nested MIME structures - + Args: payload: Gmail API message payload - + Returns: Extracted email body as plain text """ - + # Handle multipart emails (most common) if "parts" in payload: return self._extract_from_parts(payload["parts"]) - + # Handle single-part emails elif "body" in payload and "data" in payload["body"]: mime_type = payload.get("mimeType", "") data = payload["body"]["data"] return self._decode_body_data(data, mime_type) - + return "" - + def _extract_from_parts(self, parts: list) -> str: """ Extract body from multipart email structure. - + Priority: 1. text/plain (preferred) 2. text/html (strip tags) 3. Nested parts (recurse) """ - + # First pass: look for text/plain for part in parts: mime_type = part.get("mimeType", "") - + if mime_type == "text/plain": data = part.get("body", {}).get("data", "") if data: body = self._decode_body_data(data, mime_type) if body: return body - + # Second pass: look for text/html for part in parts: mime_type = part.get("mimeType", "") - + if mime_type == "text/html": data = part.get("body", {}).get("data", "") if data: body = self._decode_body_data(data, mime_type) if body: return body - + # Third pass: recurse into nested parts for part in parts: if "parts" in part: body = self._extract_from_parts(part["parts"]) if body: return body - + return "" - + def _decode_body_data(self, data: str, mime_type: str) -> str: """ Decode base64-encoded email body data. - + Args: data: Base64-encoded body data mime_type: MIME type (text/plain or text/html) - + Returns: Decoded and cleaned text """ - + if not data: return "" - + try: # Gmail API uses URL-safe base64 encoding decoded_bytes = base64.urlsafe_b64decode(data) decoded_text = decoded_bytes.decode("utf-8", errors="ignore") - + # Clean based on MIME type if mime_type == "text/html": decoded_text = self._strip_html_tags(decoded_text) - + return decoded_text.strip() - + except Exception as e: logger.error(f"Error decoding body data: {e}") return "" - + def _strip_html_tags(self, html_text: str) -> str: """ Strip HTML tags and extract plain text. - - Basic implementation - good enough for email bodies. + + Comprehensive implementation handles: + - DOCTYPE and XML declarations + - Script and style blocks + - HTML comments and CDATA + - MSO (Microsoft Office) conditional comments + - Inline styles + - All HTML tags """ - + if not html_text: return "" - - # Decode HTML entities first - text = unescape(html_text) - + + text = html_text + + # Remove DOCTYPE declarations () + text = re.sub(r"]*>", "", text, flags=re.IGNORECASE) + # Remove XML declarations () + text = re.sub(r"<\?[^>]*\?>", "", text, flags=re.IGNORECASE) + # Remove CDATA sections + text = re.sub(r"", "", text, flags=re.DOTALL) + # Remove HTML comments () + text = re.sub(r"", "", text, flags=re.DOTALL) + # Remove MSO (Microsoft Office) conditional comments + text = re.sub( + r"\[if[^\]]*\].*?\[endif\]", "", text, flags=re.DOTALL | re.IGNORECASE + ) + # Remove script and style tags with their content - text = re.sub(r']*>.*?', '', text, flags=re.DOTALL | re.IGNORECASE) - text = re.sub(r']*>.*?', '', text, flags=re.DOTALL | re.IGNORECASE) - + text = re.sub( + r"]*>.*?", "", text, flags=re.DOTALL | re.IGNORECASE + ) + text = re.sub( + r"]*>.*?", "", text, flags=re.DOTALL | re.IGNORECASE + ) + + # Remove inline styles (before removing tags to prevent garbage) + text = re.sub( + r'\s*style\s*=\s*["\'][^"\']*["\']', "", text, flags=re.IGNORECASE + ) + # Replace common block elements with newlines - text = re.sub(r'', '\n', text, flags=re.IGNORECASE) - text = re.sub(r'', '\n', text, flags=re.IGNORECASE) - + text = re.sub(r"", "\n", text, flags=re.IGNORECASE) + text = re.sub(r"", "\n", text, flags=re.IGNORECASE) + # Remove all remaining HTML tags - text = re.sub(r'<[^>]+>', '', text) - + text = re.sub(r"<[^>]+>", "", text) + + # Decode HTML entities + text = unescape(text) + # Clean up whitespace - text = re.sub(r'\n\s*\n', '\n\n', text) - text = re.sub(r' +', ' ', text) - + text = re.sub(r"\n\s*\n", "\n\n", text) + text = re.sub(r" +", " ", text) + return text.strip() - + def _has_attachments(self, payload: dict) -> bool: """Check if email has attachments (returns boolean only).""" attachments = self._get_attachments(payload) return len(attachments) > 0 - + def _get_attachments(self, payload: dict, attachments: list = None) -> list: """ Extract attachment metadata from email payload. - + Args: payload: Gmail API message payload attachments: List to accumulate attachments (for recursion) - + Returns: List of attachment dicts with: filename, mimeType, size, attachmentId """ if attachments is None: attachments = [] - + if "parts" in payload: for part in payload["parts"]: filename = part.get("filename", "") - + # Part has a filename and body.attachmentId - it's an attachment if filename and part.get("body", {}).get("attachmentId"): attachment_info = { @@ -387,13 +413,13 @@ class GmailProcessor: "attachmentId": part.get("body", {}).get("attachmentId"), } attachments.append(attachment_info) - + # Recursively check nested parts if "parts" in part: self._get_attachments(part, attachments) - + return attachments - + def _create_document_text( self, subject: str, @@ -402,23 +428,23 @@ class GmailProcessor: to_email: str, date_readable: str, body: str, - snippet: str + snippet: str, ) -> str: """ Create clean document text for embedding. - + Simple, clean format without headers (better for semantic search). """ - + # Use cleaned body if available, otherwise snippet content = body if body else snippet - + # Build clean document - just subject and content (no "Email Subject:" labels) if subject and subject != "(No Subject)": document_text = f"{subject}\n\n{content}" else: document_text = content - + return document_text.strip() @@ -430,10 +456,10 @@ class GmailProcessor: def create_sample_gmail_response() -> dict: """ Create a sample Gmail API response for testing. - + This is useful for unit tests and development. """ - + return { "id": "msg_18c2a3b4d5e6f7g8", "threadId": "thread_12345", @@ -452,20 +478,20 @@ def create_sample_gmail_response() -> dict: "data": base64.urlsafe_b64encode( b"Hi team,\n\nI wanted to discuss our Q4 budget allocation.\n\nBest,\nJohn" ).decode() - } - } + }, + }, } def test_processor(): """Quick test function to verify processor works""" - + processor = GmailProcessor() sample_data = create_sample_gmail_response() - + try: result = processor.parse_email(sample_data, "test_user_123") - + print("✅ GmailProcessor Test Results:") print(f" Email ID: {result['email_id']}") print(f" Subject: {result['metadata']['subject']}") @@ -475,9 +501,9 @@ def test_processor(): print(f" Has attachments: {result['metadata']['has_attachments']}") print(f"\n Document text preview:") print(f" {result['document_text'][:200]}...") - + return True - + except Exception as e: print(f"❌ Test failed: {e}") return False @@ -486,4 +512,3 @@ def test_processor(): if __name__ == "__main__": # Run test when executed directly test_processor() -