diff --git a/parsers/__init__.py b/parsers/__init__.py index e69de29..2a10239 100644 --- a/parsers/__init__.py +++ b/parsers/__init__.py @@ -0,0 +1,6 @@ +from .base import ParserManager +from .html_parser import HTMLDocParser +from .pdf_parser import PDFDocParser +from .email_parser import EmailDocParser +from .cable_parser import CableDocParser +from .text_parser import TextDocParser diff --git a/parsers/base.py b/parsers/base.py new file mode 100644 index 0000000..5df479a --- /dev/null +++ b/parsers/base.py @@ -0,0 +1,161 @@ +"""Base parser with subprocess isolation for Mosaic document processing.""" +import os +import logging +import multiprocessing +from multiprocessing import Process, Queue +from typing import Optional +from dataclasses import dataclass, field + +logger = logging.getLogger(__name__) + +# Default limits +DEFAULT_MEMORY_LIMIT_MB = 512 +DEFAULT_TIMEOUT_SECONDS = 60 + + +@dataclass +class ParseResult: + """Result from a parser subprocess.""" + success: bool = False + text: str = "" + metadata: dict = field(default_factory=dict) + doc_type: str = "" + title: str = "" + char_count: int = 0 + error: Optional[str] = None + needs_ocr: bool = False + + +class ParserManager: + """Manages document parsing with subprocess isolation. + + All parsers run in subprocess workers with: + - Memory limits (resource.setrlimit) + - Timeouts + - No access to config, SQLite, or network + """ + + def __init__(self, memory_limit_mb: int = DEFAULT_MEMORY_LIMIT_MB, + timeout_seconds: int = DEFAULT_TIMEOUT_SECONDS, + min_chars_per_page: int = 50, + max_pdf_pages: int = 200): + self.memory_limit_mb = memory_limit_mb + self.timeout_seconds = timeout_seconds + self.min_chars_per_page = min_chars_per_page + self.max_pdf_pages = max_pdf_pages + self._parsers = {} + self._register_parsers() + + def _register_parsers(self): + """Register available parsers by doc_type.""" + from .html_parser import HTMLDocParser + from .pdf_parser import PDFDocParser + from .email_parser import EmailDocParser + from .cable_parser import CableDocParser + from .text_parser import TextDocParser + + self._parsers = { + 'html': HTMLDocParser(), + 'pdf': PDFDocParser(max_pages=self.max_pdf_pages, + min_chars_per_page=self.min_chars_per_page), + 'email': EmailDocParser(), + 'cable': CableDocParser(), + 'text': TextDocParser(), + } + + def parse(self, file_path: str, doc_type: str, + content_hash: Optional[str] = None) -> ParseResult: + """Parse a document in a subprocess worker. + + Args: + file_path: Path to the document file + doc_type: Document type (html, pdf, email, cable, text) + content_hash: Expected hash for integrity verification + + Returns: + ParseResult with extracted text and metadata + """ + if not os.path.exists(file_path): + return ParseResult(error=f"File not found: {file_path}") + + # Verify hash before parsing + if content_hash: + from utils.cache import DownloadCache + actual_hash = DownloadCache.compute_hash(file_path) + if actual_hash != content_hash: + return ParseResult( + error=f"Hash mismatch: expected {content_hash}, got {actual_hash}" + ) + + parser = self._parsers.get(doc_type) + if not parser: + # Try to detect type + parser = self._detect_parser(file_path) + if not parser: + return ParseResult(error=f"No parser for doc_type: {doc_type}") + + # Run in subprocess + result_queue = Queue() + proc = Process( + target=self._worker, + args=(parser, file_path, result_queue, self.memory_limit_mb) + ) + proc.start() + proc.join(timeout=self.timeout_seconds) + + if proc.is_alive(): + proc.terminate() + proc.join(timeout=5) + if proc.is_alive(): + proc.kill() + return ParseResult(error=f"Parser timed out after {self.timeout_seconds}s") + + if result_queue.empty(): + return ParseResult(error="Parser process returned no result") + + result = result_queue.get_nowait() + + # Check for OCR needed (low char count in PDFs) + if doc_type == 'pdf' and result.success: + if result.char_count > 0 and result.metadata.get('page_count', 1) > 0: + chars_per_page = result.char_count / result.metadata.get('page_count', 1) + if chars_per_page < self.min_chars_per_page: + result.needs_ocr = True + logger.info("Document may need OCR: %.1f chars/page", chars_per_page) + + return result + + @staticmethod + def _worker(parser, file_path: str, result_queue: Queue, memory_limit_mb: int): + """Subprocess worker that runs the parser with resource limits.""" + try: + # Set memory limit + try: + import resource + mem_bytes = memory_limit_mb * 1024 * 1024 + resource.setrlimit(resource.RLIMIT_AS, (mem_bytes, mem_bytes)) + except (ImportError, ValueError, OSError) as e: + # resource module may not be available on all platforms + pass + + result = parser.parse(file_path) + result_queue.put(result) + except MemoryError: + result_queue.put(ParseResult(error="Parser exceeded memory limit")) + except Exception as e: + result_queue.put(ParseResult(error=f"Parser error: {str(e)}")) + + def _detect_parser(self, file_path: str) -> Optional[object]: + """Try to detect the right parser from file content.""" + try: + from utils.sanitize import validate_content_type + content_type = validate_content_type(file_path) + if 'html' in content_type: + return self._parsers['html'] + elif 'pdf' in content_type: + return self._parsers['pdf'] + elif 'text' in content_type: + return self._parsers['text'] + except Exception: + pass + return self._parsers.get('text') diff --git a/parsers/cable_parser.py b/parsers/cable_parser.py new file mode 100644 index 0000000..e7f8904 --- /dev/null +++ b/parsers/cable_parser.py @@ -0,0 +1,228 @@ +"""Diplomatic cable parser for Mosaic — structured field extraction. + +Cables have a specific format with fields like SUBJECT, ORIGIN, +CLASSIFICATION, TAGS, and a multi-section body. +""" +import re +import logging +from typing import Optional + +from .base import ParseResult + +logger = logging.getLogger(__name__) + +# Cable field patterns +CABLE_FIELDS = { + 'VZCZCXYZ': 'transmission_id', + 'DE ': 'from_code', + 'ZNR': 'precedence', + 'ZNY': 'precedence', + 'FM ': 'from', + 'TO ': 'to', + 'INFO ': 'info', + 'BT': 'break', + 'UNCLAS': 'classification', + 'CONFIDENTIAL': 'classification', + 'SECRET': 'classification', + 'SIPDIS': 'distribution', + 'E.O.': 'executive_order', + 'TAGS:': 'tags', + 'SUBJECT:': 'subject', + 'REF:': 'reference', + 'REFS:': 'reference', +} + +# Classification levels +CLASSIFICATIONS = ['UNCLASSIFIED', 'CONFIDENTIAL', 'SECRET', 'SECRET//NOFORN', + 'CONFIDENTIAL//NOFORN', 'UNCLAS', 'UNCLAS SECTION'] + +# Section markers in cable bodies +SECTION_PATTERN = re.compile(r'^\d+\.\s+\((?:U|C|S|SBU)\)', re.MULTILINE) + +# ANSI escape pattern +ANSI_ESCAPE = re.compile(r'\x1b\[[0-9;]*[a-zA-Z]|\x1b\].*?\x07') + + +class CableDocParser: + """Parse diplomatic cables into structured fields.""" + + def parse(self, file_path: str) -> ParseResult: + """Parse a diplomatic cable. + + Extracts structured fields (SUBJECT, ORIGIN, CLASSIFICATION, TAGS, etc.) + as discrete fields BEFORE any chunking. + """ + try: + with open(file_path, 'r', encoding='utf-8', errors='replace') as f: + content = f.read() + + # Check if this is HTML (WikiLeaks format) or raw cable + if ' ParseResult: + """Parse a raw-format diplomatic cable.""" + metadata = {} + + lines = content.split('\n') + body_start = 0 + + # Extract header fields + for i, line in enumerate(lines): + line_stripped = line.strip() + if not line_stripped: + continue + + # Classification + for cls in CLASSIFICATIONS: + if line_stripped.startswith(cls): + metadata['CLASSIFICATION'] = cls + break + + # Named fields + if line_stripped.startswith('SUBJECT:'): + metadata['SUBJECT'] = line_stripped[8:].strip() + elif line_stripped.startswith('TAGS:'): + metadata['TAGS'] = line_stripped[5:].strip() + elif line_stripped.startswith('REF:') or line_stripped.startswith('REFS:'): + ref_start = 4 if line_stripped.startswith('REF:') else 5 + metadata['REFERENCE'] = line_stripped[ref_start:].strip() + elif line_stripped.startswith('E.O.'): + metadata['EXECUTIVE_ORDER'] = line_stripped[4:].strip() + elif line_stripped.startswith('FM '): + metadata['ORIGIN'] = line_stripped[3:].strip() + elif line_stripped.startswith('TO '): + metadata.setdefault('TO', []).append(line_stripped[3:].strip()) + + # Detect body start (after numbered paragraph or after header block) + if SECTION_PATTERN.match(line_stripped): + body_start = i + break + if line_stripped == 'BT' and i > 5: + body_start = i + 1 + break + + if body_start == 0: + # Fallback: body starts after a double blank line + for i in range(len(lines) - 1): + if not lines[i].strip() and not lines[i+1].strip(): + body_start = i + 2 + break + + body = '\n'.join(lines[body_start:]).strip() + + # Extract sections from body + sections = self._extract_sections(body) + if sections: + metadata['sections'] = sections + + # Build formatted text + text_parts = [] + for key in ('SUBJECT', 'ORIGIN', 'CLASSIFICATION', 'TAGS', 'REFERENCE'): + if key in metadata: + val = metadata[key] + if isinstance(val, list): + val = ', '.join(val) + text_parts.append(f"{key}: {val}") + + text_parts.append('') + text_parts.append(body) + + text = '\n'.join(text_parts) + text = ANSI_ESCAPE.sub('', text) + + return ParseResult( + success=True, + text=text, + metadata=metadata, + doc_type='cable', + title=metadata.get('SUBJECT', ''), + char_count=len(text), + ) + + def _parse_html_cable(self, content: str) -> ParseResult: + """Parse a WikiLeaks HTML-format cable page.""" + try: + from bs4 import BeautifulSoup + soup = BeautifulSoup(content, 'html.parser') + + metadata = {} + + # WikiLeaks cable pages have specific CSS classes/structures + # Try common patterns + title_el = soup.find('title') + if title_el: + metadata['SUBJECT'] = title_el.get_text(strip=True) + + # Look for cable metadata in page + for label in ('origin', 'classification', 'tags', 'subject', 'date'): + el = soup.find(class_=re.compile(label, re.I)) + if el: + metadata[label.upper()] = el.get_text(strip=True) + + # Extract main content + body_el = (soup.find(id='cable-text') or + soup.find(class_='cable-text') or + soup.find('pre') or + soup.find(class_='content')) + + if body_el: + body = body_el.get_text(separator='\n', strip=True) + else: + body = soup.get_text(separator='\n', strip=True) + + # Build text + text_parts = [] + for key in ('SUBJECT', 'ORIGIN', 'CLASSIFICATION', 'TAGS'): + if key in metadata: + text_parts.append(f"{key}: {metadata[key]}") + text_parts.append('') + text_parts.append(body) + + text = '\n'.join(text_parts) + text = ANSI_ESCAPE.sub('', text) + + return ParseResult( + success=True, + text=text, + metadata=metadata, + doc_type='cable', + title=metadata.get('SUBJECT', ''), + char_count=len(text), + ) + + except Exception as e: + return ParseResult(error=f"HTML cable parse error: {str(e)}") + + def _extract_sections(self, body: str) -> list[dict]: + """Extract numbered sections from cable body.""" + sections = [] + current_num = None + current_text = [] + + for line in body.split('\n'): + match = SECTION_PATTERN.match(line.strip()) + if match: + if current_num is not None: + sections.append({ + 'number': current_num, + 'text': '\n'.join(current_text).strip() + }) + # Extract section number and classification + current_num = line.strip().split('.')[0] + current_text = [line] + elif current_num is not None: + current_text.append(line) + + if current_num is not None: + sections.append({ + 'number': current_num, + 'text': '\n'.join(current_text).strip() + }) + + return sections diff --git a/parsers/email_parser.py b/parsers/email_parser.py new file mode 100644 index 0000000..830b978 --- /dev/null +++ b/parsers/email_parser.py @@ -0,0 +1,207 @@ +"""Email document parser for Mosaic — stdlib email parsing. + +SECURITY: Attachments are metadata-only. Never execute, open, or auto-extract +attachment contents. +""" +import email +import email.policy +import re +import hashlib +import logging +from typing import Optional + +from .base import ParseResult + +logger = logging.getLogger(__name__) + +# Suspicious MIME types that warrant flagging +SUSPICIOUS_MIME_TYPES = { + 'application/x-executable', + 'application/x-msdos-program', + 'application/x-msdownload', + 'application/vnd.ms-excel.sheet.macroEnabled.12', + 'application/vnd.ms-word.document.macroEnabled.12', + 'application/x-shellscript', + 'application/javascript', + 'application/x-bat', + 'application/x-com', + 'application/x-dosexec', +} + +# ANSI escape pattern +ANSI_ESCAPE = re.compile(r'\x1b\[[0-9;]*[a-zA-Z]|\x1b\].*?\x07') + + +class EmailDocParser: + """Parse email documents — headers + body, attachments as metadata only.""" + + def parse(self, file_path: str) -> ParseResult: + """Parse an email file (EML/MBOX format or HTML email page).""" + try: + with open(file_path, 'r', encoding='utf-8', errors='replace') as f: + content = f.read() + + # Try standard email parsing first + msg = email.message_from_string(content, policy=email.policy.default) + + # If it doesn't look like a real email, try HTML extraction + if not msg['From'] and not msg['Subject']: + return self._parse_html_email(content, file_path) + + return self._parse_email_message(msg) + + except Exception as e: + return ParseResult(error=f"Email parse error: {str(e)}") + + def _parse_email_message(self, msg) -> ParseResult: + """Parse a standard email.Message object.""" + # Extract headers + headers = {} + for key in ('From', 'To', 'Cc', 'Bcc', 'Subject', 'Date', + 'Message-ID', 'In-Reply-To', 'References'): + val = msg.get(key) + if val: + headers[key] = str(val) + + # Extract body + body_parts = [] + if msg.is_multipart(): + for part in msg.walk(): + content_type = part.get_content_type() + disposition = str(part.get('Content-Disposition', '')) + + if 'attachment' in disposition: + continue # Skip attachments (metadata extracted separately) + + if content_type == 'text/plain': + payload = part.get_payload(decode=True) + if payload: + body_parts.append(payload.decode('utf-8', errors='replace')) + elif content_type == 'text/html': + payload = part.get_payload(decode=True) + if payload: + # Simple HTML → text + text = self._html_to_text(payload.decode('utf-8', errors='replace')) + body_parts.append(text) + else: + payload = msg.get_payload(decode=True) + if payload: + body_parts.append(payload.decode('utf-8', errors='replace')) + + body = '\n\n'.join(body_parts) + + # Extract attachment metadata (NEVER the content) + attachments = self._extract_attachment_metadata(msg) + + # Build formatted output + text_parts = [] + for key, val in headers.items(): + text_parts.append(f"{key}: {val}") + text_parts.append('') # blank line + text_parts.append(body) + + text = '\n'.join(text_parts) + # Strip ANSI + text = ANSI_ESCAPE.sub('', text) + + metadata = { + 'headers': headers, + 'attachments': attachments, + 'attachment_count': len(attachments), + 'has_suspicious_attachments': any( + a.get('suspicious') for a in attachments + ), + } + + return ParseResult( + success=True, + text=text, + metadata=metadata, + doc_type='email', + title=headers.get('Subject', ''), + char_count=len(text), + ) + + def _extract_attachment_metadata(self, msg) -> list[dict]: + """Extract attachment metadata ONLY — never content.""" + attachments = [] + if not msg.is_multipart(): + return attachments + + for part in msg.walk(): + disposition = str(part.get('Content-Disposition', '')) + if 'attachment' not in disposition: + continue + + filename = part.get_filename() or 'unnamed' + content_type = part.get_content_type() + payload = part.get_payload(decode=True) + size = len(payload) if payload else 0 + md5 = hashlib.md5(payload).hexdigest() if payload else '' + + att = { + 'filename': filename, + 'content_type': content_type, + 'size': size, + 'md5': md5, + 'suspicious': content_type in SUSPICIOUS_MIME_TYPES, + } + + if att['suspicious']: + logger.warning("Suspicious attachment: %s (%s)", filename, content_type) + + attachments.append(att) + + return attachments + + def _parse_html_email(self, content: str, file_path: str) -> ParseResult: + """Parse an HTML page that contains email content (e.g., WikiLeaks email viewer).""" + try: + from bs4 import BeautifulSoup + soup = BeautifulSoup(content, 'html.parser') + + # Try to extract email fields from HTML structure + metadata = {} + text_parts = [] + + # Look for common WikiLeaks email display patterns + for field in ('from', 'to', 'cc', 'subject', 'date'): + el = soup.find(class_=re.compile(field, re.I)) + if el: + metadata[field] = el.get_text(strip=True) + text_parts.append(f"{field.title()}: {metadata[field]}") + + # Get body + body = soup.find(class_=re.compile('body|content|message', re.I)) + if body: + text_parts.append('') + text_parts.append(body.get_text(separator='\n', strip=True)) + else: + # Fallback: get all text + text_parts.append('') + text_parts.append(soup.get_text(separator='\n', strip=True)) + + text = '\n'.join(text_parts) + text = ANSI_ESCAPE.sub('', text) + + return ParseResult( + success=True, + text=text, + metadata={'headers': metadata, 'attachments': [], 'parsed_from_html': True}, + doc_type='email', + title=metadata.get('subject', ''), + char_count=len(text), + ) + except Exception as e: + return ParseResult(error=f"HTML email parse error: {str(e)}") + + @staticmethod + def _html_to_text(html: str) -> str: + """Simple HTML to text conversion.""" + try: + from bs4 import BeautifulSoup + soup = BeautifulSoup(html, 'html.parser') + return soup.get_text(separator='\n', strip=True) + except Exception: + # Fallback: strip tags with regex + return re.sub(r'<[^>]+>', '', html) diff --git a/parsers/html_parser.py b/parsers/html_parser.py new file mode 100644 index 0000000..cf182e0 --- /dev/null +++ b/parsers/html_parser.py @@ -0,0 +1,125 @@ +"""HTML document parser for Mosaic — BS4 with html.parser backend.""" +import re +import logging + +from bs4 import BeautifulSoup + +from .base import ParseResult + +logger = logging.getLogger(__name__) + +# Elements to remove (navigation, scripts, styles) +REMOVE_TAGS = ['script', 'style', 'nav', 'footer', 'header', 'aside', + 'noscript', 'iframe', 'svg'] + +# ANSI escape sequence pattern +ANSI_ESCAPE = re.compile(r'\x1b\[[0-9;]*[a-zA-Z]|\x1b\].*?\x07') + + +class HTMLDocParser: + """Parse HTML documents into structured text.""" + + def parse(self, file_path: str) -> ParseResult: + """Parse an HTML file. + + Uses BeautifulSoup with 'html.parser' backend (explicitly specified, + not auto-detected) to extract structured text. + """ + try: + with open(file_path, 'r', encoding='utf-8', errors='replace') as f: + raw_html = f.read() + + soup = BeautifulSoup(raw_html, 'html.parser') + + # Extract title + title = "" + title_tag = soup.find('title') + if title_tag: + title = title_tag.get_text(strip=True) + + # Extract metadata + metadata = self._extract_metadata(soup) + + # Remove unwanted elements + for tag_name in REMOVE_TAGS: + for tag in soup.find_all(tag_name): + tag.decompose() + + # Extract text preserving structure + text = self._extract_structured_text(soup) + + # Strip ANSI escape sequences + text = ANSI_ESCAPE.sub('', text) + + # Extract tables separately + tables = self._extract_tables(soup) + if tables: + metadata['tables'] = tables + + char_count = len(text) + + return ParseResult( + success=True, + text=text, + metadata=metadata, + doc_type='html', + title=title, + char_count=char_count, + ) + + except Exception as e: + return ParseResult(error=f"HTML parse error: {str(e)}") + + def _extract_metadata(self, soup: BeautifulSoup) -> dict: + """Extract metadata from HTML meta tags.""" + metadata = {} + for meta in soup.find_all('meta'): + name = meta.get('name', meta.get('property', '')) + content = meta.get('content', '') + if name and content: + metadata[name] = content + return metadata + + def _extract_structured_text(self, soup: BeautifulSoup) -> str: + """Extract text preserving heading structure.""" + parts = [] + + for element in soup.find_all(['h1', 'h2', 'h3', 'h4', 'h5', 'h6', 'p', + 'li', 'td', 'th', 'pre', 'blockquote', 'div']): + text = element.get_text(separator=' ', strip=True) + if not text: + continue + + tag = element.name + if tag.startswith('h'): + level = int(tag[1]) + prefix = '#' * level + parts.append(f"\n{prefix} {text}\n") + elif tag == 'li': + parts.append(f" - {text}") + elif tag == 'pre': + parts.append(f"\n```\n{text}\n```\n") + elif tag == 'blockquote': + parts.append(f"> {text}") + else: + parts.append(text) + + result = '\n'.join(parts) + # Collapse excessive whitespace + result = re.sub(r'\n{3,}', '\n\n', result) + return result.strip() + + def _extract_tables(self, soup: BeautifulSoup) -> list[dict]: + """Extract tables as structured data.""" + tables = [] + for table in soup.find_all('table'): + rows = [] + for tr in table.find_all('tr'): + cells = [] + for td in tr.find_all(['td', 'th']): + cells.append(td.get_text(strip=True)) + if cells: + rows.append(cells) + if rows: + tables.append({'rows': rows}) + return tables diff --git a/parsers/pdf_parser.py b/parsers/pdf_parser.py new file mode 100644 index 0000000..24c1890 --- /dev/null +++ b/parsers/pdf_parser.py @@ -0,0 +1,82 @@ +"""PDF document parser for Mosaic — PyMuPDF text extraction.""" +import logging +from typing import Optional + +from .base import ParseResult + +logger = logging.getLogger(__name__) + + +class PDFDocParser: + """Parse PDF documents using PyMuPDF (fitz) for text extraction.""" + + def __init__(self, max_pages: int = 200, min_chars_per_page: int = 50): + self.max_pages = max_pages + self.min_chars_per_page = min_chars_per_page + + def parse(self, file_path: str) -> ParseResult: + """Parse a PDF file. + + Extracts text page by page. Large docs (>max_pages) are split. + After extraction, checks chars/page ratio for OCR detection. + """ + try: + import fitz # PyMuPDF + except ImportError: + return ParseResult(error="PyMuPDF (fitz) not installed") + + try: + doc = fitz.open(file_path) + total_pages = len(doc) + metadata = { + 'page_count': total_pages, + 'format': doc.metadata.get('format', ''), + 'title': doc.metadata.get('title', ''), + 'author': doc.metadata.get('author', ''), + 'subject': doc.metadata.get('subject', ''), + 'creator': doc.metadata.get('creator', ''), + } + + title = metadata['title'] or '' + + # Limit pages + pages_to_process = min(total_pages, self.max_pages) + if total_pages > self.max_pages: + logger.warning("PDF has %d pages, processing first %d", + total_pages, self.max_pages) + metadata['truncated'] = True + metadata['truncated_at_page'] = self.max_pages + + # Extract text page by page with form feed separators + text_parts = [] + for page_num in range(pages_to_process): + page = doc[page_num] + page_text = page.get_text("text") + if page_text.strip(): + text_parts.append(page_text.strip()) + + doc.close() + + text = '\f'.join(text_parts) # Form feed between pages + char_count = len(text) + + # Check for OCR need + needs_ocr = False + if pages_to_process > 0: + chars_per_page = char_count / pages_to_process + if chars_per_page < self.min_chars_per_page: + needs_ocr = True + metadata['chars_per_page'] = chars_per_page + + return ParseResult( + success=True, + text=text, + metadata=metadata, + doc_type='pdf', + title=title, + char_count=char_count, + needs_ocr=needs_ocr, + ) + + except Exception as e: + return ParseResult(error=f"PDF parse error: {str(e)}") diff --git a/parsers/text_parser.py b/parsers/text_parser.py new file mode 100644 index 0000000..1a0300f --- /dev/null +++ b/parsers/text_parser.py @@ -0,0 +1,102 @@ +"""Plain text / markdown parser for Mosaic.""" +import re +import logging + +from .base import ParseResult + +logger = logging.getLogger(__name__) + +# ANSI escape pattern +ANSI_ESCAPE = re.compile(r'\x1b\[[0-9;]*[a-zA-Z]|\x1b\].*?\x07') + + +class TextDocParser: + """Parse plain text and markdown documents.""" + + def parse(self, file_path: str) -> ParseResult: + """Parse a plain text file.""" + try: + # Try UTF-8 first, fall back to latin-1 + try: + with open(file_path, 'r', encoding='utf-8') as f: + text = f.read() + except UnicodeDecodeError: + with open(file_path, 'r', encoding='latin-1') as f: + text = f.read() + + # Strip ANSI escape sequences + text = ANSI_ESCAPE.sub('', text) + + # Detect if it's markdown + is_markdown = self._detect_markdown(text, file_path) + + # Extract basic metadata + metadata = { + 'is_markdown': is_markdown, + } + + # Try to extract a title + title = self._extract_title(text) + + # Basic section detection + sections = self._detect_sections(text) + if sections: + metadata['section_count'] = len(sections) + + return ParseResult( + success=True, + text=text.strip(), + metadata=metadata, + doc_type='text', + title=title, + char_count=len(text), + ) + + except Exception as e: + return ParseResult(error=f"Text parse error: {str(e)}") + + def _detect_markdown(self, text: str, file_path: str) -> bool: + """Detect if text is markdown.""" + if file_path.endswith(('.md', '.markdown', '.mkd')): + return True + # Check for markdown indicators + md_patterns = [ + r'^#{1,6}\s', # Headers + r'^\*\*[^*]+\*\*', # Bold + r'^\- ', # List items + r'^\d+\. ', # Numbered list + r'\[.+\]\(.+\)', # Links + ] + matches = sum(1 for p in md_patterns if re.search(p, text[:2000], re.MULTILINE)) + return matches >= 2 + + def _extract_title(self, text: str) -> str: + """Extract title from first heading or first line.""" + lines = text.strip().split('\n') + for line in lines[:5]: + line = line.strip() + if not line: + continue + # Markdown heading + match = re.match(r'^#{1,3}\s+(.+)', line) + if match: + return match.group(1).strip() + # First non-empty line + if len(line) < 200: + return line + break + return "" + + def _detect_sections(self, text: str) -> list[str]: + """Detect section headings in text.""" + sections = [] + for line in text.split('\n'): + line = line.strip() + # Markdown headings + if re.match(r'^#{1,6}\s+', line): + sections.append(line) + # All-caps headings + elif (line.isupper() and 3 < len(line) < 100 and + not line.startswith(('---', '===', '***'))): + sections.append(line) + return sections