Add parser engine: base with subprocess isolation, HTML/PDF/email/cable/text parsers
This commit is contained in:
@@ -0,0 +1,102 @@
|
||||
"""Plain text / markdown parser for Mosaic."""
|
||||
import re
|
||||
import logging
|
||||
|
||||
from .base import ParseResult
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# ANSI escape pattern
|
||||
ANSI_ESCAPE = re.compile(r'\x1b\[[0-9;]*[a-zA-Z]|\x1b\].*?\x07')
|
||||
|
||||
|
||||
class TextDocParser:
|
||||
"""Parse plain text and markdown documents."""
|
||||
|
||||
def parse(self, file_path: str) -> ParseResult:
|
||||
"""Parse a plain text file."""
|
||||
try:
|
||||
# Try UTF-8 first, fall back to latin-1
|
||||
try:
|
||||
with open(file_path, 'r', encoding='utf-8') as f:
|
||||
text = f.read()
|
||||
except UnicodeDecodeError:
|
||||
with open(file_path, 'r', encoding='latin-1') as f:
|
||||
text = f.read()
|
||||
|
||||
# Strip ANSI escape sequences
|
||||
text = ANSI_ESCAPE.sub('', text)
|
||||
|
||||
# Detect if it's markdown
|
||||
is_markdown = self._detect_markdown(text, file_path)
|
||||
|
||||
# Extract basic metadata
|
||||
metadata = {
|
||||
'is_markdown': is_markdown,
|
||||
}
|
||||
|
||||
# Try to extract a title
|
||||
title = self._extract_title(text)
|
||||
|
||||
# Basic section detection
|
||||
sections = self._detect_sections(text)
|
||||
if sections:
|
||||
metadata['section_count'] = len(sections)
|
||||
|
||||
return ParseResult(
|
||||
success=True,
|
||||
text=text.strip(),
|
||||
metadata=metadata,
|
||||
doc_type='text',
|
||||
title=title,
|
||||
char_count=len(text),
|
||||
)
|
||||
|
||||
except Exception as e:
|
||||
return ParseResult(error=f"Text parse error: {str(e)}")
|
||||
|
||||
def _detect_markdown(self, text: str, file_path: str) -> bool:
|
||||
"""Detect if text is markdown."""
|
||||
if file_path.endswith(('.md', '.markdown', '.mkd')):
|
||||
return True
|
||||
# Check for markdown indicators
|
||||
md_patterns = [
|
||||
r'^#{1,6}\s', # Headers
|
||||
r'^\*\*[^*]+\*\*', # Bold
|
||||
r'^\- ', # List items
|
||||
r'^\d+\. ', # Numbered list
|
||||
r'\[.+\]\(.+\)', # Links
|
||||
]
|
||||
matches = sum(1 for p in md_patterns if re.search(p, text[:2000], re.MULTILINE))
|
||||
return matches >= 2
|
||||
|
||||
def _extract_title(self, text: str) -> str:
|
||||
"""Extract title from first heading or first line."""
|
||||
lines = text.strip().split('\n')
|
||||
for line in lines[:5]:
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
# Markdown heading
|
||||
match = re.match(r'^#{1,3}\s+(.+)', line)
|
||||
if match:
|
||||
return match.group(1).strip()
|
||||
# First non-empty line
|
||||
if len(line) < 200:
|
||||
return line
|
||||
break
|
||||
return ""
|
||||
|
||||
def _detect_sections(self, text: str) -> list[str]:
|
||||
"""Detect section headings in text."""
|
||||
sections = []
|
||||
for line in text.split('\n'):
|
||||
line = line.strip()
|
||||
# Markdown headings
|
||||
if re.match(r'^#{1,6}\s+', line):
|
||||
sections.append(line)
|
||||
# All-caps headings
|
||||
elif (line.isupper() and 3 < len(line) < 100 and
|
||||
not line.startswith(('---', '===', '***'))):
|
||||
sections.append(line)
|
||||
return sections
|
||||
Reference in New Issue
Block a user