"""Plain text and markdown file extractor. Handles .md and .txt files. Reads as UTF-8 by default, with fallback charset detection for non-UTF-8 encoded files. """ import logging from pathlib import Path from .base import ExtractionResult logger = logging.getLogger(__name__) SUPPORTED_EXTENSIONS = {".md", ".txt"} class PlainTextExtractor: """Extracts text from plain text and markdown files.""" def can_handle(self, file_path: Path) -> bool: """Return True for .md and .txt files.""" return file_path.suffix.lower() in SUPPORTED_EXTENSIONS def extract(self, file_path: Path) -> ExtractionResult: """Read file as UTF-8, falling back to charset detection.""" text = self._read_with_fallback(file_path) return ExtractionResult( text=text, source_file=file_path, extraction_method="direct", metadata={"encoding": "utf-8"}, ) def _read_with_fallback(self, file_path: Path) -> str: """Try UTF-8 first, then attempt charset detection.""" try: return file_path.read_text(encoding="utf-8") except UnicodeDecodeError: logger.warning( "UTF-8 decode failed for %s, attempting charset detection", file_path ) return self._read_with_detection(file_path) def _read_with_detection(self, file_path: Path) -> str: """Detect encoding and read file.""" raw = file_path.read_bytes() # Try common encodings before giving up for encoding in ("latin-1", "cp1252", "iso-8859-1"): try: return raw.decode(encoding) except (UnicodeDecodeError, LookupError): continue # Last resort: decode with replacement characters logger.warning("Could not detect encoding for %s, using replacement", file_path) return raw.decode("utf-8", errors="replace")