feat: group private apps into andreknie-privat
This commit is contained in:
@@ -0,0 +1,57 @@
|
||||
"""DOCX text extraction using python-docx.
|
||||
|
||||
Preserves heading structure and paragraph breaks.
|
||||
"""
|
||||
|
||||
import logging
|
||||
from pathlib import Path
|
||||
|
||||
from .base import ExtractionResult
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class DocxExtractor:
|
||||
"""Extracts text from .docx files preserving structure."""
|
||||
|
||||
def can_handle(self, file_path: Path) -> bool:
|
||||
"""Return True for .docx files."""
|
||||
return file_path.suffix.lower() == ".docx"
|
||||
|
||||
def extract(self, file_path: Path) -> ExtractionResult:
|
||||
"""Extract text from DOCX preserving headings and paragraphs."""
|
||||
try:
|
||||
from docx import Document
|
||||
except ImportError:
|
||||
raise ImportError(
|
||||
"python-docx is required for DOCX extraction. "
|
||||
"Install with: pip install python-docx"
|
||||
)
|
||||
|
||||
doc = Document(str(file_path))
|
||||
parts: list[str] = []
|
||||
|
||||
for paragraph in doc.paragraphs:
|
||||
text = paragraph.text.strip()
|
||||
if not text:
|
||||
continue
|
||||
|
||||
# Preserve heading structure with markdown-style headers
|
||||
style_name = paragraph.style.name.lower() if paragraph.style else ""
|
||||
if "heading 1" in style_name:
|
||||
parts.append(f"# {text}")
|
||||
elif "heading 2" in style_name:
|
||||
parts.append(f"## {text}")
|
||||
elif "heading 3" in style_name:
|
||||
parts.append(f"### {text}")
|
||||
elif "heading 4" in style_name:
|
||||
parts.append(f"#### {text}")
|
||||
else:
|
||||
parts.append(text)
|
||||
|
||||
return ExtractionResult(
|
||||
text="\n\n".join(parts),
|
||||
source_file=file_path,
|
||||
extraction_method="direct",
|
||||
metadata={"paragraph_count": len(parts)},
|
||||
)
|
||||
Reference in New Issue
Block a user