Bahn: aisupport, Analyse-O2C-C2S, awesome-bahn-mcp-servers, beam-mcp,
Confluence_Bot, db-planet-mcp-server, O2C-Harness, project-audit,
Projekt-KIQ-HP, teamlandkarte-mcp
Dhive: Jury-Voting
Privat: CV, NoteGraph (NOTE: NoteGraph needs complete redo after consolidation)
Shared: AI-Orchestrator, OrgMyLife, power_skills_and_more
Shared/references: symphony (read-only)
Bahn repos remain available as independent remotes - this monorepo
pulls them in via subtree, the originals are untouched.
32 lines
954 B
Python
32 lines
954 B
Python
|
|
import os
|
|
import glob
|
|
from pypdf import PdfReader
|
|
from docx import Document
|
|
|
|
os.makedirs('extracted_text', exist_ok=True)
|
|
|
|
for filepath in glob.glob('source_Data/*.*'):
|
|
filename = os.path.basename(filepath)
|
|
output_path = f'extracted_text/{filename}.txt'
|
|
print(f'Extracting {filename}...')
|
|
|
|
try:
|
|
if filename.endswith('.pdf'):
|
|
reader = PdfReader(filepath)
|
|
text = ''
|
|
for page in reader.pages:
|
|
text += page.extract_text() + '\n'
|
|
with open(output_path, 'w', encoding='utf-8') as f:
|
|
f.write(text)
|
|
|
|
elif filename.endswith('.docx'):
|
|
doc = Document(filepath)
|
|
text = '\n'.join([p.text for p in doc.paragraphs])
|
|
with open(output_path, 'w', encoding='utf-8') as f:
|
|
f.write(text)
|
|
except Exception as e:
|
|
print(f"Error parsing {filename}: {e}")
|
|
|
|
print("Done extracting!")
|