/** * LinkedIn Data Export → Homepage Content Extractor * * Reads the LinkedIn data export and produces content files for andreknie.de: * - Articles (HTML) → content/kniepunkt/ (Markdown) + content/posts/ (Markdown) * - Shares (CSV) → content/posts/ (Markdown) for substantial posts * * Usage: node scripts/extract-linkedin.js */ import { readFileSync, readdirSync, writeFileSync, mkdirSync, existsSync } from 'fs' import { join, basename } from 'path' // --- Configuration --- const EXPORT_DIR = join(process.cwd(), 'sources', 'Complete_LinkedInDataExport_05-31-2026') const ARTICLES_DIR = join(EXPORT_DIR, 'Articles', 'Articles') const SHARES_FILE = join(EXPORT_DIR, 'Shares_547037907.csv') const OUTPUT_KNIEPUNKT = join(process.cwd(), 'andreknie.de', 'content', 'kniepunkt') const OUTPUT_POSTS = join(process.cwd(), 'andreknie.de', 'content', 'posts') // Minimum character length for a share to be considered "substantial" const MIN_POST_LENGTH = 150 // --- Helpers --- function ensureDir(dir) { if (!existsSync(dir)) mkdirSync(dir, { recursive: true }) } function slugify(text) { return text .toLowerCase() .replace(/ä/g, 'ae').replace(/ö/g, 'oe').replace(/ü/g, 'ue').replace(/ß/g, 'ss') .replace(/[^a-z0-9]+/g, '-') .replace(/^-+|-+$/g, '') .substring(0, 80) } /** * Strip HTML tags and decode entities, preserving paragraph breaks. */ function htmlToMarkdown(html) { let md = html // Remove style/head tags entirely .replace(/]*>[\s\S]*?<\/style>/gi, '') .replace(/]*>[\s\S]*?<\/head>/gi, '') // Convert headers .replace(/]*>([\s\S]*?)<\/h1>/gi, '\n# $1\n') .replace(/]*>([\s\S]*?)<\/h2>/gi, '\n## $1\n') .replace(/]*>([\s\S]*?)<\/h3>/gi, '\n### $1\n') // Convert links .replace(/]+href="([^"]*)"[^>]*>([\s\S]*?)<\/a>/gi, '[$2]($1)') // Convert lists .replace(/]*>

([\s\S]*?)<\/p><\/li>/gi, '- $1\n') .replace(/]*>([\s\S]*?)<\/li>/gi, '- $1\n') .replace(/<\/?[uo]l[^>]*>/gi, '\n') // Convert paragraphs and divs to double newlines .replace(/<\/p>/gi, '\n\n') .replace(/]*>/gi, '') .replace(/<\/div>/gi, '\n\n') .replace(/]*>/gi, '') // Convert bold/italic .replace(/]*>([\s\S]*?)<\/strong>/gi, '**$1**') .replace(/]*>([\s\S]*?)<\/em>/gi, '*$1*') // Convert blockquotes .replace(/]*>([\s\S]*?)<\/blockquote>/gi, '\n> $1\n') // Convert br .replace(//gi, '\n') // Remove remaining HTML tags .replace(/<[^>]+>/g, '') // Decode HTML entities .replace(/&/g, '&') .replace(/</g, '<') .replace(/>/g, '>') .replace(/"/g, '"') .replace(/'/g, "'") .replace(/ /g, ' ') // Clean up whitespace .replace(/\n{3,}/g, '\n\n') .trim() return md } /** * Extract metadata from an article HTML file. */ function parseArticle(filepath) { const html = readFileSync(filepath, 'utf-8') // Extract title from or <h1> const titleMatch = html.match(/<title>([\s\S]*?)<\/title>/i) || html.match(/<h1[^>]*><a[^>]*>([\s\S]*?)<\/a><\/h1>/i) || html.match(/<h1[^>]*>([\s\S]*?)<\/h1>/i) const title = titleMatch ? titleMatch[1].replace(/<[^>]+>/g, '').trim() : basename(filepath, '.html') // Extract dates const createdMatch = html.match(/class="created"[^>]*>Created on ([\d-]+\s[\d:]+)/i) const publishedMatch = html.match(/class="published"[^>]*>Published on ([\d-]+\s[\d:]+)/i) const date = publishedMatch ? publishedMatch[1].split(' ')[0] : (createdMatch ? createdMatch[1].split(' ')[0] : null) // Extract body content (everything inside <body> after metadata) const bodyMatch = html.match(/<body[^>]*>([\s\S]*)<\/body>/i) const bodyHtml = bodyMatch ? bodyMatch[1] : '' // Remove the series logo/title/description and metadata lines from body const cleanBody = bodyHtml .replace(/<img[^>]*class="series-logo"[^>]*>/gi, '') .replace(/<h3>[\s\S]*?<\/h3>/i, '') // series title block .replace(/<img[^>]*alt=""[^>]*>/gi, '') // cover images with empty alt .replace(/<p class="created">[^<]*<\/p>/gi, '') .replace(/<p class="published">[^<]*<\/p>/gi, '') .replace(/<h1[^>]*>[\s\S]*?<\/h1>/gi, '') // title already in frontmatter const body = htmlToMarkdown(cleanBody) // Extract LinkedIn URL from h1 link const urlMatch = html.match(/<h1[^>]*><a href="([^"]+)"/i) const linkedinUrl = urlMatch ? urlMatch[1] : null return { title, date, body, linkedinUrl, filename: basename(filepath) } } /** * Detect if an article is a Kniepunkt issue. */ function detectKniepunktNumber(title, filename) { // From title: "KNIEPUNKT 039: ..." or "Kniepunkt #12 ..." const titleMatch = title.match(/kniepunkt\s*#?0*(\d+)/i) if (titleMatch) return parseInt(titleMatch[1], 10) // From filename: "kniepunkt-039-..." const fileMatch = filename.match(/kniepunkt-0*(\d+)/i) if (fileMatch) return parseInt(fileMatch[1], 10) return null } /** * Auto-generate tags from content using keyword matching. */ function autoTag(title, body) { const text = (title + ' ' + body).toLowerCase() const tagMap = { 'ki': ['künstliche intelligenz', 'ki-', 'ki ', 'artificial intelligence', ' ai ', 'machine learning', 'maschinelles lernen'], 'sovereignty': ['souverän', 'sovereign', 'gaia-x', 'eu-hosting', 'dsgvo'], 'ai-act': ['ai act', 'ai-act', 'regulierung'], 'mittelstand': ['mittelstand', 'kmu', 'kleine und mittlere'], 'leadership': ['führung', 'leadership', 'shared leadership', 'jobsharing'], 'change': ['change', 'transformation', 'wandel'], 'energy': ['energie', 'energieeffizienz', 'druckluft'], 'ethics': ['ethik', 'verantwortung', 'bias', 'fairness'], 'education': ['bildung', 'universität', 'hochschule', 'studier'], 'automation': ['automatisierung', 'automation', 'roboter'], 'data': ['daten', 'data', 'datenschutz'], } const tags = [] for (const [tag, keywords] of Object.entries(tagMap)) { if (keywords.some(kw => text.includes(kw))) { tags.push(tag) } } // Always add 'ki' for Kniepunkt articles if (title.toLowerCase().includes('kniepunkt') && !tags.includes('ki')) { tags.push('ki') } return tags.length > 0 ? tags : ['ki'] } /** * Parse the Shares CSV (handles multi-line fields with quotes). */ function parseSharesCsv(filepath) { const raw = readFileSync(filepath, 'utf-8') const lines = raw.split('\n') const shares = [] let currentRow = '' let inQuotes = false // Skip header for (let i = 1; i < lines.length; i++) { currentRow += (currentRow ? '\n' : '') + lines[i] // Count unescaped quotes to track if we're inside a quoted field const quoteCount = (currentRow.match(/"/g) || []).length inQuotes = quoteCount % 2 !== 0 if (!inQuotes) { // Parse complete row const parsed = parseCsvRow(currentRow) if (parsed) shares.push(parsed) currentRow = '' } } return shares } function parseCsvRow(row) { // CSV format: Date,ShareLink,ShareCommentary,SharedUrl,MediaUrl,Visibility const fields = [] let current = '' let inQuotes = false for (let i = 0; i < row.length; i++) { const char = row[i] if (char === '"') { if (inQuotes && row[i + 1] === '"') { current += '"' i++ // skip escaped quote } else { inQuotes = !inQuotes } } else if (char === ',' && !inQuotes) { fields.push(current) current = '' } else { current += char } } fields.push(current) if (fields.length < 4) return null return { date: fields[0]?.trim(), shareLink: fields[1]?.trim(), commentary: fields[2]?.trim().replace(/""/g, '"'), sharedUrl: fields[3]?.trim(), mediaUrl: fields[4]?.trim(), visibility: fields[5]?.trim(), } } // --- Main Extraction --- function extractArticles() { console.log('\n📄 Extracting Articles...') ensureDir(OUTPUT_KNIEPUNKT) ensureDir(OUTPUT_POSTS) const files = readdirSync(ARTICLES_DIR).filter(f => f.endsWith('.html')) let kniepunktCount = 0 let postCount = 0 for (const file of files) { const filepath = join(ARTICLES_DIR, file) const article = parseArticle(filepath) if (!article.date) { console.warn(` ⚠️ No date found: ${file}`) continue } const kniepunktNum = detectKniepunktNumber(article.title, file) const tags = autoTag(article.title, article.body) if (kniepunktNum) { // Kniepunkt issue const paddedNum = String(kniepunktNum).padStart(3, '0') const slug = `${paddedNum}-${slugify(article.title.replace(/kniepunkt\s*#?\d+:?\s*/i, ''))}` const frontmatter = [ '---', `title: "${article.title.replace(/"/g, '\\"')}"`, `issue: ${kniepunktNum}`, `date: ${article.date}`, `tags: [${tags.join(', ')}]`, `summary: ""`, `visibility: primary`, `source: linkedin-article`, article.linkedinUrl ? `linkedin_url: "${article.linkedinUrl}"` : null, '---', ].filter(Boolean).join('\n') const output = `${frontmatter}\n\n${article.body}\n` const outputPath = join(OUTPUT_KNIEPUNKT, `${slug}.md`) writeFileSync(outputPath, output, 'utf-8') kniepunktCount++ } else { // Regular article → post const slug = `${article.date}-${slugify(article.title)}` const frontmatter = [ '---', `title: "${article.title.replace(/"/g, '\\"')}"`, `date: ${article.date}`, `tags: [${tags.join(', ')}]`, `summary: ""`, `source: linkedin-article`, `visibility: primary`, article.linkedinUrl ? `linkedin_url: "${article.linkedinUrl}"` : null, '---', ].filter(Boolean).join('\n') const output = `${frontmatter}\n\n${article.body}\n` const outputPath = join(OUTPUT_POSTS, `${slug}.md`) writeFileSync(outputPath, output, 'utf-8') postCount++ } } console.log(` ✅ ${kniepunktCount} Kniepunkt issues → content/kniepunkt/`) console.log(` ✅ ${postCount} articles → content/posts/`) } function extractShares() { console.log('\n📢 Extracting Shares (Posts)...') ensureDir(OUTPUT_POSTS) ensureDir(OUTPUT_KNIEPUNKT) const shares = parseSharesCsv(SHARES_FILE) console.log(` 📊 Total shares in export: ${shares.length}`) let postCount = 0 let kniepunktCount = 0 let skipped = 0 for (const share of shares) { if (!share.commentary || share.commentary.length < MIN_POST_LENGTH) { skipped++ continue } // Skip reposts of articles (they'll be extracted from Articles/) if (share.sharedUrl && share.sharedUrl.includes('linkedin.com/pulse/')) { skipped++ continue } const date = share.date ? share.date.split(' ')[0] : null if (!date) { skipped++; continue } const commentary = share.commentary const tags = autoTag('', commentary) // Check if this is a Kniepunkt post (issues 1-11 that were regular posts) const kniepunktMatch = commentary.match(/kniepunkt\s*#?0*(\d+)/i) const kniepunktNum = kniepunktMatch ? parseInt(kniepunktMatch[1], 10) : null if (kniepunktNum && kniepunktNum <= 11) { // Kniepunkt as post (issues 1-11) const paddedNum = String(kniepunktNum).padStart(3, '0') // Try to extract a title from the first line or generate one const firstLine = commentary.split('\n')[0].substring(0, 100) const title = `Kniepunkt #${kniepunktNum}` const slug = `${paddedNum}-post-${slugify(firstLine)}` const frontmatter = [ '---', `title: "${title}"`, `issue: ${kniepunktNum}`, `date: ${date}`, `tags: [${tags.join(', ')}]`, `summary: ""`, `visibility: primary`, `source: linkedin-post`, share.shareLink ? `linkedin_url: "${share.shareLink}"` : null, '---', ].filter(Boolean).join('\n') const output = `${frontmatter}\n\n${commentary}\n` const outputPath = join(OUTPUT_KNIEPUNKT, `${slug}.md`) writeFileSync(outputPath, output, 'utf-8') kniepunktCount++ } else { // Regular post const firstLine = commentary.split('\n')[0].replace(/["""]/g, '').substring(0, 80).trim() const title = firstLine || 'Untitled Post' const slug = `${date}-${slugify(title)}` const frontmatter = [ '---', `title: "${title.replace(/"/g, '\\"')}"`, `date: ${date}`, `tags: [${tags.join(', ')}]`, `summary: ""`, `source: linkedin-post`, `visibility: primary`, share.shareLink ? `linkedin_url: "${share.shareLink}"` : null, '---', ].filter(Boolean).join('\n') const output = `${frontmatter}\n\n${commentary}\n` const outputPath = join(OUTPUT_POSTS, `${slug}.md`) writeFileSync(outputPath, output, 'utf-8') postCount++ } } console.log(` ✅ ${kniepunktCount} Kniepunkt posts (issues as posts) → content/kniepunkt/`) console.log(` ✅ ${postCount} substantial posts → content/posts/`) console.log(` ⏭️ ${skipped} skipped (too short, reposts, or article shares)`) } // --- Run --- console.log('🚀 LinkedIn Export → Homepage Content Extractor') console.log(` Source: ${EXPORT_DIR}`) console.log(` Target: andreknie.de/content/`) // Remove sample files first const sampleKniepunkt = join(OUTPUT_KNIEPUNKT, '001-sample.md') const samplePost = join(OUTPUT_POSTS, 'sample-post.md') if (existsSync(sampleKniepunkt)) { const { unlinkSync } = await import('fs') unlinkSync(sampleKniepunkt) } if (existsSync(samplePost)) { const { unlinkSync } = await import('fs') unlinkSync(samplePost) } extractArticles() extractShares() console.log('\n✨ Done! Review the generated files in andreknie.de/content/') console.log(' - Check content/kniepunkt/ for column issues') console.log(' - Check content/posts/ for articles and posts') console.log(' - Fill in empty "summary" fields for key content') console.log(' - Adjust "visibility" to "secondary" for non-AI topics')