familyfedie-website/scripts/extract-content.mjs
Loyyd d74495ddef
Some checks are pending
/ deploy (push) Waiting to run
new source of truth
2026-06-11 10:01:45 +02:00

198 lines
6.6 KiB
JavaScript

import fs from "node:fs";
import path from "node:path";
const repoRoot = process.cwd();
const legacyHtmlRoot = path.join(repoRoot, "archive-source");
const contentRoot = path.join(repoRoot, "content");
const contentReadme = `# Content
This directory is the future source of truth for editable site content.
These Markdown files were extracted from the exported HTML posts in \`archive-source/\`.
The body content intentionally keeps its original HTML markup so the migration
does not lose formatting, links, headings, or embedded images.
- \`content/speeches/<year>/\` - Rev. Sun Myung Moon speech posts grouped by speech year
- \`content/speeches/mrs-hak-ja-han-moon/\` - Mrs. Hak Ja Han Moon speech posts
- \`content/blog/<year>/\` - non-speech blog posts grouped by published year
Regenerate this content from \`archive-source/blog/\` with:
\`\`\`bash
npm run extract:content
\`\`\`
`;
function walk(dir) {
return fs.readdirSync(dir, { withFileTypes: true }).flatMap((entry) => {
const fullPath = path.join(dir, entry.name);
if (entry.isDirectory()) {
return walk(fullPath);
}
return fullPath.endsWith(".html") ? [fullPath] : [];
});
}
function decodeEntities(value = "") {
return value
.replace(/&#8211;/g, "-")
.replace(/&#8212;/g, "-")
.replace(/&#8216;/g, "'")
.replace(/&#8217;/g, "'")
.replace(/&#8220;/g, '"')
.replace(/&#8221;/g, '"')
.replace(/&raquo;/g, "»")
.replace(/&laquo;/g, "«")
.replace(/&amp;/g, "&")
.replace(/&quot;/g, '"')
.replace(/&#039;/g, "'")
.replace(/&nbsp;/g, " ")
.replace(/&hellip;/g, "...")
.replace(/<[^>]*>/g, "")
.replace(/\s+/g, " ")
.trim();
}
function yamlString(value) {
return JSON.stringify(value ?? "");
}
function yamlArray(key, values) {
if (!values.length) {
return `${key}: []`;
}
return [`${key}:`, ...values.map((value) => ` - ${yamlString(value)}`)].join("\n");
}
function matchFirst(html, pattern) {
return html.match(pattern)?.[1] ?? "";
}
function extractLinks(html, containerClass) {
const startMatch = html.match(new RegExp(`<[^>]*class=["'][^"']*${containerClass}[^"']*["'][^>]*>`));
if (!startMatch?.index) {
return [];
}
const start = startMatch.index + startMatch[0].length;
const afterStart = html.slice(start);
const endIndex = afterStart.search(containerClass === "footer-tags" ? /<\/div>/ : /<\/span>/);
const body = endIndex === -1 ? afterStart : afterStart.slice(0, endIndex);
return Array.from(body.matchAll(/<a\s+[^>]*href=["']([^"']+)["'][^>]*>([\s\S]*?)<\/a>/g)).map((match) => ({
href: match[1],
label: decodeEntities(match[2]),
}));
}
function extractEntryContent(html) {
const content = matchFirst(html, /<div class="entry-content">\s*([\s\S]*?)\s*<\/div><!-- \.entry-content -->/);
return content
.replace(/<div class="pdfprnt-buttons[\s\S]*?<\/div>/g, "")
.replace(/\n[ \t]+/g, "\n")
.trim();
}
function slugFromPath(filePath) {
return path.basename(filePath, ".html");
}
function relativeUrl(filePath) {
return `/${path.relative(legacyHtmlRoot, filePath).split(path.sep).join("/")}`;
}
function destinationFor(post) {
const speechCategory = post.categories.find((category) => category.href.startsWith("/speeches/categories/"));
if (speechCategory) {
const year =
speechCategory.href.match(/(?:^|-)(1\d{3}|20\d{2})(?:-|\.html$)/)?.[1] ??
speechCategory.label.match(/\b(1\d{3}|20\d{2})\b/)?.[1];
const collection =
year ??
path.basename(speechCategory.href, ".html").replace(/^rev-sun-myung-moon$/, "rev-sun-myung-moon-general");
return path.join(contentRoot, "speeches", collection, `${post.slug}.md`);
}
const publishYear = post.published.match(/^\d{4}/)?.[0] ?? "undated";
return path.join(contentRoot, "blog", publishYear, `${post.slug}.md`);
}
function extractPost(filePath) {
const html = fs.readFileSync(filePath, "utf8");
const entryClass = matchFirst(html, /<div id="post-[^"]*" class="([^"]*post type-post[^"]*)">/);
if (!entryClass) {
return null;
}
const title = decodeEntities(matchFirst(html, /<h1 class="entry-title">([\s\S]*?)<\/h1>/));
const published = matchFirst(html, /<time class="onDate date published" datetime="([^"]+)"/);
const updated = matchFirst(html, /<time class="updated"\s+datetime="([^"]+)"/);
const categories = extractLinks(html, "bl_categ");
const tags = extractLinks(html, "footer-tags");
const content = extractEntryContent(html);
const slug = slugFromPath(filePath);
const source = relativeUrl(filePath);
return {
title,
published,
updated,
source,
slug,
categories,
tags,
content,
};
}
function renderMarkdown(post) {
const speechCategory = post.categories.find((category) => category.href.startsWith("/speeches/categories/"));
const type = speechCategory ? "speech" : "blog";
const speechYear =
speechCategory?.href.match(/(?:^|-)(1\d{3}|20\d{2})(?:-|\.html$)/)?.[1] ??
speechCategory?.label.match(/\b(1\d{3}|20\d{2})\b/)?.[1];
const speechCollection =
speechCategory && !speechYear
? path.basename(speechCategory.href, ".html").replace(/^rev-sun-myung-moon$/, "rev-sun-myung-moon-general")
: null;
const categoryLabels = post.categories.map((category) => category.label);
const tagLabels = post.tags.map((tag) => tag.label);
const frontmatter = [
"---",
`title: ${yamlString(post.title)}`,
`type: ${yamlString(type)}`,
`published: ${yamlString(post.published)}`,
`updated: ${yamlString(post.updated)}`,
`source: ${yamlString(post.source)}`,
speechYear ? `speechYear: ${yamlString(speechYear)}` : null,
speechCollection ? `speechCollection: ${yamlString(speechCollection)}` : null,
yamlArray("categories", categoryLabels),
yamlArray("tags", tagLabels),
"---",
]
.filter(Boolean)
.join("\n");
return `${frontmatter}\n\n${post.content}\n`;
}
fs.rmSync(contentRoot, { recursive: true, force: true });
const postFiles = walk(path.join(legacyHtmlRoot, "blog")).filter((filePath) => /\/\d{4}\/\d{2}\/\d{2}\//.test(filePath));
const posts = postFiles.map(extractPost).filter(Boolean);
for (const post of posts) {
const destination = destinationFor(post);
fs.mkdirSync(path.dirname(destination), { recursive: true });
fs.writeFileSync(destination, renderMarkdown(post));
}
fs.writeFileSync(path.join(contentRoot, "README.md"), contentReadme);
const speechCount = posts.filter((post) => destinationFor(post).includes(`${path.sep}speeches${path.sep}`)).length;
const blogCount = posts.length - speechCount;
console.log(`Extracted ${posts.length} posts to content/.`);
console.log(`- speeches: ${speechCount}`);
console.log(`- blog: ${blogCount}`);