This is an automated email from the ASF dual-hosted git repository.
davsclaus pushed a commit to branch main
in repository https://gitbox.apache.org/repos/asf/camel-website.git
The following commit(s) were added to refs/heads/main by this push:
new 429b321a chore: share the page-to-Markdown converter between the page
and index generators
429b321a is described below
commit 429b321a91b09188adb259d0e927611ddf86f3e5
Author: Claus Ibsen <[email protected]>
AuthorDate: Fri Sep 18 19:43:11 2026 +0200
chore: share the page-to-Markdown converter between the page and index
generators
html-index.js carried its own copy of the content extraction and link
rewriting in generate-markdown.js, so every fix to one had to be repeated
in the other (most recently the eyebrow label). convertPage and
trimBlogPost move to gulp/helpers/convert-page.js and both generators
use it; the index pages also gain the jsdom repair of malformed HTML and
the table of contents removal that only the page converter had.
Closes #1777
---
.../convert-page.js} | 101 +--------------------
gulp/helpers/html-index.js | 45 +--------
gulp/tasks/generate-markdown.js | 99 +-------------------
...erate-markdown-test.js => convert-page-test.js} | 2 +-
4 files changed, 6 insertions(+), 241 deletions(-)
diff --git a/gulp/tasks/generate-markdown.js b/gulp/helpers/convert-page.js
similarity index 56%
copy from gulp/tasks/generate-markdown.js
copy to gulp/helpers/convert-page.js
index d6d191b4..ac593aae 100644
--- a/gulp/tasks/generate-markdown.js
+++ b/gulp/helpers/convert-page.js
@@ -1,103 +1,5 @@
-const fs = require('fs');
const { JSDOM } = require('jsdom');
const { parse, valid } = require('node-html-parser');
-const { createTurndownService } = require('../helpers/turndown-config');
-const { generateToonSitemaps } = require('../helpers/toon-format');
-const { generateLlmsTxt } = require('../helpers/llms-txt');
-const { generateReleasesIndex, generateBlogIndex } =
require('../helpers/rss-feed');
-const { generateAllIndexes } = require('../helpers/html-index');
-
-/**
- * Generates Markdown (.md) files from HTML files for LLM consumption.
- * This task converts HTML documentation pages to Markdown format, making them
- * accessible to LLMs as per https://llmstxt.org/ specification.
- *
- * For each .html file, it creates a corresponding .md file with:
- * - Only the main article content (excluding nav, header, footer)
- * - Clean Markdown formatting using Turndown
- * - GitHub-flavored Markdown for tables and code blocks
- *
- * Hugo renders every website page as <page>/index.html, so an index.html is
converted too
- * when it is a content page (has an article.doc): the .md then sits next to
it as
- * <page>/index.md, which keeps the page's relative links valid. List and
section pages
- * (home, download, community, ...) have no article.doc and are skipped. Blog
posts are
- * content pages too and are reduced to their title, byline and body, see
trimBlogPost.
- */
-async function generateMarkdown() {
- const turndownService = createTurndownService();
-
- // Keep track of processed files for llms.txt
- const processedPages = [];
-
- const glob = require('glob');
-
- // Get all HTML files
- const htmlFiles = glob.sync('public/**/*.html', {
- ignore: [
- 'public/404.html',
- 'public/releases/**/index.html' // release pages are converted by
generateAllIndexes below
- ]
- });
-
- let processedCount = 0;
- const totalFiles = htmlFiles.length;
- const BATCH_SIZE = 500; // Process in batches to avoid memory issues
-
- console.log(`Found ${totalFiles} HTML files to convert`);
-
- // Process files in batches
- for (let i = 0; i < htmlFiles.length; i += BATCH_SIZE) {
- const batch = htmlFiles.slice(i, i + BATCH_SIZE);
-
- for (const htmlFile of batch) {
- try {
- const htmlContent = fs.readFileSync(htmlFile, 'utf8');
- const articleOnly = htmlFile.endsWith('/index.html');
- const { markdown, repaired } = convertPage(htmlContent,
turndownService, { articleOnly });
-
- if (repaired) {
- console.warn(`Repaired malformed HTML in ${htmlFile}, fix the
mis-nested markup in its source`);
- }
-
- if (markdown === null) {
- if (!articleOnly) {
- console.warn(`Skipping ${htmlFile}: no main content found`);
- }
- continue;
- }
-
- // Write .md file (replace .html extension with .md)
- const mdFile = htmlFile.replace(/\.html$/, '.md');
- fs.writeFileSync(mdFile, markdown, 'utf8');
-
- // Track for llms.txt (convert to URL path)
- const urlPath = htmlFile.replace('public/', '/').replace('.html',
'.md');
- processedPages.push(urlPath);
-
- processedCount++;
- } catch (error) {
- console.error(`Error processing ${htmlFile}:`, error.message);
- }
- }
- }
-
- console.log(`\nSuccessfully generated ${processedCount} Markdown files`);
-
- // Generate llms.txt file
- generateLlmsTxt(processedPages);
-
- // Generate toon format sitemaps
- await generateToonSitemaps();
-
- // Generate toon format for releases RSS feed
- await generateReleasesIndex();
-
- // Generate toon format for blog RSS feed
- await generateBlogIndex();
-
- // Generate all other index files
- await generateAllIndexes();
-}
/**
* Converts the main content of one rendered HTML page to Markdown.
@@ -194,5 +96,4 @@ function trimBlogPost(article) {
}
}
-module.exports = generateMarkdown;
-module.exports.convertPage = convertPage;
+module.exports = { convertPage };
diff --git a/gulp/helpers/html-index.js b/gulp/helpers/html-index.js
index 03af79ed..4f300e44 100644
--- a/gulp/helpers/html-index.js
+++ b/gulp/helpers/html-index.js
@@ -1,5 +1,5 @@
const fs = require('fs');
-const { parse } = require('node-html-parser');
+const { convertPage } = require('./convert-page');
const { createTurndownService } = require('./turndown-config');
/**
@@ -21,51 +21,12 @@ async function generateHtmlIndex(config) {
}
const htmlContent = fs.readFileSync(htmlPath, 'utf8');
- const root = parse(htmlContent);
+ let { markdown } = convertPage(htmlContent, createTurndownService());
- // Create turndown service
- const turndownService = createTurndownService();
-
- // Extract only the main article content
- let mainContent = root.querySelector('article.doc') ||
- root.querySelector('main') ||
- root.querySelector('.article') ||
- root.querySelector('article');
-
- if (!mainContent) {
+ if (markdown === null) {
return;
}
- // Remove navigation elements and the eyebrow label above the title
- const elementsToRemove = mainContent.querySelectorAll('nav, header,
footer, .nav, .navbar, .toolbar, .doc-eyebrow');
- elementsToRemove.forEach(el => el.remove());
-
- // Remove anchor links
- const anchors = mainContent.querySelectorAll('a.anchor');
- anchors.forEach(el => el.remove());
-
- // Clean up table cells by unwrapping div.content and div.paragraph
wrappers
- const tableCells = mainContent.querySelectorAll('td.tableblock,
th.tableblock');
- tableCells.forEach(cell => {
- let html = cell.innerHTML;
- // Unwrap <div class="content"><div
class="paragraph"><p>...</p></div></div>
- html = html.replace(/<div class="content"><div
class="paragraph">\s*<p>(.*?)<\/p>\s*<\/div><\/div>/gs, '$1');
- // Unwrap <div class="content"><div id="..."
class="paragraph"><p>...</p></div></div>
- html = html.replace(/<div
class="content"><div[^>]*class="paragraph"[^>]*>\s*<p>(.*?)<\/p>\s*<\/div><\/div>/gs,
'$1');
- // Also handle simple <p class="tableblock">...</p> wrappers
- html = html.replace(/<p class="tableblock">(.*?)<\/p>/gs, '$1');
- cell.set_content(html);
- });
-
- // Convert to Markdown
- let markdown = turndownService.turndown(mainContent.innerHTML);
-
- // Update links to point to .md files instead of .html
- // Replace https://camel.apache.org/**/*.html with
https://camel.apache.org/**/*.md
- markdown =
markdown.replace(/(https:\/\/camel\.apache\.org\/[^)\s]*?)\.html/g, '$1.md');
- // Replace relative links *.html with *.md
- markdown = markdown.replace(/\[([^\]]+)\]\(([^)]+?)\.html\)/g,
'[$1]($2.md)');
-
// Add header if title and description provided
if (title && description) {
markdown = `# ${title}\n\n${description}\n\n${markdown}`;
diff --git a/gulp/tasks/generate-markdown.js b/gulp/tasks/generate-markdown.js
index d6d191b4..de3aa02e 100644
--- a/gulp/tasks/generate-markdown.js
+++ b/gulp/tasks/generate-markdown.js
@@ -1,6 +1,5 @@
const fs = require('fs');
-const { JSDOM } = require('jsdom');
-const { parse, valid } = require('node-html-parser');
+const { convertPage } = require('../helpers/convert-page');
const { createTurndownService } = require('../helpers/turndown-config');
const { generateToonSitemaps } = require('../helpers/toon-format');
const { generateLlmsTxt } = require('../helpers/llms-txt');
@@ -99,100 +98,4 @@ async function generateMarkdown() {
await generateAllIndexes();
}
-/**
- * Converts the main content of one rendered HTML page to Markdown.
- *
- * @param {string} htmlContent the full HTML page
- * @param {TurndownService} turndownService configured Turndown instance
- * @param {{articleOnly?: boolean}} [options] articleOnly: only accept an
article.doc as the main
- * content, so list and section pages that merely have a <main> are not
converted
- * @returns {{markdown: string|null, repaired: boolean}} the Markdown (null
when the page has no
- * main content), and whether the HTML was malformed and had to be repaired
before parsing
- */
-function convertPage(htmlContent, turndownService, { articleOnly = false } =
{}) {
- // node-html-parser cannot repair mis-nested inline tags (Asciidoctor emits
them for a `*` inside
- // backticks): it unwraps every unclosed ancestor, article.doc included. Let
jsdom's HTML5 parser
- // repair such pages the way browsers do; it is much slower, so only
malformed pages go through it.
- const repaired = !valid(htmlContent);
- const root = parse(repaired ? new JSDOM(htmlContent).serialize() :
htmlContent);
-
- // Extract only the main article content
- // Try different selectors based on Antora and Hugo structure
- let mainContent = root.querySelector('article.doc');
- if (!mainContent && !articleOnly) {
- mainContent = root.querySelector('main') ||
- root.querySelector('.article') ||
- root.querySelector('article');
- }
-
- if (!mainContent) {
- return { markdown: null, repaired };
- }
-
- if (mainContent.classList.contains('post')) {
- trimBlogPost(mainContent);
- }
-
- // Remove navigation elements, headers, footers, the embedded table of
contents and the eyebrow
- // label above the title (the Antora UI puts the component title there, e.g.
"User manual")
- const elementsToRemove = mainContent.querySelectorAll('nav, header, footer,
.nav, .navbar, .toolbar, aside.toc, .doc-eyebrow');
- elementsToRemove.forEach(el => el.remove());
-
- // Remove anchor links (they are just UI navigation aids)
- const anchors = mainContent.querySelectorAll('a.anchor');
- anchors.forEach(el => el.remove());
-
- // Clean up table cells by unwrapping div.content and div.paragraph wrappers
- const tableCells = mainContent.querySelectorAll('td.tableblock,
th.tableblock');
- tableCells.forEach(cell => {
- let html = cell.innerHTML;
- // Unwrap <div class="content"><div
class="paragraph"><p>...</p></div></div>
- html = html.replace(/<div class="content"><div
class="paragraph">\s*<p>(.*?)<\/p>\s*<\/div><\/div>/gs, '$1');
- // Unwrap <div class="content"><div id="..."
class="paragraph"><p>...</p></div></div>
- html = html.replace(/<div
class="content"><div[^>]*class="paragraph"[^>]*>\s*<p>(.*?)<\/p>\s*<\/div><\/div>/gs,
'$1');
- // Also handle simple <p class="tableblock">...</p> wrappers
- html = html.replace(/<p class="tableblock">(.*?)<\/p>/gs, '$1');
- cell.set_content(html);
- });
-
- // Convert to Markdown
- let markdown = turndownService.turndown(mainContent.innerHTML);
-
- // Update links to point to .md files instead of .html
- // Replace https://camel.apache.org/**/*.html with
https://camel.apache.org/**/*.md
- markdown =
markdown.replace(/(https:\/\/camel\.apache\.org\/[^)\s]*?)\.html/g, '$1.md');
- // Replace relative links *.html with *.md
- markdown = markdown.replace(/\[([^\]]+)\]\(([^)]+?)\.html\)/g,
'[$1]($2.md)');
-
- return { markdown, repaired };
-}
-
-/**
- * Reduces a rendered blog post (layouts/blog/post.html) to its title, lead
and body, with the
- * date, authors and categories from the rail folded into one line under the
title. The rail
- * itself (avatars, share links, table of contents, previous/next), the "All
posts" link, the
- * featured image and the related posts are page chrome and are dropped.
- *
- * @param {HTMLElement} article the article.post element
- */
-function trimBlogPost(article) {
- const date =
article.querySelector('time.post-date')?.getAttribute('datetime');
- const authors = article.querySelectorAll('.post-author-name').map(el =>
el.text.trim());
- const categories = article.querySelectorAll('.post-tags a').map(el =>
el.text.trim());
-
- article.querySelectorAll('a.post-back, .post-tags, aside.post-rail,
img.featured, section.post-related')
- .forEach(el => el.remove());
-
- const byline = [
- date && `Published ${date}`,
- authors.length && `by ${authors.join(', ')}`,
- categories.length && `in ${categories.join(', ')}`
- ].filter(Boolean).join(' ');
- const title = article.querySelector('h1.post-title');
- if (byline && title) {
- title.insertAdjacentHTML('afterend', `<p>${byline}</p>`);
- }
-}
-
module.exports = generateMarkdown;
-module.exports.convertPage = convertPage;
diff --git a/test/generate-markdown-test.js b/test/convert-page-test.js
similarity index 99%
rename from test/generate-markdown-test.js
rename to test/convert-page-test.js
index 47236cdb..fbd86010 100644
--- a/test/generate-markdown-test.js
+++ b/test/convert-page-test.js
@@ -3,7 +3,7 @@
const assert = require('node:assert/strict')
const test = require('node:test')
-const { convertPage } = require('../gulp/tasks/generate-markdown')
+const { convertPage } = require('../gulp/helpers/convert-page')
const { createTurndownService } = require('../gulp/helpers/turndown-config')
function page (body) {