This is an automated email from the ASF dual-hosted git repository.

davsclaus pushed a commit to branch main
in repository https://gitbox.apache.org/repos/asf/camel-website.git


The following commit(s) were added to refs/heads/main by this push:
     new 429b321a chore: share the page-to-Markdown converter between the page 
and index generators
429b321a is described below

commit 429b321a91b09188adb259d0e927611ddf86f3e5
Author: Claus Ibsen <[email protected]>
AuthorDate: Fri Sep 18 19:43:11 2026 +0200

    chore: share the page-to-Markdown converter between the page and index 
generators
    
    html-index.js carried its own copy of the content extraction and link
    rewriting in generate-markdown.js, so every fix to one had to be repeated
    in the other (most recently the eyebrow label). convertPage and
    trimBlogPost move to gulp/helpers/convert-page.js and both generators
    use it; the index pages also gain the jsdom repair of malformed HTML and
    the table of contents removal that only the page converter had.
    
    Closes #1777
---
 .../convert-page.js}                               | 101 +--------------------
 gulp/helpers/html-index.js                         |  45 +--------
 gulp/tasks/generate-markdown.js                    |  99 +-------------------
 ...erate-markdown-test.js => convert-page-test.js} |   2 +-
 4 files changed, 6 insertions(+), 241 deletions(-)

diff --git a/gulp/tasks/generate-markdown.js b/gulp/helpers/convert-page.js
similarity index 56%
copy from gulp/tasks/generate-markdown.js
copy to gulp/helpers/convert-page.js
index d6d191b4..ac593aae 100644
--- a/gulp/tasks/generate-markdown.js
+++ b/gulp/helpers/convert-page.js
@@ -1,103 +1,5 @@
-const fs = require('fs');
 const { JSDOM } = require('jsdom');
 const { parse, valid } = require('node-html-parser');
-const { createTurndownService } = require('../helpers/turndown-config');
-const { generateToonSitemaps } = require('../helpers/toon-format');
-const { generateLlmsTxt } = require('../helpers/llms-txt');
-const { generateReleasesIndex, generateBlogIndex } = 
require('../helpers/rss-feed');
-const { generateAllIndexes } = require('../helpers/html-index');
-
-/**
- * Generates Markdown (.md) files from HTML files for LLM consumption.
- * This task converts HTML documentation pages to Markdown format, making them
- * accessible to LLMs as per https://llmstxt.org/ specification.
- *
- * For each .html file, it creates a corresponding .md file with:
- * - Only the main article content (excluding nav, header, footer)
- * - Clean Markdown formatting using Turndown
- * - GitHub-flavored Markdown for tables and code blocks
- *
- * Hugo renders every website page as <page>/index.html, so an index.html is 
converted too
- * when it is a content page (has an article.doc): the .md then sits next to 
it as
- * <page>/index.md, which keeps the page's relative links valid. List and 
section pages
- * (home, download, community, ...) have no article.doc and are skipped. Blog 
posts are
- * content pages too and are reduced to their title, byline and body, see 
trimBlogPost.
- */
-async function generateMarkdown() {
-  const turndownService = createTurndownService();
-
-  // Keep track of processed files for llms.txt
-  const processedPages = [];
-
-  const glob = require('glob');
-
-  // Get all HTML files
-  const htmlFiles = glob.sync('public/**/*.html', {
-    ignore: [
-      'public/404.html',
-      'public/releases/**/index.html' // release pages are converted by 
generateAllIndexes below
-    ]
-  });
-
-  let processedCount = 0;
-  const totalFiles = htmlFiles.length;
-  const BATCH_SIZE = 500; // Process in batches to avoid memory issues
-
-  console.log(`Found ${totalFiles} HTML files to convert`);
-
-  // Process files in batches
-  for (let i = 0; i < htmlFiles.length; i += BATCH_SIZE) {
-    const batch = htmlFiles.slice(i, i + BATCH_SIZE);
-
-    for (const htmlFile of batch) {
-      try {
-        const htmlContent = fs.readFileSync(htmlFile, 'utf8');
-        const articleOnly = htmlFile.endsWith('/index.html');
-        const { markdown, repaired } = convertPage(htmlContent, 
turndownService, { articleOnly });
-
-        if (repaired) {
-          console.warn(`Repaired malformed HTML in ${htmlFile}, fix the 
mis-nested markup in its source`);
-        }
-
-        if (markdown === null) {
-          if (!articleOnly) {
-            console.warn(`Skipping ${htmlFile}: no main content found`);
-          }
-          continue;
-        }
-
-        // Write .md file (replace .html extension with .md)
-        const mdFile = htmlFile.replace(/\.html$/, '.md');
-        fs.writeFileSync(mdFile, markdown, 'utf8');
-
-        // Track for llms.txt (convert to URL path)
-        const urlPath = htmlFile.replace('public/', '/').replace('.html', 
'.md');
-        processedPages.push(urlPath);
-
-        processedCount++;
-      } catch (error) {
-        console.error(`Error processing ${htmlFile}:`, error.message);
-      }
-    }
-  }
-
-  console.log(`\nSuccessfully generated ${processedCount} Markdown files`);
-
-  // Generate llms.txt file
-  generateLlmsTxt(processedPages);
-
-  // Generate toon format sitemaps
-  await generateToonSitemaps();
-
-  // Generate toon format for releases RSS feed
-  await generateReleasesIndex();
-
-  // Generate toon format for blog RSS feed
-  await generateBlogIndex();
-
-  // Generate all other index files
-  await generateAllIndexes();
-}
 
 /**
  * Converts the main content of one rendered HTML page to Markdown.
@@ -194,5 +96,4 @@ function trimBlogPost(article) {
   }
 }
 
-module.exports = generateMarkdown;
-module.exports.convertPage = convertPage;
+module.exports = { convertPage };
diff --git a/gulp/helpers/html-index.js b/gulp/helpers/html-index.js
index 03af79ed..4f300e44 100644
--- a/gulp/helpers/html-index.js
+++ b/gulp/helpers/html-index.js
@@ -1,5 +1,5 @@
 const fs = require('fs');
-const { parse } = require('node-html-parser');
+const { convertPage } = require('./convert-page');
 const { createTurndownService } = require('./turndown-config');
 
 /**
@@ -21,51 +21,12 @@ async function generateHtmlIndex(config) {
     }
 
     const htmlContent = fs.readFileSync(htmlPath, 'utf8');
-    const root = parse(htmlContent);
+    let { markdown } = convertPage(htmlContent, createTurndownService());
 
-    // Create turndown service
-    const turndownService = createTurndownService();
-
-    // Extract only the main article content
-    let mainContent = root.querySelector('article.doc') ||
-                     root.querySelector('main') ||
-                     root.querySelector('.article') ||
-                     root.querySelector('article');
-
-    if (!mainContent) {
+    if (markdown === null) {
       return;
     }
 
-    // Remove navigation elements and the eyebrow label above the title
-    const elementsToRemove = mainContent.querySelectorAll('nav, header, 
footer, .nav, .navbar, .toolbar, .doc-eyebrow');
-    elementsToRemove.forEach(el => el.remove());
-
-    // Remove anchor links
-    const anchors = mainContent.querySelectorAll('a.anchor');
-    anchors.forEach(el => el.remove());
-
-    // Clean up table cells by unwrapping div.content and div.paragraph 
wrappers
-    const tableCells = mainContent.querySelectorAll('td.tableblock, 
th.tableblock');
-    tableCells.forEach(cell => {
-      let html = cell.innerHTML;
-      // Unwrap <div class="content"><div 
class="paragraph"><p>...</p></div></div>
-      html = html.replace(/<div class="content"><div 
class="paragraph">\s*<p>(.*?)<\/p>\s*<\/div><\/div>/gs, '$1');
-      // Unwrap <div class="content"><div id="..." 
class="paragraph"><p>...</p></div></div>
-      html = html.replace(/<div 
class="content"><div[^>]*class="paragraph"[^>]*>\s*<p>(.*?)<\/p>\s*<\/div><\/div>/gs,
 '$1');
-      // Also handle simple <p class="tableblock">...</p> wrappers
-      html = html.replace(/<p class="tableblock">(.*?)<\/p>/gs, '$1');
-      cell.set_content(html);
-    });
-
-    // Convert to Markdown
-    let markdown = turndownService.turndown(mainContent.innerHTML);
-
-    // Update links to point to .md files instead of .html
-    // Replace https://camel.apache.org/**/*.html with 
https://camel.apache.org/**/*.md
-    markdown = 
markdown.replace(/(https:\/\/camel\.apache\.org\/[^)\s]*?)\.html/g, '$1.md');
-    // Replace relative links *.html with *.md
-    markdown = markdown.replace(/\[([^\]]+)\]\(([^)]+?)\.html\)/g, 
'[$1]($2.md)');
-
     // Add header if title and description provided
     if (title && description) {
       markdown = `# ${title}\n\n${description}\n\n${markdown}`;
diff --git a/gulp/tasks/generate-markdown.js b/gulp/tasks/generate-markdown.js
index d6d191b4..de3aa02e 100644
--- a/gulp/tasks/generate-markdown.js
+++ b/gulp/tasks/generate-markdown.js
@@ -1,6 +1,5 @@
 const fs = require('fs');
-const { JSDOM } = require('jsdom');
-const { parse, valid } = require('node-html-parser');
+const { convertPage } = require('../helpers/convert-page');
 const { createTurndownService } = require('../helpers/turndown-config');
 const { generateToonSitemaps } = require('../helpers/toon-format');
 const { generateLlmsTxt } = require('../helpers/llms-txt');
@@ -99,100 +98,4 @@ async function generateMarkdown() {
   await generateAllIndexes();
 }
 
-/**
- * Converts the main content of one rendered HTML page to Markdown.
- *
- * @param {string} htmlContent the full HTML page
- * @param {TurndownService} turndownService configured Turndown instance
- * @param {{articleOnly?: boolean}} [options] articleOnly: only accept an 
article.doc as the main
- *   content, so list and section pages that merely have a <main> are not 
converted
- * @returns {{markdown: string|null, repaired: boolean}} the Markdown (null 
when the page has no
- *   main content), and whether the HTML was malformed and had to be repaired 
before parsing
- */
-function convertPage(htmlContent, turndownService, { articleOnly = false } = 
{}) {
-  // node-html-parser cannot repair mis-nested inline tags (Asciidoctor emits 
them for a `*` inside
-  // backticks): it unwraps every unclosed ancestor, article.doc included. Let 
jsdom's HTML5 parser
-  // repair such pages the way browsers do; it is much slower, so only 
malformed pages go through it.
-  const repaired = !valid(htmlContent);
-  const root = parse(repaired ? new JSDOM(htmlContent).serialize() : 
htmlContent);
-
-  // Extract only the main article content
-  // Try different selectors based on Antora and Hugo structure
-  let mainContent = root.querySelector('article.doc');
-  if (!mainContent && !articleOnly) {
-    mainContent = root.querySelector('main') ||
-                  root.querySelector('.article') ||
-                  root.querySelector('article');
-  }
-
-  if (!mainContent) {
-    return { markdown: null, repaired };
-  }
-
-  if (mainContent.classList.contains('post')) {
-    trimBlogPost(mainContent);
-  }
-
-  // Remove navigation elements, headers, footers, the embedded table of 
contents and the eyebrow
-  // label above the title (the Antora UI puts the component title there, e.g. 
"User manual")
-  const elementsToRemove = mainContent.querySelectorAll('nav, header, footer, 
.nav, .navbar, .toolbar, aside.toc, .doc-eyebrow');
-  elementsToRemove.forEach(el => el.remove());
-
-  // Remove anchor links (they are just UI navigation aids)
-  const anchors = mainContent.querySelectorAll('a.anchor');
-  anchors.forEach(el => el.remove());
-
-  // Clean up table cells by unwrapping div.content and div.paragraph wrappers
-  const tableCells = mainContent.querySelectorAll('td.tableblock, 
th.tableblock');
-  tableCells.forEach(cell => {
-    let html = cell.innerHTML;
-    // Unwrap <div class="content"><div 
class="paragraph"><p>...</p></div></div>
-    html = html.replace(/<div class="content"><div 
class="paragraph">\s*<p>(.*?)<\/p>\s*<\/div><\/div>/gs, '$1');
-    // Unwrap <div class="content"><div id="..." 
class="paragraph"><p>...</p></div></div>
-    html = html.replace(/<div 
class="content"><div[^>]*class="paragraph"[^>]*>\s*<p>(.*?)<\/p>\s*<\/div><\/div>/gs,
 '$1');
-    // Also handle simple <p class="tableblock">...</p> wrappers
-    html = html.replace(/<p class="tableblock">(.*?)<\/p>/gs, '$1');
-    cell.set_content(html);
-  });
-
-  // Convert to Markdown
-  let markdown = turndownService.turndown(mainContent.innerHTML);
-
-  // Update links to point to .md files instead of .html
-  // Replace https://camel.apache.org/**/*.html with 
https://camel.apache.org/**/*.md
-  markdown = 
markdown.replace(/(https:\/\/camel\.apache\.org\/[^)\s]*?)\.html/g, '$1.md');
-  // Replace relative links *.html with *.md
-  markdown = markdown.replace(/\[([^\]]+)\]\(([^)]+?)\.html\)/g, 
'[$1]($2.md)');
-
-  return { markdown, repaired };
-}
-
-/**
- * Reduces a rendered blog post (layouts/blog/post.html) to its title, lead 
and body, with the
- * date, authors and categories from the rail folded into one line under the 
title. The rail
- * itself (avatars, share links, table of contents, previous/next), the "All 
posts" link, the
- * featured image and the related posts are page chrome and are dropped.
- *
- * @param {HTMLElement} article the article.post element
- */
-function trimBlogPost(article) {
-  const date = 
article.querySelector('time.post-date')?.getAttribute('datetime');
-  const authors = article.querySelectorAll('.post-author-name').map(el => 
el.text.trim());
-  const categories = article.querySelectorAll('.post-tags a').map(el => 
el.text.trim());
-
-  article.querySelectorAll('a.post-back, .post-tags, aside.post-rail, 
img.featured, section.post-related')
-    .forEach(el => el.remove());
-
-  const byline = [
-    date && `Published ${date}`,
-    authors.length && `by ${authors.join(', ')}`,
-    categories.length && `in ${categories.join(', ')}`
-  ].filter(Boolean).join(' ');
-  const title = article.querySelector('h1.post-title');
-  if (byline && title) {
-    title.insertAdjacentHTML('afterend', `<p>${byline}</p>`);
-  }
-}
-
 module.exports = generateMarkdown;
-module.exports.convertPage = convertPage;
diff --git a/test/generate-markdown-test.js b/test/convert-page-test.js
similarity index 99%
rename from test/generate-markdown-test.js
rename to test/convert-page-test.js
index 47236cdb..fbd86010 100644
--- a/test/generate-markdown-test.js
+++ b/test/convert-page-test.js
@@ -3,7 +3,7 @@
 const assert = require('node:assert/strict')
 const test = require('node:test')
 
-const { convertPage } = require('../gulp/tasks/generate-markdown')
+const { convertPage } = require('../gulp/helpers/convert-page')
 const { createTurndownService } = require('../gulp/helpers/turndown-config')
 
 function page (body) {

Reply via email to