From 33480cfd6cc8f9533551e58c38f9670282081213 Mon Sep 17 00:00:00 2001 From: Karl Kemister-Sheppard Date: Wed, 23 Sep 2026 16:45:27 +1000 Subject: [PATCH 1/8] TINYDOC-3613: Date each page from its source history with an Antora extension. --- .gitignore | 3 + antora-playbook.yml | 3 + lib/antora-extension-page-dates.js | 262 +++++++++++++++++++++++++++++ 3 files changed, 268 insertions(+) create mode 100644 lib/antora-extension-page-dates.js diff --git a/.gitignore b/.gitignore index 490a27221c..4e7ade760d 100644 --- a/.gitignore +++ b/.gitignore @@ -14,6 +14,9 @@ vendor # Antora tmp files _site/ +# Build caches (page date history clones and the page date map) +.cache/ + # Cursor setup (local only, do not push) .cursor/ AGENTS.md diff --git a/antora-playbook.yml b/antora-playbook.yml index 1bd25a4e0d..9361056236 100644 --- a/antora-playbook.yml +++ b/antora-playbook.yml @@ -16,5 +16,8 @@ ui: asciidoc: extensions: - '@tinymce/antora-extension-livedemos' +antora: + extensions: + - require: ./lib/antora-extension-page-dates.js runtime: fetch: true diff --git a/lib/antora-extension-page-dates.js b/lib/antora-extension-page-dates.js new file mode 100644 index 0000000000..f6acbd1ae5 --- /dev/null +++ b/lib/antora-extension-page-dates.js @@ -0,0 +1,262 @@ +'use strict' + +/** + * Antora extension: records when each page's content last changed. + * + * For every publishable page, the date is read from the git history of the + * content source Antora built the page from. The source is identified by the + * page's own origin (repository URL, ref and commit hash), never by the URL + * segment of the published page: `latest` maps to `tinymce/8` today and will + * not always. + * + * A page's content includes the files it pulls in with `include::`, so a + * page's `last_updated` is the newest commit date of the page source or any + * file it includes, directly or transitively. `first_published` is the oldest + * commit date of the page source file alone. + * + * Generated API reference pages (`modules/ROOT/pages/apis/`) are committed to + * the content branches when the API reference is regenerated for a release, + * so they follow the same rule: their date is the date the reference was last + * regenerated. + * + * A page with no commit history (for example, an uncommitted file in a local + * worktree build) falls back to the build clock and is marked `source: build`. + * + * The dates are: + * - set on each page as the `page-last-updated`, `page-first-published` and + * `page-date-source` attributes, for the UI templates; + * - applied to the `lastmod` of every sitemap entry; + * - written to a JSON map, for the post-build markdown and LLM generators. + * + * Configuration (all optional): + * cache_dir: directory for the history clones (default: /.cache/page-dates) + * output: path of the JSON map (default: /.cache/page-dates.json) + */ + +const { execFile } = require('node:child_process') +const { createHash } = require('node:crypto') +const fs = require('node:fs') +const path = require('node:path') +const { promisify } = require('node:util') + +const run = promisify(execFile) + +const SCHEMA = 1 +const RECORD_SEPARATOR = '\x1e' +const INCLUDE_RX = /^include::([^[\n]+)\[/gm + +const toIsoSeconds = (date) => new Date(date).toISOString().replace(/\.\d{3}Z$/, 'Z') +const newest = (a, b) => (!a || (b && b > a) ? b : a) + +const git = async (args, opts = {}) => { + const { stdout } = await run('git', args, { maxBuffer: 256 * 1024 * 1024, ...opts }) + return stdout +} + +const isRemoteUrl = (url) => /^(https?|ssh|git):\/\//.test(url || '') || /^[^/]+@[^:]+:/.test(url || '') + +// Return a git dir that contains the full history of `origin.refhash`, or null. +// Antora fetches remote content at depth 1, so its own cache cannot be used; +// a blobless clone carries every commit and tree without the file contents. +// Several origins (one per branch) share a repository, so the clone and any +// fetch run once per URL. +const createHistoryResolver = (cacheDir, logger) => { + const clones = new Map() + const fetches = new Map() + + const cloneOnce = (url) => { + if (!clones.has(url)) { + clones.set(url, (async () => { + const key = createHash('sha1').update(url).digest('hex').slice(0, 16) + const gitdir = path.join(cacheDir, `${key}.git`) + if (!fs.existsSync(gitdir)) { + logger.info(`Cloning history of ${url}`) + fs.mkdirSync(cacheDir, { recursive: true }) + await git(['clone', '--quiet', '--bare', '--filter=blob:none', '--no-tags', url, gitdir]) + } + return ['--git-dir', gitdir] + })()) + } + return clones.get(url) + } + + const fetchOnce = (url, gitArgs) => { + if (!fetches.has(url)) { + logger.info(`Fetching history of ${url}`) + fetches.set(url, git([...gitArgs, 'fetch', '--quiet', '--filter=blob:none', '--no-tags', 'origin', '+refs/heads/*:refs/heads/*'])) + } + return fetches.get(url) + } + + return async (origin) => { + if (origin.worktree) return { gitArgs: ['-C', origin.worktree], ref: 'HEAD', worktree: origin.worktree } + + if (origin.gitdir && fs.existsSync(origin.gitdir)) { + const shallow = (await git(['--git-dir', origin.gitdir, 'rev-parse', '--is-shallow-repository']).catch(() => 'true')).trim() + if (shallow === 'false') return { gitArgs: ['--git-dir', origin.gitdir], ref: origin.refhash } + } + + if (!isRemoteUrl(origin.url)) return null + + const gitArgs = await cloneOnce(origin.url) + const hasCommit = () => git([...gitArgs, 'cat-file', '-e', `${origin.refhash}^{commit}`]).then(() => true, () => false) + if (!(await hasCommit())) { + await fetchOnce(origin.url, gitArgs) + if (!(await hasCommit())) throw new Error(`Commit ${origin.refhash} not found in ${origin.url}`) + } + return { gitArgs, ref: origin.refhash } + } +} + +// One pass over the history of a ref: path -> { newest, oldest } commit date. +const readFileDates = async ({ gitArgs, ref, worktree }, startPath) => { + const out = await git([...gitArgs, 'log', '--no-renames', '--full-history', `--format=${RECORD_SEPARATOR}%cI`, '--name-only', ref, '--', startPath || '.']) + const dates = new Map() + for (const record of out.split(RECORD_SEPARATOR)) { + const [date, ...files] = record.split('\n') + if (!date) continue + const iso = toIsoSeconds(date) + for (const file of files) { + if (!file) continue + const entry = dates.get(file) + if (!entry) dates.set(file, { newest: iso, oldest: iso }) + else { + if (iso > entry.newest) entry.newest = iso + if (iso < entry.oldest) entry.oldest = iso + } + } + } + if (worktree) { + // Uncommitted changes have no commit date; they are dated by the build clock. + const status = await git([...gitArgs, 'status', '--porcelain', '--', startPath || '.']) + for (const line of status.split('\n')) { + const file = line.slice(3).replace(/^.* -> /, '') + if (file) dates.set(file, { uncommitted: true }) + } + } + return dates +} + +const originKey = (origin) => `${origin.url || origin.worktree || origin.gitdir}#${origin.refhash || origin.refname}` +const repoPath = (file) => path.posix.join(file.src.origin.startPath || '', file.src.path) + +// Resolve the include:: targets of a file to catalog files. Targets that use +// attribute references which cannot be resolved are skipped. +const resolveIncludes = (file, contentCatalog, attributes) => { + const text = file.contents.toString() + const found = [] + for (const [, rawTarget] of text.matchAll(INCLUDE_RX)) { + const target = rawTarget.trim().replace(/\{([\w-]+)\}/g, (ref, name) => (name in attributes ? attributes[name] : ref)) + if (/[{}]/.test(target) || /^https?:/.test(target)) continue + let resolved + if (target.includes('$')) { + resolved = contentCatalog.resolveResource(target, file.src, 'partial', ['partial', 'example', 'page', 'attachment']) + } else { + const relative = path.posix.normalize(path.posix.join(path.posix.dirname(file.src.path), target)) + resolved = contentCatalog.getFiles().find((f) => + f.src.path === relative && f.src.component === file.src.component && f.src.version === file.src.version) + } + if (resolved && resolved !== file) found.push(resolved) + } + return found +} + +module.exports.register = function ({ config = {} }) { + const logger = this.getLogger('page-dates-extension') + let pageDates + + this.once('contentClassified', async ({ playbook, contentCatalog }) => { + const playbookDir = playbook.dir || process.cwd() + const cacheDir = path.resolve(playbookDir, config.cacheDir || '.cache/page-dates') + const buildClock = toIsoSeconds(Date.now()) + const pages = contentCatalog.getPages((page) => page.pub) + + // Read each content source's history once. + const resolveHistoryRepo = createHistoryResolver(cacheDir, logger) + const histories = new Map() + for (const page of pages) { + const origin = page.src.origin + if (!origin || histories.has(originKey(origin))) continue + histories.set(originKey(origin), (async () => { + const repo = await resolveHistoryRepo(origin) + return repo ? readFileDates(repo, origin.startPath) : new Map() + })()) + } + const datesFor = async (file) => { + const origin = file.src.origin + if (!origin) return undefined + return (await histories.get(originKey(origin)))?.get(repoPath(file)) + } + + // Newest date across a file and everything it includes, transitively. + const contentDates = new Map() + const newestContentDate = async (file, attributes, seen = new Set()) => { + if (contentDates.has(file)) return contentDates.get(file) + if (seen.has(file)) return undefined + seen.add(file) + const own = await datesFor(file) + let date = own?.uncommitted ? buildClock : own?.newest + for (const included of resolveIncludes(file, contentCatalog, attributes)) { + date = newest(date, await newestContentDate(included, attributes, seen)) + } + contentDates.set(file, date) + return date + } + + pageDates = new Map() + for (const page of pages) { + const attributes = page.src.origin?.descriptor?.asciidoc?.attributes || {} + const own = await datesFor(page) + const lastUpdated = await newestContentDate(page, attributes) + const committed = !!own && !own.uncommitted && !!lastUpdated + pageDates.set(page, { + last_updated: committed ? lastUpdated : buildClock, + first_published: committed ? own.oldest : buildClock, + source: committed ? 'git' : 'build', + ref: page.src.origin?.refname || null, + path: page.src.origin ? repoPath(page) : page.src.path, + }) + } + + const fallbacks = [...pageDates.values()].filter((d) => d.source === 'build').length + logger.info(`Dated ${pageDates.size} pages from git history (${fallbacks} dated by the build clock)`) + }) + + this.once('documentsConverted', ({ contentCatalog }) => { + for (const [page, dates] of pageDates) { + const attributes = page.asciidoc?.attributes + if (!attributes) continue + attributes['page-last-updated'] = dates.last_updated + attributes['page-first-published'] = dates.first_published + attributes['page-date-source'] = dates.source + } + }) + + this.once('beforePublish', ({ playbook, siteCatalog }) => { + const byUrl = new Map([...pageDates].map(([page, dates]) => [page.pub.url, dates])) + const siteUrl = (playbook.site.url || '').replace(/\/$/, '') + + // Antora stamps every sitemap entry with the build time; replace it with the page's date. + for (const file of siteCatalog.getFiles()) { + if (!/^sitemap(-[^/]+)?\.xml$/.test(file.out?.path || '')) continue + const xml = file.contents.toString().replace( + /([^<]+)<\/loc>(\s*)[^<]*<\/lastmod>/g, + (match, loc, space) => { + const dates = byUrl.get(loc.startsWith(siteUrl) ? loc.slice(siteUrl.length) : loc) + return dates ? `${loc}${space}${dates.last_updated}` : match + } + ) + file.contents = Buffer.from(xml) + } + + const output = path.resolve(playbook.dir || process.cwd(), config.output || '.cache/page-dates.json') + const map = { + schema: SCHEMA, + generated: toIsoSeconds(Date.now()), + pages: Object.fromEntries([...byUrl].sort(([a], [b]) => a.localeCompare(b))), + } + fs.mkdirSync(path.dirname(output), { recursive: true }) + fs.writeFileSync(output, JSON.stringify(map, null, 2) + '\n') + logger.info(`Wrote page date map to ${path.relative(process.cwd(), output)}`) + }) +} From 12e9a2d40648cff0db40e0d8a9c05661941709d9 Mon Sep 17 00:00:00 2001 From: Karl Kemister-Sheppard Date: Wed, 23 Sep 2026 16:51:38 +1000 Subject: [PATCH 2/8] TINYDOC-3613: Add citation frontmatter to generated markdown and stop raw links leaking into it. --- scripts/generate-markdown.mjs | 177 ++++++++++++++++++++++++++++++---- 1 file changed, 159 insertions(+), 18 deletions(-) diff --git a/scripts/generate-markdown.mjs b/scripts/generate-markdown.mjs index d5f703b4d4..bac7bd9fec 100644 --- a/scripts/generate-markdown.mjs +++ b/scripts/generate-markdown.mjs @@ -6,6 +6,10 @@ * * Usage: node scripts/generate-markdown.mjs [buildDir] * Default buildDir = build/site + * + * Each page's last_updated date comes from the page date map written by the + * page dates Antora extension (lib/antora-extension-page-dates.js). Set + * PAGE_DATES_FILE to read the map from somewhere other than the default. */ import { readdir, readFile, writeFile, mkdir } from 'node:fs/promises'; @@ -19,6 +23,11 @@ import { encode } from 'gpt-tokenizer'; // --------------------------------------------------------------------------- const BUILD_DIR = process.argv[2] || 'build/site'; +const PAGE_DATES_FILE = process.env.PAGE_DATES_FILE || '.cache/page-dates.json'; +const SITE_URL = 'https://www.tiny.cloud/docs'; + +// Not a documentation page: the site's 404 page is served for any missing path. +const EXCLUDED_PAGES = new Set([ '404.html' ]); const ADMONITION_TYPES = [ 'note', 'warning', 'tip', 'important', 'caution' ]; @@ -122,6 +131,46 @@ const buildListItem = (td, doc) => { return li; }; +// The converter writes [text](href) only when a link holds a single, trimmed +// text node; any other link (inline code, emphasis, surrounding whitespace) +// is written out as a raw element. Flatten each link's content to +// one text node, keeping inline code and emphasis as markdown. +const codeSpan = (text) => + text.includes('`') ? '`` ' + text + ' ``' : '`' + text + '`'; + +const inlineMarkdown = (node) => { + if (node.nodeType === node.TEXT_NODE) return node.textContent; + if (node.nodeType !== node.ELEMENT_NODE) return ''; + + const inner = () => [ ...node.childNodes ].map(inlineMarkdown).join(''); + switch (node.tagName.toLowerCase()) { + case 'code': return codeSpan(node.textContent); + case 'strong': case 'b': return inner().trim() ? `**${inner().trim()}**` : ''; + case 'em': case 'i': return inner().trim() ? `*${inner().trim()}*` : ''; + case 'img': return `![${node.getAttribute('alt') ?? ''}](${node.getAttribute('src') ?? ''})`; + case 'br': return ' '; + default: return inner(); + } +}; + +const flattenLinkContent = (article, doc) => { + // Inline anchors ([[id]]) carry no href and mean nothing in markdown. + article.querySelectorAll('a:not([href])').forEach((anchor) => { + anchor.replaceWith(...anchor.childNodes); + }); + + article.querySelectorAll('a[href]').forEach((link) => { + if (link.closest('pre')) return; + + const text = inlineMarkdown(link).replace(/\s+/g, ' ').trim(); + if (text) { + link.replaceChildren(doc.createTextNode(text)); + } else { + link.remove(); + } + }); +}; + const rewriteCardTables = (article, doc) => { article.querySelectorAll('table.tableblock').forEach((table) => { if (!isCardLayoutTable(table)) return; @@ -148,6 +197,7 @@ const TRANSFORMS = [ rewriteLiveDemos, stripHeadingAnchors, rewriteCardTables, + flattenLinkContent, ]; const preprocess = (articleEl, doc) => { @@ -171,10 +221,22 @@ const D2M_OPTIONS = (dom) => ({ const fixBlankAnchors = (md) => md.replace(/about:blank#/g, '#'); +// Occurrences of " { + const walker = dom.window.document.createTreeWalker(article, dom.window.NodeFilter.SHOW_TEXT); + let count = 0; + for (let node = walker.nextNode(); node; node = walker.nextNode()) { + if (!node.parentElement?.closest('pre, code')) { + count += (node.textContent.match(/ { const article = preprocess(articleEl, dom.window.document); const raw = convertHtmlToMarkdown(article.innerHTML, D2M_OPTIONS(dom)); - return fixBlankAnchors(raw); + return { markdown: fixBlankAnchors(raw), literalLinkText: countLiteralLinkText(article, dom) }; }; // --------------------------------------------------------------------------- @@ -182,10 +244,21 @@ const toMarkdown = (articleEl, dom) => { // --------------------------------------------------------------------------- const escapeYaml = (s) => - s.replace(/\\/g, '\\\\').replace(/"/g, '\\"'); - -const buildFrontmatter = (title, tokens) => - `---\ntitle: "${escapeYaml(title)}"\ntokens: ${tokens}\n---\n`; + s.replace(/\\/g, '\\\\').replace(/"/g, '\\"').replace(/\s*\n\s*/g, ' '); + +const buildFrontmatter = (page) => + [ + '---', + `title: "${escapeYaml(page.title)}"`, + `description: "${escapeYaml(page.description)}"`, + `canonical_url: "${escapeYaml(page.canonical_url)}"`, + `md_url: "${escapeYaml(page.md_url)}"`, + `version: "${escapeYaml(page.version)}"`, + `last_updated: "${escapeYaml(page.last_updated)}"`, + `tokens: ${page.tokens}`, + '---', + '', + ].join('\n'); // --------------------------------------------------------------------------- // Title extraction @@ -196,6 +269,50 @@ const extractTitle = (doc) => ?? doc.querySelector('title')?.textContent?.trim()?.replace(/ \|.*$/, '') ?? 'Untitled'; +const extractDescription = (doc) => + doc.querySelector('meta[name="description"]')?.getAttribute('content')?.trim() ?? ''; + +// Antora points an older version's canonical URL at the newest version of the +// page, so the markdown URL is derived from the page's own path instead. +const extractCanonicalUrl = (doc, pagePath) => + doc.querySelector('link[rel="canonical"]')?.getAttribute('href') ?? SITE_URL + pagePath; + +// /tinymce/// +const versionOf = (pagePath) => pagePath.split('/')[2] ?? ''; + +// --------------------------------------------------------------------------- +// Page dates +// --------------------------------------------------------------------------- + +const loadPageDates = async () => { + try { + return JSON.parse(await readFile(PAGE_DATES_FILE, 'utf-8')).pages; + } catch (err) { + throw new Error(`Cannot read the page date map at ${PAGE_DATES_FILE} (${err.message}). ` + + 'It is written by the page dates Antora extension; build the site with the playbook first.'); + } +}; + +// --------------------------------------------------------------------------- +// Checks +// --------------------------------------------------------------------------- + +// Count raw elements outside fenced code blocks and inline code spans. +// Code examples may legitimately contain HTML links, and a page may show one as +// literal text; converted links must not. +const countRawLinks = (markdown, literalLinkText) => { + let inFence = false; + let count = 0; + for (const line of markdown.split('\n')) { + if (/^\s*(```|~~~)/.test(line)) { + inFence = !inFence; + } else if (!inFence) { + count += (line.replace(/`[^`]*`/g, '').match(/ { const full = join(dir, entry.name); return entry.isDirectory() ? collectHtmlFiles(full) - : entry.name.endsWith('.html') ? [ full ] : []; + : entry.name.endsWith('.html') && !(dir === BUILD_DIR && EXCLUDED_PAGES.has(entry.name)) ? [ full ] : []; }) ); return nested.flat(); @@ -217,23 +334,35 @@ const collectHtmlFiles = async (dir) => { // Single-page conversion // --------------------------------------------------------------------------- -const convertPage = async (htmlPath) => { +const convertPage = async (htmlPath, pageDates) => { const html = await readFile(htmlPath, 'utf-8'); const dom = new JSDOM(html); - const articleEl = dom.window.document.querySelector('article.doc'); + const doc = dom.window.document; + const articleEl = doc.querySelector('article.doc'); if (!articleEl) return null; - const title = extractTitle(dom.window.document); - const markdown = toMarkdown(articleEl, dom); - const tokens = encode(markdown, { allowedSpecial: 'all' }).length; - const content = buildFrontmatter(title, tokens) + markdown + '\n'; - const mdPath = htmlPath.replace(/\.html$/, '.md'); - - await mkdir(dirname(mdPath), { recursive: true }); - await writeFile(mdPath, content, 'utf-8'); + const path = '/' + relative(BUILD_DIR, dirname(htmlPath)) + '/'; + const { markdown, literalLinkText } = toMarkdown(articleEl, dom); + const page = { + path, + title: extractTitle(doc), + description: extractDescription(doc), + canonical_url: extractCanonicalUrl(doc, path), + md_url: SITE_URL + path + 'index.md', + version: versionOf(path), + last_updated: pageDates[path]?.last_updated, + tokens: encode(markdown, { allowedSpecial: 'all' }).length, + rawLinks: countRawLinks(markdown, literalLinkText), + }; + + if (page.last_updated) { + const mdPath = htmlPath.replace(/\.html$/, '.md'); + await mkdir(dirname(mdPath), { recursive: true }); + await writeFile(mdPath, buildFrontmatter(page) + markdown + '\n', 'utf-8'); + } - return { path: '/' + relative(BUILD_DIR, dirname(htmlPath)) + '/', tokens }; + return page; }; // --------------------------------------------------------------------------- @@ -243,14 +372,26 @@ const convertPage = async (htmlPath) => { const main = async () => { console.log(`Generating markdown siblings in ${BUILD_DIR} …`); + const pageDates = await loadPageDates(); const htmlFiles = await collectHtmlFiles(BUILD_DIR); const pages = []; for (const htmlPath of htmlFiles) { - const result = await convertPage(htmlPath); + const result = await convertPage(htmlPath, pageDates); if (result) pages.push(result); } + const undated = pages.filter((page) => !page.last_updated).map((page) => page.path); + if (undated.length) { + throw new Error(`${undated.length} page(s) have no entry in the page date map:\n ${undated.join('\n ')}`); + } + + const withRawLinks = pages.filter((page) => page.rawLinks > 0); + if (withRawLinks.length) { + throw new Error(`${withRawLinks.length} page(s) contain raw elements outside code:\n ` + + withRawLinks.map((page) => `${page.path} (${page.rawLinks})`).join('\n ')); + } + const manifest = Object.fromEntries(pages.map(({ path, tokens }) => [ path, tokens ])); const manifestPath = join(BUILD_DIR, '_markdown-manifest.json'); await writeFile(manifestPath, JSON.stringify(manifest, null, 2) + '\n', 'utf-8'); From 228a215e726c4dd051ece695fb96d05dd7ebdac5 Mon Sep 17 00:00:00 2001 From: Karl Kemister-Sheppard Date: Wed, 23 Sep 2026 17:02:29 +1000 Subject: [PATCH 3/8] TINYDOC-3613: Version the markdown manifest and give each entry its citation fields. --- scripts/generate-markdown.mjs | 14 +++++++++++++- 1 file changed, 13 insertions(+), 1 deletion(-) diff --git a/scripts/generate-markdown.mjs b/scripts/generate-markdown.mjs index bac7bd9fec..445e85850f 100644 --- a/scripts/generate-markdown.mjs +++ b/scripts/generate-markdown.mjs @@ -24,6 +24,9 @@ import { encode } from 'gpt-tokenizer'; const BUILD_DIR = process.argv[2] || 'build/site'; const PAGE_DATES_FILE = process.env.PAGE_DATES_FILE || '.cache/page-dates.json'; + +// Version of the _markdown-manifest.json shape. Increment on any breaking change. +const MANIFEST_SCHEMA = 1; const SITE_URL = 'https://www.tiny.cloud/docs'; // Not a documentation page: the site's 404 page is served for any missing path. @@ -392,7 +395,16 @@ const main = async () => { withRawLinks.map((page) => `${page.path} (${page.rawLinks})`).join('\n ')); } - const manifest = Object.fromEntries(pages.map(({ path, tokens }) => [ path, tokens ])); + const manifest = { + schema: MANIFEST_SCHEMA, + generated: new Date().toISOString().replace(/\.\d{3}Z$/, 'Z'), + pages: Object.fromEntries( + pages + .sort((a, b) => a.path.localeCompare(b.path)) + .map(({ path, title, description, md_url, version, last_updated, tokens }) => + [ path, { title, description, md_url, version, last_updated, tokens } ]) + ), + }; const manifestPath = join(BUILD_DIR, '_markdown-manifest.json'); await writeFile(manifestPath, JSON.stringify(manifest, null, 2) + '\n', 'utf-8'); From 726eaca383bf95e458485f013010fb5a2a787072 Mon Sep 17 00:00:00 2001 From: Karl Kemister-Sheppard Date: Wed, 23 Sep 2026 17:06:58 +1000 Subject: [PATCH 4/8] TINYDOC-3613: Read LLM file titles from the generated markdown instead of fetching production. --- -scripts/generate-llm-files.js | 504 ++++----------------------- .github/workflows/deploy_docs_v2.yml | 6 +- package.json | 6 +- 3 files changed, 74 insertions(+), 442 deletions(-) diff --git a/-scripts/generate-llm-files.js b/-scripts/generate-llm-files.js index 7215092f33..c23d78d499 100755 --- a/-scripts/generate-llm-files.js +++ b/-scripts/generate-llm-files.js @@ -1,32 +1,23 @@ #!/usr/bin/env node /** - * Script to generate llms.txt and llms-full.txt files from sitemap.xml - * + * Generates llms.txt and llms-full.txt from a built site. + * * Usage: - * node -scripts/generate-llm-files.js [sitemap-path-or-url] - * - * Defaults to build/site/sitemap.xml (local) or can use remote URL + * node -scripts/generate-llm-files.js [buildDir] + * + * buildDir defaults to build/site. Run it after the markdown step + * (yarn build:markdown), because page titles are read from each page's + * generated index.md rather than fetched from the published site. */ const fs = require('fs'); const path = require('path'); -const https = require('https'); -const http = require('http'); -const sanitizeHtml = require('sanitize-html'); const BASE_URL = 'https://www.tiny.cloud/docs/tinymce/latest'; const DOCS_ROOT_URL = 'https://www.tiny.cloud/docs'; -const ATTACHMENTS_DIR = path.join(__dirname, '../modules/ROOT/attachments'); const DEFAULT_BUILD_DIR = path.join(__dirname, '../build/site'); -// Resolve where the sitemap is read from and where the generated files are written. -// The argument may be a build directory, a sitemap file, or a remote sitemap URL: -// generate-llm-files build/site -> reads build/site/sitemap.xml, writes build/site -// generate-llm-files build/site/sitemap.xml -> writes build/site -// generate-llm-files https://.../sitemap.xml -> writes modules/ROOT/attachments -// The remote form preserves the manual workflow, where the output is reviewed and -// committed. The local forms write into the build so the deploy needs no commit. // The generated files describe a single documentation version — parseSitemap() keeps // only the URLs under BASE_URL — so they must be published to that version's // attachments directory and no other. The path is derived from BASE_URL rather than @@ -52,12 +43,6 @@ function versionAttachmentDir(buildDir) { // root. The fallback keeps /docs/llms.txt correct — and keeps copy-llms-files.sh's // existence check passing — if the version segment ever changes. function writeGenerated(outputDir, filename, contents) { - if (path.resolve(outputDir) === path.resolve(ATTACHMENTS_DIR)) { - const target = path.join(outputDir, filename); - fs.writeFileSync(target, contents); - return [target]; - } - const attachmentDir = versionAttachmentDir(outputDir); if (!attachmentDir) { @@ -72,24 +57,6 @@ function writeGenerated(outputDir, filename, contents) { return [target]; } -function resolveTargets(arg) { - if (!arg) { - return { sitemap: path.join(DEFAULT_BUILD_DIR, 'sitemap.xml'), outputDir: DEFAULT_BUILD_DIR }; - } - - if (arg.startsWith('http://') || arg.startsWith('https://')) { - return { sitemap: arg, outputDir: ATTACHMENTS_DIR }; - } - - const resolved = path.resolve(arg); - - if (fs.existsSync(resolved) && fs.statSync(resolved).isDirectory()) { - return { sitemap: path.join(resolved, 'sitemap.xml'), outputDir: resolved }; - } - - return { sitemap: resolved, outputDir: path.dirname(resolved) }; -} - // Convert a trailing-slash doc URL to its Markdown endpoint. // e.g. https://www.tiny.cloud/docs/tinymce/latest/basic-setup/ // -> https://www.tiny.cloud/docs/tinymce/latest/basic-setup/index.md @@ -102,38 +69,14 @@ function toMarkdownEndpoints(content) { ); } -// Fetch sitemap from URL or file -async function getSitemap(source) { - if (source.startsWith('http://') || source.startsWith('https://')) { - return new Promise((resolve, reject) => { - const client = source.startsWith('https') ? https : http; - client.get(source, (res) => { - let data = ''; - res.on('data', (chunk) => { data += chunk; }); - res.on('end', () => resolve(data)); - }).on('error', reject); - }); - } else { - // Validate file path to prevent path traversal - const resolvedPath = path.resolve(source); - const projectRoot = path.resolve(__dirname, '..'); - - // Ensure the resolved path is within the project directory - if (!resolvedPath.startsWith(projectRoot)) { - throw new Error(`Invalid sitemap path: ${source}. Path must be within the project directory.`); - } - - if (!fs.existsSync(resolvedPath)) { - throw new Error(`Sitemap not found: ${source}\nPlease run 'yarn antora ./antora-playbook.yml' first to generate the site, or provide a URL.`); - } - - // Only allow .xml files - if (!resolvedPath.endsWith('.xml')) { - throw new Error(`Invalid file type: ${source}. Only .xml files are allowed.`); - } - - return fs.readFileSync(resolvedPath, 'utf8'); +function readSitemap(buildDir) { + const sitemapPath = path.join(buildDir, 'sitemap.xml'); + + if (!fs.existsSync(sitemapPath)) { + throw new Error(`Sitemap not found: ${sitemapPath}\nBuild the site first: yarn antora ./antora-playbook.yml`); } + + return fs.readFileSync(sitemapPath, 'utf8'); } // Parse sitemap.xml to extract all URLs @@ -162,331 +105,49 @@ function getUrlPath(url) { return match ? match[1].replace(/\/$/, '') : ''; } -// Fetch H1 title from a page URL -async function fetchH1Title(url) { - return new Promise((resolve) => { - // Validate URL to prevent SSRF - only allow tiny.cloud domains - if (!url.startsWith('https://www.tiny.cloud/') && !url.startsWith('http://www.tiny.cloud/')) { - resolve(null); - return; - } - - const client = url.startsWith('https') ? https : http; - - // codeql[js/file-access-to-http]: URL is validated to only allow tiny.cloud domains, preventing SSRF attacks - const req = client.get(url, (res) => { - // Check for error status codes (404, 500, etc.) - if (res.statusCode >= 400) { - resolve(null); - return; - } - - let data = ''; - - res.on('data', (chunk) => { data += chunk; }); - - res.on('end', () => { - try { - // Extract H1 tag using regex - look for

or

- const h1Match = data.match(/]*>(.*?)<\/h1>/i); - - if (h1Match && h1Match[1]) { - // Clean up the title using a well-tested HTML sanitization library - let title = h1Match[1]; - - // First, use sanitize-html to strip all HTML tags and attributes while preserving text. - // This avoids fragile hand-written tag parsing and multi-character sanitization pitfalls. - title = sanitizeHtml(title, { - allowedTags: [], - allowedAttributes: {}, - textFilter: (text) => text - }); - - // Then, defensively remove any remaining angle brackets and script/protocol keywords - // to ensure no HTML-like or script-related fragments remain. - title = title - .replace(/[<>]/g, '') - .replace(/(?:javascript|data|vbscript)\s*:?/gi, '') - .replace(/\bscript\b/gi, '') - .trim(); - - // At this point, title is plain text with no angle brackets or script/protocol keywords - // Additionally, defensively strip any residual script/protocol keywords that could - // be used for injection even after angle brackets and colons have been removed - title = title.replace(/\b(?:script|javascript|vbscript|data)\b/gi, ''); - - // Decode HTML entities safely - decode all entities to plain text - // Order matters: decode '&' last to avoid double-unescaping - // Decode all specific entities first, then & at the end - title = title - .replace(/ /g, ' ') - .replace(/"/g, '"') - .replace(/'/g, "'") - .replace(/’/g, "'") // Right single quotation mark (apostrophe) - .replace(/‘/g, "'") // Left single quotation mark - .replace(/“/g, '"') // Left double quotation mark - .replace(/”/g, '"') // Right double quotation mark - .replace(/–/g, '–') // En dash - .replace(/—/g, '—') // Em dash - .replace(/ /g, ' ') // Non-breaking space - // Decode numeric entities ({ format) - safe as we've already removed HTML tags - .replace(/&#(\d+);/g, (match, dec) => { - const code = parseInt(dec, 10); - // Only decode safe character codes (printable ASCII and valid Unicode) - if ((code >= 32 && code <= 126) || (code >= 160 && code <= 1114111)) { - return String.fromCharCode(code); - } - return match; // Keep entity if unsafe - }) - // Decode hex entities ( format) - .replace(/&#x([0-9a-fA-F]+);/gi, (match, hex) => { - const code = parseInt(hex, 16); - // Only decode safe character codes - if ((code >= 32 && code <= 126) || (code >= 160 && code <= 1114111)) { - return String.fromCharCode(code); - } - return match; // Keep entity if unsafe - }) - // Decode remaining named entities (after numeric/hex to avoid conflicts) - .replace(/</g, '<') // Safe: HTML tags already removed - .replace(/>/g, '>') // Safe: HTML tags already removed - .replace(/&/g, '&') // Decode '&' last to prevent double-unescaping - .trim(); - - // Remove extra whitespace - title = title.replace(/\s+/g, ' '); - - // Filter out error page titles - const errorPatterns = [ - /couldn't find the page/i, - /page not found/i, - /404/i, - /error/i, - /not found/i - ]; - - if (errorPatterns.some(pattern => pattern.test(title))) { - resolve(null); - return; - } - - if (title) { - resolve(title); - } else { - resolve(null); - } - } else { - // Fallback to generated title if no H1 found - resolve(null); - } - } catch (error) { - // If parsing fails, fallback to generated title - resolve(null); - } - }); - }); - - req.on('error', (error) => { - // If fetch fails, fallback to generated title - resolve(null); - }); - - req.setTimeout(10000, () => { - req.destroy(); - resolve(null); - }); - }); +// The markdown sibling of a page URL, in the build directory. +function markdownPathFor(buildDir, url) { + return path.join(buildDir, url.slice(DOCS_ROOT_URL.length), 'index.md'); } -// Batch fetch H1 titles with rate limiting -async function fetchH1TitlesBatch(urls, batchSize = 10, delay = 100) { - const results = new Map(); - - for (let i = 0; i < urls.length; i += batchSize) { - const batch = urls.slice(i, i + batchSize); - const promises = batch.map(async (url) => { - const title = await fetchH1Title(url); - return { url, title }; - }); - - const batchResults = await Promise.all(promises); - batchResults.forEach(({ url, title }) => { - results.set(url, title); - }); - - // Rate limiting - wait between batches - if (i + batchSize < urls.length) { - await new Promise(resolve => setTimeout(resolve, delay)); +// Parse the frontmatter written by scripts/generate-markdown.mjs: one +// `key: "string"` or `key: number` pair per line, strings escaped by escapeYaml. +function parseFrontmatter(markdown) { + const match = markdown.match(/^---\n([\s\S]*?)\n---\n/); + if (!match) return null; + + const fields = {}; + for (const line of match[1].split('\n')) { + const pair = line.match(/^([a-z_]+): (?:"((?:[^"\\]|\\.)*)"|(\d+))$/); + if (pair) { + fields[pair[1]] = pair[3] !== undefined ? Number(pair[3]) : pair[2].replace(/\\(.)/g, '$1'); } } - - return results; + return { fields, body: markdown.slice(match[0].length) }; } -// Generate a descriptive title from URL path -// NOTE: This script does A LOT of string matching against URL paths. If you encounter -// weirdness (e.g. wrong titles, wrong categories), search for the relevant path strings -// in this file to locate and fix the matching logic. -function generateTitleFromPath(urlPath) { - if (!urlPath) return 'Home'; - - const pathTitleMap = { - '': 'Home', - 'index': 'Home', - 'getting-started': 'Getting Started', - 'introduction-to-tinymce': 'Introduction to TinyMCE', - 'installation': 'Installation', - 'cloud-quick-start': 'Cloud Quick Start', - 'npm-projects': 'NPM Projects Quick Start', - 'zip-install': 'ZIP Installation Quick Start', - 'installation-cloud': 'Cloud', - 'installation-self-hosted': 'Self-hosted', - 'installation-zip': 'ZIP', - 'basic-setup': 'Basic Setup', - 'work-with-plugins': 'Using plugins to extend TinyMCE', - 'filter-content': 'Content Filtering', - 'content-filtering': 'Content filtering', - 'localize-your-language': 'Localization', - 'spell-checking': 'Spell Checking', - 'editor-content-css': 'CSS for rendering content', - 'url-handling': 'URL Handling', - 'plugins': 'Plugins', - 'table': 'Table', - 'table-options': 'Table options', - 'image': 'Image', - 'link': 'Link', - 'lists': 'Lists', - 'code': 'Code', - 'codesample': 'Code Sample', - 'advtable': 'Enhanced Tables', - 'advcode': 'Enhanced Code Editor', - 'editimage': 'Image Editing', - 'linkchecker': 'Link Checker', - 'a11ychecker': 'Accessibility Checker', - 'upgrading': 'Upgrading TinyMCE', - 'migration-guides': 'Migration Guides Overview', - 'migration-from-7x': 'Migration from 7.x', - 'migration-from-6x': 'Migration from 6.x', - 'migration-from-6x-to-8x': 'Migration from 6.x to 8.x', - 'migration-from-5x': 'Migration from 5.x', - 'migration-from-5x-to-8x': 'Migration from 5.x to 8.x', - 'migration-from-4x': 'Migration from 4.x', - 'migration-from-4x-to-8x': 'Migration from 4.x to 8.x', - 'migration-from-froala': 'Migrating from Froala', - 'examples': 'Examples', - 'how-to-guides': 'How To Guides', - 'release-notes': 'Release Notes', - 'changelog': 'Changelog', - 'accessibility': 'Accessibility', - 'security': 'Security guide', - 'support': 'Support', - 'react': 'React', - 'react-cloud': 'React Cloud', - 'react-pm-host': 'React Package Manager', - 'react-pm-bundle': 'React Package Manager (with bundling)', - 'react-zip-host': 'React ZIP', - 'react-zip-bundle': 'React ZIP (with bundling)', - 'react-ref': 'Technical reference (React)', - 'vue': 'Vue.js', - 'vue-cloud': 'Vue Cloud', - 'vue-pm': 'Vue Package Manager', - 'vue-pm-bundle': 'Vue Package Manager (with bundling)', - 'vue-zip': 'Vue ZIP', - 'vue-ref': 'Technical reference (Vue)', - 'angular': 'Angular', - 'angular-cloud': 'Angular Cloud', - 'angular-pm': 'Angular Package Manager', - 'angular-pm-bundle': 'Angular Package Manager (with bundling)', - 'angular-zip': 'Angular ZIP', - 'angular-zip-bundle': 'Angular ZIP (with bundling)', - 'angular-ref': 'Technical reference (Angular)', - 'blazor': 'Blazor', - 'blazor-cloud': 'Blazor Cloud', - 'blazor-pm': 'Blazor Package Manager', - 'blazor-zip': 'Blazor ZIP', - 'blazor-ref': 'Technical reference (Blazor)', - 'svelte': 'Svelte', - 'svelte-cloud': 'Svelte Cloud', - 'svelte-pm': 'Svelte Package Manager', - 'svelte-pm-bundle': 'Svelte Package Manager (with bundling)', - 'svelte-zip': 'Svelte ZIP', - 'svelte-ref': 'Technical reference (Svelte)', - 'webcomponent': 'Web Component', - 'webcomponent-cloud': 'Web Component Cloud', - 'webcomponent-pm': 'Web Component Package Manager', - 'webcomponent-zip': 'Web Component ZIP', - 'webcomponent-ref': 'Technical reference (Web Component)', - 'jquery': 'jQuery', - 'jquery-cloud': 'jQuery Cloud', - 'jquery-pm': 'jQuery Package Manager', - 'django': 'Django', - 'django-cloud': 'Django Cloud', - 'django-zip': 'Django ZIP', - 'laravel': 'Laravel', - 'laravel-tiny-cloud': 'Laravel Cloud', - 'laravel-composer-install': 'Laravel Composer', - 'laravel-zip-install': 'Laravel ZIP', - 'rails': 'Ruby on Rails', - 'rails-cloud': 'Rails Cloud', - 'rails-third-party': 'Rails Package Manager', - 'rails-zip': 'Rails ZIP', - 'expressjs-pm': 'Node.js + Express', - 'bootstrap': 'Bootstrap', - 'bootstrap-cloud': 'Bootstrap Cloud', - 'bootstrap-zip': 'Bootstrap ZIP', - 'php-projects': 'PHP Projects', - 'dotnet-projects': '.NET Projects', - 'wordpress': 'WordPress', - 'shadow-dom': 'Shadow DOM', - 'swing': 'Java Swing' - }; - - if (pathTitleMap[urlPath]) { - return pathTitleMap[urlPath]; +// Read every page's generated markdown. Every sitemap page must have one: the +// markdown step skips a page with no article element, and a page missing here +// would silently drop out of both LLM files. +function readPages(buildDir, urls) { + const missing = []; + const pages = new Map(); + + for (const url of urls) { + const mdPath = markdownPathFor(buildDir, url); + const parsed = fs.existsSync(mdPath) ? parseFrontmatter(fs.readFileSync(mdPath, 'utf8')) : null; + if (!parsed || !parsed.fields.title) { + missing.push(url); + } else { + pages.set(url, parsed); + } } - // Handle release notes - if (urlPath.match(/^\d+\.\d+\.\d+-release-notes$/)) { - const version = urlPath.replace('-release-notes', ''); - return `TinyMCE ${version}`; + if (missing.length) { + throw new Error(`${missing.length} sitemap page(s) have no generated markdown (run yarn build:markdown first):\n ${missing.join('\n ')}`); } - // Handle bundling entries (order matters: more specific keys first) - const bundlingTitles = { - 'webpack-cjs-npm': 'CommonJS and NPM (Webpack)', - 'webpack-es6-npm': 'ES6 and NPM (Webpack)', - 'webpack-cjs-download': 'CommonJS and a .zip archive (Webpack)', - 'webpack-es6-download': 'ES6 and a .zip archive (Webpack)', - 'rollup-es6-npm': 'ES6 and npm (Rollup)', - 'rollup-es6-download': 'ES6 and a .zip archive (Rollup)', - 'vite-es6-npm': 'ES6 and NPM (Vite)', - 'browserify-cjs-npm': 'CommonJS and npm (Browserify)', - 'browserify-cjs-download': 'CommonJS and a .zip archive (Browserify)' - }; - const bundlingMatch = Object.keys(bundlingTitles).find((k) => urlPath.includes(k)); - if (bundlingMatch) { - return bundlingTitles[bundlingMatch]; - } - - // Handle API references - if (urlPath.startsWith('apis/')) { - const apiPath = urlPath.replace('apis/', ''); - return apiPath.split('/').pop().replace(/\.adoc$/, ''); - } - - // Convert kebab-case to Title Case - const words = urlPath - .replace(/-/g, ' ') - .split(' ') - .map(word => { - // Handle acronyms - if (word.toLowerCase() === 'npm' || word.toLowerCase() === 'cjs' || word.toLowerCase() === 'es6') { - return word.toUpperCase(); - } - return word.charAt(0).toUpperCase() + word.slice(1).toLowerCase(); - }); - - return words.join(' '); + return pages; } // Categorize URL based on path @@ -762,44 +423,20 @@ function makeTitlesUnique(entries) { } // Generate llms-full.txt -async function generateLLMsFullTxt(urls) { - console.log(`Fetching H1 titles from ${urls.length} pages...`); - console.log('This may take a few minutes...'); - - // Fetch H1 titles from all pages - const h1Titles = await fetchH1TitlesBatch(urls, 10, 100); - - let fetchedCount = 0; - let fallbackCount = 0; - - // Process all URLs +function generateLLMsFullTxt(urls, pages) { const entries = urls.map(url => { const urlPath = getUrlPath(url); - const fetchedTitle = h1Titles.get(url); - const generatedTitle = generateTitleFromPath(urlPath); - - // Use fetched H1 if available, otherwise fallback to generated - const title = fetchedTitle || generatedTitle; - - if (fetchedTitle) { - fetchedCount++; - } else { - fallbackCount++; - } - const catInfo = categorizeUrl(urlPath); return { url, urlPath, - title, + title: pages.get(url).fields.title, category: catInfo.category, subcategory: catInfo.subcategory }; }); - console.log(`✓ Fetched ${fetchedCount} H1 titles, ${fallbackCount} used fallback titles`); - // Remove duplicate URLs (keep first occurrence) - should already be unique from parseSitemap, but double-check const seenUrls = new Set(); const uniqueEntries = entries.filter(entry => { @@ -1324,38 +961,35 @@ For a complete list of all ${urls.length} documentation pages, see [llms-full.tx } // Main execution -async function main() { - const { sitemap: sitemapSource, outputDir } = resolveTargets(process.argv[2]); +function main() { + const buildDir = path.resolve(process.argv[2] || DEFAULT_BUILD_DIR); console.log('Generating LLM files...'); - console.log(`Using sitemap: ${sitemapSource}`); - console.log(`Writing to: ${outputDir}`); + console.log(`Using build: ${buildDir}`); - if (!fs.existsSync(outputDir)) { - throw new Error(`Output directory does not exist: ${outputDir}`); + if (!fs.existsSync(buildDir) || !fs.statSync(buildDir).isDirectory()) { + throw new Error(`Build directory does not exist: ${buildDir}`); } - try { - const sitemapContent = await getSitemap(sitemapSource); - const urls = parseSitemap(sitemapContent); - console.log(`Found ${urls.length} unique URLs in sitemap`); - - const llmsTxt = toMarkdownEndpoints(generateLLMsTxt(urls)); - const llmsTxtPaths = writeGenerated(outputDir, 'llms.txt', llmsTxt); - llmsTxtPaths.forEach((p) => console.log(`✓ Wrote ${p}`)); + const urls = parseSitemap(readSitemap(buildDir)); + console.log(`Found ${urls.length} unique URLs in sitemap`); - const llmsFullTxt = toMarkdownEndpoints(await generateLLMsFullTxt(urls)); - const llmsFullPaths = writeGenerated(outputDir, 'llms-full.txt', llmsFullTxt); - llmsFullPaths.forEach((p) => console.log(`✓ Wrote ${p}`)); - + const pages = readPages(buildDir, urls); + + const llmsTxt = toMarkdownEndpoints(generateLLMsTxt(urls)); + writeGenerated(buildDir, 'llms.txt', llmsTxt).forEach((p) => console.log(`✓ Wrote ${p}`)); + + const llmsFullTxt = toMarkdownEndpoints(generateLLMsFullTxt(urls, pages)); + writeGenerated(buildDir, 'llms-full.txt', llmsFullTxt).forEach((p) => console.log(`✓ Wrote ${p}`)); +} + +if (require.main === module) { + try { + main(); } catch (error) { console.error('Error:', error.message); process.exit(1); } } -if (require.main === module) { - main(); -} - -module.exports = { generateLLMsTxt, generateLLMsFullTxt, parseSitemap, getSitemap, fetchH1Title, fetchH1TitlesBatch }; +module.exports = { generateLLMsTxt, generateLLMsFullTxt, parseSitemap, parseFrontmatter }; diff --git a/.github/workflows/deploy_docs_v2.yml b/.github/workflows/deploy_docs_v2.yml index 1e3b13253f..dd0f83ee29 100644 --- a/.github/workflows/deploy_docs_v2.yml +++ b/.github/workflows/deploy_docs_v2.yml @@ -50,15 +50,15 @@ jobs: - name: Build Website run: yarn antora ./antora-playbook.yml + - name: Generate markdown for agents + run: yarn build:markdown build/site + - name: Generate LLM files run: yarn generate-llm-files build/site - name: Copy llms.txt files to root run: ./-scripts/copy-llms-files.sh build/site - - name: Generate markdown for agents - run: yarn build:markdown build/site - - name: Rename site folder to docs run: | mv ./build/site ./build/docs diff --git a/package.json b/package.json index 86f08b005c..439db09bca 100644 --- a/package.json +++ b/package.json @@ -17,8 +17,7 @@ "server": "http-server build/site/ --port 4000", "serve": "npm-run-all -p nodemon-dev server", "start": "yarn clean && yarn serve", - "generate-llm-files": "node ./-scripts/generate-llm-files.js", - "generate-llm-files-from-url": "node ./-scripts/generate-llm-files.js https://www.tiny.cloud/docs/antora-sitemap.xml" + "generate-llm-files": "node ./-scripts/generate-llm-files.js" }, "author": "Tiny Technologies Inc", "license": "CC-BY-NC-SA-3.0", @@ -46,7 +45,6 @@ "http-server": "^14.1.1", "jsdom": "^24.1.0", "nodemon": "^3.1.10", - "npm-run-all": "^4.1.5", - "sanitize-html": "^2.13.0" + "npm-run-all": "^4.1.5" } } From 9eb3fb860d6e116f9addad4791bad7b7bd5e6680 Mon Sep 17 00:00:00 2001 From: Karl Kemister-Sheppard Date: Wed, 23 Sep 2026 17:13:37 +1000 Subject: [PATCH 5/8] TINYDOC-3613: Publish every page's content in llms-full.txt and fail the build on a short file. --- -scripts/generate-llm-files.js | 109 ++++++++++++++++++++++++++++++--- 1 file changed, 101 insertions(+), 8 deletions(-) diff --git a/-scripts/generate-llm-files.js b/-scripts/generate-llm-files.js index c23d78d499..5e0bfd72ff 100755 --- a/-scripts/generate-llm-files.js +++ b/-scripts/generate-llm-files.js @@ -18,6 +18,14 @@ const BASE_URL = 'https://www.tiny.cloud/docs/tinymce/latest'; const DOCS_ROOT_URL = 'https://www.tiny.cloud/docs'; const DEFAULT_BUILD_DIR = path.join(__dirname, '../build/site'); +// llms-full.txt carries every page's content, measured at about 3.5 MB. Anything +// under this floor means content is missing, so the build fails rather than +// publishing an index of links. +const LLMS_FULL_MIN_BYTES = 2 * 1024 * 1024; +// A phrase from the body of the basic-setup page, not from its URL or title, so a +// file of links alone cannot pass. +const LLMS_FULL_SENTINEL = 'The four most common configuration options for TinyMCE are'; + // The generated files describe a single documentation version — parseSitemap() keeps // only the URLs under BASE_URL — so they must be published to that version's // attachments directory and no other. The path is derived from BASE_URL rather than @@ -150,6 +158,63 @@ function readPages(buildDir, urls) { return pages; } +// --------------------------------------------------------------------------- +// Page content for llms-full.txt +// --------------------------------------------------------------------------- + +// Apply fn to each line outside fenced code blocks. +function mapProseLines(markdown, fn) { + let inFence = false; + return markdown.split('\n').map((line) => { + if (/^\s*(```|~~~)/.test(line)) { + inFence = !inFence; + return line; + } + return inFence ? line : fn(line); + }).join('\n'); +} + +// Relative links in a page body resolve against that page's URL, so once the +// page is inlined they must be made absolute. Inline code spans are left alone. +function absolutizeLinks(markdown, pageUrl) { + return mapProseLines(markdown, (line) => + line.split(/(`[^`]*`)/).map((part, i) => i % 2 ? part : part.replace( + /(\]\()([^)\s]+)(\))/g, + (match, open, target, close) => { + if (/^[a-z][a-z0-9+.-]*:/i.test(target)) return match; + try { + return open + new URL(target, pageUrl).href + close; + } catch { + return match; + } + } + )).join('') + ); +} + +// Each page is inlined under a level-2 heading, so its own headings move down one +// level, and its level-1 title is dropped in favour of the section header. +function demoteHeadings(markdown) { + return mapProseLines(markdown, (line) => { + const heading = line.match(/^(#{1,6}) (.*)$/); + if (!heading) return line; + return `${'#'.repeat(Math.min(heading[1].length + 1, 6))} ${heading[2]}`; + }); +} + +function renderPageSection({ fields, body }) { + const content = body.replace(/^\s*# [^\n]*\n+/, '').trim(); + return [ + `## ${fields.title}`, + '', + `Source: ${fields.canonical_url}`, + `Last updated: ${fields.last_updated}`, + '', + demoteHeadings(absolutizeLinks(content, fields.canonical_url)), + '', + ].join('\n'); +} + // Categorize URL based on path function categorizeUrl(urlPath) { // Getting Started & Installation @@ -423,7 +488,7 @@ function makeTitlesUnique(entries) { } // Generate llms-full.txt -function generateLLMsFullTxt(urls, pages) { +function generateLLMsFullTxt(urls, pages, generated) { const entries = urls.map(url => { const urlPath = getUrlPath(url); const catInfo = categorizeUrl(urlPath); @@ -482,6 +547,8 @@ function generateLLMsFullTxt(urls, pages) { // Build content let content = `# TinyMCE Documentation - Complete Reference +> The complete content of the TinyMCE 8 documentation, grouped by topic. Generated ${generated}. + ## Overview TinyMCE is a rich text editor that provides a WYSIWYG editing experience. The latest stable version is TinyMCE 8, released in July 2025. @@ -733,7 +800,7 @@ export default { // Complete Documentation Index content += `## Complete Documentation Index\n\n`; - content += `This section provides a complete list of all ${uniqueEntries.length} documentation pages available in TinyMCE 8, organized by category. This comprehensive index ensures LLMs have access to every documentation page, reducing the risk of hallucinations or missing important details.\n\n`; + content += `This section lists all ${uniqueEntries.length} documentation pages available in TinyMCE 8, organized by category. The full content of every page follows the index, in the same order, each under its own heading with its source URL and last updated date.\n\n`; // Output categories in specific order const categoryStructure = [ @@ -802,14 +869,38 @@ export default { } }); - return content; + // Page content, in index order: one level-1 heading per category, one level-2 + // heading per page. Only the index is rewritten to markdown endpoints; page + // bodies, including their code samples, are inlined as generated. + let pagesContent = ''; + categoryStructure.forEach(({ category, subcategory }) => { + const key = subcategory ? `${category}::${subcategory}` : category; + if (!categorized.has(key)) return; + + pagesContent += `\n# ${subcategory ? `${category}: ${subcategory}` : category}\n\n`; + categorized.get(key).forEach((entry) => { + pagesContent += renderPageSection(pages.get(entry.url)) + '\n'; + }); + }); + + return toMarkdownEndpoints(content) + pagesContent; +} + +// Fail the build rather than publish a file of links. +function assertFullText(llmsFullTxt) { + const bytes = Buffer.byteLength(llmsFullTxt); + const problems = []; + if (bytes < LLMS_FULL_MIN_BYTES) problems.push(`is ${bytes} bytes, under the ${LLMS_FULL_MIN_BYTES}-byte floor`); + if (!llmsFullTxt.includes(LLMS_FULL_SENTINEL)) problems.push(`does not contain the basic-setup sentinel "${LLMS_FULL_SENTINEL}"`); + if (/^---$/m.test(llmsFullTxt)) problems.push('contains a frontmatter fence (---)'); + if (problems.length) throw new Error(`llms-full.txt ${problems.join('; ')}`); } // Generate llms.txt (curated, simplified version) -function generateLLMsTxt(urls) { +function generateLLMsTxt(urls, generated) { return `# TinyMCE Documentation -> Rich text editor for web applications. The latest stable version is TinyMCE 8. +> Rich text editor for web applications. The latest stable version is TinyMCE 8. Generated ${generated}. TinyMCE is a powerful, flexible WYSIWYG rich text editor that can be integrated into any web application. @@ -955,7 +1046,7 @@ Add "use context7" to any prompt for live TinyMCE documentation lookups. ## Complete Documentation -For a complete list of all ${urls.length} documentation pages, see [llms-full.txt](${DOCS_ROOT_URL}/llms-full.txt). +For the full content of all ${urls.length} documentation pages in one file, see [llms-full.txt](${DOCS_ROOT_URL}/llms-full.txt). `; } @@ -975,11 +1066,13 @@ function main() { console.log(`Found ${urls.length} unique URLs in sitemap`); const pages = readPages(buildDir, urls); + const generated = new Date().toISOString().replace(/\.\d{3}Z$/, 'Z'); - const llmsTxt = toMarkdownEndpoints(generateLLMsTxt(urls)); + const llmsTxt = toMarkdownEndpoints(generateLLMsTxt(urls, generated)); writeGenerated(buildDir, 'llms.txt', llmsTxt).forEach((p) => console.log(`✓ Wrote ${p}`)); - const llmsFullTxt = toMarkdownEndpoints(generateLLMsFullTxt(urls, pages)); + const llmsFullTxt = generateLLMsFullTxt(urls, pages, generated); + assertFullText(llmsFullTxt); writeGenerated(buildDir, 'llms-full.txt', llmsFullTxt).forEach((p) => console.log(`✓ Wrote ${p}`)); } From a0b478cea4a1c20bc220309802f335bd44e7c63d Mon Sep 17 00:00:00 2001 From: Karl Kemister-Sheppard Date: Wed, 23 Sep 2026 17:22:44 +1000 Subject: [PATCH 6/8] TINYDOC-3613: Publish AGENTS.md, sitemap.md and changes.json, and state the license key requirement in the LLM files. --- -scripts/generate-llm-files.js | 243 ++++++++++++++++++++++++++++++--- 1 file changed, 221 insertions(+), 22 deletions(-) diff --git a/-scripts/generate-llm-files.js b/-scripts/generate-llm-files.js index 5e0bfd72ff..8d5001eb8c 100755 --- a/-scripts/generate-llm-files.js +++ b/-scripts/generate-llm-files.js @@ -723,6 +723,7 @@ tinymce.init({