From fb083f3850c242e14613de517811a3818bef98db Mon Sep 17 00:00:00 2001 From: Kevin Date: Fri, 28 Aug 2026 04:43:46 -0700 Subject: [PATCH] docs(gen_help_html): generate Typst format #40665 Problem: gen_help_html.lua only outputs HTML, but Typst is useful for producing PDFs. Solution: Add gen_one_typ() and ts_node_to_typ(), which mirror the existing gen_one_html() and ts_node_to_html(). --- src/gen/gen_help_html.lua | 259 +++++++++++++++++++++++++++++++++++++- 1 file changed, 255 insertions(+), 4 deletions(-) diff --git a/src/gen/gen_help_html.lua b/src/gen/gen_help_html.lua index 414c9e2814..6c82031496 100644 --- a/src/gen/gen_help_html.lua +++ b/src/gen/gen_help_html.lua @@ -1,4 +1,4 @@ ---- Converts Nvim :help files to HTML. Validates |tag| links and document syntax (parser errors). +--- Converts Nvim :help files to alternate formats (HTML or Typst). Validates |tag| links and document syntax (parser errors). -- -- USAGE (For CI/local testing purposes): Simply `make lintdoc`, which basically does the following: -- 1. :helptags ALL @@ -12,6 +12,15 @@ -- 3. cd target/dir/ && jekyll serve --host 0.0.0.0 -- 4. Visit http://localhost:4000/…/help.txt.html -- +-- USAGE (GENERATE PDF VIA TYPST): +-- 1. `:helptags ALL` +-- 2. mkdir usr_manual pdf_docs +-- 3. cp runtime/doc/usr_*.txt usr_manual/ +-- 4. nvim -V1 -es --clean +"lua require('src.gen.gen_help_html').gen('typ', './usr_manual', 'pdf_docs')" +q +-- 5. mv pdf_docs/usr_toc.typ pdf_docs/user_manual.typ && cat pdf_docs/usr_*.typ >> pdf_docs/user_manual.typ +-- 6. typst compile pdf_docs/user_manual.typ +-- 7. open pdf_docs/user_manual.pdf +-- -- USAGE (VALIDATE): -- 1. nvim -V1 -es +"lua require('src.gen.gen_help_html').validate('./runtime/doc')" +q -- - validate() is 10x faster than gen(), so it is used in CI. @@ -163,6 +172,12 @@ local function html_esc(s) return (s and string.gsub(s, '[&<>{}]', html_entity) or nil) end +--- Escape a string to make it valid Typst +---@type fun(s: string): string +local function typ_esc(s) + return (s and string.gsub(s, '([%[%]@#\\/<*`_^$])', '\\%1') or '') +end + local function url_encode(s) -- Credit: tpope / vim-unimpaired -- NOTE: these chars intentionally *not* escaped: ' ( ) @@ -220,7 +235,7 @@ local function fix_url(url) return fixed_url, removed_chars end ---- Checks if a given line is a "noise" line that doesn't look good in HTML form. +--- Checks if a given line is a "noise" line that doesn't look good in non-vimdoc formats local function is_noise(line, noise_lines) if -- First line is always noise. @@ -840,6 +855,199 @@ local function ts_node_to_html(root, level, lang_tree, headings, opt, stats) end end +-- Generate Typst from node `root` recursively. +---@param root TSNode +---@param level integer +---@param lang_tree TSTree +---@param opt table +---@param stats table +local function ts_node_to_typ(root, level, lang_tree, opt, stats) + local function node_text(node, ws_) + node = node or root + ws_ = (ws_ == nil or ws_ == true) and getws(node, opt.buf) or '' + return string.format('%s%s', ws_, vim.treesitter.get_node_text(node, opt.buf)) + end + + -- Gets leading whitespace of `node`. + local function ws(node) + node = node or root + local ws_ = getws(node, opt.buf) + -- XXX: first node of a (line) includes whitespace, even after + -- https://github.com/neovim/tree-sitter-vimdoc/pull/31 ? + if ws_ == '' then + ws_ = vim.treesitter.get_node_text(node, opt.buf):match('^%s+') or '' + end + return ws_ + end + + local node_name = (root.named and root:named()) and root:type() or nil + local prev, next_ = get_prev_next(root) + -- Parent kind (string). + local parent = root:parent() and root:parent():type() or nil + + local text = '' + local trimmed ---@type string + if root:named_child_count() == 0 or node_name == 'ERROR' then + text = node_text() + trimmed = typ_esc(trim(text)) + + -- A URL is a "word" directly under a "url" + -- We don't want any typ escaping there, as typ won't try to parse the + -- value as a command + if parent ~= 'url' then + text = typ_esc(text) + end + else + -- Process children and join them with whitespace. + for node, _ in root:iter_children() do + if node:named() then + local r = ts_node_to_typ(node, level + 1, lang_tree, opt, stats) + text = string.format('%s%s', text, r) + end + end + trimmed = trim(text) + end + + if node_name == 'help_file' then -- root node + return text + elseif node_name == 'url' then + local fixed_url, removed_chars = fix_url(trimmed) + return ('%s#link("%s")%s'):format(ws(), fixed_url, removed_chars) + elseif node_name == 'word' or node_name == 'uppercase_name' then + return text + elseif node_name == 'note' then + return ('*%s*'):format(text) + elseif node_name == 'h1' or node_name == 'h2' or node_name == 'h3' then + if is_noise(text, stats.noise_lines) then + return '' -- Discard common "noise" lines. + end + -- Remove tags from ToC text. + local heading_node = first(root, 'heading') + local hname = trim(node_text(heading_node):gsub('%*.*%*', '')) + if not heading_node or hname == '' then + return '' -- Spurious "===" or "---" in the help doc. + end + + local heading_level = tonumber(string.sub(node_name, -1)) + if not heading_level then + print(('Cannot parse heading level from node name %s'):format(node_name)) + return '' + end + return ('%s %s'):format(string.rep('=', heading_level), trimmed) + elseif node_name == 'heading' then + return trimmed + elseif node_name == 'column_heading' then + if root:has_error() then + return text + end + -- In help files column_heading is used for a lot more than column headings. + -- Sometimes it's some sort of title (see usr_toc.txt), and sometimes it's + -- some example text that really should be a code block (see usr_02.txt) + -- Using a raw block allows us to match all use cases fairly well + return ('```\n%s\n```\n'):format(trimmed) + elseif node_name == 'block' then + if is_blank(text) then + return '' + end + return ('%s\n'):format(text) + elseif node_name == 'line' then + if + (parent ~= 'codeblock' or parent ~= 'code') + and (is_blank(text) or is_noise(text, stats.noise_lines)) + then + return '' -- Discard common "noise" lines. + end + -- XXX: Avoid newlines (too much whitespace) after block elements in old (preformatted) layout. + local old_after_block = opt.old + and root:child(0) + and vim.list_contains({ 'column_heading', 'h1', 'h2', 'h3' }, root:child(0):type()) + + if parent == 'codeblock' or parent == 'code' then + return text + elseif old_after_block then + return trim(text) + else + -- It would be nice to let the text flow when possible + -- But many lines do need an explicit line break (e.g. in the ToC) + return string.format('%s \\ \n', text) + end + elseif parent == 'line_li' and node_name == 'prefix' then + return '' + elseif node_name == 'line_li' then + -- Just writing the text - indentation needs improvements + return text + elseif node_name == 'taglink' or node_name == 'optionlink' then + -- Links are just text for now, because dealing with broken links is + -- difficult. Broken links are expected when generating a PDF of the user + -- manual, since all links to the reference don't have a destination. + -- Ideally, links with invalid destinations would be allowed and just be + -- normal text. + return (' #underline[%s]'):format(text) + elseif vim.list_contains({ 'codespan', 'keycode' }, node_name) then + if root:has_error() then + return text + end + return ('%s`%s`'):format(ws(), trimmed) + elseif node_name == 'argument' then + return ('%s`%s`'):format(ws(), trim(node_text(root))) + elseif node_name == 'codeblock' then + return text + elseif node_name == 'language' then + language = node_text(root) + return '' + elseif node_name == 'code' then -- Highlighted codeblock (child). + if is_blank(text) then + return '' + end + local code ---@type string + if language then + code = ('```%s\n%s\n```'):format(language, trim(trim_indent(text), 2)) + language = nil + else + code = ('\n```\n%s\n```\n'):format(trim(trim_indent(text), 2)) + end + return code + elseif node_name == 'tag' then -- anchor, h4 pseudo-heading + if root:has_error() then + return text + end + local in_heading = vim.list_contains({ 'h1', 'h2', 'h3' }, parent) + local tagname = node_text(root:child(1), false) + if vim.tbl_count(stats.first_tags) < 2 then + -- Force the first 2 tags in the doc to be anchored at the main heading. + table.insert(stats.first_tags, tagname) + return '' + end + -- NOTE: the <%s> syntax doesn't work because it terminates the heading + local s = ('%s%s #label("%s")'):format(ws(), text, tagname) + if opt.old then + s = fix_tab_after_conceal(s, node_text(root:next_sibling())) + end + + if in_heading and prev ~= 'tag' then + -- Add fractional spacing to right-align the tags in a heading + -- TODO those right-aligned tags should also be styled differently + -- (i.e. not like a header) + return (' #h(1fr) %s'):format(s) + end + return s + elseif node_name == 'delimiter' then + return '\n' + elseif node_name == 'modeline' then + return '' + else -- Unknown token. + local sample_text = level > 0 and getbuflinestr(root, opt.buf, 3) or '[top level!]' + if node_name == 'ERROR' then + if ignore_parse_error(opt.fname, trimmed) then + return text + end + table.insert(stats.parse_errors, sample_text) + end + print(('Unexpected token %s: %s'):format(node_name, sample_text)) + return '' + end +end + --- @param dir string e.g. '$VIMRUNTIME/doc' --- @param include string[]|nil --- @return string[] @@ -1008,6 +1216,49 @@ local function gen_one_html(fname, text, to_fname, old, commit) return html, stats end +--- Generates Typst from one :help file `fname` and writes the result to `to_fname`. +--- +--- @param fname string Source :help file. +--- @param text string|nil Source :help file contents, or nil to read `fname`. +--- @param to_fname string Destination .typ file +--- @param old boolean Preformat paragraphs (for old :help files which are full of arbitrary whitespace) +--- +--- @return string Typst output +--- @return table stats +local function gen_one_typ(fname, text, to_fname, old, commit) + local stats = { + noise_lines = {}, + parse_errors = {}, + first_tags = {}, -- Track the first few tags in doc. + } + local lang_tree, buf = parse_buf(fname, text) + ---@type nvim.gen_help_html.heading[] + local title = to_titlecase(basename_noext(fname)) + + local typ = title + for _, tree in ipairs(lang_tree:trees()) do + typ = typ + .. ( + ts_node_to_typ( + tree:root(), + 0, + tree, + { buf = buf, old = old, fname = fname, to_fname = to_fname, indent = 1 }, + stats + ) + ) + end + + -- Add page break at the bottom, so that merging .typ files before conversion + -- results in page breaks in the right places + local page_break = '#pagebreak()' + typ = ('%s\n%s'):format(typ, page_break) + + vim.cmd('q!') + lang_tree:destroy() + return typ, stats +end + --- Generates a JSON map of tags to URL-encoded `filename#anchor` locations. --- ---@param fname string @@ -1207,9 +1458,9 @@ function M.gen(output_format, help_dir, to_dir, include, commit, parser_path) end -- Map output formats to the function that outputs one file in that format - -- NOTE: Only .html for now, but .typ is coming local generators = { html = gen_one_html, + typ = gen_one_typ, } local gen_one = generators[output_format] @@ -1245,7 +1496,7 @@ function M.gen(output_format, help_dir, to_dir, include, commit, parser_path) err_count = err_count + #stats.parse_errors end - print(('\ngenerated %d html pages'):format(#helpfiles + redirects_count)) + print(('\ngenerated %d pages'):format(#helpfiles + redirects_count)) print(('total errors: %d'):format(err_count)) -- Why aren't the netrw tags found in neovim/docs/ CI? print(('invalid tags: %s'):format(vim.inspect(invalid_links)))