Files
llm-bridge/src/markdown.rs
T
bruno 7b5aa6fcc9
CI / fmt, clippy and tests (ubuntu-latest) (push) Successful in 21m0s
CI / browser end-to-end (Chromium + fixture) (push) Skipped
CI / fmt, clippy and tests (windows-latest) (push) Canceled after 0s
Add initial llm-bridge project
2026-09-18 18:01:07 -04:00

1381 lines
45 KiB
Rust
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
//! HTML to Markdown conversion for captured answers.
//!
//! The gateway reads answers from a signed-in browser page, so the text it
//! forwards was written for a human looking at rendered HTML. `innerText` keeps
//! the words and drops the structure: a table collapses into a column of
//! unrelated lines and a code block loses its fences, so the next model in the
//! chain cannot tell code from prose. Capturing `innerHTML` instead keeps that
//! structure, and this module turns it back into Markdown.
//!
//! The converter is deliberately self-contained: no browser, no network and no
//! HTML parser crate, because the markup is whatever a third party UI happens
//! to ship. The tokenizer is tolerant by design, so malformed markup degrades
//! into the readable text around it, and both parsing and rendering run on
//! explicit stacks, so no input can overflow the stack, spin forever or panic.
/// Depth beyond which start tags are ignored.
///
/// Real answers nest a handful of levels; the cap only stops a hostile page
/// from making the converter allocate without bound.
const MAX_DEPTH: usize = 128;
/// Elements whose content is text rather than markup, so it is skipped whole.
const RAW_TEXT_ELEMENTS: [&str; 4] = ["script", "style", "textarea", "title"];
/// Elements that never have children and never need a closing tag.
const VOID_ELEMENTS: [&str; 14] = [
"area", "base", "br", "col", "embed", "hr", "img", "input", "link", "meta", "param", "source",
"track", "wbr",
];
/// Elements that are a block of their own and are followed by a blank line.
const BLOCK_ELEMENTS: [&str; 12] = [
"p",
"div",
"section",
"article",
"header",
"footer",
"main",
"aside",
"figure",
"figcaption",
"dd",
"dt",
];
/// Convert the inner HTML of one answer node into Markdown.
pub fn html_to_markdown(html: &str) -> String {
if html.trim().is_empty() {
return String::new();
}
let root = parse(html);
tidy(&render(&root.children))
}
/// True when the HTML contains a construct a plain text extraction would lose
/// (a fenced code block, a table row, a list item, a heading, a link).
pub fn has_rich_structure(html: &str) -> bool {
// Callers use this to choose between `innerText` and `innerHTML`, so a cheap
// approximate scan is enough; an exact answer would mean parsing twice.
const MARKERS: [&str; 12] = [
"<pre", "<table", "<ul", "<ol", "<li", "<h1", "<h2", "<h3", "<h4", "<h5", "<h6", "<a ",
];
let lower = html.to_ascii_lowercase();
MARKERS.iter().any(|marker| lower.contains(marker))
}
/// Collapse the whitespace of a Markdown document without touching fenced code
/// blocks: trim trailing spaces per line, drop runs of more than two blank
/// lines down to one blank line, and end with exactly one newline (or an empty
/// string when there is no content).
pub fn tidy(markdown: &str) -> String {
let mut out = String::with_capacity(markdown.len());
let mut pending_blank = false;
let mut fence: Option<String> = None;
for raw_line in markdown.split('\n') {
// A carriage return only survives a CRLF document: line ending noise.
let line = raw_line.strip_suffix('\r').unwrap_or(raw_line);
if let Some(opening) = fence.as_deref() {
// Inside a fence the text is code the user wrote, so it is copied
// through untouched, blank lines, indentation and trailing spaces
// included.
out.push_str(line);
out.push('\n');
if closes_fence(line, opening) {
fence = None;
}
continue;
}
let line = line.trim_end_matches([' ', '\t']);
if line.is_empty() {
// Remembered rather than written: a blank line only exists when
// something follows it, which also drops the ones at the end.
pending_blank = true;
continue;
}
if let Some(opening) = opening_fence(line) {
fence = Some(opening);
}
if pending_blank && !out.is_empty() {
out.push('\n');
}
pending_blank = false;
out.push_str(line);
out.push('\n');
}
out
}
/// The run of backticks that opens a fenced block, when a line starts one.
fn opening_fence(line: &str) -> Option<String> {
let run: String = line
.trim_start()
.chars()
.take_while(|character| *character == '`')
.collect();
if run.len() >= 3 {
Some(run)
} else {
None
}
}
/// True when a line closes the fence it is inside.
fn closes_fence(line: &str, opening: &str) -> bool {
let trimmed = line.trim();
trimmed.len() >= opening.len() && trimmed.chars().all(|character| character == '`')
}
/// One node of the parsed document.
#[derive(Debug)]
enum Node {
Element(Element),
/// Entity-decoded text, otherwise exactly as written: whitespace is only
/// collapsed when the text is rendered, so a fenced block still reads it
/// verbatim.
Text(String),
}
/// One element: a lowercased name, decoded attributes and its children.
#[derive(Debug, Default)]
struct Element {
name: String,
attributes: Vec<(String, String)>,
children: Vec<Node>,
}
impl Element {
/// The value of an attribute, matched by its lowercased name.
fn attr(&self, name: &str) -> Option<&str> {
self.attributes
.iter()
.find(|(key, _)| key == name)
.map(|(_, value)| value.as_str())
}
}
/// Parse HTML into a tree, tolerating anything.
///
/// Text that no rule understands still ends up somewhere sensible, and an
/// unterminated construct simply ends the document: a broken page must never
/// cost an answer that is already half readable.
fn parse(html: &str) -> Element {
let mut stack: Vec<Element> = vec![Element::default()];
let bytes = html.as_bytes();
let mut cursor = 0usize;
while cursor < bytes.len() {
if bytes.get(cursor) != Some(&b'<') {
let start = cursor;
while let Some(byte) = bytes.get(cursor) {
if *byte == b'<' {
break;
}
cursor += 1;
}
if let Some(text) = html.get(start..cursor) {
push_text_node(&mut stack, text);
}
continue;
}
let rest = html.get(cursor..).unwrap_or("");
// Comments, doctypes and processing instructions carry no answer text.
if rest.starts_with("<!--") {
cursor = match rest.find("-->") {
Some(offset) => cursor + offset + 3,
None => bytes.len(),
};
continue;
}
if rest.starts_with("<!") || rest.starts_with("<?") {
cursor = match rest.find('>') {
Some(offset) => cursor + offset + 1,
None => bytes.len(),
};
continue;
}
if rest.starts_with("</") {
let name_start = cursor + 2;
let mut end = name_start;
while let Some(byte) = bytes.get(end) {
if byte.is_ascii_whitespace() || *byte == b'>' {
break;
}
end += 1;
}
let name = html.get(name_start..end).unwrap_or("").to_ascii_lowercase();
cursor = match html.get(end..).and_then(|tail| tail.find('>')) {
Some(offset) => end + offset + 1,
None => bytes.len(),
};
close_element(&mut stack, &name);
continue;
}
if rest
.as_bytes()
.get(1)
.is_some_and(|byte| byte.is_ascii_alphabetic())
{
let (name, attributes, self_closing, end) = read_start_tag(html, cursor);
cursor = end;
if name.is_empty() {
continue;
}
if RAW_TEXT_ELEMENTS.contains(&name.as_str()) && !self_closing {
cursor = skip_raw_text(html, end, &name);
continue;
}
let element = Element {
name: name.clone(),
attributes,
children: Vec::new(),
};
if self_closing || VOID_ELEMENTS.contains(&name.as_str()) || stack.len() >= MAX_DEPTH {
// A void element, a self-closing tag, or markup nested deeper
// than any real answer goes: the content is kept, the wrapper
// is not.
if let Some(parent) = stack.last_mut() {
parent.children.push(Node::Element(element));
}
} else {
stack.push(element);
}
continue;
}
// A `<` that opens nothing is text, which is what a browser shows too.
push_text_node(&mut stack, "<");
cursor += 1;
}
close_until(&mut stack, 1);
stack.pop().unwrap_or_default()
}
/// Read one start tag: its lowercased name, its decoded attributes, whether it
/// closes itself, and the position just past the final `>`.
fn read_start_tag(html: &str, start: usize) -> (String, Vec<(String, String)>, bool, usize) {
let bytes = html.as_bytes();
let mut cursor = start + 1;
let name_start = cursor;
while let Some(byte) = bytes.get(cursor) {
if byte.is_ascii_whitespace() || *byte == b'>' || *byte == b'/' {
break;
}
cursor += 1;
}
let name = html
.get(name_start..cursor)
.unwrap_or("")
.to_ascii_lowercase();
let mut attributes: Vec<(String, String)> = Vec::new();
let mut self_closing = false;
loop {
let before = cursor;
while let Some(byte) = bytes.get(cursor) {
if byte.is_ascii_whitespace() {
cursor += 1;
} else {
break;
}
}
match bytes.get(cursor).copied() {
None => break,
Some(b'>') => {
cursor += 1;
break;
}
Some(b'/') => {
self_closing = true;
cursor += 1;
}
Some(_) => {
let attribute_start = cursor;
while let Some(byte) = bytes.get(cursor) {
if byte.is_ascii_whitespace() || *byte == b'=' || *byte == b'>' || *byte == b'/'
{
break;
}
cursor += 1;
}
let attribute = html
.get(attribute_start..cursor)
.unwrap_or("")
.to_ascii_lowercase();
while let Some(byte) = bytes.get(cursor) {
if byte.is_ascii_whitespace() {
cursor += 1;
} else {
break;
}
}
let mut value = String::new();
if bytes.get(cursor) == Some(&b'=') {
cursor += 1;
while let Some(byte) = bytes.get(cursor) {
if byte.is_ascii_whitespace() {
cursor += 1;
} else {
break;
}
}
match bytes.get(cursor).copied() {
Some(quote) if quote == b'"' || quote == b'\'' => {
cursor += 1;
let value_start = cursor;
while let Some(byte) = bytes.get(cursor) {
if *byte == quote {
break;
}
cursor += 1;
}
value = html.get(value_start..cursor).unwrap_or("").to_string();
// An unterminated quote runs to the end of the
// document; the value stays a best effort.
if bytes.get(cursor) == Some(&quote) {
cursor += 1;
}
}
None | Some(b'>') => {}
Some(_) => {
let value_start = cursor;
while let Some(byte) = bytes.get(cursor) {
if byte.is_ascii_whitespace() || *byte == b'>' {
break;
}
cursor += 1;
}
value = html.get(value_start..cursor).unwrap_or("").to_string();
}
}
}
if !attribute.is_empty() {
attributes.push((attribute, decode_entities(&value)));
}
}
}
// Every turn consumes at least one byte; the guard is what keeps that
// true even for markup this reader has not thought of.
if cursor == before {
cursor += 1;
}
}
(name, attributes, self_closing, cursor)
}
/// Append a text node, decoding its entities on the way in.
fn push_text_node(stack: &mut [Element], text: &str) {
if let Some(parent) = stack.last_mut() {
parent.children.push(Node::Text(decode_entities(text)));
}
}
/// Close the innermost open element with this name.
///
/// A stray end tag is ignored, while one that closes an outer element also
/// closes the elements it still contains, which is what a browser does with
/// mismatched markup.
fn close_element(stack: &mut Vec<Element>, name: &str) {
if name.is_empty() {
return;
}
if let Some(index) = stack.iter().rposition(|element| element.name == name) {
close_until(stack, index);
}
}
/// Pop open elements down to `keep`, attaching each one to its new parent.
fn close_until(stack: &mut Vec<Element>, keep: usize) {
while stack.len() > keep {
match stack.pop() {
Some(element) => {
if let Some(parent) = stack.last_mut() {
parent.children.push(Node::Element(element));
}
}
None => return,
}
}
}
/// Skip the content of a raw text element, returning the position just past its
/// end tag.
fn skip_raw_text(html: &str, from: usize, name: &str) -> usize {
let needle = format!("</{name}");
match find_ascii_case_insensitive(html.as_bytes(), needle.as_bytes(), from) {
Some(index) => match html.get(index..).and_then(|tail| tail.find('>')) {
Some(offset) => index + offset + 1,
None => html.len(),
},
None => html.len(),
}
}
/// Find `needle` in `haystack` from `from`, comparing ASCII case-insensitively.
fn find_ascii_case_insensitive(haystack: &[u8], needle: &[u8], from: usize) -> Option<usize> {
if needle.is_empty() || haystack.len() < needle.len() {
return None;
}
let last = haystack.len() - needle.len();
let mut index = from;
while index <= last {
let window = haystack.get(index..index + needle.len())?;
if window.eq_ignore_ascii_case(needle) {
return Some(index);
}
index += 1;
}
None
}
/// Decode the HTML entities that show up in captured answers.
///
/// Anything unrecognised is left exactly as it was written, because a stray
/// ampersand in prose is far more common than an entity this table has missed.
fn decode_entities(text: &str) -> String {
if !text.contains('&') {
return text.to_string();
}
let mut out = String::with_capacity(text.len());
let mut rest = text;
while let Some(index) = rest.find('&') {
out.push_str(rest.get(..index).unwrap_or(""));
let tail = rest.get(index..).unwrap_or("");
match decode_one(tail) {
Some((decoded, consumed)) => {
out.push(decoded);
rest = tail.get(consumed..).unwrap_or("");
}
None => {
out.push('&');
rest = tail.get(1..).unwrap_or("");
}
}
}
out.push_str(rest);
out
}
/// Decode the single entity a string starts with, with its length in bytes.
fn decode_one(tail: &str) -> Option<(char, usize)> {
let body = tail.strip_prefix('&')?;
let end = body.find(';')?;
let name = body.get(..end)?;
// A bare ampersand must not swallow the rest of a sentence, so only a short
// reference is considered at all.
if name.is_empty() || name.len() > 32 {
return None;
}
let decoded = if let Some(digits) = name.strip_prefix('#') {
let code = match digits.strip_prefix(['x', 'X']) {
Some(hex) => u32::from_str_radix(hex, 16).ok()?,
None => digits.parse::<u32>().ok()?,
};
match char::from_u32(code) {
// Out of range or a lone surrogate: the replacement character is
// the honest answer, and it cannot panic.
Some(character) if character != '\0' => character,
_ => '\u{fffd}',
}
} else {
named_entity(name)?
};
Some((decoded, end + 2))
}
/// The character behind a named entity, or `None` when it is not one we know.
fn named_entity(name: &str) -> Option<char> {
// Matching ignores case: pages spell `&AMP;` and `&Nbsp;` too.
let character = match name.to_ascii_lowercase().as_str() {
"amp" => '&',
"lt" => '<',
"gt" => '>',
"quot" => '"',
"apos" => '\'',
"nbsp" => '\u{a0}',
"mdash" => '—',
"ndash" => '–',
"hellip" => '…',
"laquo" => '«',
"raquo" => '»',
"times" => '×',
"copy" => '©',
"reg" => '®',
"deg" => '°',
"middot" => '·',
"bull" => '•',
"eacute" => 'é',
"egrave" => 'è',
"agrave" => 'à',
"ccedil" => 'ç',
"uuml" => 'ü',
"ouml" => 'ö',
"auml" => 'ä',
_ => return None,
};
Some(character)
}
/// Collapse runs of layout whitespace into a single space, the way HTML itself
/// does. A non-breaking space is content rather than layout and survives.
fn collapse_whitespace(text: &str) -> String {
let mut out = String::with_capacity(text.len());
let mut in_space = false;
for character in text.chars() {
match character {
' ' | '\t' | '\n' | '\r' | '\u{c}' => {
if !in_space {
out.push(' ');
in_space = true;
}
}
_ => {
out.push(character);
in_space = false;
}
}
}
out
}
/// One unit of work for the iterative renderer.
enum Work<'a> {
/// Render a node of the parsed tree.
Node(&'a Node),
/// Append literal Markdown.
Emit(String),
/// Start a new line when the current one already holds something.
Break,
/// Push a buffer that the matching `Finish` will consume.
Capture(Capture),
/// Close the innermost capture and turn it into Markdown.
Finish,
}
/// Why a buffer is being captured: the text has to be complete before it can be
/// wrapped (a link label), prefixed (a quote or a list item) or cut into cells.
enum Capture {
Link { href: String },
Quote,
Item { ordered: bool, index: usize },
Row { header: bool, cells: Vec<String> },
Cell { colspan: usize },
}
/// Render a slice of nodes into Markdown.
///
/// The walk runs on an explicit stack of work items - a node becomes the items
/// that produce its Markdown - so the traversal stays flat however deeply the
/// document nests.
fn render(nodes: &[Node]) -> String {
let mut buffers: Vec<String> = vec![String::new()];
let mut captures: Vec<Capture> = Vec::new();
let mut work: Vec<Work<'_>> = nodes.iter().rev().map(Work::Node).collect();
while let Some(item) = work.pop() {
match item {
Work::Emit(text) => push_text(&mut buffers, &text),
Work::Break => ensure_newline(&mut buffers),
Work::Capture(capture) => {
captures.push(capture);
buffers.push(String::new());
}
Work::Finish => finish_capture(&mut buffers, &mut captures),
Work::Node(Node::Text(text)) => {
let collapsed = collapse_whitespace(text);
// The indentation of the markup must not become a leading space
// on a fresh line, while on a line in progress it is what keeps
// `a <b>b</b>` from running into `ab`.
let collapsed = if at_line_start(&buffers) {
collapsed.trim_start_matches(' ')
} else {
collapsed.as_str()
};
if !collapsed.is_empty() {
push_text(&mut buffers, collapsed);
}
}
Work::Node(Node::Element(element)) => push_element(element, &mut work),
}
}
// A capture left open would swallow its text, so anything still pending is
// unwound into the buffer below it.
while buffers.len() > 1 {
finish_capture(&mut buffers, &mut captures);
}
buffers.pop().unwrap_or_default()
}
/// Turn one element into the work items that render it.
fn push_element<'a>(element: &'a Element, work: &mut Vec<Work<'a>>) {
let mut parts: Vec<Work<'a>> = Vec::new();
match element.name.as_str() {
// Markdown's hard break is two trailing spaces. `tidy` trims them, so
// the caller sees a plain newline, which is what a reader expects.
"br" => parts.push(Work::Emit(" \n".to_string())),
"hr" => {
parts.push(Work::Break);
parts.push(Work::Emit("---\n\n".to_string()));
}
"img" => {
let image = image_markdown(element);
if !image.is_empty() {
parts.push(Work::Emit(image));
}
}
// Fenced blocks and inline code spans read their own text, so their
// children are never walked as markup.
"pre" => {
parts.push(Work::Break);
parts.push(Work::Emit(fenced_block(element)));
}
"code" => {
let span = inline_code(element);
if !span.is_empty() {
parts.push(Work::Emit(span));
}
}
"blockquote" => {
parts.push(Work::Break);
parts.push(Work::Capture(Capture::Quote));
parts.extend(children(element));
parts.push(Work::Finish);
}
"ul" | "ol" => {
parts.push(Work::Break);
push_list(element, element.name == "ol", &mut parts);
}
"table" => {
parts.push(Work::Break);
push_table(element, &mut parts);
}
"a" => {
let href = element.attr("href").unwrap_or_default().to_string();
parts.push(Work::Capture(Capture::Link { href }));
parts.extend(children(element));
parts.push(Work::Finish);
}
"h1" | "h2" | "h3" | "h4" | "h5" | "h6" => {
let level = heading_level(&element.name);
parts.push(Work::Break);
parts.push(Work::Emit(format!("{} ", "#".repeat(level))));
parts.extend(children(element));
parts.push(Work::Emit("\n\n".to_string()));
}
"strong" | "b" => parts.extend(wrapped("**", element)),
"em" | "i" => parts.extend(wrapped("*", element)),
"del" | "s" | "strike" => parts.extend(wrapped("~~", element)),
name if BLOCK_ELEMENTS.contains(&name) => {
parts.push(Work::Break);
parts.extend(children(element));
parts.push(Work::Emit("\n\n".to_string()));
}
// Unknown elements, and cells or items met outside their table or list,
// keep their content and nothing else.
_ => parts.extend(children(element)),
}
// The stack is popped from the back, so the parts go on in reverse.
for part in parts.into_iter().rev() {
work.push(part);
}
}
/// The heading level of an `h1`..`h6` element.
fn heading_level(name: &str) -> usize {
name.as_bytes()
.last()
.map_or(1, |digit| usize::from(digit.saturating_sub(b'0')))
.clamp(1, 6)
}
/// Emphasised or struck-through content: a marker on either side.
fn wrapped<'a>(marker: &str, element: &'a Element) -> Vec<Work<'a>> {
let mut parts = vec![Work::Emit(marker.to_string())];
parts.extend(children(element));
parts.push(Work::Emit(marker.to_string()));
parts
}
/// One work item per child, in document order.
fn children<'a>(element: &'a Element) -> Vec<Work<'a>> {
element.children.iter().map(Work::Node).collect()
}
/// Add the items that render a `ul` or `ol`.
fn push_list<'a>(element: &'a Element, ordered: bool, parts: &mut Vec<Work<'a>>) {
let mut index = 0usize;
for child in &element.children {
match child {
Node::Element(item) if item.name == "li" => {
parts.push(Work::Capture(Capture::Item { ordered, index }));
parts.extend(children(item));
parts.push(Work::Finish);
index += 1;
}
// Anything a list holds besides its items is kept as it is.
other => parts.push(Work::Node(other)),
}
}
// A list is followed by a blank line, so the next block starts cleanly.
parts.push(Work::Emit("\n".to_string()));
}
/// Add the items that render a `table`.
fn push_table<'a>(element: &'a Element, parts: &mut Vec<Work<'a>>) {
let rows = table_rows(element);
if rows.is_empty() {
// A table with no rows holds nothing a reader would miss.
return;
}
// Markdown tables need a header row; when the markup names none, the first
// row takes that role, which is what a reader of the page would assume.
let has_header = rows.iter().any(|(_, header)| *header);
for (index, (row, header)) in rows.iter().enumerate() {
let header = *header || (!has_header && index == 0);
parts.push(Work::Capture(Capture::Row {
header,
cells: Vec::new(),
}));
for cell in row_cells(row) {
parts.push(Work::Capture(Capture::Cell {
colspan: colspan(cell),
}));
parts.extend(children(cell));
parts.push(Work::Finish);
}
parts.push(Work::Finish);
}
parts.push(Work::Emit("\n".to_string()));
}
/// The rows of a table in document order, each with the flag telling whether it
/// is a header row (inside `thead`, or holding `th` cells).
fn table_rows(table: &Element) -> Vec<(&Element, bool)> {
let mut rows = Vec::new();
for child in &table.children {
let Node::Element(child) = child else {
continue;
};
match child.name.as_str() {
"tr" => rows.push((child, row_has_th(child))),
"thead" | "tbody" | "tfoot" => {
let in_header = child.name == "thead";
for row in &child.children {
if let Node::Element(row) = row {
if row.name == "tr" {
rows.push((row, in_header || row_has_th(row)));
}
}
}
}
_ => {}
}
}
rows
}
/// True when a row holds header cells of its own.
fn row_has_th(row: &Element) -> bool {
row_cells(row).iter().any(|cell| cell.name == "th")
}
/// The cells of a row, in document order.
fn row_cells(row: &Element) -> Vec<&Element> {
row.children
.iter()
.filter_map(|child| match child {
Node::Element(cell) if cell.name == "th" || cell.name == "td" => Some(cell),
_ => None,
})
.collect()
}
/// How many columns a cell covers; a missing or silly value means one, and the
/// cap keeps a hostile `colspan` from asking for an enormous row.
fn colspan(cell: &Element) -> usize {
cell.attr("colspan")
.and_then(|value| value.trim().parse::<usize>().ok())
.filter(|span| *span > 0)
.map_or(1, |span| span.min(64))
}
/// Close the innermost capture and append its Markdown to the buffer below it.
fn finish_capture(buffers: &mut Vec<String>, captures: &mut Vec<Capture>) {
let text = match buffers.pop() {
Some(text) => text,
None => return,
};
match captures.pop() {
Some(Capture::Link { href }) => {
let label = text.trim();
let rendered = if href.is_empty() {
label.to_string()
} else if label == href {
href
} else if label.is_empty() {
String::new()
} else {
format!("[{label}]({href})")
};
push_text(buffers, &rendered);
}
Some(Capture::Quote) => {
let mut quoted = String::new();
let body = text.trim();
if !body.is_empty() {
for line in body.split('\n') {
let line = line.trim_end();
if line.is_empty() {
quoted.push_str(">\n");
} else {
quoted.push_str("> ");
quoted.push_str(line);
quoted.push('\n');
}
}
quoted.push('\n');
}
push_text(buffers, &quoted);
}
Some(Capture::Item { ordered, index }) => {
// The bullet shares its line with the first line of the item; the
// rest continues indented under it, two spaces per level.
let bullet = if ordered {
format!("{}.", index + 1)
} else {
"-".to_string()
};
let mut rendered = String::new();
for (number, line) in text.trim().split('\n').enumerate() {
let line = line.trim_end();
if number == 0 {
rendered.push_str(&bullet);
rendered.push(' ');
rendered.push_str(line.trim_start());
} else if !line.is_empty() {
rendered.push_str(" ");
rendered.push_str(line);
}
rendered.push('\n');
}
push_text(buffers, &rendered);
}
Some(Capture::Row { header, cells }) => {
let mut rendered = String::new();
if !cells.is_empty() {
rendered.push_str("| ");
rendered.push_str(&cells.join(" | "));
rendered.push_str(" |\n");
if header {
rendered.push_str("| ");
rendered.push_str(&vec!["---"; cells.len()].join(" | "));
rendered.push_str(" |\n");
}
}
push_text(buffers, &rendered);
}
Some(Capture::Cell { colspan }) => {
// A cell is a single line: newlines inside it become spaces, and a
// pipe would otherwise cut the row into extra columns.
let cell = text
.split_whitespace()
.collect::<Vec<_>>()
.join(" ")
.replace('|', "\\|");
match captures.last_mut() {
Some(Capture::Row { cells, .. }) => {
for _ in 0..colspan {
cells.push(cell.clone());
}
}
_ => push_text(buffers, &cell),
}
}
None => push_text(buffers, &text),
}
}
/// An image with its alternative text, as Markdown.
fn image_markdown(element: &Element) -> String {
let alt = element.attr("alt").unwrap_or_default();
let src = element.attr("src").unwrap_or_default();
if alt.is_empty() && src.is_empty() {
// A decorative spacer: nothing a reader would miss.
return String::new();
}
format!("![{alt}]({src})")
}
/// An inline code span: one backtick, or two when the text contains one.
fn inline_code(element: &Element) -> String {
let text = text_content(element);
let text = text.trim();
if text.is_empty() {
return String::new();
}
let ticks = if text.contains('`') { "``" } else { "`" };
format!("{ticks}{text}{ticks}")
}
/// A `pre` as a fenced block, named with the language of an inner `code`
/// element when the page declares one.
fn fenced_block(element: &Element) -> String {
let text = text_content(element);
let text = text.trim_matches(['\n', '\r']);
let language = code_language(element).unwrap_or_default();
let fence = fence_for(text);
format!("{fence}{language}\n{text}\n{fence}\n\n")
}
/// A fence longer than any run of backticks in the text, and at least three.
fn fence_for(text: &str) -> String {
let mut longest = 0usize;
let mut run = 0usize;
for character in text.chars() {
if character == '`' {
run += 1;
longest = longest.max(run);
} else {
run = 0;
}
}
"`".repeat(longest.max(2) + 1)
}
/// The language named by `class="language-xxx"` on the `code` child of a `pre`.
fn code_language(element: &Element) -> Option<String> {
let code = element.children.iter().find_map(|child| match child {
Node::Element(child) if child.name == "code" => Some(child),
_ => None,
})?;
code.attr("class")?
.split_whitespace()
.find_map(|token| token.strip_prefix("language-"))
.filter(|language| !language.is_empty())
.map(|language| language.to_string())
}
/// The text of a subtree, with its whitespace exactly as written.
///
/// Fenced blocks and code spans use this instead of the renderer, because
/// Markdown code is literal text rather than markup.
fn text_content(element: &Element) -> String {
let mut out = String::new();
let mut stack: Vec<&Node> = element.children.iter().rev().collect();
while let Some(node) = stack.pop() {
match node {
Node::Text(text) => out.push_str(text),
Node::Element(child) => {
if child.name == "br" {
out.push('\n');
}
stack.extend(child.children.iter().rev());
}
}
}
out
}
/// Append text to the innermost buffer.
fn push_text(buffers: &mut [String], text: &str) {
if let Some(buffer) = buffers.last_mut() {
buffer.push_str(text);
}
}
/// End the current line when it already holds something.
fn ensure_newline(buffers: &mut [String]) {
if let Some(buffer) = buffers.last_mut() {
if !buffer.is_empty() && !buffer.ends_with('\n') {
buffer.push('\n');
}
}
}
/// True when nothing has been written on the current line yet.
fn at_line_start(buffers: &[String]) -> bool {
buffers
.last()
.is_none_or(|buffer| buffer.is_empty() || buffer.ends_with('\n'))
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn empty_input_produces_empty_output() {
assert_eq!(html_to_markdown(""), "");
assert_eq!(html_to_markdown(" \n\t "), "");
}
#[test]
fn paragraphs_keep_their_text_without_the_markup() {
assert_eq!(html_to_markdown("<p>Hello world</p>"), "Hello world\n");
}
#[test]
fn two_paragraphs_are_separated_by_a_blank_line() {
assert_eq!(html_to_markdown("<p>One</p><p>Two</p>"), "One\n\nTwo\n");
}
#[test]
fn layout_whitespace_never_leaves_blank_runs() {
assert_eq!(html_to_markdown("<p>a</p>\n\n\n<p>b</p>\n\n\n"), "a\n\nb\n");
}
#[test]
fn line_break_ends_the_line() {
// The renderer writes Markdown's two-space hard break; `tidy` trims the
// trailing spaces, so the caller sees a plain newline.
assert_eq!(html_to_markdown("<p>one<br>two</p>"), "one\ntwo\n");
assert_eq!(html_to_markdown("<p>one<br/>two</p>"), "one\ntwo\n");
}
#[test]
fn headings_use_one_hash_per_level() {
assert_eq!(
html_to_markdown("<h1>Title</h1><h4>Deep</h4>"),
"# Title\n\n#### Deep\n"
);
}
#[test]
fn emphasis_becomes_markdown_markers() {
let html =
"<p><strong>bold</strong> <b>also bold</b> <em>italic</em> <i>also</i> <del>gone</del></p>";
assert_eq!(
html_to_markdown(html),
"**bold** **also bold** *italic* *also* ~~gone~~\n"
);
}
#[test]
fn inline_code_uses_one_backtick() {
assert_eq!(
html_to_markdown("<p>run <code>cargo test</code> now</p>"),
"run `cargo test` now\n"
);
}
#[test]
fn inline_code_containing_a_backtick_uses_two() {
assert_eq!(
html_to_markdown("<p><code>x `y` z</code></p>"),
"``x `y` z``\n"
);
}
#[test]
fn fenced_code_block_without_a_language() {
assert_eq!(
html_to_markdown("<pre>fn main() {}</pre>"),
"```\nfn main() {}\n```\n"
);
}
#[test]
fn fenced_code_block_takes_the_language_from_the_code_class() {
let html = r#"<pre><code class="language-rust hljs">let x = 1;</code></pre>"#;
assert_eq!(html_to_markdown(html), "```rust\nlet x = 1;\n```\n");
}
#[test]
fn fenced_code_block_leaves_its_text_alone() {
let html = "<pre><code>a &lt; b &amp;&amp; c\n indented\n</code></pre>";
assert_eq!(
html_to_markdown(html),
"```\na < b && c\n indented\n```\n"
);
}
#[test]
fn fence_grows_when_the_code_contains_backticks() {
assert_eq!(
html_to_markdown("<pre>a ``` b</pre>"),
"````\na ``` b\n````\n"
);
}
#[test]
fn named_and_numeric_entities_are_decoded() {
assert_eq!(
html_to_markdown("<p>a &amp; b &lt;tag&gt; &#65;&#x42;</p>"),
"a & b <tag> AB\n"
);
assert_eq!(html_to_markdown("<p>&#x1F600;</p>"), "\u{1f600}\n");
}
#[test]
fn punctuation_entities_are_decoded() {
assert_eq!(
html_to_markdown("<p>5 &times; 3 &mdash; caf&eacute;</p>"),
"5 × 3 — café\n"
);
}
#[test]
fn unknown_entities_are_left_alone() {
assert_eq!(
html_to_markdown("<p>a &unknown; b & c</p>"),
"a &unknown; b & c\n"
);
}
#[test]
fn invalid_numeric_references_do_not_panic() {
assert_eq!(
html_to_markdown("<p>&#xD800;&#x110000;</p>"),
"\u{fffd}\u{fffd}\n"
);
}
#[test]
fn non_breaking_spaces_are_kept() {
assert_eq!(html_to_markdown("<p>a&nbsp;b</p>"), "a\u{a0}b\n");
}
#[test]
fn attributes_are_decoded() {
let html = r#"<a href="https://x.test/?a=1&amp;b=2">docs</a>"#;
assert_eq!(html_to_markdown(html), "[docs](https://x.test/?a=1&b=2)\n");
}
#[test]
fn attribute_names_are_case_insensitive() {
let html = r#"<a HREF="https://e.test">https://e.test</a>"#;
assert_eq!(html_to_markdown(html), "https://e.test\n");
}
#[test]
fn attribute_values_may_be_quoted_or_bare() {
assert_eq!(
html_to_markdown("<a href='https://s.test'>x</a>"),
"[x](https://s.test)\n"
);
assert_eq!(
html_to_markdown("<img src=/u/pic.png alt=pic>"),
"![pic](/u/pic.png)\n"
);
}
#[test]
fn nested_lists_are_indented() {
let html = "<ul><li>one<ul><li>inner</li></ul></li><li>two</li></ul>";
assert_eq!(html_to_markdown(html), "- one\n - inner\n- two\n");
}
#[test]
fn ordered_list_items_are_numbered() {
assert_eq!(
html_to_markdown("<ol><li>first</li><li>second</li></ol>"),
"1. first\n2. second\n"
);
}
#[test]
fn list_item_blocks_continue_on_indented_lines() {
assert_eq!(
html_to_markdown("<ul><li>first<p>second</p></li></ul>"),
"- first\n second\n"
);
}
#[test]
fn blockquote_prefixes_every_line() {
let html = "<blockquote><p>first</p><p>second</p></blockquote>";
assert_eq!(html_to_markdown(html), "> first\n>\n> second\n");
}
#[test]
fn link_with_a_different_text_uses_the_markdown_form() {
let html = r#"<p>see <a href="https://e.test">the docs</a></p>"#;
assert_eq!(html_to_markdown(html), "see [the docs](https://e.test)\n");
}
#[test]
fn link_whose_text_is_its_href_collapses_to_the_url() {
let html = r#"<p><a href="https://e.test">https://e.test</a></p>"#;
assert_eq!(html_to_markdown(html), "https://e.test\n");
}
#[test]
fn image_uses_its_alt_text_and_source() {
let html = r#"<p><img src="/pic.png" alt="A picture"></p>"#;
assert_eq!(html_to_markdown(html), "![A picture](/pic.png)\n");
}
#[test]
fn horizontal_rule_gets_its_own_line() {
assert_eq!(
html_to_markdown("<p>above</p><hr><p>below</p>"),
"above\n\n---\n\nbelow\n"
);
}
#[test]
fn table_with_a_header_becomes_a_github_table() {
let html = "<table><thead><tr><th>Name</th><th>Value</th></tr></thead>\
<tbody><tr><td>a</td><td>1</td></tr></tbody></table>";
assert_eq!(
html_to_markdown(html),
"| Name | Value |\n| --- | --- |\n| a | 1 |\n"
);
}
#[test]
fn table_without_a_header_promotes_the_first_row() {
let html = "<table><tr><td>h1</td><td>h2</td></tr><tr><td>a</td><td>b</td></tr></table>";
assert_eq!(
html_to_markdown(html),
"| h1 | h2 |\n| --- | --- |\n| a | b |\n"
);
}
#[test]
fn pipes_inside_a_cell_are_escaped() {
let html = "<table><tr><th>a</th><th>b</th></tr><tr><td>x | y</td><td>z</td></tr></table>";
assert_eq!(
html_to_markdown(html),
"| a | b |\n| --- | --- |\n| x \\| y | z |\n"
);
}
#[test]
fn colspan_repeats_the_cell_content() {
let html =
r#"<table><tr><th>a</th><th>b</th></tr><tr><td colspan="2">span</td></tr></table>"#;
assert_eq!(
html_to_markdown(html),
"| a | b |\n| --- | --- |\n| span | span |\n"
);
}
#[test]
fn a_table_without_rows_produces_nothing() {
assert_eq!(html_to_markdown("<table></table>"), "");
}
#[test]
fn unknown_elements_keep_their_content() {
let html = r#"<answer><span data-x="1">text</span></answer>"#;
assert_eq!(html_to_markdown(html), "text\n");
}
#[test]
fn comments_doctypes_and_raw_text_are_dropped() {
let html = "<!doctype html><!-- hidden --><style>p{color:red}</style>\
<script>var x = 1;</script><p>kept</p>";
assert_eq!(html_to_markdown(html), "kept\n");
}
#[test]
fn an_unclosed_tag_keeps_the_text() {
assert_eq!(html_to_markdown("<p>hello <b>world"), "hello **world**\n");
}
#[test]
fn a_stray_angle_bracket_is_text() {
assert_eq!(
html_to_markdown("<p>2 < 3 and 4 > 1</p>"),
"2 < 3 and 4 > 1\n"
);
}
#[test]
fn an_unterminated_attribute_quote_stops_at_the_end() {
// The quoted value runs to the end of the document: everything before it
// is still rendered and nothing hangs.
let html = "<p>visible</p><a href=\"broken>tail";
assert_eq!(html_to_markdown(html), "visible\n");
}
#[test]
fn a_stray_end_tag_is_ignored() {
assert_eq!(html_to_markdown("<p><b>bold</i></p>"), "**bold**\n");
}
#[test]
fn unterminated_constructs_terminate() {
assert_eq!(html_to_markdown("<>&&&"), "<>&&&\n");
assert_eq!(html_to_markdown("<p>a</p><!-- open"), "a\n");
assert_eq!(html_to_markdown("<p>a"), "a\n");
}
#[test]
fn deeply_nested_markup_is_bounded() {
let mut html = String::new();
for _ in 0..5_000 {
html.push_str("<div>");
}
html.push_str("deep");
assert_eq!(html_to_markdown(&html), "deep\n");
}
#[test]
fn tidy_trims_trailing_spaces_and_collapses_blank_lines() {
assert_eq!(tidy("a \n\n\n\nb \n\n\n"), "a\n\nb\n");
}
#[test]
fn tidy_ends_with_exactly_one_newline() {
assert_eq!(tidy("text"), "text\n");
assert_eq!(tidy("text\n\n\n"), "text\n");
}
#[test]
fn tidy_of_whitespace_only_input_is_empty() {
assert_eq!(tidy(""), "");
assert_eq!(tidy("\n\n \n"), "");
}
#[test]
fn tidy_does_not_touch_fenced_code_blocks() {
let input = "text \n\n```rust\nlet a = 1; \n\n\nlet b = 2;\n```\n\n\n";
let expected = "text\n\n```rust\nlet a = 1; \n\n\nlet b = 2;\n```\n";
assert_eq!(tidy(input), expected);
}
#[test]
fn has_rich_structure_detects_structure() {
assert!(has_rich_structure("<pre>code</pre>"));
assert!(has_rich_structure("<TABLE><tr><td>x</td></tr></TABLE>"));
assert!(has_rich_structure("<ul><li>a</li></ul>"));
assert!(has_rich_structure("<ol><li>a</li></ol>"));
assert!(has_rich_structure("<h3>title</h3>"));
assert!(has_rich_structure(
"see <a href=\"https://e.test\">the docs</a>"
));
}
#[test]
fn has_rich_structure_is_false_for_plain_markup() {
assert!(!has_rich_structure(""));
assert!(!has_rich_structure("just words"));
assert!(!has_rich_structure("<p>just <b>words</b></p>"));
assert!(!has_rich_structure("<div><span>text</span></div>"));
}
}