//! HTML to Markdown conversion for captured answers. //! //! The gateway reads answers from a signed-in browser page, so the text it //! forwards was written for a human looking at rendered HTML. `innerText` keeps //! the words and drops the structure: a table collapses into a column of //! unrelated lines and a code block loses its fences, so the next model in the //! chain cannot tell code from prose. Capturing `innerHTML` instead keeps that //! structure, and this module turns it back into Markdown. //! //! The converter is deliberately self-contained: no browser, no network and no //! HTML parser crate, because the markup is whatever a third party UI happens //! to ship. The tokenizer is tolerant by design, so malformed markup degrades //! into the readable text around it, and both parsing and rendering run on //! explicit stacks, so no input can overflow the stack, spin forever or panic. /// Depth beyond which start tags are ignored. /// /// Real answers nest a handful of levels; the cap only stops a hostile page /// from making the converter allocate without bound. const MAX_DEPTH: usize = 128; /// Elements whose content is text rather than markup, so it is skipped whole. const RAW_TEXT_ELEMENTS: [&str; 4] = ["script", "style", "textarea", "title"]; /// Elements that never have children and never need a closing tag. const VOID_ELEMENTS: [&str; 14] = [ "area", "base", "br", "col", "embed", "hr", "img", "input", "link", "meta", "param", "source", "track", "wbr", ]; /// Elements that are a block of their own and are followed by a blank line. const BLOCK_ELEMENTS: [&str; 12] = [ "p", "div", "section", "article", "header", "footer", "main", "aside", "figure", "figcaption", "dd", "dt", ]; /// Convert the inner HTML of one answer node into Markdown. pub fn html_to_markdown(html: &str) -> String { if html.trim().is_empty() { return String::new(); } let root = parse(html); tidy(&render(&root.children)) } /// True when the HTML contains a construct a plain text extraction would lose /// (a fenced code block, a table row, a list item, a heading, a link). pub fn has_rich_structure(html: &str) -> bool { // Callers use this to choose between `innerText` and `innerHTML`, so a cheap // approximate scan is enough; an exact answer would mean parsing twice. const MARKERS: [&str; 12] = [ "
String {
let mut out = String::with_capacity(markdown.len());
let mut pending_blank = false;
let mut fence: Option = None;
for raw_line in markdown.split('\n') {
// A carriage return only survives a CRLF document: line ending noise.
let line = raw_line.strip_suffix('\r').unwrap_or(raw_line);
if let Some(opening) = fence.as_deref() {
// Inside a fence the text is code the user wrote, so it is copied
// through untouched, blank lines, indentation and trailing spaces
// included.
out.push_str(line);
out.push('\n');
if closes_fence(line, opening) {
fence = None;
}
continue;
}
let line = line.trim_end_matches([' ', '\t']);
if line.is_empty() {
// Remembered rather than written: a blank line only exists when
// something follows it, which also drops the ones at the end.
pending_blank = true;
continue;
}
if let Some(opening) = opening_fence(line) {
fence = Some(opening);
}
if pending_blank && !out.is_empty() {
out.push('\n');
}
pending_blank = false;
out.push_str(line);
out.push('\n');
}
out
}
/// The run of backticks that opens a fenced block, when a line starts one.
fn opening_fence(line: &str) -> Option {
let run: String = line
.trim_start()
.chars()
.take_while(|character| *character == '`')
.collect();
if run.len() >= 3 {
Some(run)
} else {
None
}
}
/// True when a line closes the fence it is inside.
fn closes_fence(line: &str, opening: &str) -> bool {
let trimmed = line.trim();
trimmed.len() >= opening.len() && trimmed.chars().all(|character| character == '`')
}
/// One node of the parsed document.
#[derive(Debug)]
enum Node {
Element(Element),
/// Entity-decoded text, otherwise exactly as written: whitespace is only
/// collapsed when the text is rendered, so a fenced block still reads it
/// verbatim.
Text(String),
}
/// One element: a lowercased name, decoded attributes and its children.
#[derive(Debug, Default)]
struct Element {
name: String,
attributes: Vec<(String, String)>,
children: Vec,
}
impl Element {
/// The value of an attribute, matched by its lowercased name.
fn attr(&self, name: &str) -> Option<&str> {
self.attributes
.iter()
.find(|(key, _)| key == name)
.map(|(_, value)| value.as_str())
}
}
/// Parse HTML into a tree, tolerating anything.
///
/// Text that no rule understands still ends up somewhere sensible, and an
/// unterminated construct simply ends the document: a broken page must never
/// cost an answer that is already half readable.
fn parse(html: &str) -> Element {
let mut stack: Vec = vec![Element::default()];
let bytes = html.as_bytes();
let mut cursor = 0usize;
while cursor < bytes.len() {
if bytes.get(cursor) != Some(&b'<') {
let start = cursor;
while let Some(byte) = bytes.get(cursor) {
if *byte == b'<' {
break;
}
cursor += 1;
}
if let Some(text) = html.get(start..cursor) {
push_text_node(&mut stack, text);
}
continue;
}
let rest = html.get(cursor..).unwrap_or("");
// Comments, doctypes and processing instructions carry no answer text.
if rest.starts_with("") {
Some(offset) => cursor + offset + 3,
None => bytes.len(),
};
continue;
}
if rest.starts_with("') {
Some(offset) => cursor + offset + 1,
None => bytes.len(),
};
continue;
}
if rest.starts_with("") {
let name_start = cursor + 2;
let mut end = name_start;
while let Some(byte) = bytes.get(end) {
if byte.is_ascii_whitespace() || *byte == b'>' {
break;
}
end += 1;
}
let name = html.get(name_start..end).unwrap_or("").to_ascii_lowercase();
cursor = match html.get(end..).and_then(|tail| tail.find('>')) {
Some(offset) => end + offset + 1,
None => bytes.len(),
};
close_element(&mut stack, &name);
continue;
}
if rest
.as_bytes()
.get(1)
.is_some_and(|byte| byte.is_ascii_alphabetic())
{
let (name, attributes, self_closing, end) = read_start_tag(html, cursor);
cursor = end;
if name.is_empty() {
continue;
}
if RAW_TEXT_ELEMENTS.contains(&name.as_str()) && !self_closing {
cursor = skip_raw_text(html, end, &name);
continue;
}
let element = Element {
name: name.clone(),
attributes,
children: Vec::new(),
};
if self_closing || VOID_ELEMENTS.contains(&name.as_str()) || stack.len() >= MAX_DEPTH {
// A void element, a self-closing tag, or markup nested deeper
// than any real answer goes: the content is kept, the wrapper
// is not.
if let Some(parent) = stack.last_mut() {
parent.children.push(Node::Element(element));
}
} else {
stack.push(element);
}
continue;
}
// A `<` that opens nothing is text, which is what a browser shows too.
push_text_node(&mut stack, "<");
cursor += 1;
}
close_until(&mut stack, 1);
stack.pop().unwrap_or_default()
}
/// Read one start tag: its lowercased name, its decoded attributes, whether it
/// closes itself, and the position just past the final `>`.
fn read_start_tag(html: &str, start: usize) -> (String, Vec<(String, String)>, bool, usize) {
let bytes = html.as_bytes();
let mut cursor = start + 1;
let name_start = cursor;
while let Some(byte) = bytes.get(cursor) {
if byte.is_ascii_whitespace() || *byte == b'>' || *byte == b'/' {
break;
}
cursor += 1;
}
let name = html
.get(name_start..cursor)
.unwrap_or("")
.to_ascii_lowercase();
let mut attributes: Vec<(String, String)> = Vec::new();
let mut self_closing = false;
loop {
let before = cursor;
while let Some(byte) = bytes.get(cursor) {
if byte.is_ascii_whitespace() {
cursor += 1;
} else {
break;
}
}
match bytes.get(cursor).copied() {
None => break,
Some(b'>') => {
cursor += 1;
break;
}
Some(b'/') => {
self_closing = true;
cursor += 1;
}
Some(_) => {
let attribute_start = cursor;
while let Some(byte) = bytes.get(cursor) {
if byte.is_ascii_whitespace() || *byte == b'=' || *byte == b'>' || *byte == b'/'
{
break;
}
cursor += 1;
}
let attribute = html
.get(attribute_start..cursor)
.unwrap_or("")
.to_ascii_lowercase();
while let Some(byte) = bytes.get(cursor) {
if byte.is_ascii_whitespace() {
cursor += 1;
} else {
break;
}
}
let mut value = String::new();
if bytes.get(cursor) == Some(&b'=') {
cursor += 1;
while let Some(byte) = bytes.get(cursor) {
if byte.is_ascii_whitespace() {
cursor += 1;
} else {
break;
}
}
match bytes.get(cursor).copied() {
Some(quote) if quote == b'"' || quote == b'\'' => {
cursor += 1;
let value_start = cursor;
while let Some(byte) = bytes.get(cursor) {
if *byte == quote {
break;
}
cursor += 1;
}
value = html.get(value_start..cursor).unwrap_or("").to_string();
// An unterminated quote runs to the end of the
// document; the value stays a best effort.
if bytes.get(cursor) == Some("e) {
cursor += 1;
}
}
None | Some(b'>') => {}
Some(_) => {
let value_start = cursor;
while let Some(byte) = bytes.get(cursor) {
if byte.is_ascii_whitespace() || *byte == b'>' {
break;
}
cursor += 1;
}
value = html.get(value_start..cursor).unwrap_or("").to_string();
}
}
}
if !attribute.is_empty() {
attributes.push((attribute, decode_entities(&value)));
}
}
}
// Every turn consumes at least one byte; the guard is what keeps that
// true even for markup this reader has not thought of.
if cursor == before {
cursor += 1;
}
}
(name, attributes, self_closing, cursor)
}
/// Append a text node, decoding its entities on the way in.
fn push_text_node(stack: &mut [Element], text: &str) {
if let Some(parent) = stack.last_mut() {
parent.children.push(Node::Text(decode_entities(text)));
}
}
/// Close the innermost open element with this name.
///
/// A stray end tag is ignored, while one that closes an outer element also
/// closes the elements it still contains, which is what a browser does with
/// mismatched markup.
fn close_element(stack: &mut Vec, name: &str) {
if name.is_empty() {
return;
}
if let Some(index) = stack.iter().rposition(|element| element.name == name) {
close_until(stack, index);
}
}
/// Pop open elements down to `keep`, attaching each one to its new parent.
fn close_until(stack: &mut Vec, keep: usize) {
while stack.len() > keep {
match stack.pop() {
Some(element) => {
if let Some(parent) = stack.last_mut() {
parent.children.push(Node::Element(element));
}
}
None => return,
}
}
}
/// Skip the content of a raw text element, returning the position just past its
/// end tag.
fn skip_raw_text(html: &str, from: usize, name: &str) -> usize {
let needle = format!("{name}");
match find_ascii_case_insensitive(html.as_bytes(), needle.as_bytes(), from) {
Some(index) => match html.get(index..).and_then(|tail| tail.find('>')) {
Some(offset) => index + offset + 1,
None => html.len(),
},
None => html.len(),
}
}
/// Find `needle` in `haystack` from `from`, comparing ASCII case-insensitively.
fn find_ascii_case_insensitive(haystack: &[u8], needle: &[u8], from: usize) -> Option {
if needle.is_empty() || haystack.len() < needle.len() {
return None;
}
let last = haystack.len() - needle.len();
let mut index = from;
while index <= last {
let window = haystack.get(index..index + needle.len())?;
if window.eq_ignore_ascii_case(needle) {
return Some(index);
}
index += 1;
}
None
}
/// Decode the HTML entities that show up in captured answers.
///
/// Anything unrecognised is left exactly as it was written, because a stray
/// ampersand in prose is far more common than an entity this table has missed.
fn decode_entities(text: &str) -> String {
if !text.contains('&') {
return text.to_string();
}
let mut out = String::with_capacity(text.len());
let mut rest = text;
while let Some(index) = rest.find('&') {
out.push_str(rest.get(..index).unwrap_or(""));
let tail = rest.get(index..).unwrap_or("");
match decode_one(tail) {
Some((decoded, consumed)) => {
out.push(decoded);
rest = tail.get(consumed..).unwrap_or("");
}
None => {
out.push('&');
rest = tail.get(1..).unwrap_or("");
}
}
}
out.push_str(rest);
out
}
/// Decode the single entity a string starts with, with its length in bytes.
fn decode_one(tail: &str) -> Option<(char, usize)> {
let body = tail.strip_prefix('&')?;
let end = body.find(';')?;
let name = body.get(..end)?;
// A bare ampersand must not swallow the rest of a sentence, so only a short
// reference is considered at all.
if name.is_empty() || name.len() > 32 {
return None;
}
let decoded = if let Some(digits) = name.strip_prefix('#') {
let code = match digits.strip_prefix(['x', 'X']) {
Some(hex) => u32::from_str_radix(hex, 16).ok()?,
None => digits.parse::().ok()?,
};
match char::from_u32(code) {
// Out of range or a lone surrogate: the replacement character is
// the honest answer, and it cannot panic.
Some(character) if character != '\0' => character,
_ => '\u{fffd}',
}
} else {
named_entity(name)?
};
Some((decoded, end + 2))
}
/// The character behind a named entity, or `None` when it is not one we know.
fn named_entity(name: &str) -> Option {
// Matching ignores case: pages spell `&` and `&Nbsp;` too.
let character = match name.to_ascii_lowercase().as_str() {
"amp" => '&',
"lt" => '<',
"gt" => '>',
"quot" => '"',
"apos" => '\'',
"nbsp" => '\u{a0}',
"mdash" => '—',
"ndash" => '–',
"hellip" => '…',
"laquo" => '«',
"raquo" => '»',
"times" => '×',
"copy" => '©',
"reg" => '®',
"deg" => '°',
"middot" => '·',
"bull" => '•',
"eacute" => 'é',
"egrave" => 'è',
"agrave" => 'à',
"ccedil" => 'ç',
"uuml" => 'ü',
"ouml" => 'ö',
"auml" => 'ä',
_ => return None,
};
Some(character)
}
/// Collapse runs of layout whitespace into a single space, the way HTML itself
/// does. A non-breaking space is content rather than layout and survives.
fn collapse_whitespace(text: &str) -> String {
let mut out = String::with_capacity(text.len());
let mut in_space = false;
for character in text.chars() {
match character {
' ' | '\t' | '\n' | '\r' | '\u{c}' => {
if !in_space {
out.push(' ');
in_space = true;
}
}
_ => {
out.push(character);
in_space = false;
}
}
}
out
}
/// One unit of work for the iterative renderer.
enum Work<'a> {
/// Render a node of the parsed tree.
Node(&'a Node),
/// Append literal Markdown.
Emit(String),
/// Start a new line when the current one already holds something.
Break,
/// Push a buffer that the matching `Finish` will consume.
Capture(Capture),
/// Close the innermost capture and turn it into Markdown.
Finish,
}
/// Why a buffer is being captured: the text has to be complete before it can be
/// wrapped (a link label), prefixed (a quote or a list item) or cut into cells.
enum Capture {
Link { href: String },
Quote,
Item { ordered: bool, index: usize },
Row { header: bool, cells: Vec },
Cell { colspan: usize },
}
/// Render a slice of nodes into Markdown.
///
/// The walk runs on an explicit stack of work items - a node becomes the items
/// that produce its Markdown - so the traversal stays flat however deeply the
/// document nests.
fn render(nodes: &[Node]) -> String {
let mut buffers: Vec = vec![String::new()];
let mut captures: Vec = Vec::new();
let mut work: Vec> = nodes.iter().rev().map(Work::Node).collect();
while let Some(item) = work.pop() {
match item {
Work::Emit(text) => push_text(&mut buffers, &text),
Work::Break => ensure_newline(&mut buffers),
Work::Capture(capture) => {
captures.push(capture);
buffers.push(String::new());
}
Work::Finish => finish_capture(&mut buffers, &mut captures),
Work::Node(Node::Text(text)) => {
let collapsed = collapse_whitespace(text);
// The indentation of the markup must not become a leading space
// on a fresh line, while on a line in progress it is what keeps
// `a b` from running into `ab`.
let collapsed = if at_line_start(&buffers) {
collapsed.trim_start_matches(' ')
} else {
collapsed.as_str()
};
if !collapsed.is_empty() {
push_text(&mut buffers, collapsed);
}
}
Work::Node(Node::Element(element)) => push_element(element, &mut work),
}
}
// A capture left open would swallow its text, so anything still pending is
// unwound into the buffer below it.
while buffers.len() > 1 {
finish_capture(&mut buffers, &mut captures);
}
buffers.pop().unwrap_or_default()
}
/// Turn one element into the work items that render it.
fn push_element<'a>(element: &'a Element, work: &mut Vec>) {
let mut parts: Vec> = Vec::new();
match element.name.as_str() {
// Markdown's hard break is two trailing spaces. `tidy` trims them, so
// the caller sees a plain newline, which is what a reader expects.
"br" => parts.push(Work::Emit(" \n".to_string())),
"hr" => {
parts.push(Work::Break);
parts.push(Work::Emit("---\n\n".to_string()));
}
"img" => {
let image = image_markdown(element);
if !image.is_empty() {
parts.push(Work::Emit(image));
}
}
// Fenced blocks and inline code spans read their own text, so their
// children are never walked as markup.
"pre" => {
parts.push(Work::Break);
parts.push(Work::Emit(fenced_block(element)));
}
"code" => {
let span = inline_code(element);
if !span.is_empty() {
parts.push(Work::Emit(span));
}
}
"blockquote" => {
parts.push(Work::Break);
parts.push(Work::Capture(Capture::Quote));
parts.extend(children(element));
parts.push(Work::Finish);
}
"ul" | "ol" => {
parts.push(Work::Break);
push_list(element, element.name == "ol", &mut parts);
}
"table" => {
parts.push(Work::Break);
push_table(element, &mut parts);
}
"a" => {
let href = element.attr("href").unwrap_or_default().to_string();
parts.push(Work::Capture(Capture::Link { href }));
parts.extend(children(element));
parts.push(Work::Finish);
}
"h1" | "h2" | "h3" | "h4" | "h5" | "h6" => {
let level = heading_level(&element.name);
parts.push(Work::Break);
parts.push(Work::Emit(format!("{} ", "#".repeat(level))));
parts.extend(children(element));
parts.push(Work::Emit("\n\n".to_string()));
}
"strong" | "b" => parts.extend(wrapped("**", element)),
"em" | "i" => parts.extend(wrapped("*", element)),
"del" | "s" | "strike" => parts.extend(wrapped("~~", element)),
name if BLOCK_ELEMENTS.contains(&name) => {
parts.push(Work::Break);
parts.extend(children(element));
parts.push(Work::Emit("\n\n".to_string()));
}
// Unknown elements, and cells or items met outside their table or list,
// keep their content and nothing else.
_ => parts.extend(children(element)),
}
// The stack is popped from the back, so the parts go on in reverse.
for part in parts.into_iter().rev() {
work.push(part);
}
}
/// The heading level of an `h1`..`h6` element.
fn heading_level(name: &str) -> usize {
name.as_bytes()
.last()
.map_or(1, |digit| usize::from(digit.saturating_sub(b'0')))
.clamp(1, 6)
}
/// Emphasised or struck-through content: a marker on either side.
fn wrapped<'a>(marker: &str, element: &'a Element) -> Vec> {
let mut parts = vec![Work::Emit(marker.to_string())];
parts.extend(children(element));
parts.push(Work::Emit(marker.to_string()));
parts
}
/// One work item per child, in document order.
fn children<'a>(element: &'a Element) -> Vec> {
element.children.iter().map(Work::Node).collect()
}
/// Add the items that render a `ul` or `ol`.
fn push_list<'a>(element: &'a Element, ordered: bool, parts: &mut Vec>) {
let mut index = 0usize;
for child in &element.children {
match child {
Node::Element(item) if item.name == "li" => {
parts.push(Work::Capture(Capture::Item { ordered, index }));
parts.extend(children(item));
parts.push(Work::Finish);
index += 1;
}
// Anything a list holds besides its items is kept as it is.
other => parts.push(Work::Node(other)),
}
}
// A list is followed by a blank line, so the next block starts cleanly.
parts.push(Work::Emit("\n".to_string()));
}
/// Add the items that render a `table`.
fn push_table<'a>(element: &'a Element, parts: &mut Vec>) {
let rows = table_rows(element);
if rows.is_empty() {
// A table with no rows holds nothing a reader would miss.
return;
}
// Markdown tables need a header row; when the markup names none, the first
// row takes that role, which is what a reader of the page would assume.
let has_header = rows.iter().any(|(_, header)| *header);
for (index, (row, header)) in rows.iter().enumerate() {
let header = *header || (!has_header && index == 0);
parts.push(Work::Capture(Capture::Row {
header,
cells: Vec::new(),
}));
for cell in row_cells(row) {
parts.push(Work::Capture(Capture::Cell {
colspan: colspan(cell),
}));
parts.extend(children(cell));
parts.push(Work::Finish);
}
parts.push(Work::Finish);
}
parts.push(Work::Emit("\n".to_string()));
}
/// The rows of a table in document order, each with the flag telling whether it
/// is a header row (inside `thead`, or holding `th` cells).
fn table_rows(table: &Element) -> Vec<(&Element, bool)> {
let mut rows = Vec::new();
for child in &table.children {
let Node::Element(child) = child else {
continue;
};
match child.name.as_str() {
"tr" => rows.push((child, row_has_th(child))),
"thead" | "tbody" | "tfoot" => {
let in_header = child.name == "thead";
for row in &child.children {
if let Node::Element(row) = row {
if row.name == "tr" {
rows.push((row, in_header || row_has_th(row)));
}
}
}
}
_ => {}
}
}
rows
}
/// True when a row holds header cells of its own.
fn row_has_th(row: &Element) -> bool {
row_cells(row).iter().any(|cell| cell.name == "th")
}
/// The cells of a row, in document order.
fn row_cells(row: &Element) -> Vec<&Element> {
row.children
.iter()
.filter_map(|child| match child {
Node::Element(cell) if cell.name == "th" || cell.name == "td" => Some(cell),
_ => None,
})
.collect()
}
/// How many columns a cell covers; a missing or silly value means one, and the
/// cap keeps a hostile `colspan` from asking for an enormous row.
fn colspan(cell: &Element) -> usize {
cell.attr("colspan")
.and_then(|value| value.trim().parse::().ok())
.filter(|span| *span > 0)
.map_or(1, |span| span.min(64))
}
/// Close the innermost capture and append its Markdown to the buffer below it.
fn finish_capture(buffers: &mut Vec, captures: &mut Vec) {
let text = match buffers.pop() {
Some(text) => text,
None => return,
};
match captures.pop() {
Some(Capture::Link { href }) => {
let label = text.trim();
let rendered = if href.is_empty() {
label.to_string()
} else if label == href {
href
} else if label.is_empty() {
String::new()
} else {
format!("[{label}]({href})")
};
push_text(buffers, &rendered);
}
Some(Capture::Quote) => {
let mut quoted = String::new();
let body = text.trim();
if !body.is_empty() {
for line in body.split('\n') {
let line = line.trim_end();
if line.is_empty() {
quoted.push_str(">\n");
} else {
quoted.push_str("> ");
quoted.push_str(line);
quoted.push('\n');
}
}
quoted.push('\n');
}
push_text(buffers, "ed);
}
Some(Capture::Item { ordered, index }) => {
// The bullet shares its line with the first line of the item; the
// rest continues indented under it, two spaces per level.
let bullet = if ordered {
format!("{}.", index + 1)
} else {
"-".to_string()
};
let mut rendered = String::new();
for (number, line) in text.trim().split('\n').enumerate() {
let line = line.trim_end();
if number == 0 {
rendered.push_str(&bullet);
rendered.push(' ');
rendered.push_str(line.trim_start());
} else if !line.is_empty() {
rendered.push_str(" ");
rendered.push_str(line);
}
rendered.push('\n');
}
push_text(buffers, &rendered);
}
Some(Capture::Row { header, cells }) => {
let mut rendered = String::new();
if !cells.is_empty() {
rendered.push_str("| ");
rendered.push_str(&cells.join(" | "));
rendered.push_str(" |\n");
if header {
rendered.push_str("| ");
rendered.push_str(&vec!["---"; cells.len()].join(" | "));
rendered.push_str(" |\n");
}
}
push_text(buffers, &rendered);
}
Some(Capture::Cell { colspan }) => {
// A cell is a single line: newlines inside it become spaces, and a
// pipe would otherwise cut the row into extra columns.
let cell = text
.split_whitespace()
.collect::>()
.join(" ")
.replace('|', "\\|");
match captures.last_mut() {
Some(Capture::Row { cells, .. }) => {
for _ in 0..colspan {
cells.push(cell.clone());
}
}
_ => push_text(buffers, &cell),
}
}
None => push_text(buffers, &text),
}
}
/// An image with its alternative text, as Markdown.
fn image_markdown(element: &Element) -> String {
let alt = element.attr("alt").unwrap_or_default();
let src = element.attr("src").unwrap_or_default();
if alt.is_empty() && src.is_empty() {
// A decorative spacer: nothing a reader would miss.
return String::new();
}
format!("")
}
/// An inline code span: one backtick, or two when the text contains one.
fn inline_code(element: &Element) -> String {
let text = text_content(element);
let text = text.trim();
if text.is_empty() {
return String::new();
}
let ticks = if text.contains('`') { "``" } else { "`" };
format!("{ticks}{text}{ticks}")
}
/// A `pre` as a fenced block, named with the language of an inner `code`
/// element when the page declares one.
fn fenced_block(element: &Element) -> String {
let text = text_content(element);
let text = text.trim_matches(['\n', '\r']);
let language = code_language(element).unwrap_or_default();
let fence = fence_for(text);
format!("{fence}{language}\n{text}\n{fence}\n\n")
}
/// A fence longer than any run of backticks in the text, and at least three.
fn fence_for(text: &str) -> String {
let mut longest = 0usize;
let mut run = 0usize;
for character in text.chars() {
if character == '`' {
run += 1;
longest = longest.max(run);
} else {
run = 0;
}
}
"`".repeat(longest.max(2) + 1)
}
/// The language named by `class="language-xxx"` on the `code` child of a `pre`.
fn code_language(element: &Element) -> Option {
let code = element.children.iter().find_map(|child| match child {
Node::Element(child) if child.name == "code" => Some(child),
_ => None,
})?;
code.attr("class")?
.split_whitespace()
.find_map(|token| token.strip_prefix("language-"))
.filter(|language| !language.is_empty())
.map(|language| language.to_string())
}
/// The text of a subtree, with its whitespace exactly as written.
///
/// Fenced blocks and code spans use this instead of the renderer, because
/// Markdown code is literal text rather than markup.
fn text_content(element: &Element) -> String {
let mut out = String::new();
let mut stack: Vec<&Node> = element.children.iter().rev().collect();
while let Some(node) = stack.pop() {
match node {
Node::Text(text) => out.push_str(text),
Node::Element(child) => {
if child.name == "br" {
out.push('\n');
}
stack.extend(child.children.iter().rev());
}
}
}
out
}
/// Append text to the innermost buffer.
fn push_text(buffers: &mut [String], text: &str) {
if let Some(buffer) = buffers.last_mut() {
buffer.push_str(text);
}
}
/// End the current line when it already holds something.
fn ensure_newline(buffers: &mut [String]) {
if let Some(buffer) = buffers.last_mut() {
if !buffer.is_empty() && !buffer.ends_with('\n') {
buffer.push('\n');
}
}
}
/// True when nothing has been written on the current line yet.
fn at_line_start(buffers: &[String]) -> bool {
buffers
.last()
.is_none_or(|buffer| buffer.is_empty() || buffer.ends_with('\n'))
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn empty_input_produces_empty_output() {
assert_eq!(html_to_markdown(""), "");
assert_eq!(html_to_markdown(" \n\t "), "");
}
#[test]
fn paragraphs_keep_their_text_without_the_markup() {
assert_eq!(html_to_markdown("Hello world
"), "Hello world\n");
}
#[test]
fn two_paragraphs_are_separated_by_a_blank_line() {
assert_eq!(html_to_markdown("One
Two
"), "One\n\nTwo\n");
}
#[test]
fn layout_whitespace_never_leaves_blank_runs() {
assert_eq!(html_to_markdown("a
\n\n\nb
\n\n\n"), "a\n\nb\n");
}
#[test]
fn line_break_ends_the_line() {
// The renderer writes Markdown's two-space hard break; `tidy` trims the
// trailing spaces, so the caller sees a plain newline.
assert_eq!(html_to_markdown("one
two
"), "one\ntwo\n");
assert_eq!(html_to_markdown("one
two
"), "one\ntwo\n");
}
#[test]
fn headings_use_one_hash_per_level() {
assert_eq!(
html_to_markdown("Title
Deep
"),
"# Title\n\n#### Deep\n"
);
}
#[test]
fn emphasis_becomes_markdown_markers() {
let html =
"bold also bold italic also gone
";
assert_eq!(
html_to_markdown(html),
"**bold** **also bold** *italic* *also* ~~gone~~\n"
);
}
#[test]
fn inline_code_uses_one_backtick() {
assert_eq!(
html_to_markdown("run cargo test now
"),
"run `cargo test` now\n"
);
}
#[test]
fn inline_code_containing_a_backtick_uses_two() {
assert_eq!(
html_to_markdown("x `y` z
"),
"``x `y` z``\n"
);
}
#[test]
fn fenced_code_block_without_a_language() {
assert_eq!(
html_to_markdown("fn main() {}"),
"```\nfn main() {}\n```\n"
);
}
#[test]
fn fenced_code_block_takes_the_language_from_the_code_class() {
let html = r#"let x = 1;
"#;
assert_eq!(html_to_markdown(html), "```rust\nlet x = 1;\n```\n");
}
#[test]
fn fenced_code_block_leaves_its_text_alone() {
let html = "a < b && c\n indented\n
";
assert_eq!(
html_to_markdown(html),
"```\na < b && c\n indented\n```\n"
);
}
#[test]
fn fence_grows_when_the_code_contains_backticks() {
assert_eq!(
html_to_markdown("a ``` b
"),
"````\na ``` b\n````\n"
);
}
#[test]
fn named_and_numeric_entities_are_decoded() {
assert_eq!(
html_to_markdown("a & b <tag> AB
"),
"a & b AB\n"
);
assert_eq!(html_to_markdown("😀
"), "\u{1f600}\n");
}
#[test]
fn punctuation_entities_are_decoded() {
assert_eq!(
html_to_markdown("5 × 3 — café
"),
"5 × 3 — café\n"
);
}
#[test]
fn unknown_entities_are_left_alone() {
assert_eq!(
html_to_markdown("a &unknown; b & c
"),
"a &unknown; b & c\n"
);
}
#[test]
fn invalid_numeric_references_do_not_panic() {
assert_eq!(
html_to_markdown(""),
"\u{fffd}\u{fffd}\n"
);
}
#[test]
fn non_breaking_spaces_are_kept() {
assert_eq!(html_to_markdown("a b
"), "a\u{a0}b\n");
}
#[test]
fn attributes_are_decoded() {
let html = r#"docs"#;
assert_eq!(html_to_markdown(html), "[docs](https://x.test/?a=1&b=2)\n");
}
#[test]
fn attribute_names_are_case_insensitive() {
let html = r#"https://e.test"#;
assert_eq!(html_to_markdown(html), "https://e.test\n");
}
#[test]
fn attribute_values_may_be_quoted_or_bare() {
assert_eq!(
html_to_markdown("x"),
"[x](https://s.test)\n"
);
assert_eq!(
html_to_markdown("
"),
"\n"
);
}
#[test]
fn nested_lists_are_indented() {
let html = "- one
- inner
- two
";
assert_eq!(html_to_markdown(html), "- one\n - inner\n- two\n");
}
#[test]
fn ordered_list_items_are_numbered() {
assert_eq!(
html_to_markdown("- first
- second
"),
"1. first\n2. second\n"
);
}
#[test]
fn list_item_blocks_continue_on_indented_lines() {
assert_eq!(
html_to_markdown("- first
second
"),
"- first\n second\n"
);
}
#[test]
fn blockquote_prefixes_every_line() {
let html = "first
second
";
assert_eq!(html_to_markdown(html), "> first\n>\n> second\n");
}
#[test]
fn link_with_a_different_text_uses_the_markdown_form() {
let html = r#"see the docs
"#;
assert_eq!(html_to_markdown(html), "see [the docs](https://e.test)\n");
}
#[test]
fn link_whose_text_is_its_href_collapses_to_the_url() {
let html = r#""#;
assert_eq!(html_to_markdown(html), "https://e.test\n");
}
#[test]
fn image_uses_its_alt_text_and_source() {
let html = r#"
"#;
assert_eq!(html_to_markdown(html), "\n");
}
#[test]
fn horizontal_rule_gets_its_own_line() {
assert_eq!(
html_to_markdown("above
below
"),
"above\n\n---\n\nbelow\n"
);
}
#[test]
fn table_with_a_header_becomes_a_github_table() {
let html = "Name Value \
a 1
";
assert_eq!(
html_to_markdown(html),
"| Name | Value |\n| --- | --- |\n| a | 1 |\n"
);
}
#[test]
fn table_without_a_header_promotes_the_first_row() {
let html = "h1 h2 a b
";
assert_eq!(
html_to_markdown(html),
"| h1 | h2 |\n| --- | --- |\n| a | b |\n"
);
}
#[test]
fn pipes_inside_a_cell_are_escaped() {
let html = "a b x | y z
";
assert_eq!(
html_to_markdown(html),
"| a | b |\n| --- | --- |\n| x \\| y | z |\n"
);
}
#[test]
fn colspan_repeats_the_cell_content() {
let html =
r#"a b span
"#;
assert_eq!(
html_to_markdown(html),
"| a | b |\n| --- | --- |\n| span | span |\n"
);
}
#[test]
fn a_table_without_rows_produces_nothing() {
assert_eq!(html_to_markdown("
"), "");
}
#[test]
fn unknown_elements_keep_their_content() {
let html = r#"text "#;
assert_eq!(html_to_markdown(html), "text\n");
}
#[test]
fn comments_doctypes_and_raw_text_are_dropped() {
let html = "\
kept
";
assert_eq!(html_to_markdown(html), "kept\n");
}
#[test]
fn an_unclosed_tag_keeps_the_text() {
assert_eq!(html_to_markdown("hello world"), "hello **world**\n");
}
#[test]
fn a_stray_angle_bracket_is_text() {
assert_eq!(
html_to_markdown("
2 < 3 and 4 > 1
"),
"2 < 3 and 4 > 1\n"
);
}
#[test]
fn an_unterminated_attribute_quote_stops_at_the_end() {
// The quoted value runs to the end of the document: everything before it
// is still rendered and nothing hangs.
let html = "visible
tail";
assert_eq!(html_to_markdown(html), "visible\n");
}
#[test]
fn a_stray_end_tag_is_ignored() {
assert_eq!(html_to_markdown("bold
"), "**bold**\n");
}
#[test]
fn unterminated_constructs_terminate() {
assert_eq!(html_to_markdown("<>&&&"), "<>&&&\n");
assert_eq!(html_to_markdown("a