//! HTML to Markdown conversion for captured answers. //! //! The gateway reads answers from a signed-in browser page, so the text it //! forwards was written for a human looking at rendered HTML. `innerText` keeps //! the words and drops the structure: a table collapses into a column of //! unrelated lines and a code block loses its fences, so the next model in the //! chain cannot tell code from prose. Capturing `innerHTML` instead keeps that //! structure, and this module turns it back into Markdown. //! //! The converter is deliberately self-contained: no browser, no network and no //! HTML parser crate, because the markup is whatever a third party UI happens //! to ship. The tokenizer is tolerant by design, so malformed markup degrades //! into the readable text around it, and both parsing and rendering run on //! explicit stacks, so no input can overflow the stack, spin forever or panic. /// Depth beyond which start tags are ignored. /// /// Real answers nest a handful of levels; the cap only stops a hostile page /// from making the converter allocate without bound. const MAX_DEPTH: usize = 128; /// Elements whose content is text rather than markup, so it is skipped whole. const RAW_TEXT_ELEMENTS: [&str; 4] = ["script", "style", "textarea", "title"]; /// Elements that never have children and never need a closing tag. const VOID_ELEMENTS: [&str; 14] = [ "area", "base", "br", "col", "embed", "hr", "img", "input", "link", "meta", "param", "source", "track", "wbr", ]; /// Elements that are a block of their own and are followed by a blank line. const BLOCK_ELEMENTS: [&str; 12] = [ "p", "div", "section", "article", "header", "footer", "main", "aside", "figure", "figcaption", "dd", "dt", ]; /// Convert the inner HTML of one answer node into Markdown. pub fn html_to_markdown(html: &str) -> String { if html.trim().is_empty() { return String::new(); } let root = parse(html); tidy(&render(&root.children)) } /// True when the HTML contains a construct a plain text extraction would lose /// (a fenced code block, a table row, a list item, a heading, a link). pub fn has_rich_structure(html: &str) -> bool { // Callers use this to choose between `innerText` and `innerHTML`, so a cheap // approximate scan is enough; an exact answer would mean parsing twice. const MARKERS: [&str; 12] = [ " String { let mut out = String::with_capacity(markdown.len()); let mut pending_blank = false; let mut fence: Option = None; for raw_line in markdown.split('\n') { // A carriage return only survives a CRLF document: line ending noise. let line = raw_line.strip_suffix('\r').unwrap_or(raw_line); if let Some(opening) = fence.as_deref() { // Inside a fence the text is code the user wrote, so it is copied // through untouched, blank lines, indentation and trailing spaces // included. out.push_str(line); out.push('\n'); if closes_fence(line, opening) { fence = None; } continue; } let line = line.trim_end_matches([' ', '\t']); if line.is_empty() { // Remembered rather than written: a blank line only exists when // something follows it, which also drops the ones at the end. pending_blank = true; continue; } if let Some(opening) = opening_fence(line) { fence = Some(opening); } if pending_blank && !out.is_empty() { out.push('\n'); } pending_blank = false; out.push_str(line); out.push('\n'); } out } /// The run of backticks that opens a fenced block, when a line starts one. fn opening_fence(line: &str) -> Option { let run: String = line .trim_start() .chars() .take_while(|character| *character == '`') .collect(); if run.len() >= 3 { Some(run) } else { None } } /// True when a line closes the fence it is inside. fn closes_fence(line: &str, opening: &str) -> bool { let trimmed = line.trim(); trimmed.len() >= opening.len() && trimmed.chars().all(|character| character == '`') } /// One node of the parsed document. #[derive(Debug)] enum Node { Element(Element), /// Entity-decoded text, otherwise exactly as written: whitespace is only /// collapsed when the text is rendered, so a fenced block still reads it /// verbatim. Text(String), } /// One element: a lowercased name, decoded attributes and its children. #[derive(Debug, Default)] struct Element { name: String, attributes: Vec<(String, String)>, children: Vec, } impl Element { /// The value of an attribute, matched by its lowercased name. fn attr(&self, name: &str) -> Option<&str> { self.attributes .iter() .find(|(key, _)| key == name) .map(|(_, value)| value.as_str()) } } /// Parse HTML into a tree, tolerating anything. /// /// Text that no rule understands still ends up somewhere sensible, and an /// unterminated construct simply ends the document: a broken page must never /// cost an answer that is already half readable. fn parse(html: &str) -> Element { let mut stack: Vec = vec![Element::default()]; let bytes = html.as_bytes(); let mut cursor = 0usize; while cursor < bytes.len() { if bytes.get(cursor) != Some(&b'<') { let start = cursor; while let Some(byte) = bytes.get(cursor) { if *byte == b'<' { break; } cursor += 1; } if let Some(text) = html.get(start..cursor) { push_text_node(&mut stack, text); } continue; } let rest = html.get(cursor..).unwrap_or(""); // Comments, doctypes and processing instructions carry no answer text. if rest.starts_with("") { Some(offset) => cursor + offset + 3, None => bytes.len(), }; continue; } if rest.starts_with("') { Some(offset) => cursor + offset + 1, None => bytes.len(), }; continue; } if rest.starts_with("' { break; } end += 1; } let name = html.get(name_start..end).unwrap_or("").to_ascii_lowercase(); cursor = match html.get(end..).and_then(|tail| tail.find('>')) { Some(offset) => end + offset + 1, None => bytes.len(), }; close_element(&mut stack, &name); continue; } if rest .as_bytes() .get(1) .is_some_and(|byte| byte.is_ascii_alphabetic()) { let (name, attributes, self_closing, end) = read_start_tag(html, cursor); cursor = end; if name.is_empty() { continue; } if RAW_TEXT_ELEMENTS.contains(&name.as_str()) && !self_closing { cursor = skip_raw_text(html, end, &name); continue; } let element = Element { name: name.clone(), attributes, children: Vec::new(), }; if self_closing || VOID_ELEMENTS.contains(&name.as_str()) || stack.len() >= MAX_DEPTH { // A void element, a self-closing tag, or markup nested deeper // than any real answer goes: the content is kept, the wrapper // is not. if let Some(parent) = stack.last_mut() { parent.children.push(Node::Element(element)); } } else { stack.push(element); } continue; } // A `<` that opens nothing is text, which is what a browser shows too. push_text_node(&mut stack, "<"); cursor += 1; } close_until(&mut stack, 1); stack.pop().unwrap_or_default() } /// Read one start tag: its lowercased name, its decoded attributes, whether it /// closes itself, and the position just past the final `>`. fn read_start_tag(html: &str, start: usize) -> (String, Vec<(String, String)>, bool, usize) { let bytes = html.as_bytes(); let mut cursor = start + 1; let name_start = cursor; while let Some(byte) = bytes.get(cursor) { if byte.is_ascii_whitespace() || *byte == b'>' || *byte == b'/' { break; } cursor += 1; } let name = html .get(name_start..cursor) .unwrap_or("") .to_ascii_lowercase(); let mut attributes: Vec<(String, String)> = Vec::new(); let mut self_closing = false; loop { let before = cursor; while let Some(byte) = bytes.get(cursor) { if byte.is_ascii_whitespace() { cursor += 1; } else { break; } } match bytes.get(cursor).copied() { None => break, Some(b'>') => { cursor += 1; break; } Some(b'/') => { self_closing = true; cursor += 1; } Some(_) => { let attribute_start = cursor; while let Some(byte) = bytes.get(cursor) { if byte.is_ascii_whitespace() || *byte == b'=' || *byte == b'>' || *byte == b'/' { break; } cursor += 1; } let attribute = html .get(attribute_start..cursor) .unwrap_or("") .to_ascii_lowercase(); while let Some(byte) = bytes.get(cursor) { if byte.is_ascii_whitespace() { cursor += 1; } else { break; } } let mut value = String::new(); if bytes.get(cursor) == Some(&b'=') { cursor += 1; while let Some(byte) = bytes.get(cursor) { if byte.is_ascii_whitespace() { cursor += 1; } else { break; } } match bytes.get(cursor).copied() { Some(quote) if quote == b'"' || quote == b'\'' => { cursor += 1; let value_start = cursor; while let Some(byte) = bytes.get(cursor) { if *byte == quote { break; } cursor += 1; } value = html.get(value_start..cursor).unwrap_or("").to_string(); // An unterminated quote runs to the end of the // document; the value stays a best effort. if bytes.get(cursor) == Some("e) { cursor += 1; } } None | Some(b'>') => {} Some(_) => { let value_start = cursor; while let Some(byte) = bytes.get(cursor) { if byte.is_ascii_whitespace() || *byte == b'>' { break; } cursor += 1; } value = html.get(value_start..cursor).unwrap_or("").to_string(); } } } if !attribute.is_empty() { attributes.push((attribute, decode_entities(&value))); } } } // Every turn consumes at least one byte; the guard is what keeps that // true even for markup this reader has not thought of. if cursor == before { cursor += 1; } } (name, attributes, self_closing, cursor) } /// Append a text node, decoding its entities on the way in. fn push_text_node(stack: &mut [Element], text: &str) { if let Some(parent) = stack.last_mut() { parent.children.push(Node::Text(decode_entities(text))); } } /// Close the innermost open element with this name. /// /// A stray end tag is ignored, while one that closes an outer element also /// closes the elements it still contains, which is what a browser does with /// mismatched markup. fn close_element(stack: &mut Vec, name: &str) { if name.is_empty() { return; } if let Some(index) = stack.iter().rposition(|element| element.name == name) { close_until(stack, index); } } /// Pop open elements down to `keep`, attaching each one to its new parent. fn close_until(stack: &mut Vec, keep: usize) { while stack.len() > keep { match stack.pop() { Some(element) => { if let Some(parent) = stack.last_mut() { parent.children.push(Node::Element(element)); } } None => return, } } } /// Skip the content of a raw text element, returning the position just past its /// end tag. fn skip_raw_text(html: &str, from: usize, name: &str) -> usize { let needle = format!(" match html.get(index..).and_then(|tail| tail.find('>')) { Some(offset) => index + offset + 1, None => html.len(), }, None => html.len(), } } /// Find `needle` in `haystack` from `from`, comparing ASCII case-insensitively. fn find_ascii_case_insensitive(haystack: &[u8], needle: &[u8], from: usize) -> Option { if needle.is_empty() || haystack.len() < needle.len() { return None; } let last = haystack.len() - needle.len(); let mut index = from; while index <= last { let window = haystack.get(index..index + needle.len())?; if window.eq_ignore_ascii_case(needle) { return Some(index); } index += 1; } None } /// Decode the HTML entities that show up in captured answers. /// /// Anything unrecognised is left exactly as it was written, because a stray /// ampersand in prose is far more common than an entity this table has missed. fn decode_entities(text: &str) -> String { if !text.contains('&') { return text.to_string(); } let mut out = String::with_capacity(text.len()); let mut rest = text; while let Some(index) = rest.find('&') { out.push_str(rest.get(..index).unwrap_or("")); let tail = rest.get(index..).unwrap_or(""); match decode_one(tail) { Some((decoded, consumed)) => { out.push(decoded); rest = tail.get(consumed..).unwrap_or(""); } None => { out.push('&'); rest = tail.get(1..).unwrap_or(""); } } } out.push_str(rest); out } /// Decode the single entity a string starts with, with its length in bytes. fn decode_one(tail: &str) -> Option<(char, usize)> { let body = tail.strip_prefix('&')?; let end = body.find(';')?; let name = body.get(..end)?; // A bare ampersand must not swallow the rest of a sentence, so only a short // reference is considered at all. if name.is_empty() || name.len() > 32 { return None; } let decoded = if let Some(digits) = name.strip_prefix('#') { let code = match digits.strip_prefix(['x', 'X']) { Some(hex) => u32::from_str_radix(hex, 16).ok()?, None => digits.parse::().ok()?, }; match char::from_u32(code) { // Out of range or a lone surrogate: the replacement character is // the honest answer, and it cannot panic. Some(character) if character != '\0' => character, _ => '\u{fffd}', } } else { named_entity(name)? }; Some((decoded, end + 2)) } /// The character behind a named entity, or `None` when it is not one we know. fn named_entity(name: &str) -> Option { // Matching ignores case: pages spell `&` and `&Nbsp;` too. let character = match name.to_ascii_lowercase().as_str() { "amp" => '&', "lt" => '<', "gt" => '>', "quot" => '"', "apos" => '\'', "nbsp" => '\u{a0}', "mdash" => '—', "ndash" => '–', "hellip" => '…', "laquo" => '«', "raquo" => '»', "times" => '×', "copy" => '©', "reg" => '®', "deg" => '°', "middot" => '·', "bull" => '•', "eacute" => 'é', "egrave" => 'è', "agrave" => 'à', "ccedil" => 'ç', "uuml" => 'ü', "ouml" => 'ö', "auml" => 'ä', _ => return None, }; Some(character) } /// Collapse runs of layout whitespace into a single space, the way HTML itself /// does. A non-breaking space is content rather than layout and survives. fn collapse_whitespace(text: &str) -> String { let mut out = String::with_capacity(text.len()); let mut in_space = false; for character in text.chars() { match character { ' ' | '\t' | '\n' | '\r' | '\u{c}' => { if !in_space { out.push(' '); in_space = true; } } _ => { out.push(character); in_space = false; } } } out } /// One unit of work for the iterative renderer. enum Work<'a> { /// Render a node of the parsed tree. Node(&'a Node), /// Append literal Markdown. Emit(String), /// Start a new line when the current one already holds something. Break, /// Push a buffer that the matching `Finish` will consume. Capture(Capture), /// Close the innermost capture and turn it into Markdown. Finish, } /// Why a buffer is being captured: the text has to be complete before it can be /// wrapped (a link label), prefixed (a quote or a list item) or cut into cells. enum Capture { Link { href: String }, Quote, Item { ordered: bool, index: usize }, Row { header: bool, cells: Vec }, Cell { colspan: usize }, } /// Render a slice of nodes into Markdown. /// /// The walk runs on an explicit stack of work items - a node becomes the items /// that produce its Markdown - so the traversal stays flat however deeply the /// document nests. fn render(nodes: &[Node]) -> String { let mut buffers: Vec = vec![String::new()]; let mut captures: Vec = Vec::new(); let mut work: Vec> = nodes.iter().rev().map(Work::Node).collect(); while let Some(item) = work.pop() { match item { Work::Emit(text) => push_text(&mut buffers, &text), Work::Break => ensure_newline(&mut buffers), Work::Capture(capture) => { captures.push(capture); buffers.push(String::new()); } Work::Finish => finish_capture(&mut buffers, &mut captures), Work::Node(Node::Text(text)) => { let collapsed = collapse_whitespace(text); // The indentation of the markup must not become a leading space // on a fresh line, while on a line in progress it is what keeps // `a b` from running into `ab`. let collapsed = if at_line_start(&buffers) { collapsed.trim_start_matches(' ') } else { collapsed.as_str() }; if !collapsed.is_empty() { push_text(&mut buffers, collapsed); } } Work::Node(Node::Element(element)) => push_element(element, &mut work), } } // A capture left open would swallow its text, so anything still pending is // unwound into the buffer below it. while buffers.len() > 1 { finish_capture(&mut buffers, &mut captures); } buffers.pop().unwrap_or_default() } /// Turn one element into the work items that render it. fn push_element<'a>(element: &'a Element, work: &mut Vec>) { let mut parts: Vec> = Vec::new(); match element.name.as_str() { // Markdown's hard break is two trailing spaces. `tidy` trims them, so // the caller sees a plain newline, which is what a reader expects. "br" => parts.push(Work::Emit(" \n".to_string())), "hr" => { parts.push(Work::Break); parts.push(Work::Emit("---\n\n".to_string())); } "img" => { let image = image_markdown(element); if !image.is_empty() { parts.push(Work::Emit(image)); } } // Fenced blocks and inline code spans read their own text, so their // children are never walked as markup. "pre" => { parts.push(Work::Break); parts.push(Work::Emit(fenced_block(element))); } "code" => { let span = inline_code(element); if !span.is_empty() { parts.push(Work::Emit(span)); } } "blockquote" => { parts.push(Work::Break); parts.push(Work::Capture(Capture::Quote)); parts.extend(children(element)); parts.push(Work::Finish); } "ul" | "ol" => { parts.push(Work::Break); push_list(element, element.name == "ol", &mut parts); } "table" => { parts.push(Work::Break); push_table(element, &mut parts); } "a" => { let href = element.attr("href").unwrap_or_default().to_string(); parts.push(Work::Capture(Capture::Link { href })); parts.extend(children(element)); parts.push(Work::Finish); } "h1" | "h2" | "h3" | "h4" | "h5" | "h6" => { let level = heading_level(&element.name); parts.push(Work::Break); parts.push(Work::Emit(format!("{} ", "#".repeat(level)))); parts.extend(children(element)); parts.push(Work::Emit("\n\n".to_string())); } "strong" | "b" => parts.extend(wrapped("**", element)), "em" | "i" => parts.extend(wrapped("*", element)), "del" | "s" | "strike" => parts.extend(wrapped("~~", element)), name if BLOCK_ELEMENTS.contains(&name) => { parts.push(Work::Break); parts.extend(children(element)); parts.push(Work::Emit("\n\n".to_string())); } // Unknown elements, and cells or items met outside their table or list, // keep their content and nothing else. _ => parts.extend(children(element)), } // The stack is popped from the back, so the parts go on in reverse. for part in parts.into_iter().rev() { work.push(part); } } /// The heading level of an `h1`..`h6` element. fn heading_level(name: &str) -> usize { name.as_bytes() .last() .map_or(1, |digit| usize::from(digit.saturating_sub(b'0'))) .clamp(1, 6) } /// Emphasised or struck-through content: a marker on either side. fn wrapped<'a>(marker: &str, element: &'a Element) -> Vec> { let mut parts = vec![Work::Emit(marker.to_string())]; parts.extend(children(element)); parts.push(Work::Emit(marker.to_string())); parts } /// One work item per child, in document order. fn children<'a>(element: &'a Element) -> Vec> { element.children.iter().map(Work::Node).collect() } /// Add the items that render a `ul` or `ol`. fn push_list<'a>(element: &'a Element, ordered: bool, parts: &mut Vec>) { let mut index = 0usize; for child in &element.children { match child { Node::Element(item) if item.name == "li" => { parts.push(Work::Capture(Capture::Item { ordered, index })); parts.extend(children(item)); parts.push(Work::Finish); index += 1; } // Anything a list holds besides its items is kept as it is. other => parts.push(Work::Node(other)), } } // A list is followed by a blank line, so the next block starts cleanly. parts.push(Work::Emit("\n".to_string())); } /// Add the items that render a `table`. fn push_table<'a>(element: &'a Element, parts: &mut Vec>) { let rows = table_rows(element); if rows.is_empty() { // A table with no rows holds nothing a reader would miss. return; } // Markdown tables need a header row; when the markup names none, the first // row takes that role, which is what a reader of the page would assume. let has_header = rows.iter().any(|(_, header)| *header); for (index, (row, header)) in rows.iter().enumerate() { let header = *header || (!has_header && index == 0); parts.push(Work::Capture(Capture::Row { header, cells: Vec::new(), })); for cell in row_cells(row) { parts.push(Work::Capture(Capture::Cell { colspan: colspan(cell), })); parts.extend(children(cell)); parts.push(Work::Finish); } parts.push(Work::Finish); } parts.push(Work::Emit("\n".to_string())); } /// The rows of a table in document order, each with the flag telling whether it /// is a header row (inside `thead`, or holding `th` cells). fn table_rows(table: &Element) -> Vec<(&Element, bool)> { let mut rows = Vec::new(); for child in &table.children { let Node::Element(child) = child else { continue; }; match child.name.as_str() { "tr" => rows.push((child, row_has_th(child))), "thead" | "tbody" | "tfoot" => { let in_header = child.name == "thead"; for row in &child.children { if let Node::Element(row) = row { if row.name == "tr" { rows.push((row, in_header || row_has_th(row))); } } } } _ => {} } } rows } /// True when a row holds header cells of its own. fn row_has_th(row: &Element) -> bool { row_cells(row).iter().any(|cell| cell.name == "th") } /// The cells of a row, in document order. fn row_cells(row: &Element) -> Vec<&Element> { row.children .iter() .filter_map(|child| match child { Node::Element(cell) if cell.name == "th" || cell.name == "td" => Some(cell), _ => None, }) .collect() } /// How many columns a cell covers; a missing or silly value means one, and the /// cap keeps a hostile `colspan` from asking for an enormous row. fn colspan(cell: &Element) -> usize { cell.attr("colspan") .and_then(|value| value.trim().parse::().ok()) .filter(|span| *span > 0) .map_or(1, |span| span.min(64)) } /// Close the innermost capture and append its Markdown to the buffer below it. fn finish_capture(buffers: &mut Vec, captures: &mut Vec) { let text = match buffers.pop() { Some(text) => text, None => return, }; match captures.pop() { Some(Capture::Link { href }) => { let label = text.trim(); let rendered = if href.is_empty() { label.to_string() } else if label == href { href } else if label.is_empty() { String::new() } else { format!("[{label}]({href})") }; push_text(buffers, &rendered); } Some(Capture::Quote) => { let mut quoted = String::new(); let body = text.trim(); if !body.is_empty() { for line in body.split('\n') { let line = line.trim_end(); if line.is_empty() { quoted.push_str(">\n"); } else { quoted.push_str("> "); quoted.push_str(line); quoted.push('\n'); } } quoted.push('\n'); } push_text(buffers, "ed); } Some(Capture::Item { ordered, index }) => { // The bullet shares its line with the first line of the item; the // rest continues indented under it, two spaces per level. let bullet = if ordered { format!("{}.", index + 1) } else { "-".to_string() }; let mut rendered = String::new(); for (number, line) in text.trim().split('\n').enumerate() { let line = line.trim_end(); if number == 0 { rendered.push_str(&bullet); rendered.push(' '); rendered.push_str(line.trim_start()); } else if !line.is_empty() { rendered.push_str(" "); rendered.push_str(line); } rendered.push('\n'); } push_text(buffers, &rendered); } Some(Capture::Row { header, cells }) => { let mut rendered = String::new(); if !cells.is_empty() { rendered.push_str("| "); rendered.push_str(&cells.join(" | ")); rendered.push_str(" |\n"); if header { rendered.push_str("| "); rendered.push_str(&vec!["---"; cells.len()].join(" | ")); rendered.push_str(" |\n"); } } push_text(buffers, &rendered); } Some(Capture::Cell { colspan }) => { // A cell is a single line: newlines inside it become spaces, and a // pipe would otherwise cut the row into extra columns. let cell = text .split_whitespace() .collect::>() .join(" ") .replace('|', "\\|"); match captures.last_mut() { Some(Capture::Row { cells, .. }) => { for _ in 0..colspan { cells.push(cell.clone()); } } _ => push_text(buffers, &cell), } } None => push_text(buffers, &text), } } /// An image with its alternative text, as Markdown. fn image_markdown(element: &Element) -> String { let alt = element.attr("alt").unwrap_or_default(); let src = element.attr("src").unwrap_or_default(); if alt.is_empty() && src.is_empty() { // A decorative spacer: nothing a reader would miss. return String::new(); } format!("![{alt}]({src})") } /// An inline code span: one backtick, or two when the text contains one. fn inline_code(element: &Element) -> String { let text = text_content(element); let text = text.trim(); if text.is_empty() { return String::new(); } let ticks = if text.contains('`') { "``" } else { "`" }; format!("{ticks}{text}{ticks}") } /// A `pre` as a fenced block, named with the language of an inner `code` /// element when the page declares one. fn fenced_block(element: &Element) -> String { let text = text_content(element); let text = text.trim_matches(['\n', '\r']); let language = code_language(element).unwrap_or_default(); let fence = fence_for(text); format!("{fence}{language}\n{text}\n{fence}\n\n") } /// A fence longer than any run of backticks in the text, and at least three. fn fence_for(text: &str) -> String { let mut longest = 0usize; let mut run = 0usize; for character in text.chars() { if character == '`' { run += 1; longest = longest.max(run); } else { run = 0; } } "`".repeat(longest.max(2) + 1) } /// The language named by `class="language-xxx"` on the `code` child of a `pre`. fn code_language(element: &Element) -> Option { let code = element.children.iter().find_map(|child| match child { Node::Element(child) if child.name == "code" => Some(child), _ => None, })?; code.attr("class")? .split_whitespace() .find_map(|token| token.strip_prefix("language-")) .filter(|language| !language.is_empty()) .map(|language| language.to_string()) } /// The text of a subtree, with its whitespace exactly as written. /// /// Fenced blocks and code spans use this instead of the renderer, because /// Markdown code is literal text rather than markup. fn text_content(element: &Element) -> String { let mut out = String::new(); let mut stack: Vec<&Node> = element.children.iter().rev().collect(); while let Some(node) = stack.pop() { match node { Node::Text(text) => out.push_str(text), Node::Element(child) => { if child.name == "br" { out.push('\n'); } stack.extend(child.children.iter().rev()); } } } out } /// Append text to the innermost buffer. fn push_text(buffers: &mut [String], text: &str) { if let Some(buffer) = buffers.last_mut() { buffer.push_str(text); } } /// End the current line when it already holds something. fn ensure_newline(buffers: &mut [String]) { if let Some(buffer) = buffers.last_mut() { if !buffer.is_empty() && !buffer.ends_with('\n') { buffer.push('\n'); } } } /// True when nothing has been written on the current line yet. fn at_line_start(buffers: &[String]) -> bool { buffers .last() .is_none_or(|buffer| buffer.is_empty() || buffer.ends_with('\n')) } #[cfg(test)] mod tests { use super::*; #[test] fn empty_input_produces_empty_output() { assert_eq!(html_to_markdown(""), ""); assert_eq!(html_to_markdown(" \n\t "), ""); } #[test] fn paragraphs_keep_their_text_without_the_markup() { assert_eq!(html_to_markdown("

Hello world

"), "Hello world\n"); } #[test] fn two_paragraphs_are_separated_by_a_blank_line() { assert_eq!(html_to_markdown("

One

Two

"), "One\n\nTwo\n"); } #[test] fn layout_whitespace_never_leaves_blank_runs() { assert_eq!(html_to_markdown("

a

\n\n\n

b

\n\n\n"), "a\n\nb\n"); } #[test] fn line_break_ends_the_line() { // The renderer writes Markdown's two-space hard break; `tidy` trims the // trailing spaces, so the caller sees a plain newline. assert_eq!(html_to_markdown("

one
two

"), "one\ntwo\n"); assert_eq!(html_to_markdown("

one
two

"), "one\ntwo\n"); } #[test] fn headings_use_one_hash_per_level() { assert_eq!( html_to_markdown("

Title

Deep

"), "# Title\n\n#### Deep\n" ); } #[test] fn emphasis_becomes_markdown_markers() { let html = "

bold also bold italic also gone

"; assert_eq!( html_to_markdown(html), "**bold** **also bold** *italic* *also* ~~gone~~\n" ); } #[test] fn inline_code_uses_one_backtick() { assert_eq!( html_to_markdown("

run cargo test now

"), "run `cargo test` now\n" ); } #[test] fn inline_code_containing_a_backtick_uses_two() { assert_eq!( html_to_markdown("

x `y` z

"), "``x `y` z``\n" ); } #[test] fn fenced_code_block_without_a_language() { assert_eq!( html_to_markdown("
fn main() {}
"), "```\nfn main() {}\n```\n" ); } #[test] fn fenced_code_block_takes_the_language_from_the_code_class() { let html = r#"
let x = 1;
"#; assert_eq!(html_to_markdown(html), "```rust\nlet x = 1;\n```\n"); } #[test] fn fenced_code_block_leaves_its_text_alone() { let html = "
a < b && c\n   indented\n
"; assert_eq!( html_to_markdown(html), "```\na < b && c\n indented\n```\n" ); } #[test] fn fence_grows_when_the_code_contains_backticks() { assert_eq!( html_to_markdown("
a ``` b
"), "````\na ``` b\n````\n" ); } #[test] fn named_and_numeric_entities_are_decoded() { assert_eq!( html_to_markdown("

a & b <tag> AB

"), "a & b AB\n" ); assert_eq!(html_to_markdown("

😀

"), "\u{1f600}\n"); } #[test] fn punctuation_entities_are_decoded() { assert_eq!( html_to_markdown("

5 × 3 — café

"), "5 × 3 — café\n" ); } #[test] fn unknown_entities_are_left_alone() { assert_eq!( html_to_markdown("

a &unknown; b & c

"), "a &unknown; b & c\n" ); } #[test] fn invalid_numeric_references_do_not_panic() { assert_eq!( html_to_markdown("

��

"), "\u{fffd}\u{fffd}\n" ); } #[test] fn non_breaking_spaces_are_kept() { assert_eq!(html_to_markdown("

a b

"), "a\u{a0}b\n"); } #[test] fn attributes_are_decoded() { let html = r#"docs"#; assert_eq!(html_to_markdown(html), "[docs](https://x.test/?a=1&b=2)\n"); } #[test] fn attribute_names_are_case_insensitive() { let html = r#"https://e.test"#; assert_eq!(html_to_markdown(html), "https://e.test\n"); } #[test] fn attribute_values_may_be_quoted_or_bare() { assert_eq!( html_to_markdown("x"), "[x](https://s.test)\n" ); assert_eq!( html_to_markdown("pic"), "![pic](/u/pic.png)\n" ); } #[test] fn nested_lists_are_indented() { let html = "
  • one
    • inner
  • two
"; assert_eq!(html_to_markdown(html), "- one\n - inner\n- two\n"); } #[test] fn ordered_list_items_are_numbered() { assert_eq!( html_to_markdown("
  1. first
  2. second
"), "1. first\n2. second\n" ); } #[test] fn list_item_blocks_continue_on_indented_lines() { assert_eq!( html_to_markdown("
  • first

    second

"), "- first\n second\n" ); } #[test] fn blockquote_prefixes_every_line() { let html = "

first

second

"; assert_eq!(html_to_markdown(html), "> first\n>\n> second\n"); } #[test] fn link_with_a_different_text_uses_the_markdown_form() { let html = r#"

see the docs

"#; assert_eq!(html_to_markdown(html), "see [the docs](https://e.test)\n"); } #[test] fn link_whose_text_is_its_href_collapses_to_the_url() { let html = r#"

https://e.test

"#; assert_eq!(html_to_markdown(html), "https://e.test\n"); } #[test] fn image_uses_its_alt_text_and_source() { let html = r#"

A picture

"#; assert_eq!(html_to_markdown(html), "![A picture](/pic.png)\n"); } #[test] fn horizontal_rule_gets_its_own_line() { assert_eq!( html_to_markdown("

above


below

"), "above\n\n---\n\nbelow\n" ); } #[test] fn table_with_a_header_becomes_a_github_table() { let html = "\
NameValue
a1
"; assert_eq!( html_to_markdown(html), "| Name | Value |\n| --- | --- |\n| a | 1 |\n" ); } #[test] fn table_without_a_header_promotes_the_first_row() { let html = "
h1h2
ab
"; assert_eq!( html_to_markdown(html), "| h1 | h2 |\n| --- | --- |\n| a | b |\n" ); } #[test] fn pipes_inside_a_cell_are_escaped() { let html = "
ab
x | yz
"; assert_eq!( html_to_markdown(html), "| a | b |\n| --- | --- |\n| x \\| y | z |\n" ); } #[test] fn colspan_repeats_the_cell_content() { let html = r#"
ab
span
"#; assert_eq!( html_to_markdown(html), "| a | b |\n| --- | --- |\n| span | span |\n" ); } #[test] fn a_table_without_rows_produces_nothing() { assert_eq!(html_to_markdown("
"), ""); } #[test] fn unknown_elements_keep_their_content() { let html = r#"text"#; assert_eq!(html_to_markdown(html), "text\n"); } #[test] fn comments_doctypes_and_raw_text_are_dropped() { let html = "\

kept

"; assert_eq!(html_to_markdown(html), "kept\n"); } #[test] fn an_unclosed_tag_keeps_the_text() { assert_eq!(html_to_markdown("

hello world"), "hello **world**\n"); } #[test] fn a_stray_angle_bracket_is_text() { assert_eq!( html_to_markdown("

2 < 3 and 4 > 1

"), "2 < 3 and 4 > 1\n" ); } #[test] fn an_unterminated_attribute_quote_stops_at_the_end() { // The quoted value runs to the end of the document: everything before it // is still rendered and nothing hangs. let html = "

visible

tail"; assert_eq!(html_to_markdown(html), "visible\n"); } #[test] fn a_stray_end_tag_is_ignored() { assert_eq!(html_to_markdown("

bold

"), "**bold**\n"); } #[test] fn unterminated_constructs_terminate() { assert_eq!(html_to_markdown("<>&&&"), "<>&&&\n"); assert_eq!(html_to_markdown("

a