1381 lines
45 KiB
Rust
1381 lines
45 KiB
Rust
//! HTML to Markdown conversion for captured answers.
|
||
//!
|
||
//! The gateway reads answers from a signed-in browser page, so the text it
|
||
//! forwards was written for a human looking at rendered HTML. `innerText` keeps
|
||
//! the words and drops the structure: a table collapses into a column of
|
||
//! unrelated lines and a code block loses its fences, so the next model in the
|
||
//! chain cannot tell code from prose. Capturing `innerHTML` instead keeps that
|
||
//! structure, and this module turns it back into Markdown.
|
||
//!
|
||
//! The converter is deliberately self-contained: no browser, no network and no
|
||
//! HTML parser crate, because the markup is whatever a third party UI happens
|
||
//! to ship. The tokenizer is tolerant by design, so malformed markup degrades
|
||
//! into the readable text around it, and both parsing and rendering run on
|
||
//! explicit stacks, so no input can overflow the stack, spin forever or panic.
|
||
|
||
/// Depth beyond which start tags are ignored.
|
||
///
|
||
/// Real answers nest a handful of levels; the cap only stops a hostile page
|
||
/// from making the converter allocate without bound.
|
||
const MAX_DEPTH: usize = 128;
|
||
|
||
/// Elements whose content is text rather than markup, so it is skipped whole.
|
||
const RAW_TEXT_ELEMENTS: [&str; 4] = ["script", "style", "textarea", "title"];
|
||
|
||
/// Elements that never have children and never need a closing tag.
|
||
const VOID_ELEMENTS: [&str; 14] = [
|
||
"area", "base", "br", "col", "embed", "hr", "img", "input", "link", "meta", "param", "source",
|
||
"track", "wbr",
|
||
];
|
||
|
||
/// Elements that are a block of their own and are followed by a blank line.
|
||
const BLOCK_ELEMENTS: [&str; 12] = [
|
||
"p",
|
||
"div",
|
||
"section",
|
||
"article",
|
||
"header",
|
||
"footer",
|
||
"main",
|
||
"aside",
|
||
"figure",
|
||
"figcaption",
|
||
"dd",
|
||
"dt",
|
||
];
|
||
|
||
/// Convert the inner HTML of one answer node into Markdown.
|
||
pub fn html_to_markdown(html: &str) -> String {
|
||
if html.trim().is_empty() {
|
||
return String::new();
|
||
}
|
||
let root = parse(html);
|
||
tidy(&render(&root.children))
|
||
}
|
||
|
||
/// True when the HTML contains a construct a plain text extraction would lose
|
||
/// (a fenced code block, a table row, a list item, a heading, a link).
|
||
pub fn has_rich_structure(html: &str) -> bool {
|
||
// Callers use this to choose between `innerText` and `innerHTML`, so a cheap
|
||
// approximate scan is enough; an exact answer would mean parsing twice.
|
||
const MARKERS: [&str; 12] = [
|
||
"<pre", "<table", "<ul", "<ol", "<li", "<h1", "<h2", "<h3", "<h4", "<h5", "<h6", "<a ",
|
||
];
|
||
let lower = html.to_ascii_lowercase();
|
||
MARKERS.iter().any(|marker| lower.contains(marker))
|
||
}
|
||
|
||
/// Collapse the whitespace of a Markdown document without touching fenced code
|
||
/// blocks: trim trailing spaces per line, drop runs of more than two blank
|
||
/// lines down to one blank line, and end with exactly one newline (or an empty
|
||
/// string when there is no content).
|
||
pub fn tidy(markdown: &str) -> String {
|
||
let mut out = String::with_capacity(markdown.len());
|
||
let mut pending_blank = false;
|
||
let mut fence: Option<String> = None;
|
||
|
||
for raw_line in markdown.split('\n') {
|
||
// A carriage return only survives a CRLF document: line ending noise.
|
||
let line = raw_line.strip_suffix('\r').unwrap_or(raw_line);
|
||
|
||
if let Some(opening) = fence.as_deref() {
|
||
// Inside a fence the text is code the user wrote, so it is copied
|
||
// through untouched, blank lines, indentation and trailing spaces
|
||
// included.
|
||
out.push_str(line);
|
||
out.push('\n');
|
||
if closes_fence(line, opening) {
|
||
fence = None;
|
||
}
|
||
continue;
|
||
}
|
||
|
||
let line = line.trim_end_matches([' ', '\t']);
|
||
if line.is_empty() {
|
||
// Remembered rather than written: a blank line only exists when
|
||
// something follows it, which also drops the ones at the end.
|
||
pending_blank = true;
|
||
continue;
|
||
}
|
||
if let Some(opening) = opening_fence(line) {
|
||
fence = Some(opening);
|
||
}
|
||
if pending_blank && !out.is_empty() {
|
||
out.push('\n');
|
||
}
|
||
pending_blank = false;
|
||
out.push_str(line);
|
||
out.push('\n');
|
||
}
|
||
|
||
out
|
||
}
|
||
|
||
/// The run of backticks that opens a fenced block, when a line starts one.
|
||
fn opening_fence(line: &str) -> Option<String> {
|
||
let run: String = line
|
||
.trim_start()
|
||
.chars()
|
||
.take_while(|character| *character == '`')
|
||
.collect();
|
||
if run.len() >= 3 {
|
||
Some(run)
|
||
} else {
|
||
None
|
||
}
|
||
}
|
||
|
||
/// True when a line closes the fence it is inside.
|
||
fn closes_fence(line: &str, opening: &str) -> bool {
|
||
let trimmed = line.trim();
|
||
trimmed.len() >= opening.len() && trimmed.chars().all(|character| character == '`')
|
||
}
|
||
|
||
/// One node of the parsed document.
|
||
#[derive(Debug)]
|
||
enum Node {
|
||
Element(Element),
|
||
/// Entity-decoded text, otherwise exactly as written: whitespace is only
|
||
/// collapsed when the text is rendered, so a fenced block still reads it
|
||
/// verbatim.
|
||
Text(String),
|
||
}
|
||
|
||
/// One element: a lowercased name, decoded attributes and its children.
|
||
#[derive(Debug, Default)]
|
||
struct Element {
|
||
name: String,
|
||
attributes: Vec<(String, String)>,
|
||
children: Vec<Node>,
|
||
}
|
||
|
||
impl Element {
|
||
/// The value of an attribute, matched by its lowercased name.
|
||
fn attr(&self, name: &str) -> Option<&str> {
|
||
self.attributes
|
||
.iter()
|
||
.find(|(key, _)| key == name)
|
||
.map(|(_, value)| value.as_str())
|
||
}
|
||
}
|
||
|
||
/// Parse HTML into a tree, tolerating anything.
|
||
///
|
||
/// Text that no rule understands still ends up somewhere sensible, and an
|
||
/// unterminated construct simply ends the document: a broken page must never
|
||
/// cost an answer that is already half readable.
|
||
fn parse(html: &str) -> Element {
|
||
let mut stack: Vec<Element> = vec![Element::default()];
|
||
let bytes = html.as_bytes();
|
||
let mut cursor = 0usize;
|
||
|
||
while cursor < bytes.len() {
|
||
if bytes.get(cursor) != Some(&b'<') {
|
||
let start = cursor;
|
||
while let Some(byte) = bytes.get(cursor) {
|
||
if *byte == b'<' {
|
||
break;
|
||
}
|
||
cursor += 1;
|
||
}
|
||
if let Some(text) = html.get(start..cursor) {
|
||
push_text_node(&mut stack, text);
|
||
}
|
||
continue;
|
||
}
|
||
|
||
let rest = html.get(cursor..).unwrap_or("");
|
||
|
||
// Comments, doctypes and processing instructions carry no answer text.
|
||
if rest.starts_with("<!--") {
|
||
cursor = match rest.find("-->") {
|
||
Some(offset) => cursor + offset + 3,
|
||
None => bytes.len(),
|
||
};
|
||
continue;
|
||
}
|
||
if rest.starts_with("<!") || rest.starts_with("<?") {
|
||
cursor = match rest.find('>') {
|
||
Some(offset) => cursor + offset + 1,
|
||
None => bytes.len(),
|
||
};
|
||
continue;
|
||
}
|
||
|
||
if rest.starts_with("</") {
|
||
let name_start = cursor + 2;
|
||
let mut end = name_start;
|
||
while let Some(byte) = bytes.get(end) {
|
||
if byte.is_ascii_whitespace() || *byte == b'>' {
|
||
break;
|
||
}
|
||
end += 1;
|
||
}
|
||
let name = html.get(name_start..end).unwrap_or("").to_ascii_lowercase();
|
||
cursor = match html.get(end..).and_then(|tail| tail.find('>')) {
|
||
Some(offset) => end + offset + 1,
|
||
None => bytes.len(),
|
||
};
|
||
close_element(&mut stack, &name);
|
||
continue;
|
||
}
|
||
|
||
if rest
|
||
.as_bytes()
|
||
.get(1)
|
||
.is_some_and(|byte| byte.is_ascii_alphabetic())
|
||
{
|
||
let (name, attributes, self_closing, end) = read_start_tag(html, cursor);
|
||
cursor = end;
|
||
if name.is_empty() {
|
||
continue;
|
||
}
|
||
if RAW_TEXT_ELEMENTS.contains(&name.as_str()) && !self_closing {
|
||
cursor = skip_raw_text(html, end, &name);
|
||
continue;
|
||
}
|
||
let element = Element {
|
||
name: name.clone(),
|
||
attributes,
|
||
children: Vec::new(),
|
||
};
|
||
if self_closing || VOID_ELEMENTS.contains(&name.as_str()) || stack.len() >= MAX_DEPTH {
|
||
// A void element, a self-closing tag, or markup nested deeper
|
||
// than any real answer goes: the content is kept, the wrapper
|
||
// is not.
|
||
if let Some(parent) = stack.last_mut() {
|
||
parent.children.push(Node::Element(element));
|
||
}
|
||
} else {
|
||
stack.push(element);
|
||
}
|
||
continue;
|
||
}
|
||
|
||
// A `<` that opens nothing is text, which is what a browser shows too.
|
||
push_text_node(&mut stack, "<");
|
||
cursor += 1;
|
||
}
|
||
|
||
close_until(&mut stack, 1);
|
||
stack.pop().unwrap_or_default()
|
||
}
|
||
|
||
/// Read one start tag: its lowercased name, its decoded attributes, whether it
|
||
/// closes itself, and the position just past the final `>`.
|
||
fn read_start_tag(html: &str, start: usize) -> (String, Vec<(String, String)>, bool, usize) {
|
||
let bytes = html.as_bytes();
|
||
let mut cursor = start + 1;
|
||
let name_start = cursor;
|
||
while let Some(byte) = bytes.get(cursor) {
|
||
if byte.is_ascii_whitespace() || *byte == b'>' || *byte == b'/' {
|
||
break;
|
||
}
|
||
cursor += 1;
|
||
}
|
||
let name = html
|
||
.get(name_start..cursor)
|
||
.unwrap_or("")
|
||
.to_ascii_lowercase();
|
||
|
||
let mut attributes: Vec<(String, String)> = Vec::new();
|
||
let mut self_closing = false;
|
||
|
||
loop {
|
||
let before = cursor;
|
||
while let Some(byte) = bytes.get(cursor) {
|
||
if byte.is_ascii_whitespace() {
|
||
cursor += 1;
|
||
} else {
|
||
break;
|
||
}
|
||
}
|
||
|
||
match bytes.get(cursor).copied() {
|
||
None => break,
|
||
Some(b'>') => {
|
||
cursor += 1;
|
||
break;
|
||
}
|
||
Some(b'/') => {
|
||
self_closing = true;
|
||
cursor += 1;
|
||
}
|
||
Some(_) => {
|
||
let attribute_start = cursor;
|
||
while let Some(byte) = bytes.get(cursor) {
|
||
if byte.is_ascii_whitespace() || *byte == b'=' || *byte == b'>' || *byte == b'/'
|
||
{
|
||
break;
|
||
}
|
||
cursor += 1;
|
||
}
|
||
let attribute = html
|
||
.get(attribute_start..cursor)
|
||
.unwrap_or("")
|
||
.to_ascii_lowercase();
|
||
while let Some(byte) = bytes.get(cursor) {
|
||
if byte.is_ascii_whitespace() {
|
||
cursor += 1;
|
||
} else {
|
||
break;
|
||
}
|
||
}
|
||
|
||
let mut value = String::new();
|
||
if bytes.get(cursor) == Some(&b'=') {
|
||
cursor += 1;
|
||
while let Some(byte) = bytes.get(cursor) {
|
||
if byte.is_ascii_whitespace() {
|
||
cursor += 1;
|
||
} else {
|
||
break;
|
||
}
|
||
}
|
||
match bytes.get(cursor).copied() {
|
||
Some(quote) if quote == b'"' || quote == b'\'' => {
|
||
cursor += 1;
|
||
let value_start = cursor;
|
||
while let Some(byte) = bytes.get(cursor) {
|
||
if *byte == quote {
|
||
break;
|
||
}
|
||
cursor += 1;
|
||
}
|
||
value = html.get(value_start..cursor).unwrap_or("").to_string();
|
||
// An unterminated quote runs to the end of the
|
||
// document; the value stays a best effort.
|
||
if bytes.get(cursor) == Some("e) {
|
||
cursor += 1;
|
||
}
|
||
}
|
||
None | Some(b'>') => {}
|
||
Some(_) => {
|
||
let value_start = cursor;
|
||
while let Some(byte) = bytes.get(cursor) {
|
||
if byte.is_ascii_whitespace() || *byte == b'>' {
|
||
break;
|
||
}
|
||
cursor += 1;
|
||
}
|
||
value = html.get(value_start..cursor).unwrap_or("").to_string();
|
||
}
|
||
}
|
||
}
|
||
|
||
if !attribute.is_empty() {
|
||
attributes.push((attribute, decode_entities(&value)));
|
||
}
|
||
}
|
||
}
|
||
|
||
// Every turn consumes at least one byte; the guard is what keeps that
|
||
// true even for markup this reader has not thought of.
|
||
if cursor == before {
|
||
cursor += 1;
|
||
}
|
||
}
|
||
|
||
(name, attributes, self_closing, cursor)
|
||
}
|
||
|
||
/// Append a text node, decoding its entities on the way in.
|
||
fn push_text_node(stack: &mut [Element], text: &str) {
|
||
if let Some(parent) = stack.last_mut() {
|
||
parent.children.push(Node::Text(decode_entities(text)));
|
||
}
|
||
}
|
||
|
||
/// Close the innermost open element with this name.
|
||
///
|
||
/// A stray end tag is ignored, while one that closes an outer element also
|
||
/// closes the elements it still contains, which is what a browser does with
|
||
/// mismatched markup.
|
||
fn close_element(stack: &mut Vec<Element>, name: &str) {
|
||
if name.is_empty() {
|
||
return;
|
||
}
|
||
if let Some(index) = stack.iter().rposition(|element| element.name == name) {
|
||
close_until(stack, index);
|
||
}
|
||
}
|
||
|
||
/// Pop open elements down to `keep`, attaching each one to its new parent.
|
||
fn close_until(stack: &mut Vec<Element>, keep: usize) {
|
||
while stack.len() > keep {
|
||
match stack.pop() {
|
||
Some(element) => {
|
||
if let Some(parent) = stack.last_mut() {
|
||
parent.children.push(Node::Element(element));
|
||
}
|
||
}
|
||
None => return,
|
||
}
|
||
}
|
||
}
|
||
|
||
/// Skip the content of a raw text element, returning the position just past its
|
||
/// end tag.
|
||
fn skip_raw_text(html: &str, from: usize, name: &str) -> usize {
|
||
let needle = format!("</{name}");
|
||
match find_ascii_case_insensitive(html.as_bytes(), needle.as_bytes(), from) {
|
||
Some(index) => match html.get(index..).and_then(|tail| tail.find('>')) {
|
||
Some(offset) => index + offset + 1,
|
||
None => html.len(),
|
||
},
|
||
None => html.len(),
|
||
}
|
||
}
|
||
|
||
/// Find `needle` in `haystack` from `from`, comparing ASCII case-insensitively.
|
||
fn find_ascii_case_insensitive(haystack: &[u8], needle: &[u8], from: usize) -> Option<usize> {
|
||
if needle.is_empty() || haystack.len() < needle.len() {
|
||
return None;
|
||
}
|
||
let last = haystack.len() - needle.len();
|
||
let mut index = from;
|
||
while index <= last {
|
||
let window = haystack.get(index..index + needle.len())?;
|
||
if window.eq_ignore_ascii_case(needle) {
|
||
return Some(index);
|
||
}
|
||
index += 1;
|
||
}
|
||
None
|
||
}
|
||
|
||
/// Decode the HTML entities that show up in captured answers.
|
||
///
|
||
/// Anything unrecognised is left exactly as it was written, because a stray
|
||
/// ampersand in prose is far more common than an entity this table has missed.
|
||
fn decode_entities(text: &str) -> String {
|
||
if !text.contains('&') {
|
||
return text.to_string();
|
||
}
|
||
let mut out = String::with_capacity(text.len());
|
||
let mut rest = text;
|
||
while let Some(index) = rest.find('&') {
|
||
out.push_str(rest.get(..index).unwrap_or(""));
|
||
let tail = rest.get(index..).unwrap_or("");
|
||
match decode_one(tail) {
|
||
Some((decoded, consumed)) => {
|
||
out.push(decoded);
|
||
rest = tail.get(consumed..).unwrap_or("");
|
||
}
|
||
None => {
|
||
out.push('&');
|
||
rest = tail.get(1..).unwrap_or("");
|
||
}
|
||
}
|
||
}
|
||
out.push_str(rest);
|
||
out
|
||
}
|
||
|
||
/// Decode the single entity a string starts with, with its length in bytes.
|
||
fn decode_one(tail: &str) -> Option<(char, usize)> {
|
||
let body = tail.strip_prefix('&')?;
|
||
let end = body.find(';')?;
|
||
let name = body.get(..end)?;
|
||
// A bare ampersand must not swallow the rest of a sentence, so only a short
|
||
// reference is considered at all.
|
||
if name.is_empty() || name.len() > 32 {
|
||
return None;
|
||
}
|
||
|
||
let decoded = if let Some(digits) = name.strip_prefix('#') {
|
||
let code = match digits.strip_prefix(['x', 'X']) {
|
||
Some(hex) => u32::from_str_radix(hex, 16).ok()?,
|
||
None => digits.parse::<u32>().ok()?,
|
||
};
|
||
match char::from_u32(code) {
|
||
// Out of range or a lone surrogate: the replacement character is
|
||
// the honest answer, and it cannot panic.
|
||
Some(character) if character != '\0' => character,
|
||
_ => '\u{fffd}',
|
||
}
|
||
} else {
|
||
named_entity(name)?
|
||
};
|
||
|
||
Some((decoded, end + 2))
|
||
}
|
||
|
||
/// The character behind a named entity, or `None` when it is not one we know.
|
||
fn named_entity(name: &str) -> Option<char> {
|
||
// Matching ignores case: pages spell `&` and `&Nbsp;` too.
|
||
let character = match name.to_ascii_lowercase().as_str() {
|
||
"amp" => '&',
|
||
"lt" => '<',
|
||
"gt" => '>',
|
||
"quot" => '"',
|
||
"apos" => '\'',
|
||
"nbsp" => '\u{a0}',
|
||
"mdash" => '—',
|
||
"ndash" => '–',
|
||
"hellip" => '…',
|
||
"laquo" => '«',
|
||
"raquo" => '»',
|
||
"times" => '×',
|
||
"copy" => '©',
|
||
"reg" => '®',
|
||
"deg" => '°',
|
||
"middot" => '·',
|
||
"bull" => '•',
|
||
"eacute" => 'é',
|
||
"egrave" => 'è',
|
||
"agrave" => 'à',
|
||
"ccedil" => 'ç',
|
||
"uuml" => 'ü',
|
||
"ouml" => 'ö',
|
||
"auml" => 'ä',
|
||
_ => return None,
|
||
};
|
||
Some(character)
|
||
}
|
||
|
||
/// Collapse runs of layout whitespace into a single space, the way HTML itself
|
||
/// does. A non-breaking space is content rather than layout and survives.
|
||
fn collapse_whitespace(text: &str) -> String {
|
||
let mut out = String::with_capacity(text.len());
|
||
let mut in_space = false;
|
||
for character in text.chars() {
|
||
match character {
|
||
' ' | '\t' | '\n' | '\r' | '\u{c}' => {
|
||
if !in_space {
|
||
out.push(' ');
|
||
in_space = true;
|
||
}
|
||
}
|
||
_ => {
|
||
out.push(character);
|
||
in_space = false;
|
||
}
|
||
}
|
||
}
|
||
out
|
||
}
|
||
|
||
/// One unit of work for the iterative renderer.
|
||
enum Work<'a> {
|
||
/// Render a node of the parsed tree.
|
||
Node(&'a Node),
|
||
/// Append literal Markdown.
|
||
Emit(String),
|
||
/// Start a new line when the current one already holds something.
|
||
Break,
|
||
/// Push a buffer that the matching `Finish` will consume.
|
||
Capture(Capture),
|
||
/// Close the innermost capture and turn it into Markdown.
|
||
Finish,
|
||
}
|
||
|
||
/// Why a buffer is being captured: the text has to be complete before it can be
|
||
/// wrapped (a link label), prefixed (a quote or a list item) or cut into cells.
|
||
enum Capture {
|
||
Link { href: String },
|
||
Quote,
|
||
Item { ordered: bool, index: usize },
|
||
Row { header: bool, cells: Vec<String> },
|
||
Cell { colspan: usize },
|
||
}
|
||
|
||
/// Render a slice of nodes into Markdown.
|
||
///
|
||
/// The walk runs on an explicit stack of work items - a node becomes the items
|
||
/// that produce its Markdown - so the traversal stays flat however deeply the
|
||
/// document nests.
|
||
fn render(nodes: &[Node]) -> String {
|
||
let mut buffers: Vec<String> = vec![String::new()];
|
||
let mut captures: Vec<Capture> = Vec::new();
|
||
let mut work: Vec<Work<'_>> = nodes.iter().rev().map(Work::Node).collect();
|
||
|
||
while let Some(item) = work.pop() {
|
||
match item {
|
||
Work::Emit(text) => push_text(&mut buffers, &text),
|
||
Work::Break => ensure_newline(&mut buffers),
|
||
Work::Capture(capture) => {
|
||
captures.push(capture);
|
||
buffers.push(String::new());
|
||
}
|
||
Work::Finish => finish_capture(&mut buffers, &mut captures),
|
||
Work::Node(Node::Text(text)) => {
|
||
let collapsed = collapse_whitespace(text);
|
||
// The indentation of the markup must not become a leading space
|
||
// on a fresh line, while on a line in progress it is what keeps
|
||
// `a <b>b</b>` from running into `ab`.
|
||
let collapsed = if at_line_start(&buffers) {
|
||
collapsed.trim_start_matches(' ')
|
||
} else {
|
||
collapsed.as_str()
|
||
};
|
||
if !collapsed.is_empty() {
|
||
push_text(&mut buffers, collapsed);
|
||
}
|
||
}
|
||
Work::Node(Node::Element(element)) => push_element(element, &mut work),
|
||
}
|
||
}
|
||
|
||
// A capture left open would swallow its text, so anything still pending is
|
||
// unwound into the buffer below it.
|
||
while buffers.len() > 1 {
|
||
finish_capture(&mut buffers, &mut captures);
|
||
}
|
||
buffers.pop().unwrap_or_default()
|
||
}
|
||
|
||
/// Turn one element into the work items that render it.
|
||
fn push_element<'a>(element: &'a Element, work: &mut Vec<Work<'a>>) {
|
||
let mut parts: Vec<Work<'a>> = Vec::new();
|
||
|
||
match element.name.as_str() {
|
||
// Markdown's hard break is two trailing spaces. `tidy` trims them, so
|
||
// the caller sees a plain newline, which is what a reader expects.
|
||
"br" => parts.push(Work::Emit(" \n".to_string())),
|
||
"hr" => {
|
||
parts.push(Work::Break);
|
||
parts.push(Work::Emit("---\n\n".to_string()));
|
||
}
|
||
"img" => {
|
||
let image = image_markdown(element);
|
||
if !image.is_empty() {
|
||
parts.push(Work::Emit(image));
|
||
}
|
||
}
|
||
// Fenced blocks and inline code spans read their own text, so their
|
||
// children are never walked as markup.
|
||
"pre" => {
|
||
parts.push(Work::Break);
|
||
parts.push(Work::Emit(fenced_block(element)));
|
||
}
|
||
"code" => {
|
||
let span = inline_code(element);
|
||
if !span.is_empty() {
|
||
parts.push(Work::Emit(span));
|
||
}
|
||
}
|
||
"blockquote" => {
|
||
parts.push(Work::Break);
|
||
parts.push(Work::Capture(Capture::Quote));
|
||
parts.extend(children(element));
|
||
parts.push(Work::Finish);
|
||
}
|
||
"ul" | "ol" => {
|
||
parts.push(Work::Break);
|
||
push_list(element, element.name == "ol", &mut parts);
|
||
}
|
||
"table" => {
|
||
parts.push(Work::Break);
|
||
push_table(element, &mut parts);
|
||
}
|
||
"a" => {
|
||
let href = element.attr("href").unwrap_or_default().to_string();
|
||
parts.push(Work::Capture(Capture::Link { href }));
|
||
parts.extend(children(element));
|
||
parts.push(Work::Finish);
|
||
}
|
||
"h1" | "h2" | "h3" | "h4" | "h5" | "h6" => {
|
||
let level = heading_level(&element.name);
|
||
parts.push(Work::Break);
|
||
parts.push(Work::Emit(format!("{} ", "#".repeat(level))));
|
||
parts.extend(children(element));
|
||
parts.push(Work::Emit("\n\n".to_string()));
|
||
}
|
||
"strong" | "b" => parts.extend(wrapped("**", element)),
|
||
"em" | "i" => parts.extend(wrapped("*", element)),
|
||
"del" | "s" | "strike" => parts.extend(wrapped("~~", element)),
|
||
name if BLOCK_ELEMENTS.contains(&name) => {
|
||
parts.push(Work::Break);
|
||
parts.extend(children(element));
|
||
parts.push(Work::Emit("\n\n".to_string()));
|
||
}
|
||
// Unknown elements, and cells or items met outside their table or list,
|
||
// keep their content and nothing else.
|
||
_ => parts.extend(children(element)),
|
||
}
|
||
|
||
// The stack is popped from the back, so the parts go on in reverse.
|
||
for part in parts.into_iter().rev() {
|
||
work.push(part);
|
||
}
|
||
}
|
||
|
||
/// The heading level of an `h1`..`h6` element.
|
||
fn heading_level(name: &str) -> usize {
|
||
name.as_bytes()
|
||
.last()
|
||
.map_or(1, |digit| usize::from(digit.saturating_sub(b'0')))
|
||
.clamp(1, 6)
|
||
}
|
||
|
||
/// Emphasised or struck-through content: a marker on either side.
|
||
fn wrapped<'a>(marker: &str, element: &'a Element) -> Vec<Work<'a>> {
|
||
let mut parts = vec![Work::Emit(marker.to_string())];
|
||
parts.extend(children(element));
|
||
parts.push(Work::Emit(marker.to_string()));
|
||
parts
|
||
}
|
||
|
||
/// One work item per child, in document order.
|
||
fn children<'a>(element: &'a Element) -> Vec<Work<'a>> {
|
||
element.children.iter().map(Work::Node).collect()
|
||
}
|
||
|
||
/// Add the items that render a `ul` or `ol`.
|
||
fn push_list<'a>(element: &'a Element, ordered: bool, parts: &mut Vec<Work<'a>>) {
|
||
let mut index = 0usize;
|
||
for child in &element.children {
|
||
match child {
|
||
Node::Element(item) if item.name == "li" => {
|
||
parts.push(Work::Capture(Capture::Item { ordered, index }));
|
||
parts.extend(children(item));
|
||
parts.push(Work::Finish);
|
||
index += 1;
|
||
}
|
||
// Anything a list holds besides its items is kept as it is.
|
||
other => parts.push(Work::Node(other)),
|
||
}
|
||
}
|
||
// A list is followed by a blank line, so the next block starts cleanly.
|
||
parts.push(Work::Emit("\n".to_string()));
|
||
}
|
||
|
||
/// Add the items that render a `table`.
|
||
fn push_table<'a>(element: &'a Element, parts: &mut Vec<Work<'a>>) {
|
||
let rows = table_rows(element);
|
||
if rows.is_empty() {
|
||
// A table with no rows holds nothing a reader would miss.
|
||
return;
|
||
}
|
||
// Markdown tables need a header row; when the markup names none, the first
|
||
// row takes that role, which is what a reader of the page would assume.
|
||
let has_header = rows.iter().any(|(_, header)| *header);
|
||
for (index, (row, header)) in rows.iter().enumerate() {
|
||
let header = *header || (!has_header && index == 0);
|
||
parts.push(Work::Capture(Capture::Row {
|
||
header,
|
||
cells: Vec::new(),
|
||
}));
|
||
for cell in row_cells(row) {
|
||
parts.push(Work::Capture(Capture::Cell {
|
||
colspan: colspan(cell),
|
||
}));
|
||
parts.extend(children(cell));
|
||
parts.push(Work::Finish);
|
||
}
|
||
parts.push(Work::Finish);
|
||
}
|
||
parts.push(Work::Emit("\n".to_string()));
|
||
}
|
||
|
||
/// The rows of a table in document order, each with the flag telling whether it
|
||
/// is a header row (inside `thead`, or holding `th` cells).
|
||
fn table_rows(table: &Element) -> Vec<(&Element, bool)> {
|
||
let mut rows = Vec::new();
|
||
for child in &table.children {
|
||
let Node::Element(child) = child else {
|
||
continue;
|
||
};
|
||
match child.name.as_str() {
|
||
"tr" => rows.push((child, row_has_th(child))),
|
||
"thead" | "tbody" | "tfoot" => {
|
||
let in_header = child.name == "thead";
|
||
for row in &child.children {
|
||
if let Node::Element(row) = row {
|
||
if row.name == "tr" {
|
||
rows.push((row, in_header || row_has_th(row)));
|
||
}
|
||
}
|
||
}
|
||
}
|
||
_ => {}
|
||
}
|
||
}
|
||
rows
|
||
}
|
||
|
||
/// True when a row holds header cells of its own.
|
||
fn row_has_th(row: &Element) -> bool {
|
||
row_cells(row).iter().any(|cell| cell.name == "th")
|
||
}
|
||
|
||
/// The cells of a row, in document order.
|
||
fn row_cells(row: &Element) -> Vec<&Element> {
|
||
row.children
|
||
.iter()
|
||
.filter_map(|child| match child {
|
||
Node::Element(cell) if cell.name == "th" || cell.name == "td" => Some(cell),
|
||
_ => None,
|
||
})
|
||
.collect()
|
||
}
|
||
|
||
/// How many columns a cell covers; a missing or silly value means one, and the
|
||
/// cap keeps a hostile `colspan` from asking for an enormous row.
|
||
fn colspan(cell: &Element) -> usize {
|
||
cell.attr("colspan")
|
||
.and_then(|value| value.trim().parse::<usize>().ok())
|
||
.filter(|span| *span > 0)
|
||
.map_or(1, |span| span.min(64))
|
||
}
|
||
|
||
/// Close the innermost capture and append its Markdown to the buffer below it.
|
||
fn finish_capture(buffers: &mut Vec<String>, captures: &mut Vec<Capture>) {
|
||
let text = match buffers.pop() {
|
||
Some(text) => text,
|
||
None => return,
|
||
};
|
||
|
||
match captures.pop() {
|
||
Some(Capture::Link { href }) => {
|
||
let label = text.trim();
|
||
let rendered = if href.is_empty() {
|
||
label.to_string()
|
||
} else if label == href {
|
||
href
|
||
} else if label.is_empty() {
|
||
String::new()
|
||
} else {
|
||
format!("[{label}]({href})")
|
||
};
|
||
push_text(buffers, &rendered);
|
||
}
|
||
Some(Capture::Quote) => {
|
||
let mut quoted = String::new();
|
||
let body = text.trim();
|
||
if !body.is_empty() {
|
||
for line in body.split('\n') {
|
||
let line = line.trim_end();
|
||
if line.is_empty() {
|
||
quoted.push_str(">\n");
|
||
} else {
|
||
quoted.push_str("> ");
|
||
quoted.push_str(line);
|
||
quoted.push('\n');
|
||
}
|
||
}
|
||
quoted.push('\n');
|
||
}
|
||
push_text(buffers, "ed);
|
||
}
|
||
Some(Capture::Item { ordered, index }) => {
|
||
// The bullet shares its line with the first line of the item; the
|
||
// rest continues indented under it, two spaces per level.
|
||
let bullet = if ordered {
|
||
format!("{}.", index + 1)
|
||
} else {
|
||
"-".to_string()
|
||
};
|
||
let mut rendered = String::new();
|
||
for (number, line) in text.trim().split('\n').enumerate() {
|
||
let line = line.trim_end();
|
||
if number == 0 {
|
||
rendered.push_str(&bullet);
|
||
rendered.push(' ');
|
||
rendered.push_str(line.trim_start());
|
||
} else if !line.is_empty() {
|
||
rendered.push_str(" ");
|
||
rendered.push_str(line);
|
||
}
|
||
rendered.push('\n');
|
||
}
|
||
push_text(buffers, &rendered);
|
||
}
|
||
Some(Capture::Row { header, cells }) => {
|
||
let mut rendered = String::new();
|
||
if !cells.is_empty() {
|
||
rendered.push_str("| ");
|
||
rendered.push_str(&cells.join(" | "));
|
||
rendered.push_str(" |\n");
|
||
if header {
|
||
rendered.push_str("| ");
|
||
rendered.push_str(&vec!["---"; cells.len()].join(" | "));
|
||
rendered.push_str(" |\n");
|
||
}
|
||
}
|
||
push_text(buffers, &rendered);
|
||
}
|
||
Some(Capture::Cell { colspan }) => {
|
||
// A cell is a single line: newlines inside it become spaces, and a
|
||
// pipe would otherwise cut the row into extra columns.
|
||
let cell = text
|
||
.split_whitespace()
|
||
.collect::<Vec<_>>()
|
||
.join(" ")
|
||
.replace('|', "\\|");
|
||
match captures.last_mut() {
|
||
Some(Capture::Row { cells, .. }) => {
|
||
for _ in 0..colspan {
|
||
cells.push(cell.clone());
|
||
}
|
||
}
|
||
_ => push_text(buffers, &cell),
|
||
}
|
||
}
|
||
None => push_text(buffers, &text),
|
||
}
|
||
}
|
||
|
||
/// An image with its alternative text, as Markdown.
|
||
fn image_markdown(element: &Element) -> String {
|
||
let alt = element.attr("alt").unwrap_or_default();
|
||
let src = element.attr("src").unwrap_or_default();
|
||
if alt.is_empty() && src.is_empty() {
|
||
// A decorative spacer: nothing a reader would miss.
|
||
return String::new();
|
||
}
|
||
format!("")
|
||
}
|
||
|
||
/// An inline code span: one backtick, or two when the text contains one.
|
||
fn inline_code(element: &Element) -> String {
|
||
let text = text_content(element);
|
||
let text = text.trim();
|
||
if text.is_empty() {
|
||
return String::new();
|
||
}
|
||
let ticks = if text.contains('`') { "``" } else { "`" };
|
||
format!("{ticks}{text}{ticks}")
|
||
}
|
||
|
||
/// A `pre` as a fenced block, named with the language of an inner `code`
|
||
/// element when the page declares one.
|
||
fn fenced_block(element: &Element) -> String {
|
||
let text = text_content(element);
|
||
let text = text.trim_matches(['\n', '\r']);
|
||
let language = code_language(element).unwrap_or_default();
|
||
let fence = fence_for(text);
|
||
format!("{fence}{language}\n{text}\n{fence}\n\n")
|
||
}
|
||
|
||
/// A fence longer than any run of backticks in the text, and at least three.
|
||
fn fence_for(text: &str) -> String {
|
||
let mut longest = 0usize;
|
||
let mut run = 0usize;
|
||
for character in text.chars() {
|
||
if character == '`' {
|
||
run += 1;
|
||
longest = longest.max(run);
|
||
} else {
|
||
run = 0;
|
||
}
|
||
}
|
||
"`".repeat(longest.max(2) + 1)
|
||
}
|
||
|
||
/// The language named by `class="language-xxx"` on the `code` child of a `pre`.
|
||
fn code_language(element: &Element) -> Option<String> {
|
||
let code = element.children.iter().find_map(|child| match child {
|
||
Node::Element(child) if child.name == "code" => Some(child),
|
||
_ => None,
|
||
})?;
|
||
code.attr("class")?
|
||
.split_whitespace()
|
||
.find_map(|token| token.strip_prefix("language-"))
|
||
.filter(|language| !language.is_empty())
|
||
.map(|language| language.to_string())
|
||
}
|
||
|
||
/// The text of a subtree, with its whitespace exactly as written.
|
||
///
|
||
/// Fenced blocks and code spans use this instead of the renderer, because
|
||
/// Markdown code is literal text rather than markup.
|
||
fn text_content(element: &Element) -> String {
|
||
let mut out = String::new();
|
||
let mut stack: Vec<&Node> = element.children.iter().rev().collect();
|
||
while let Some(node) = stack.pop() {
|
||
match node {
|
||
Node::Text(text) => out.push_str(text),
|
||
Node::Element(child) => {
|
||
if child.name == "br" {
|
||
out.push('\n');
|
||
}
|
||
stack.extend(child.children.iter().rev());
|
||
}
|
||
}
|
||
}
|
||
out
|
||
}
|
||
|
||
/// Append text to the innermost buffer.
|
||
fn push_text(buffers: &mut [String], text: &str) {
|
||
if let Some(buffer) = buffers.last_mut() {
|
||
buffer.push_str(text);
|
||
}
|
||
}
|
||
|
||
/// End the current line when it already holds something.
|
||
fn ensure_newline(buffers: &mut [String]) {
|
||
if let Some(buffer) = buffers.last_mut() {
|
||
if !buffer.is_empty() && !buffer.ends_with('\n') {
|
||
buffer.push('\n');
|
||
}
|
||
}
|
||
}
|
||
|
||
/// True when nothing has been written on the current line yet.
|
||
fn at_line_start(buffers: &[String]) -> bool {
|
||
buffers
|
||
.last()
|
||
.is_none_or(|buffer| buffer.is_empty() || buffer.ends_with('\n'))
|
||
}
|
||
|
||
#[cfg(test)]
|
||
mod tests {
|
||
use super::*;
|
||
|
||
#[test]
|
||
fn empty_input_produces_empty_output() {
|
||
assert_eq!(html_to_markdown(""), "");
|
||
assert_eq!(html_to_markdown(" \n\t "), "");
|
||
}
|
||
|
||
#[test]
|
||
fn paragraphs_keep_their_text_without_the_markup() {
|
||
assert_eq!(html_to_markdown("<p>Hello world</p>"), "Hello world\n");
|
||
}
|
||
|
||
#[test]
|
||
fn two_paragraphs_are_separated_by_a_blank_line() {
|
||
assert_eq!(html_to_markdown("<p>One</p><p>Two</p>"), "One\n\nTwo\n");
|
||
}
|
||
|
||
#[test]
|
||
fn layout_whitespace_never_leaves_blank_runs() {
|
||
assert_eq!(html_to_markdown("<p>a</p>\n\n\n<p>b</p>\n\n\n"), "a\n\nb\n");
|
||
}
|
||
|
||
#[test]
|
||
fn line_break_ends_the_line() {
|
||
// The renderer writes Markdown's two-space hard break; `tidy` trims the
|
||
// trailing spaces, so the caller sees a plain newline.
|
||
assert_eq!(html_to_markdown("<p>one<br>two</p>"), "one\ntwo\n");
|
||
assert_eq!(html_to_markdown("<p>one<br/>two</p>"), "one\ntwo\n");
|
||
}
|
||
|
||
#[test]
|
||
fn headings_use_one_hash_per_level() {
|
||
assert_eq!(
|
||
html_to_markdown("<h1>Title</h1><h4>Deep</h4>"),
|
||
"# Title\n\n#### Deep\n"
|
||
);
|
||
}
|
||
|
||
#[test]
|
||
fn emphasis_becomes_markdown_markers() {
|
||
let html =
|
||
"<p><strong>bold</strong> <b>also bold</b> <em>italic</em> <i>also</i> <del>gone</del></p>";
|
||
assert_eq!(
|
||
html_to_markdown(html),
|
||
"**bold** **also bold** *italic* *also* ~~gone~~\n"
|
||
);
|
||
}
|
||
|
||
#[test]
|
||
fn inline_code_uses_one_backtick() {
|
||
assert_eq!(
|
||
html_to_markdown("<p>run <code>cargo test</code> now</p>"),
|
||
"run `cargo test` now\n"
|
||
);
|
||
}
|
||
|
||
#[test]
|
||
fn inline_code_containing_a_backtick_uses_two() {
|
||
assert_eq!(
|
||
html_to_markdown("<p><code>x `y` z</code></p>"),
|
||
"``x `y` z``\n"
|
||
);
|
||
}
|
||
|
||
#[test]
|
||
fn fenced_code_block_without_a_language() {
|
||
assert_eq!(
|
||
html_to_markdown("<pre>fn main() {}</pre>"),
|
||
"```\nfn main() {}\n```\n"
|
||
);
|
||
}
|
||
|
||
#[test]
|
||
fn fenced_code_block_takes_the_language_from_the_code_class() {
|
||
let html = r#"<pre><code class="language-rust hljs">let x = 1;</code></pre>"#;
|
||
assert_eq!(html_to_markdown(html), "```rust\nlet x = 1;\n```\n");
|
||
}
|
||
|
||
#[test]
|
||
fn fenced_code_block_leaves_its_text_alone() {
|
||
let html = "<pre><code>a < b && c\n indented\n</code></pre>";
|
||
assert_eq!(
|
||
html_to_markdown(html),
|
||
"```\na < b && c\n indented\n```\n"
|
||
);
|
||
}
|
||
|
||
#[test]
|
||
fn fence_grows_when_the_code_contains_backticks() {
|
||
assert_eq!(
|
||
html_to_markdown("<pre>a ``` b</pre>"),
|
||
"````\na ``` b\n````\n"
|
||
);
|
||
}
|
||
|
||
#[test]
|
||
fn named_and_numeric_entities_are_decoded() {
|
||
assert_eq!(
|
||
html_to_markdown("<p>a & b <tag> AB</p>"),
|
||
"a & b <tag> AB\n"
|
||
);
|
||
assert_eq!(html_to_markdown("<p>😀</p>"), "\u{1f600}\n");
|
||
}
|
||
|
||
#[test]
|
||
fn punctuation_entities_are_decoded() {
|
||
assert_eq!(
|
||
html_to_markdown("<p>5 × 3 — café</p>"),
|
||
"5 × 3 — café\n"
|
||
);
|
||
}
|
||
|
||
#[test]
|
||
fn unknown_entities_are_left_alone() {
|
||
assert_eq!(
|
||
html_to_markdown("<p>a &unknown; b & c</p>"),
|
||
"a &unknown; b & c\n"
|
||
);
|
||
}
|
||
|
||
#[test]
|
||
fn invalid_numeric_references_do_not_panic() {
|
||
assert_eq!(
|
||
html_to_markdown("<p>��</p>"),
|
||
"\u{fffd}\u{fffd}\n"
|
||
);
|
||
}
|
||
|
||
#[test]
|
||
fn non_breaking_spaces_are_kept() {
|
||
assert_eq!(html_to_markdown("<p>a b</p>"), "a\u{a0}b\n");
|
||
}
|
||
|
||
#[test]
|
||
fn attributes_are_decoded() {
|
||
let html = r#"<a href="https://x.test/?a=1&b=2">docs</a>"#;
|
||
assert_eq!(html_to_markdown(html), "[docs](https://x.test/?a=1&b=2)\n");
|
||
}
|
||
|
||
#[test]
|
||
fn attribute_names_are_case_insensitive() {
|
||
let html = r#"<a HREF="https://e.test">https://e.test</a>"#;
|
||
assert_eq!(html_to_markdown(html), "https://e.test\n");
|
||
}
|
||
|
||
#[test]
|
||
fn attribute_values_may_be_quoted_or_bare() {
|
||
assert_eq!(
|
||
html_to_markdown("<a href='https://s.test'>x</a>"),
|
||
"[x](https://s.test)\n"
|
||
);
|
||
assert_eq!(
|
||
html_to_markdown("<img src=/u/pic.png alt=pic>"),
|
||
"\n"
|
||
);
|
||
}
|
||
|
||
#[test]
|
||
fn nested_lists_are_indented() {
|
||
let html = "<ul><li>one<ul><li>inner</li></ul></li><li>two</li></ul>";
|
||
assert_eq!(html_to_markdown(html), "- one\n - inner\n- two\n");
|
||
}
|
||
|
||
#[test]
|
||
fn ordered_list_items_are_numbered() {
|
||
assert_eq!(
|
||
html_to_markdown("<ol><li>first</li><li>second</li></ol>"),
|
||
"1. first\n2. second\n"
|
||
);
|
||
}
|
||
|
||
#[test]
|
||
fn list_item_blocks_continue_on_indented_lines() {
|
||
assert_eq!(
|
||
html_to_markdown("<ul><li>first<p>second</p></li></ul>"),
|
||
"- first\n second\n"
|
||
);
|
||
}
|
||
|
||
#[test]
|
||
fn blockquote_prefixes_every_line() {
|
||
let html = "<blockquote><p>first</p><p>second</p></blockquote>";
|
||
assert_eq!(html_to_markdown(html), "> first\n>\n> second\n");
|
||
}
|
||
|
||
#[test]
|
||
fn link_with_a_different_text_uses_the_markdown_form() {
|
||
let html = r#"<p>see <a href="https://e.test">the docs</a></p>"#;
|
||
assert_eq!(html_to_markdown(html), "see [the docs](https://e.test)\n");
|
||
}
|
||
|
||
#[test]
|
||
fn link_whose_text_is_its_href_collapses_to_the_url() {
|
||
let html = r#"<p><a href="https://e.test">https://e.test</a></p>"#;
|
||
assert_eq!(html_to_markdown(html), "https://e.test\n");
|
||
}
|
||
|
||
#[test]
|
||
fn image_uses_its_alt_text_and_source() {
|
||
let html = r#"<p><img src="/pic.png" alt="A picture"></p>"#;
|
||
assert_eq!(html_to_markdown(html), "\n");
|
||
}
|
||
|
||
#[test]
|
||
fn horizontal_rule_gets_its_own_line() {
|
||
assert_eq!(
|
||
html_to_markdown("<p>above</p><hr><p>below</p>"),
|
||
"above\n\n---\n\nbelow\n"
|
||
);
|
||
}
|
||
|
||
#[test]
|
||
fn table_with_a_header_becomes_a_github_table() {
|
||
let html = "<table><thead><tr><th>Name</th><th>Value</th></tr></thead>\
|
||
<tbody><tr><td>a</td><td>1</td></tr></tbody></table>";
|
||
assert_eq!(
|
||
html_to_markdown(html),
|
||
"| Name | Value |\n| --- | --- |\n| a | 1 |\n"
|
||
);
|
||
}
|
||
|
||
#[test]
|
||
fn table_without_a_header_promotes_the_first_row() {
|
||
let html = "<table><tr><td>h1</td><td>h2</td></tr><tr><td>a</td><td>b</td></tr></table>";
|
||
assert_eq!(
|
||
html_to_markdown(html),
|
||
"| h1 | h2 |\n| --- | --- |\n| a | b |\n"
|
||
);
|
||
}
|
||
|
||
#[test]
|
||
fn pipes_inside_a_cell_are_escaped() {
|
||
let html = "<table><tr><th>a</th><th>b</th></tr><tr><td>x | y</td><td>z</td></tr></table>";
|
||
assert_eq!(
|
||
html_to_markdown(html),
|
||
"| a | b |\n| --- | --- |\n| x \\| y | z |\n"
|
||
);
|
||
}
|
||
|
||
#[test]
|
||
fn colspan_repeats_the_cell_content() {
|
||
let html =
|
||
r#"<table><tr><th>a</th><th>b</th></tr><tr><td colspan="2">span</td></tr></table>"#;
|
||
assert_eq!(
|
||
html_to_markdown(html),
|
||
"| a | b |\n| --- | --- |\n| span | span |\n"
|
||
);
|
||
}
|
||
|
||
#[test]
|
||
fn a_table_without_rows_produces_nothing() {
|
||
assert_eq!(html_to_markdown("<table></table>"), "");
|
||
}
|
||
|
||
#[test]
|
||
fn unknown_elements_keep_their_content() {
|
||
let html = r#"<answer><span data-x="1">text</span></answer>"#;
|
||
assert_eq!(html_to_markdown(html), "text\n");
|
||
}
|
||
|
||
#[test]
|
||
fn comments_doctypes_and_raw_text_are_dropped() {
|
||
let html = "<!doctype html><!-- hidden --><style>p{color:red}</style>\
|
||
<script>var x = 1;</script><p>kept</p>";
|
||
assert_eq!(html_to_markdown(html), "kept\n");
|
||
}
|
||
|
||
#[test]
|
||
fn an_unclosed_tag_keeps_the_text() {
|
||
assert_eq!(html_to_markdown("<p>hello <b>world"), "hello **world**\n");
|
||
}
|
||
|
||
#[test]
|
||
fn a_stray_angle_bracket_is_text() {
|
||
assert_eq!(
|
||
html_to_markdown("<p>2 < 3 and 4 > 1</p>"),
|
||
"2 < 3 and 4 > 1\n"
|
||
);
|
||
}
|
||
|
||
#[test]
|
||
fn an_unterminated_attribute_quote_stops_at_the_end() {
|
||
// The quoted value runs to the end of the document: everything before it
|
||
// is still rendered and nothing hangs.
|
||
let html = "<p>visible</p><a href=\"broken>tail";
|
||
assert_eq!(html_to_markdown(html), "visible\n");
|
||
}
|
||
|
||
#[test]
|
||
fn a_stray_end_tag_is_ignored() {
|
||
assert_eq!(html_to_markdown("<p><b>bold</i></p>"), "**bold**\n");
|
||
}
|
||
|
||
#[test]
|
||
fn unterminated_constructs_terminate() {
|
||
assert_eq!(html_to_markdown("<>&&&"), "<>&&&\n");
|
||
assert_eq!(html_to_markdown("<p>a</p><!-- open"), "a\n");
|
||
assert_eq!(html_to_markdown("<p>a"), "a\n");
|
||
}
|
||
|
||
#[test]
|
||
fn deeply_nested_markup_is_bounded() {
|
||
let mut html = String::new();
|
||
for _ in 0..5_000 {
|
||
html.push_str("<div>");
|
||
}
|
||
html.push_str("deep");
|
||
assert_eq!(html_to_markdown(&html), "deep\n");
|
||
}
|
||
|
||
#[test]
|
||
fn tidy_trims_trailing_spaces_and_collapses_blank_lines() {
|
||
assert_eq!(tidy("a \n\n\n\nb \n\n\n"), "a\n\nb\n");
|
||
}
|
||
|
||
#[test]
|
||
fn tidy_ends_with_exactly_one_newline() {
|
||
assert_eq!(tidy("text"), "text\n");
|
||
assert_eq!(tidy("text\n\n\n"), "text\n");
|
||
}
|
||
|
||
#[test]
|
||
fn tidy_of_whitespace_only_input_is_empty() {
|
||
assert_eq!(tidy(""), "");
|
||
assert_eq!(tidy("\n\n \n"), "");
|
||
}
|
||
|
||
#[test]
|
||
fn tidy_does_not_touch_fenced_code_blocks() {
|
||
let input = "text \n\n```rust\nlet a = 1; \n\n\nlet b = 2;\n```\n\n\n";
|
||
let expected = "text\n\n```rust\nlet a = 1; \n\n\nlet b = 2;\n```\n";
|
||
assert_eq!(tidy(input), expected);
|
||
}
|
||
|
||
#[test]
|
||
fn has_rich_structure_detects_structure() {
|
||
assert!(has_rich_structure("<pre>code</pre>"));
|
||
assert!(has_rich_structure("<TABLE><tr><td>x</td></tr></TABLE>"));
|
||
assert!(has_rich_structure("<ul><li>a</li></ul>"));
|
||
assert!(has_rich_structure("<ol><li>a</li></ol>"));
|
||
assert!(has_rich_structure("<h3>title</h3>"));
|
||
assert!(has_rich_structure(
|
||
"see <a href=\"https://e.test\">the docs</a>"
|
||
));
|
||
}
|
||
|
||
#[test]
|
||
fn has_rich_structure_is_false_for_plain_markup() {
|
||
assert!(!has_rich_structure(""));
|
||
assert!(!has_rich_structure("just words"));
|
||
assert!(!has_rich_structure("<p>just <b>words</b></p>"));
|
||
assert!(!has_rich_structure("<div><span>text</span></div>"));
|
||
}
|
||
}
|