Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
35 changes: 35 additions & 0 deletions html5ever/src/tree_builder/mod.rs
Original file line number Diff line number Diff line change
Expand Up @@ -413,6 +413,41 @@ where
self.context_elem.borrow().is_some()
}

/// Call `f` with the [stack of open elements].
///
/// The first element is the topmost node of the stack, the root `html` element, and the last
/// one is the bottommost node, the [current node]. The stack is empty before the root element
/// has been inserted and after [`end`](TokenSink::end) has been called. When parsing a
/// fragment, the context element is not on the stack.
///
/// # Panics
///
/// The stack is borrowed while `f` runs, so processing a token with this tree builder from
/// within `f` may panic.
///
/// [stack of open elements]: https://html.spec.whatwg.org/#stack-of-open-elements
/// [current node]: https://html.spec.whatwg.org/#current-node
pub fn with_open_elements<F, R>(&self, f: F) -> R
where
F: FnOnce(&[Handle]) -> R,
{
f(&self.open_elems.borrow())
}

/// Is the [insertion mode] ["text"]?
///
/// The tree builder switches to this insertion mode after inserting an element whose contents
/// are parsed as raw text or RCDATA, such as `script`, `style`, `title` or `textarea`, and
/// switches back at the element's end tag or at the end of the input. In this insertion mode,
/// character tokens are inserted into the [current node], which is that element.
///
/// [insertion mode]: https://html.spec.whatwg.org/#insertion-mode
/// ["text"]: https://html.spec.whatwg.org/#parsing-main-incdata
/// [current node]: https://html.spec.whatwg.org/#current-node
pub fn is_in_text_insertion_mode(&self) -> bool {
self.mode.get() == InsertionMode::Text
}

/// https://html.spec.whatwg.org/multipage/#appropriate-place-for-inserting-a-node
fn appropriate_place_for_insertion(
&self,
Expand Down
170 changes: 170 additions & 0 deletions rcdom/tests/html-tree-builder-state.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,170 @@
// Copyright 2014-2017 The html5ever Project Developers. See the
// COPYRIGHT file at the top-level directory of this distribution.
//
// Licensed under the Apache License, Version 2.0 <LICENSE-APACHE or
// http://www.apache.org/licenses/LICENSE-2.0> or the MIT license
// <LICENSE-MIT or http://opensource.org/licenses/MIT>, at your
// option. This file may not be copied, modified, or distributed
// except according to those terms.

use html5ever::driver::{self, ParseOpts};
use html5ever::tendril::TendrilSink;
use html5ever::tree_builder::{TreeBuilder, TreeBuilderOpts};
use html5ever::{local_name, ns, QualName};
use markup5ever_rcdom::{Handle, NodeData, RcDom};

/// The names of the open elements, written like in the html5lib tree construction tests.
fn open_elements(tree_builder: &TreeBuilder<Handle, RcDom>) -> Vec<String> {
tree_builder.with_open_elements(|elements| {
elements
.iter()
.map(|element| {
let NodeData::Element { ref name, .. } = element.data else {
panic!("not an element");
};
match name.ns {
ns!(html) => name.local.to_string(),
ns!(svg) => format!("svg {}", name.local),
ns!(mathml) => format!("math {}", name.local),
_ => panic!("unexpected namespace"),
}
})
.collect()
})
}

/// The text of the children of the current node.
fn current_node_text(tree_builder: &TreeBuilder<Handle, RcDom>) -> String {
tree_builder.with_open_elements(|elements| {
let current_node = elements.last().expect("no current node");
let mut text = String::new();
for child in current_node.children.borrow().iter() {
if let NodeData::Text { ref contents } = child.data {
text.push_str(&contents.borrow());
}
}
text
})
}

#[test]
fn open_elements_in_document() {
let mut parser = driver::parse_document(RcDom::default(), ParseOpts::default());
parser.process("<!DOCTYPE html>".into());
assert!(open_elements(&parser.tokenizer.sink).is_empty());

parser.process("<table><td><b>".into());
assert_eq!(
open_elements(&parser.tokenizer.sink),
["html", "body", "table", "tbody", "tr", "td", "b"]
);

parser.process("</table><svg><circle>".into());
assert_eq!(
open_elements(&parser.tokenizer.sink),
["html", "body", "svg svg", "svg circle"]
);

parser.tokenizer.end();
assert!(open_elements(&parser.tokenizer.sink).is_empty());
}

#[test]
fn open_elements_in_fragment() {
let context = QualName::new(None, ns!(html), local_name!("ul"));
let mut parser = driver::parse_fragment(
RcDom::default(),
ParseOpts::default(),
context,
vec![],
true,
);
assert_eq!(open_elements(&parser.tokenizer.sink), ["html"]);

parser.process("<li><p>".into());
assert_eq!(open_elements(&parser.tokenizer.sink), ["html", "li", "p"]);
}

#[test]
fn text_insertion_mode() {
for name in [
"iframe", "noembed", "noframes", "noscript", "script", "style", "textarea", "title", "xmp",
] {
let mut parser = driver::parse_document(RcDom::default(), ParseOpts::default());
parser.process("<body>".into());
assert!(!parser.tokenizer.sink.is_in_text_insertion_mode());

parser.process(format!("<{name}>").into());
assert!(parser.tokenizer.sink.is_in_text_insertion_mode(), "{name}");
assert_eq!(
open_elements(&parser.tokenizer.sink),
["html", "body", name]
);

// Character tokens are inserted into the current node, which is that element.
parser.process("<p>".into());
assert!(parser.tokenizer.sink.is_in_text_insertion_mode(), "{name}");
assert_eq!(current_node_text(&parser.tokenizer.sink), "<p>", "{name}");

parser.process(format!("</{name}>").into());
assert!(!parser.tokenizer.sink.is_in_text_insertion_mode(), "{name}");
assert_eq!(open_elements(&parser.tokenizer.sink), ["html", "body"]);
}
}

#[test]
fn text_insertion_mode_ends_at_eof() {
let mut parser = driver::parse_document(RcDom::default(), ParseOpts::default());
parser.process("<script>".into());
assert!(parser.tokenizer.sink.is_in_text_insertion_mode());

parser.tokenizer.end();
assert!(!parser.tokenizer.sink.is_in_text_insertion_mode());
}

#[test]
fn not_text_insertion_mode() {
// The tokenizer switches to the PLAINTEXT state, but the insertion mode stays "in body".
let mut parser = driver::parse_document(RcDom::default(), ParseOpts::default());
parser.process("<plaintext>".into());
assert!(!parser.tokenizer.sink.is_in_text_insertion_mode());

// Character tokens in a table are handled in the "in table text" insertion mode.
let mut parser = driver::parse_document(RcDom::default(), ParseOpts::default());
parser.process("<table>x".into());
assert!(!parser.tokenizer.sink.is_in_text_insertion_mode());

// Foreign content.
let mut parser = driver::parse_document(RcDom::default(), ParseOpts::default());
parser.process("<svg><script>".into());
assert!(!parser.tokenizer.sink.is_in_text_insertion_mode());
parser.process("</script><style>".into());
assert!(!parser.tokenizer.sink.is_in_text_insertion_mode());

// Without scripting, the contents of `noscript` are parsed as markup.
let opts = ParseOpts {
tree_builder: TreeBuilderOpts {
scripting_enabled: false,
..Default::default()
},
..Default::default()
};
let mut parser = driver::parse_document(RcDom::default(), opts);
parser.process("<body><noscript>".into());
assert!(!parser.tokenizer.sink.is_in_text_insertion_mode());

// The tokenizer starts in the script data state, but the text is inserted into the root
// element in the "in body" insertion mode.
let context = QualName::new(None, ns!(html), local_name!("script"));
let mut parser = driver::parse_fragment(
RcDom::default(),
ParseOpts::default(),
context,
vec![],
true,
);
parser.process("<p>".into());
assert!(!parser.tokenizer.sink.is_in_text_insertion_mode());
assert_eq!(open_elements(&parser.tokenizer.sink), ["html"]);
assert_eq!(current_node_text(&parser.tokenizer.sink), "<p>");
}
Loading