diff --git a/tokenizer/README.md b/tokenizer/README.md index 66b81e8..996d1e6 100644 --- a/tokenizer/README.md +++ b/tokenizer/README.md @@ -76,6 +76,7 @@ tokens are: ["StartTag", name, {attributes}] ["EndTag", name] ["Comment", data] + ["ProcessingInstruction", data] ["Character", data] `public_id` and `system_id` are either strings or `null`. `correctness` diff --git a/tokenizer/test2.test b/tokenizer/test2.test index c29e4c3..edbec4b 100644 --- a/tokenizer/test2.test +++ b/tokenizer/test2.test @@ -183,19 +183,17 @@ { "code": "invalid-first-character-of-tag-name", "line": 1, "col": 3 } ]}, -{"description":"Simili processing instruction", +{"description":"Processing instruction", "input":"", -"output":[["Comment", "?namespace"]], -"errors":[ - { "code": "unexpected-question-mark-instead-of-tag-name", "line": 1, "col": 2 } -]}, +"output":[["ProcessingInstruction", "namespace", ""]]}, -{"description":"A bogus comment stops at >, even if preceded by two dashes", -"input":"", -"output":[["Comment", "?foo--"]], -"errors":[ - { "code": "unexpected-question-mark-instead-of-tag-name", "line": 1, "col": 2 } -]}, +{"description":"A processing instruction delimited by ?> doesn't contain the delimiting ?", +"input":"", +"output":[["ProcessingInstruction", "foo", "bar"]]}, + +{"description":"A processing instruction's data delimited by ?> contains the correct amount of ?", +"input":"", +"output":[["ProcessingInstruction", "foo", "bar?"]]}, {"description":"Unescaped <", "input":"foo < bar", diff --git a/tokenizer/test3.test b/tokenizer/test3.test index 901a581..7a6a1b9 100644 --- a/tokenizer/test3.test +++ b/tokenizer/test3.test @@ -8954,16 +8954,184 @@ {"description":"", "input":"", "output":[["Comment", "?"]], "errors":[ - { "code": "unexpected-question-mark-instead-of-tag-name", "line": 1, "col": 2 } + { "code": "invalid-first-character-of-processing-instruction-target", "line": 1, "col": 3 } ]}, {"description":"