/[suikacvs]/markup/html/whatpm/Whatpm/HTML.pm.src
Suika

Contents of /markup/html/whatpm/Whatpm/HTML.pm.src

Parent Directory Parent Directory | Revision Log Revision Log


Revision 1.203 - (show annotations) (download) (as text)
Sat Oct 4 17:16:02 2008 UTC (18 years ago) by wakaba
Branch: MAIN
Changes since 1.202: +6 -5 lines
File MIME type: application/x-wais-source
++ whatpm/t/ChangeLog	4 Oct 2008 17:15:55 -0000
2008-10-05  Wakaba  <wakaba@suika.fam.cx>

	* HTML-tree.t: New test files added.

	* Makefile: New test files added.

++ whatpm/Whatpm/ChangeLog	4 Oct 2008 17:15:20 -0000
2008-10-05  Wakaba  <wakaba@suika.fam.cx>

	* HTML.pm.src: An AAA bug fixed.

1 package Whatpm::HTML;
2 use strict;
3 our $VERSION=do{my @r=(q$Revision: 1.202 $=~/\d+/g);sprintf "%d."."%02d" x $#r,@r};
4 use Error qw(:try);
5
6 ## NOTE: This module don't check all HTML5 parse errors; character
7 ## encoding related parse errors are expected to be handled by relevant
8 ## modules.
9 ## Parse errors for control characters that are not allowed in HTML5
10 ## documents, for surrogate code points, and for noncharacter code
11 ## points, as well as U+FFFD substitions for characters whose code points
12 ## is higher than U+10FFFF may be detected by combining the parser with
13 ## the checker implemented by Whatpm::Charset::UnicodeChecker (for its
14 ## usage example, see |t/HTML-tree.t| in the Whatpm package or the
15 ## WebHACC::Language::HTML module in the WebHACC package).
16
17 ## ISSUE:
18 ## var doc = implementation.createDocument (null, null, null);
19 ## doc.write ('');
20 ## alert (doc.compatMode);
21
22 require IO::Handle;
23
24 my $HTML_NS = q<http://www.w3.org/1999/xhtml>;
25 my $MML_NS = q<http://www.w3.org/1998/Math/MathML>;
26 my $SVG_NS = q<http://www.w3.org/2000/svg>;
27 my $XLINK_NS = q<http://www.w3.org/1999/xlink>;
28 my $XML_NS = q<http://www.w3.org/XML/1998/namespace>;
29 my $XMLNS_NS = q<http://www.w3.org/2000/xmlns/>;
30
31 sub A_EL () { 0b1 }
32 sub ADDRESS_EL () { 0b10 }
33 sub BODY_EL () { 0b100 }
34 sub BUTTON_EL () { 0b1000 }
35 sub CAPTION_EL () { 0b10000 }
36 sub DD_EL () { 0b100000 }
37 sub DIV_EL () { 0b1000000 }
38 sub DT_EL () { 0b10000000 }
39 sub FORM_EL () { 0b100000000 }
40 sub FORMATTING_EL () { 0b1000000000 }
41 sub FRAMESET_EL () { 0b10000000000 }
42 sub HEADING_EL () { 0b100000000000 }
43 sub HTML_EL () { 0b1000000000000 }
44 sub LI_EL () { 0b10000000000000 }
45 sub NOBR_EL () { 0b100000000000000 }
46 sub OPTION_EL () { 0b1000000000000000 }
47 sub OPTGROUP_EL () { 0b10000000000000000 }
48 sub P_EL () { 0b100000000000000000 }
49 sub SELECT_EL () { 0b1000000000000000000 }
50 sub TABLE_EL () { 0b10000000000000000000 }
51 sub TABLE_CELL_EL () { 0b100000000000000000000 }
52 sub TABLE_ROW_EL () { 0b1000000000000000000000 }
53 sub TABLE_ROW_GROUP_EL () { 0b10000000000000000000000 }
54 sub MISC_SCOPING_EL () { 0b100000000000000000000000 }
55 sub MISC_SPECIAL_EL () { 0b1000000000000000000000000 }
56 sub FOREIGN_EL () { 0b10000000000000000000000000 }
57 sub FOREIGN_FLOW_CONTENT_EL () { 0b100000000000000000000000000 }
58 sub MML_AXML_EL () { 0b1000000000000000000000000000 }
59 sub RUBY_EL () { 0b10000000000000000000000000000 }
60 sub RUBY_COMPONENT_EL () { 0b100000000000000000000000000000 }
61
62 sub TABLE_ROWS_EL () {
63 TABLE_EL |
64 TABLE_ROW_EL |
65 TABLE_ROW_GROUP_EL
66 }
67
68 ## NOTE: Used in "generate implied end tags" algorithm.
69 ## NOTE: There is a code where a modified version of
70 ## END_TAG_OPTIONAL_EL is used in "generate implied end tags"
71 ## implementation (search for the algorithm name).
72 sub END_TAG_OPTIONAL_EL () {
73 DD_EL |
74 DT_EL |
75 LI_EL |
76 OPTION_EL |
77 OPTGROUP_EL |
78 P_EL |
79 RUBY_COMPONENT_EL
80 }
81
82 ## NOTE: Used in </body> and EOF algorithms.
83 sub ALL_END_TAG_OPTIONAL_EL () {
84 DD_EL |
85 DT_EL |
86 LI_EL |
87 P_EL |
88
89 ## ISSUE: option, optgroup, rt, rp?
90
91 BODY_EL |
92 HTML_EL |
93 TABLE_CELL_EL |
94 TABLE_ROW_EL |
95 TABLE_ROW_GROUP_EL
96 }
97
98 sub SCOPING_EL () {
99 BUTTON_EL |
100 CAPTION_EL |
101 HTML_EL |
102 TABLE_EL |
103 TABLE_CELL_EL |
104 MISC_SCOPING_EL
105 }
106
107 sub TABLE_SCOPING_EL () {
108 HTML_EL |
109 TABLE_EL
110 }
111
112 sub TABLE_ROWS_SCOPING_EL () {
113 HTML_EL |
114 TABLE_ROW_GROUP_EL
115 }
116
117 sub TABLE_ROW_SCOPING_EL () {
118 HTML_EL |
119 TABLE_ROW_EL
120 }
121
122 sub SPECIAL_EL () {
123 ADDRESS_EL |
124 BODY_EL |
125 DIV_EL |
126
127 DD_EL |
128 DT_EL |
129 LI_EL |
130 P_EL |
131
132 FORM_EL |
133 FRAMESET_EL |
134 HEADING_EL |
135 SELECT_EL |
136 TABLE_ROW_EL |
137 TABLE_ROW_GROUP_EL |
138 MISC_SPECIAL_EL
139 }
140
141 my $el_category = {
142 a => A_EL | FORMATTING_EL,
143 address => ADDRESS_EL,
144 applet => MISC_SCOPING_EL,
145 area => MISC_SPECIAL_EL,
146 article => MISC_SPECIAL_EL,
147 aside => MISC_SPECIAL_EL,
148 b => FORMATTING_EL,
149 base => MISC_SPECIAL_EL,
150 basefont => MISC_SPECIAL_EL,
151 bgsound => MISC_SPECIAL_EL,
152 big => FORMATTING_EL,
153 blockquote => MISC_SPECIAL_EL,
154 body => BODY_EL,
155 br => MISC_SPECIAL_EL,
156 button => BUTTON_EL,
157 caption => CAPTION_EL,
158 center => MISC_SPECIAL_EL,
159 col => MISC_SPECIAL_EL,
160 colgroup => MISC_SPECIAL_EL,
161 command => MISC_SPECIAL_EL,
162 datagrid => MISC_SPECIAL_EL,
163 dd => DD_EL,
164 details => MISC_SPECIAL_EL,
165 dialog => MISC_SPECIAL_EL,
166 dir => MISC_SPECIAL_EL,
167 div => DIV_EL,
168 dl => MISC_SPECIAL_EL,
169 dt => DT_EL,
170 em => FORMATTING_EL,
171 embed => MISC_SPECIAL_EL,
172 eventsource => MISC_SPECIAL_EL,
173 fieldset => MISC_SPECIAL_EL,
174 figure => MISC_SPECIAL_EL,
175 font => FORMATTING_EL,
176 footer => MISC_SPECIAL_EL,
177 form => FORM_EL,
178 frame => MISC_SPECIAL_EL,
179 frameset => FRAMESET_EL,
180 h1 => HEADING_EL,
181 h2 => HEADING_EL,
182 h3 => HEADING_EL,
183 h4 => HEADING_EL,
184 h5 => HEADING_EL,
185 h6 => HEADING_EL,
186 head => MISC_SPECIAL_EL,
187 header => MISC_SPECIAL_EL,
188 hr => MISC_SPECIAL_EL,
189 html => HTML_EL,
190 i => FORMATTING_EL,
191 iframe => MISC_SPECIAL_EL,
192 img => MISC_SPECIAL_EL,
193 #image => MISC_SPECIAL_EL, ## NOTE: Commented out in the spec.
194 input => MISC_SPECIAL_EL,
195 isindex => MISC_SPECIAL_EL,
196 li => LI_EL,
197 link => MISC_SPECIAL_EL,
198 listing => MISC_SPECIAL_EL,
199 marquee => MISC_SCOPING_EL,
200 menu => MISC_SPECIAL_EL,
201 meta => MISC_SPECIAL_EL,
202 nav => MISC_SPECIAL_EL,
203 nobr => NOBR_EL | FORMATTING_EL,
204 noembed => MISC_SPECIAL_EL,
205 noframes => MISC_SPECIAL_EL,
206 noscript => MISC_SPECIAL_EL,
207 object => MISC_SCOPING_EL,
208 ol => MISC_SPECIAL_EL,
209 optgroup => OPTGROUP_EL,
210 option => OPTION_EL,
211 p => P_EL,
212 param => MISC_SPECIAL_EL,
213 plaintext => MISC_SPECIAL_EL,
214 pre => MISC_SPECIAL_EL,
215 rp => RUBY_COMPONENT_EL,
216 rt => RUBY_COMPONENT_EL,
217 ruby => RUBY_EL,
218 s => FORMATTING_EL,
219 script => MISC_SPECIAL_EL,
220 select => SELECT_EL,
221 section => MISC_SPECIAL_EL,
222 small => FORMATTING_EL,
223 spacer => MISC_SPECIAL_EL,
224 strike => FORMATTING_EL,
225 strong => FORMATTING_EL,
226 style => MISC_SPECIAL_EL,
227 table => TABLE_EL,
228 tbody => TABLE_ROW_GROUP_EL,
229 td => TABLE_CELL_EL,
230 textarea => MISC_SPECIAL_EL,
231 tfoot => TABLE_ROW_GROUP_EL,
232 th => TABLE_CELL_EL,
233 thead => TABLE_ROW_GROUP_EL,
234 title => MISC_SPECIAL_EL,
235 tr => TABLE_ROW_EL,
236 tt => FORMATTING_EL,
237 u => FORMATTING_EL,
238 ul => MISC_SPECIAL_EL,
239 wbr => MISC_SPECIAL_EL,
240 };
241
242 my $el_category_f = {
243 $MML_NS => {
244 'annotation-xml' => MML_AXML_EL,
245 mi => FOREIGN_FLOW_CONTENT_EL,
246 mo => FOREIGN_FLOW_CONTENT_EL,
247 mn => FOREIGN_FLOW_CONTENT_EL,
248 ms => FOREIGN_FLOW_CONTENT_EL,
249 mtext => FOREIGN_FLOW_CONTENT_EL,
250 },
251 $SVG_NS => {
252 foreignObject => FOREIGN_FLOW_CONTENT_EL | MISC_SCOPING_EL,
253 desc => FOREIGN_FLOW_CONTENT_EL,
254 title => FOREIGN_FLOW_CONTENT_EL,
255 },
256 ## NOTE: In addition, FOREIGN_EL is set to non-HTML elements.
257 };
258
259 my $svg_attr_name = {
260 attributename => 'attributeName',
261 attributetype => 'attributeType',
262 basefrequency => 'baseFrequency',
263 baseprofile => 'baseProfile',
264 calcmode => 'calcMode',
265 clippathunits => 'clipPathUnits',
266 contentscripttype => 'contentScriptType',
267 contentstyletype => 'contentStyleType',
268 diffuseconstant => 'diffuseConstant',
269 edgemode => 'edgeMode',
270 externalresourcesrequired => 'externalResourcesRequired',
271 filterres => 'filterRes',
272 filterunits => 'filterUnits',
273 glyphref => 'glyphRef',
274 gradienttransform => 'gradientTransform',
275 gradientunits => 'gradientUnits',
276 kernelmatrix => 'kernelMatrix',
277 kernelunitlength => 'kernelUnitLength',
278 keypoints => 'keyPoints',
279 keysplines => 'keySplines',
280 keytimes => 'keyTimes',
281 lengthadjust => 'lengthAdjust',
282 limitingconeangle => 'limitingConeAngle',
283 markerheight => 'markerHeight',
284 markerunits => 'markerUnits',
285 markerwidth => 'markerWidth',
286 maskcontentunits => 'maskContentUnits',
287 maskunits => 'maskUnits',
288 numoctaves => 'numOctaves',
289 pathlength => 'pathLength',
290 patterncontentunits => 'patternContentUnits',
291 patterntransform => 'patternTransform',
292 patternunits => 'patternUnits',
293 pointsatx => 'pointsAtX',
294 pointsaty => 'pointsAtY',
295 pointsatz => 'pointsAtZ',
296 preservealpha => 'preserveAlpha',
297 preserveaspectratio => 'preserveAspectRatio',
298 primitiveunits => 'primitiveUnits',
299 refx => 'refX',
300 refy => 'refY',
301 repeatcount => 'repeatCount',
302 repeatdur => 'repeatDur',
303 requiredextensions => 'requiredExtensions',
304 requiredfeatures => 'requiredFeatures',
305 specularconstant => 'specularConstant',
306 specularexponent => 'specularExponent',
307 spreadmethod => 'spreadMethod',
308 startoffset => 'startOffset',
309 stddeviation => 'stdDeviation',
310 stitchtiles => 'stitchTiles',
311 surfacescale => 'surfaceScale',
312 systemlanguage => 'systemLanguage',
313 tablevalues => 'tableValues',
314 targetx => 'targetX',
315 targety => 'targetY',
316 textlength => 'textLength',
317 viewbox => 'viewBox',
318 viewtarget => 'viewTarget',
319 xchannelselector => 'xChannelSelector',
320 ychannelselector => 'yChannelSelector',
321 zoomandpan => 'zoomAndPan',
322 };
323
324 my $foreign_attr_xname = {
325 'xlink:actuate' => [$XLINK_NS, ['xlink', 'actuate']],
326 'xlink:arcrole' => [$XLINK_NS, ['xlink', 'arcrole']],
327 'xlink:href' => [$XLINK_NS, ['xlink', 'href']],
328 'xlink:role' => [$XLINK_NS, ['xlink', 'role']],
329 'xlink:show' => [$XLINK_NS, ['xlink', 'show']],
330 'xlink:title' => [$XLINK_NS, ['xlink', 'title']],
331 'xlink:type' => [$XLINK_NS, ['xlink', 'type']],
332 'xml:base' => [$XML_NS, ['xml', 'base']],
333 'xml:lang' => [$XML_NS, ['xml', 'lang']],
334 'xml:space' => [$XML_NS, ['xml', 'space']],
335 'xmlns' => [$XMLNS_NS, [undef, 'xmlns']],
336 'xmlns:xlink' => [$XMLNS_NS, ['xmlns', 'xlink']],
337 };
338
339 ## ISSUE: xmlns:xlink="non-xlink-ns" is not an error.
340
341 my $charref_map = {
342 0x0D => 0x000A,
343 0x80 => 0x20AC,
344 0x81 => 0xFFFD,
345 0x82 => 0x201A,
346 0x83 => 0x0192,
347 0x84 => 0x201E,
348 0x85 => 0x2026,
349 0x86 => 0x2020,
350 0x87 => 0x2021,
351 0x88 => 0x02C6,
352 0x89 => 0x2030,
353 0x8A => 0x0160,
354 0x8B => 0x2039,
355 0x8C => 0x0152,
356 0x8D => 0xFFFD,
357 0x8E => 0x017D,
358 0x8F => 0xFFFD,
359 0x90 => 0xFFFD,
360 0x91 => 0x2018,
361 0x92 => 0x2019,
362 0x93 => 0x201C,
363 0x94 => 0x201D,
364 0x95 => 0x2022,
365 0x96 => 0x2013,
366 0x97 => 0x2014,
367 0x98 => 0x02DC,
368 0x99 => 0x2122,
369 0x9A => 0x0161,
370 0x9B => 0x203A,
371 0x9C => 0x0153,
372 0x9D => 0xFFFD,
373 0x9E => 0x017E,
374 0x9F => 0x0178,
375 }; # $charref_map
376 $charref_map->{$_} = 0xFFFD
377 for 0x0000..0x0008, 0x000B, 0x000E..0x001F, 0x007F,
378 0xD800..0xDFFF, 0xFDD0..0xFDDF, ## ISSUE: 0xFDEF
379 0xFFFE, 0xFFFF, 0x1FFFE, 0x1FFFF, 0x2FFFE, 0x2FFFF, 0x3FFFE, 0x3FFFF,
380 0x4FFFE, 0x4FFFF, 0x5FFFE, 0x5FFFF, 0x6FFFE, 0x6FFFF, 0x7FFFE,
381 0x7FFFF, 0x8FFFE, 0x8FFFF, 0x9FFFE, 0x9FFFF, 0xAFFFE, 0xAFFFF,
382 0xBFFFE, 0xBFFFF, 0xCFFFE, 0xCFFFF, 0xDFFFE, 0xDFFFF, 0xEFFFE,
383 0xEFFFF, 0xFFFFE, 0xFFFFF, 0x10FFFE, 0x10FFFF;
384
385 ## TODO: Invoke the reset algorithm when a resettable element is
386 ## created (cf. HTML5 revision 2259).
387
388 sub parse_byte_string ($$$$;$) {
389 my $self = shift;
390 my $charset_name = shift;
391 open my $input, '<', ref $_[0] ? $_[0] : \($_[0]);
392 return $self->parse_byte_stream ($charset_name, $input, @_[1..$#_]);
393 } # parse_byte_string
394
395 sub parse_byte_stream ($$$$;$$) {
396 # my ($self, $charset_name, $byte_stream, $doc, $onerror, $get_wrapper) = @_;
397 my $self = ref $_[0] ? shift : shift->new;
398 my $charset_name = shift;
399 my $byte_stream = $_[0];
400
401 my $onerror = $_[2] || sub {
402 my (%opt) = @_;
403 warn "Parse error ($opt{type})\n";
404 };
405 $self->{parse_error} = $onerror; # updated later by parse_char_string
406
407 my $get_wrapper = $_[3] || sub ($) {
408 return $_[0]; # $_[0] = byte stream handle, returned = arg to char handle
409 };
410
411 ## HTML5 encoding sniffing algorithm
412 require Message::Charset::Info;
413 my $charset;
414 my $buffer;
415 my ($char_stream, $e_status);
416
417 SNIFFING: {
418 ## NOTE: By setting |allow_fallback| option true when the
419 ## |get_decode_handle| method is invoked, we ignore what the HTML5
420 ## spec requires, i.e. unsupported encoding should be ignored.
421 ## TODO: We should not do this unless the parser is invoked
422 ## in the conformance checking mode, in which this behavior
423 ## would be useful.
424
425 ## Step 1
426 if (defined $charset_name) {
427 $charset = Message::Charset::Info->get_by_html_name ($charset_name);
428 ## TODO: Is this ok? Transfer protocol's parameter should be
429 ## interpreted in its semantics?
430
431 ($char_stream, $e_status) = $charset->get_decode_handle
432 ($byte_stream, allow_error_reporting => 1,
433 allow_fallback => 1);
434 if ($char_stream) {
435 $self->{confident} = 1;
436 last SNIFFING;
437 } else {
438 !!!parse-error (type => 'charset:not supported',
439 layer => 'encode',
440 line => 1, column => 1,
441 value => $charset_name,
442 level => $self->{level}->{uncertain});
443 }
444 }
445
446 ## Step 2
447 my $byte_buffer = '';
448 for (1..1024) {
449 my $char = $byte_stream->getc;
450 last unless defined $char;
451 $byte_buffer .= $char;
452 } ## TODO: timeout
453
454 ## Step 3
455 if ($byte_buffer =~ /^\xFE\xFF/) {
456 $charset = Message::Charset::Info->get_by_html_name ('utf-16be');
457 ($char_stream, $e_status) = $charset->get_decode_handle
458 ($byte_stream, allow_error_reporting => 1,
459 allow_fallback => 1, byte_buffer => \$byte_buffer);
460 $self->{confident} = 1;
461 last SNIFFING;
462 } elsif ($byte_buffer =~ /^\xFF\xFE/) {
463 $charset = Message::Charset::Info->get_by_html_name ('utf-16le');
464 ($char_stream, $e_status) = $charset->get_decode_handle
465 ($byte_stream, allow_error_reporting => 1,
466 allow_fallback => 1, byte_buffer => \$byte_buffer);
467 $self->{confident} = 1;
468 last SNIFFING;
469 } elsif ($byte_buffer =~ /^\xEF\xBB\xBF/) {
470 $charset = Message::Charset::Info->get_by_html_name ('utf-8');
471 ($char_stream, $e_status) = $charset->get_decode_handle
472 ($byte_stream, allow_error_reporting => 1,
473 allow_fallback => 1, byte_buffer => \$byte_buffer);
474 $self->{confident} = 1;
475 last SNIFFING;
476 }
477
478 ## Step 4
479 ## TODO: <meta charset>
480
481 ## Step 5
482 ## TODO: from history
483
484 ## Step 6
485 require Whatpm::Charset::UniversalCharDet;
486 $charset_name = Whatpm::Charset::UniversalCharDet->detect_byte_string
487 ($byte_buffer);
488 if (defined $charset_name) {
489 $charset = Message::Charset::Info->get_by_html_name ($charset_name);
490
491 require Whatpm::Charset::DecodeHandle;
492 $buffer = Whatpm::Charset::DecodeHandle::ByteBuffer->new
493 ($byte_stream);
494 ($char_stream, $e_status) = $charset->get_decode_handle
495 ($buffer, allow_error_reporting => 1,
496 allow_fallback => 1, byte_buffer => \$byte_buffer);
497 if ($char_stream) {
498 $buffer->{buffer} = $byte_buffer;
499 !!!parse-error (type => 'sniffing:chardet',
500 text => $charset_name,
501 level => $self->{level}->{info},
502 layer => 'encode',
503 line => 1, column => 1);
504 $self->{confident} = 0;
505 last SNIFFING;
506 }
507 }
508
509 ## Step 7: default
510 ## TODO: Make this configurable.
511 $charset = Message::Charset::Info->get_by_html_name ('windows-1252');
512 ## NOTE: We choose |windows-1252| here, since |utf-8| should be
513 ## detectable in the step 6.
514 require Whatpm::Charset::DecodeHandle;
515 $buffer = Whatpm::Charset::DecodeHandle::ByteBuffer->new
516 ($byte_stream);
517 ($char_stream, $e_status)
518 = $charset->get_decode_handle ($buffer,
519 allow_error_reporting => 1,
520 allow_fallback => 1,
521 byte_buffer => \$byte_buffer);
522 $buffer->{buffer} = $byte_buffer;
523 !!!parse-error (type => 'sniffing:default',
524 text => 'windows-1252',
525 level => $self->{level}->{info},
526 line => 1, column => 1,
527 layer => 'encode');
528 $self->{confident} = 0;
529 } # SNIFFING
530
531 if ($e_status & Message::Charset::Info::FALLBACK_ENCODING_IMPL ()) {
532 $self->{input_encoding} = $charset->get_iana_name; ## TODO: Should we set actual charset decoder's encoding name?
533 !!!parse-error (type => 'chardecode:fallback',
534 #text => $self->{input_encoding},
535 level => $self->{level}->{uncertain},
536 line => 1, column => 1,
537 layer => 'encode');
538 } elsif (not ($e_status &
539 Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL ())) {
540 $self->{input_encoding} = $charset->get_iana_name;
541 !!!parse-error (type => 'chardecode:no error',
542 text => $self->{input_encoding},
543 level => $self->{level}->{uncertain},
544 line => 1, column => 1,
545 layer => 'encode');
546 } else {
547 $self->{input_encoding} = $charset->get_iana_name;
548 }
549
550 $self->{change_encoding} = sub {
551 my $self = shift;
552 $charset_name = shift;
553 my $token = shift;
554
555 $charset = Message::Charset::Info->get_by_html_name ($charset_name);
556 ($char_stream, $e_status) = $charset->get_decode_handle
557 ($byte_stream, allow_error_reporting => 1, allow_fallback => 1,
558 byte_buffer => \ $buffer->{buffer});
559
560 if ($char_stream) { # if supported
561 ## "Change the encoding" algorithm:
562
563 ## Step 1
564 if ($charset->{category} &
565 Message::Charset::Info::CHARSET_CATEGORY_UTF16 ()) {
566 $charset = Message::Charset::Info->get_by_html_name ('utf-8');
567 ($char_stream, $e_status) = $charset->get_decode_handle
568 ($byte_stream,
569 byte_buffer => \ $buffer->{buffer});
570 }
571 $charset_name = $charset->get_iana_name;
572
573 ## Step 2
574 if (defined $self->{input_encoding} and
575 $self->{input_encoding} eq $charset_name) {
576 !!!parse-error (type => 'charset label:matching',
577 text => $charset_name,
578 level => $self->{level}->{info});
579 $self->{confident} = 1;
580 return;
581 }
582
583 !!!parse-error (type => 'charset label detected',
584 text => $self->{input_encoding},
585 value => $charset_name,
586 level => $self->{level}->{warn},
587 token => $token);
588
589 ## Step 3
590 # if (can) {
591 ## change the encoding on the fly.
592 #$self->{confident} = 1;
593 #return;
594 # }
595
596 ## Step 4
597 throw Whatpm::HTML::RestartParser ();
598 }
599 }; # $self->{change_encoding}
600
601 my $char_onerror = sub {
602 my (undef, $type, %opt) = @_;
603 !!!parse-error (layer => 'encode',
604 line => $self->{line}, column => $self->{column} + 1,
605 %opt, type => $type);
606 if ($opt{octets}) {
607 ${$opt{octets}} = "\x{FFFD}"; # relacement character
608 }
609 };
610
611 my $wrapped_char_stream = $get_wrapper->($char_stream);
612 $wrapped_char_stream->onerror ($char_onerror);
613
614 my @args = ($_[1], $_[2]); # $doc, $onerror - $get_wrapper = undef;
615 my $return;
616 try {
617 $return = $self->parse_char_stream ($wrapped_char_stream, @args);
618 } catch Whatpm::HTML::RestartParser with {
619 ## NOTE: Invoked after {change_encoding}.
620
621 if ($e_status & Message::Charset::Info::FALLBACK_ENCODING_IMPL ()) {
622 $self->{input_encoding} = $charset->get_iana_name; ## TODO: Should we set actual charset decoder's encoding name?
623 !!!parse-error (type => 'chardecode:fallback',
624 level => $self->{level}->{uncertain},
625 #text => $self->{input_encoding},
626 line => 1, column => 1,
627 layer => 'encode');
628 } elsif (not ($e_status &
629 Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL ())) {
630 $self->{input_encoding} = $charset->get_iana_name;
631 !!!parse-error (type => 'chardecode:no error',
632 text => $self->{input_encoding},
633 level => $self->{level}->{uncertain},
634 line => 1, column => 1,
635 layer => 'encode');
636 } else {
637 $self->{input_encoding} = $charset->get_iana_name;
638 }
639 $self->{confident} = 1;
640
641 $wrapped_char_stream = $get_wrapper->($char_stream);
642 $wrapped_char_stream->onerror ($char_onerror);
643
644 $return = $self->parse_char_stream ($wrapped_char_stream, @args);
645 };
646 return $return;
647 } # parse_byte_stream
648
649 ## NOTE: HTML5 spec says that the encoding layer MUST NOT strip BOM
650 ## and the HTML layer MUST ignore it. However, we does strip BOM in
651 ## the encoding layer and the HTML layer does not ignore any U+FEFF,
652 ## because the core part of our HTML parser expects a string of character,
653 ## not a string of bytes or code units or anything which might contain a BOM.
654 ## Therefore, any parser interface that accepts a string of bytes,
655 ## such as |parse_byte_string| in this module, must ensure that it does
656 ## strip the BOM and never strip any ZWNBSP.
657
658 sub parse_char_string ($$$;$$) {
659 #my ($self, $s, $doc, $onerror, $get_wrapper) = @_;
660 my $self = shift;
661 my $s = ref $_[0] ? $_[0] : \($_[0]);
662 require Whatpm::Charset::DecodeHandle;
663 my $input = Whatpm::Charset::DecodeHandle::CharString->new ($s);
664 return $self->parse_char_stream ($input, @_[1..$#_]);
665 } # parse_char_string
666 *parse_string = \&parse_char_string; ## NOTE: Alias for backward compatibility.
667
668 sub parse_char_stream ($$$;$$) {
669 my $self = ref $_[0] ? shift : shift->new;
670 my $input = $_[0];
671 $self->{document} = $_[1];
672 @{$self->{document}->child_nodes} = ();
673
674 ## NOTE: |set_inner_html| copies most of this method's code
675
676 $self->{confident} = 1 unless exists $self->{confident};
677 $self->{document}->input_encoding ($self->{input_encoding})
678 if defined $self->{input_encoding};
679 ## TODO: |{input_encoding}| is needless?
680
681 $self->{line_prev} = $self->{line} = 1;
682 $self->{column_prev} = -1;
683 $self->{column} = 0;
684 $self->{set_nc} = sub {
685 my $self = shift;
686
687 my $char = '';
688 if (defined $self->{next_nc}) {
689 $char = $self->{next_nc};
690 delete $self->{next_nc};
691 $self->{nc} = ord $char;
692 } else {
693 $self->{char_buffer} = '';
694 $self->{char_buffer_pos} = 0;
695
696 my $count = $input->manakai_read_until
697 ($self->{char_buffer}, qr/[^\x00\x0A\x0D]/, $self->{char_buffer_pos});
698 if ($count) {
699 $self->{line_prev} = $self->{line};
700 $self->{column_prev} = $self->{column};
701 $self->{column}++;
702 $self->{nc}
703 = ord substr ($self->{char_buffer}, $self->{char_buffer_pos}++, 1);
704 return;
705 }
706
707 if ($input->read ($char, 1)) {
708 $self->{nc} = ord $char;
709 } else {
710 $self->{nc} = -1;
711 return;
712 }
713 }
714
715 ($self->{line_prev}, $self->{column_prev})
716 = ($self->{line}, $self->{column});
717 $self->{column}++;
718
719 if ($self->{nc} == 0x000A) { # LF
720 !!!cp ('j1');
721 $self->{line}++;
722 $self->{column} = 0;
723 } elsif ($self->{nc} == 0x000D) { # CR
724 !!!cp ('j2');
725 ## TODO: support for abort/streaming
726 my $next = '';
727 if ($input->read ($next, 1) and $next ne "\x0A") {
728 $self->{next_nc} = $next;
729 }
730 $self->{nc} = 0x000A; # LF # MUST
731 $self->{line}++;
732 $self->{column} = 0;
733 } elsif ($self->{nc} == 0x0000) { # NULL
734 !!!cp ('j4');
735 !!!parse-error (type => 'NULL');
736 $self->{nc} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
737 }
738 };
739
740 $self->{read_until} = sub {
741 #my ($scalar, $specials_range, $offset) = @_;
742 return 0 if defined $self->{next_nc};
743
744 my $pattern = qr/[^$_[1]\x00\x0A\x0D]/;
745 my $offset = $_[2] || 0;
746
747 if ($self->{char_buffer_pos} < length $self->{char_buffer}) {
748 pos ($self->{char_buffer}) = $self->{char_buffer_pos};
749 if ($self->{char_buffer} =~ /\G(?>$pattern)+/) {
750 substr ($_[0], $offset)
751 = substr ($self->{char_buffer}, $-[0], $+[0] - $-[0]);
752 my $count = $+[0] - $-[0];
753 if ($count) {
754 $self->{column} += $count;
755 $self->{char_buffer_pos} += $count;
756 $self->{line_prev} = $self->{line};
757 $self->{column_prev} = $self->{column} - 1;
758 $self->{nc} = -1;
759 }
760 return $count;
761 } else {
762 return 0;
763 }
764 } else {
765 my $count = $input->manakai_read_until ($_[0], $pattern, $_[2]);
766 if ($count) {
767 $self->{column} += $count;
768 $self->{line_prev} = $self->{line};
769 $self->{column_prev} = $self->{column} - 1;
770 $self->{nc} = -1;
771 }
772 return $count;
773 }
774 }; # $self->{read_until}
775
776 my $onerror = $_[2] || sub {
777 my (%opt) = @_;
778 my $line = $opt{token} ? $opt{token}->{line} : $opt{line};
779 my $column = $opt{token} ? $opt{token}->{column} : $opt{column};
780 warn "Parse error ($opt{type}) at line $line column $column\n";
781 };
782 $self->{parse_error} = sub {
783 $onerror->(line => $self->{line}, column => $self->{column}, @_);
784 };
785
786 my $char_onerror = sub {
787 my (undef, $type, %opt) = @_;
788 !!!parse-error (layer => 'encode',
789 line => $self->{line}, column => $self->{column} + 1,
790 %opt, type => $type);
791 }; # $char_onerror
792
793 if ($_[3]) {
794 $input = $_[3]->($input);
795 $input->onerror ($char_onerror);
796 } else {
797 $input->onerror ($char_onerror) unless defined $input->onerror;
798 }
799
800 $self->_initialize_tokenizer;
801 $self->_initialize_tree_constructor;
802 $self->_construct_tree;
803 $self->_terminate_tree_constructor;
804
805 delete $self->{parse_error}; # remove loop
806
807 return $self->{document};
808 } # parse_char_stream
809
810 sub new ($) {
811 my $class = shift;
812 my $self = bless {
813 level => {must => 'm',
814 should => 's',
815 warn => 'w',
816 info => 'i',
817 uncertain => 'u'},
818 }, $class;
819 $self->{set_nc} = sub {
820 $self->{nc} = -1;
821 };
822 $self->{parse_error} = sub {
823 #
824 };
825 $self->{change_encoding} = sub {
826 # if ($_[0] is a supported encoding) {
827 # run "change the encoding" algorithm;
828 # throw Whatpm::HTML::RestartParser (charset => $new_encoding);
829 # }
830 };
831 $self->{application_cache_selection} = sub {
832 #
833 };
834 return $self;
835 } # new
836
837 sub CM_ENTITY () { 0b001 } # & markup in data
838 sub CM_LIMITED_MARKUP () { 0b010 } # < markup in data (limited)
839 sub CM_FULL_MARKUP () { 0b100 } # < markup in data (any)
840
841 sub PLAINTEXT_CONTENT_MODEL () { 0 }
842 sub CDATA_CONTENT_MODEL () { CM_LIMITED_MARKUP }
843 sub RCDATA_CONTENT_MODEL () { CM_ENTITY | CM_LIMITED_MARKUP }
844 sub PCDATA_CONTENT_MODEL () { CM_ENTITY | CM_FULL_MARKUP }
845
846 sub DATA_STATE () { 0 }
847 #sub ENTITY_DATA_STATE () { 1 }
848 sub TAG_OPEN_STATE () { 2 }
849 sub CLOSE_TAG_OPEN_STATE () { 3 }
850 sub TAG_NAME_STATE () { 4 }
851 sub BEFORE_ATTRIBUTE_NAME_STATE () { 5 }
852 sub ATTRIBUTE_NAME_STATE () { 6 }
853 sub AFTER_ATTRIBUTE_NAME_STATE () { 7 }
854 sub BEFORE_ATTRIBUTE_VALUE_STATE () { 8 }
855 sub ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE () { 9 }
856 sub ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE () { 10 }
857 sub ATTRIBUTE_VALUE_UNQUOTED_STATE () { 11 }
858 #sub ENTITY_IN_ATTRIBUTE_VALUE_STATE () { 12 }
859 sub MARKUP_DECLARATION_OPEN_STATE () { 13 }
860 sub COMMENT_START_STATE () { 14 }
861 sub COMMENT_START_DASH_STATE () { 15 }
862 sub COMMENT_STATE () { 16 }
863 sub COMMENT_END_STATE () { 17 }
864 sub COMMENT_END_DASH_STATE () { 18 }
865 sub BOGUS_COMMENT_STATE () { 19 }
866 sub DOCTYPE_STATE () { 20 }
867 sub BEFORE_DOCTYPE_NAME_STATE () { 21 }
868 sub DOCTYPE_NAME_STATE () { 22 }
869 sub AFTER_DOCTYPE_NAME_STATE () { 23 }
870 sub BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE () { 24 }
871 sub DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE () { 25 }
872 sub DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE () { 26 }
873 sub AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE () { 27 }
874 sub BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE () { 28 }
875 sub DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE () { 29 }
876 sub DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE () { 30 }
877 sub AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE () { 31 }
878 sub BOGUS_DOCTYPE_STATE () { 32 }
879 sub AFTER_ATTRIBUTE_VALUE_QUOTED_STATE () { 33 }
880 sub SELF_CLOSING_START_TAG_STATE () { 34 }
881 sub CDATA_SECTION_STATE () { 35 }
882 sub MD_HYPHEN_STATE () { 36 } # "markup declaration open state" in the spec
883 sub MD_DOCTYPE_STATE () { 37 } # "markup declaration open state" in the spec
884 sub MD_CDATA_STATE () { 38 } # "markup declaration open state" in the spec
885 sub CDATA_RCDATA_CLOSE_TAG_STATE () { 39 } # "close tag open state" in the spec
886 sub CDATA_SECTION_MSE1_STATE () { 40 } # "CDATA section state" in the spec
887 sub CDATA_SECTION_MSE2_STATE () { 41 } # "CDATA section state" in the spec
888 sub PUBLIC_STATE () { 42 } # "after DOCTYPE name state" in the spec
889 sub SYSTEM_STATE () { 43 } # "after DOCTYPE name state" in the spec
890 ## NOTE: "Entity data state", "entity in attribute value state", and
891 ## "consume a character reference" algorithm are jointly implemented
892 ## using the following six states:
893 sub ENTITY_STATE () { 44 }
894 sub ENTITY_HASH_STATE () { 45 }
895 sub NCR_NUM_STATE () { 46 }
896 sub HEXREF_X_STATE () { 47 }
897 sub HEXREF_HEX_STATE () { 48 }
898 sub ENTITY_NAME_STATE () { 49 }
899 sub PCDATA_STATE () { 50 } # "data state" in the spec
900
901 sub DOCTYPE_TOKEN () { 1 }
902 sub COMMENT_TOKEN () { 2 }
903 sub START_TAG_TOKEN () { 3 }
904 sub END_TAG_TOKEN () { 4 }
905 sub END_OF_FILE_TOKEN () { 5 }
906 sub CHARACTER_TOKEN () { 6 }
907
908 sub AFTER_HTML_IMS () { 0b100 }
909 sub HEAD_IMS () { 0b1000 }
910 sub BODY_IMS () { 0b10000 }
911 sub BODY_TABLE_IMS () { 0b100000 }
912 sub TABLE_IMS () { 0b1000000 }
913 sub ROW_IMS () { 0b10000000 }
914 sub BODY_AFTER_IMS () { 0b100000000 }
915 sub FRAME_IMS () { 0b1000000000 }
916 sub SELECT_IMS () { 0b10000000000 }
917 sub IN_FOREIGN_CONTENT_IM () { 0b100000000000 }
918 ## NOTE: "in foreign content" insertion mode is special; it is combined
919 ## with the secondary insertion mode. In this parser, they are stored
920 ## together in the bit-or'ed form.
921
922 ## NOTE: "initial" and "before html" insertion modes have no constants.
923
924 ## NOTE: "after after body" insertion mode.
925 sub AFTER_HTML_BODY_IM () { AFTER_HTML_IMS | BODY_AFTER_IMS }
926
927 ## NOTE: "after after frameset" insertion mode.
928 sub AFTER_HTML_FRAMESET_IM () { AFTER_HTML_IMS | FRAME_IMS }
929
930 sub IN_HEAD_IM () { HEAD_IMS | 0b00 }
931 sub IN_HEAD_NOSCRIPT_IM () { HEAD_IMS | 0b01 }
932 sub AFTER_HEAD_IM () { HEAD_IMS | 0b10 }
933 sub BEFORE_HEAD_IM () { HEAD_IMS | 0b11 }
934 sub IN_BODY_IM () { BODY_IMS }
935 sub IN_CELL_IM () { BODY_IMS | BODY_TABLE_IMS | 0b01 }
936 sub IN_CAPTION_IM () { BODY_IMS | BODY_TABLE_IMS | 0b10 }
937 sub IN_ROW_IM () { TABLE_IMS | ROW_IMS | 0b01 }
938 sub IN_TABLE_BODY_IM () { TABLE_IMS | ROW_IMS | 0b10 }
939 sub IN_TABLE_IM () { TABLE_IMS }
940 sub AFTER_BODY_IM () { BODY_AFTER_IMS }
941 sub IN_FRAMESET_IM () { FRAME_IMS | 0b01 }
942 sub AFTER_FRAMESET_IM () { FRAME_IMS | 0b10 }
943 sub IN_SELECT_IM () { SELECT_IMS | 0b01 }
944 sub IN_SELECT_IN_TABLE_IM () { SELECT_IMS | 0b10 }
945 sub IN_COLUMN_GROUP_IM () { 0b10 }
946
947 ## Implementations MUST act as if state machine in the spec
948
949 sub _initialize_tokenizer ($) {
950 my $self = shift;
951 $self->{state} = DATA_STATE; # MUST
952 #$self->{s_kwd}; # state keyword - initialized when used
953 #$self->{entity__value}; # initialized when used
954 #$self->{entity__match}; # initialized when used
955 $self->{content_model} = PCDATA_CONTENT_MODEL; # be
956 undef $self->{ct}; # current token
957 undef $self->{ca}; # current attribute
958 undef $self->{last_stag_name}; # last emitted start tag name
959 #$self->{prev_state}; # initialized when used
960 delete $self->{self_closing};
961 $self->{char_buffer} = '';
962 $self->{char_buffer_pos} = 0;
963 $self->{nc} = -1; # next input character
964 #$self->{next_nc}
965 !!!next-input-character;
966 $self->{token} = [];
967 # $self->{escape}
968 } # _initialize_tokenizer
969
970 ## A token has:
971 ## ->{type} == DOCTYPE_TOKEN, START_TAG_TOKEN, END_TAG_TOKEN, COMMENT_TOKEN,
972 ## CHARACTER_TOKEN, or END_OF_FILE_TOKEN
973 ## ->{name} (DOCTYPE_TOKEN)
974 ## ->{tag_name} (START_TAG_TOKEN, END_TAG_TOKEN)
975 ## ->{pubid} (DOCTYPE_TOKEN)
976 ## ->{sysid} (DOCTYPE_TOKEN)
977 ## ->{quirks} == 1 or 0 (DOCTYPE_TOKEN): "force-quirks" flag
978 ## ->{attributes} isa HASH (START_TAG_TOKEN, END_TAG_TOKEN)
979 ## ->{name}
980 ## ->{value}
981 ## ->{has_reference} == 1 or 0
982 ## ->{data} (COMMENT_TOKEN, CHARACTER_TOKEN)
983 ## NOTE: The "self-closing flag" is hold as |$self->{self_closing}|.
984 ## |->{self_closing}| is used to save the value of |$self->{self_closing}|
985 ## while the token is pushed back to the stack.
986
987 ## Emitted token MUST immediately be handled by the tree construction state.
988
989 ## Before each step, UA MAY check to see if either one of the scripts in
990 ## "list of scripts that will execute as soon as possible" or the first
991 ## script in the "list of scripts that will execute asynchronously",
992 ## has completed loading. If one has, then it MUST be executed
993 ## and removed from the list.
994
995 ## TODO: Polytheistic slash SHOULD NOT be used. (Applied only to atheists.)
996 ## (This requirement was dropped from HTML5 spec, unfortunately.)
997
998 my $is_space = {
999 0x0009 => 1, # CHARACTER TABULATION (HT)
1000 0x000A => 1, # LINE FEED (LF)
1001 #0x000B => 0, # LINE TABULATION (VT)
1002 0x000C => 1, # FORM FEED (FF)
1003 #0x000D => 1, # CARRIAGE RETURN (CR)
1004 0x0020 => 1, # SPACE (SP)
1005 };
1006
1007 sub _get_next_token ($) {
1008 my $self = shift;
1009
1010 if ($self->{self_closing}) {
1011 !!!parse-error (type => 'nestc', token => $self->{ct});
1012 ## NOTE: The |self_closing| flag is only set by start tag token.
1013 ## In addition, when a start tag token is emitted, it is always set to
1014 ## |ct|.
1015 delete $self->{self_closing};
1016 }
1017
1018 if (@{$self->{token}}) {
1019 $self->{self_closing} = $self->{token}->[0]->{self_closing};
1020 return shift @{$self->{token}};
1021 }
1022
1023 A: {
1024 if ($self->{state} == PCDATA_STATE) {
1025 ## NOTE: Same as |DATA_STATE|, but only for |PCDATA| content model.
1026
1027 if ($self->{nc} == 0x0026) { # &
1028 !!!cp (0.1);
1029 ## NOTE: In the spec, the tokenizer is switched to the
1030 ## "entity data state". In this implementation, the tokenizer
1031 ## is switched to the |ENTITY_STATE|, which is an implementation
1032 ## of the "consume a character reference" algorithm.
1033 $self->{entity_add} = -1;
1034 $self->{prev_state} = DATA_STATE;
1035 $self->{state} = ENTITY_STATE;
1036 !!!next-input-character;
1037 redo A;
1038 } elsif ($self->{nc} == 0x003C) { # <
1039 !!!cp (0.2);
1040 $self->{state} = TAG_OPEN_STATE;
1041 !!!next-input-character;
1042 redo A;
1043 } elsif ($self->{nc} == -1) {
1044 !!!cp (0.3);
1045 !!!emit ({type => END_OF_FILE_TOKEN,
1046 line => $self->{line}, column => $self->{column}});
1047 last A; ## TODO: ok?
1048 } else {
1049 !!!cp (0.4);
1050 #
1051 }
1052
1053 # Anything else
1054 my $token = {type => CHARACTER_TOKEN,
1055 data => chr $self->{nc},
1056 line => $self->{line}, column => $self->{column},
1057 };
1058 $self->{read_until}->($token->{data}, q[<&], length $token->{data});
1059
1060 ## Stay in the state.
1061 !!!next-input-character;
1062 !!!emit ($token);
1063 redo A;
1064 } elsif ($self->{state} == DATA_STATE) {
1065 $self->{s_kwd} = '' unless defined $self->{s_kwd};
1066 if ($self->{nc} == 0x0026) { # &
1067 $self->{s_kwd} = '';
1068 if ($self->{content_model} & CM_ENTITY and # PCDATA | RCDATA
1069 not $self->{escape}) {
1070 !!!cp (1);
1071 ## NOTE: In the spec, the tokenizer is switched to the
1072 ## "entity data state". In this implementation, the tokenizer
1073 ## is switched to the |ENTITY_STATE|, which is an implementation
1074 ## of the "consume a character reference" algorithm.
1075 $self->{entity_add} = -1;
1076 $self->{prev_state} = DATA_STATE;
1077 $self->{state} = ENTITY_STATE;
1078 !!!next-input-character;
1079 redo A;
1080 } else {
1081 !!!cp (2);
1082 #
1083 }
1084 } elsif ($self->{nc} == 0x002D) { # -
1085 if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
1086 $self->{s_kwd} .= '-';
1087
1088 if ($self->{s_kwd} eq '<!--') {
1089 !!!cp (3);
1090 $self->{escape} = 1; # unless $self->{escape};
1091 $self->{s_kwd} = '--';
1092 #
1093 } elsif ($self->{s_kwd} eq '---') {
1094 !!!cp (4);
1095 $self->{s_kwd} = '--';
1096 #
1097 } else {
1098 !!!cp (5);
1099 #
1100 }
1101 }
1102
1103 #
1104 } elsif ($self->{nc} == 0x0021) { # !
1105 if (length $self->{s_kwd}) {
1106 !!!cp (5.1);
1107 $self->{s_kwd} .= '!';
1108 #
1109 } else {
1110 !!!cp (5.2);
1111 #$self->{s_kwd} = '';
1112 #
1113 }
1114 #
1115 } elsif ($self->{nc} == 0x003C) { # <
1116 if ($self->{content_model} & CM_FULL_MARKUP or # PCDATA
1117 (($self->{content_model} & CM_LIMITED_MARKUP) and # CDATA | RCDATA
1118 not $self->{escape})) {
1119 !!!cp (6);
1120 $self->{state} = TAG_OPEN_STATE;
1121 !!!next-input-character;
1122 redo A;
1123 } else {
1124 !!!cp (7);
1125 $self->{s_kwd} = '';
1126 #
1127 }
1128 } elsif ($self->{nc} == 0x003E) { # >
1129 if ($self->{escape} and
1130 ($self->{content_model} & CM_LIMITED_MARKUP)) { # RCDATA | CDATA
1131 if ($self->{s_kwd} eq '--') {
1132 !!!cp (8);
1133 delete $self->{escape};
1134 } else {
1135 !!!cp (9);
1136 }
1137 } else {
1138 !!!cp (10);
1139 }
1140
1141 $self->{s_kwd} = '';
1142 #
1143 } elsif ($self->{nc} == -1) {
1144 !!!cp (11);
1145 $self->{s_kwd} = '';
1146 !!!emit ({type => END_OF_FILE_TOKEN,
1147 line => $self->{line}, column => $self->{column}});
1148 last A; ## TODO: ok?
1149 } else {
1150 !!!cp (12);
1151 $self->{s_kwd} = '';
1152 #
1153 }
1154
1155 # Anything else
1156 my $token = {type => CHARACTER_TOKEN,
1157 data => chr $self->{nc},
1158 line => $self->{line}, column => $self->{column},
1159 };
1160 if ($self->{read_until}->($token->{data}, q[-!<>&],
1161 length $token->{data})) {
1162 $self->{s_kwd} = '';
1163 }
1164
1165 ## Stay in the data state.
1166 if ($self->{content_model} == PCDATA_CONTENT_MODEL) {
1167 !!!cp (13);
1168 $self->{state} = PCDATA_STATE;
1169 } else {
1170 !!!cp (14);
1171 ## Stay in the state.
1172 }
1173 !!!next-input-character;
1174 !!!emit ($token);
1175 redo A;
1176 } elsif ($self->{state} == TAG_OPEN_STATE) {
1177 if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
1178 if ($self->{nc} == 0x002F) { # /
1179 !!!cp (15);
1180 !!!next-input-character;
1181 $self->{state} = CLOSE_TAG_OPEN_STATE;
1182 redo A;
1183 } elsif ($self->{nc} == 0x0021) { # !
1184 !!!cp (15.1);
1185 $self->{s_kwd} = '<' unless $self->{escape};
1186 #
1187 } else {
1188 !!!cp (16);
1189 #
1190 }
1191
1192 ## reconsume
1193 $self->{state} = DATA_STATE;
1194 !!!emit ({type => CHARACTER_TOKEN, data => '<',
1195 line => $self->{line_prev},
1196 column => $self->{column_prev},
1197 });
1198 redo A;
1199 } elsif ($self->{content_model} & CM_FULL_MARKUP) { # PCDATA
1200 if ($self->{nc} == 0x0021) { # !
1201 !!!cp (17);
1202 $self->{state} = MARKUP_DECLARATION_OPEN_STATE;
1203 !!!next-input-character;
1204 redo A;
1205 } elsif ($self->{nc} == 0x002F) { # /
1206 !!!cp (18);
1207 $self->{state} = CLOSE_TAG_OPEN_STATE;
1208 !!!next-input-character;
1209 redo A;
1210 } elsif (0x0041 <= $self->{nc} and
1211 $self->{nc} <= 0x005A) { # A..Z
1212 !!!cp (19);
1213 $self->{ct}
1214 = {type => START_TAG_TOKEN,
1215 tag_name => chr ($self->{nc} + 0x0020),
1216 line => $self->{line_prev},
1217 column => $self->{column_prev}};
1218 $self->{state} = TAG_NAME_STATE;
1219 !!!next-input-character;
1220 redo A;
1221 } elsif (0x0061 <= $self->{nc} and
1222 $self->{nc} <= 0x007A) { # a..z
1223 !!!cp (20);
1224 $self->{ct} = {type => START_TAG_TOKEN,
1225 tag_name => chr ($self->{nc}),
1226 line => $self->{line_prev},
1227 column => $self->{column_prev}};
1228 $self->{state} = TAG_NAME_STATE;
1229 !!!next-input-character;
1230 redo A;
1231 } elsif ($self->{nc} == 0x003E) { # >
1232 !!!cp (21);
1233 !!!parse-error (type => 'empty start tag',
1234 line => $self->{line_prev},
1235 column => $self->{column_prev});
1236 $self->{state} = DATA_STATE;
1237 !!!next-input-character;
1238
1239 !!!emit ({type => CHARACTER_TOKEN, data => '<>',
1240 line => $self->{line_prev},
1241 column => $self->{column_prev},
1242 });
1243
1244 redo A;
1245 } elsif ($self->{nc} == 0x003F) { # ?
1246 !!!cp (22);
1247 !!!parse-error (type => 'pio',
1248 line => $self->{line_prev},
1249 column => $self->{column_prev});
1250 $self->{state} = BOGUS_COMMENT_STATE;
1251 $self->{ct} = {type => COMMENT_TOKEN, data => '',
1252 line => $self->{line_prev},
1253 column => $self->{column_prev},
1254 };
1255 ## $self->{nc} is intentionally left as is
1256 redo A;
1257 } else {
1258 !!!cp (23);
1259 !!!parse-error (type => 'bare stago',
1260 line => $self->{line_prev},
1261 column => $self->{column_prev});
1262 $self->{state} = DATA_STATE;
1263 ## reconsume
1264
1265 !!!emit ({type => CHARACTER_TOKEN, data => '<',
1266 line => $self->{line_prev},
1267 column => $self->{column_prev},
1268 });
1269
1270 redo A;
1271 }
1272 } else {
1273 die "$0: $self->{content_model} in tag open";
1274 }
1275 } elsif ($self->{state} == CLOSE_TAG_OPEN_STATE) {
1276 ## NOTE: The "close tag open state" in the spec is implemented as
1277 ## |CLOSE_TAG_OPEN_STATE| and |CDATA_RCDATA_CLOSE_TAG_STATE|.
1278
1279 my ($l, $c) = ($self->{line_prev}, $self->{column_prev} - 1); # "<"of"</"
1280 if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
1281 if (defined $self->{last_stag_name}) {
1282 $self->{state} = CDATA_RCDATA_CLOSE_TAG_STATE;
1283 $self->{s_kwd} = '';
1284 ## Reconsume.
1285 redo A;
1286 } else {
1287 ## No start tag token has ever been emitted
1288 ## NOTE: See <http://krijnhoetmer.nl/irc-logs/whatwg/20070626#l-564>.
1289 !!!cp (28);
1290 $self->{state} = DATA_STATE;
1291 ## Reconsume.
1292 !!!emit ({type => CHARACTER_TOKEN, data => '</',
1293 line => $l, column => $c,
1294 });
1295 redo A;
1296 }
1297 }
1298
1299 if (0x0041 <= $self->{nc} and
1300 $self->{nc} <= 0x005A) { # A..Z
1301 !!!cp (29);
1302 $self->{ct}
1303 = {type => END_TAG_TOKEN,
1304 tag_name => chr ($self->{nc} + 0x0020),
1305 line => $l, column => $c};
1306 $self->{state} = TAG_NAME_STATE;
1307 !!!next-input-character;
1308 redo A;
1309 } elsif (0x0061 <= $self->{nc} and
1310 $self->{nc} <= 0x007A) { # a..z
1311 !!!cp (30);
1312 $self->{ct} = {type => END_TAG_TOKEN,
1313 tag_name => chr ($self->{nc}),
1314 line => $l, column => $c};
1315 $self->{state} = TAG_NAME_STATE;
1316 !!!next-input-character;
1317 redo A;
1318 } elsif ($self->{nc} == 0x003E) { # >
1319 !!!cp (31);
1320 !!!parse-error (type => 'empty end tag',
1321 line => $self->{line_prev}, ## "<" in "</>"
1322 column => $self->{column_prev} - 1);
1323 $self->{state} = DATA_STATE;
1324 !!!next-input-character;
1325 redo A;
1326 } elsif ($self->{nc} == -1) {
1327 !!!cp (32);
1328 !!!parse-error (type => 'bare etago');
1329 $self->{state} = DATA_STATE;
1330 # reconsume
1331
1332 !!!emit ({type => CHARACTER_TOKEN, data => '</',
1333 line => $l, column => $c,
1334 });
1335
1336 redo A;
1337 } else {
1338 !!!cp (33);
1339 !!!parse-error (type => 'bogus end tag');
1340 $self->{state} = BOGUS_COMMENT_STATE;
1341 $self->{ct} = {type => COMMENT_TOKEN, data => '',
1342 line => $self->{line_prev}, # "<" of "</"
1343 column => $self->{column_prev} - 1,
1344 };
1345 ## NOTE: $self->{nc} is intentionally left as is.
1346 ## Although the "anything else" case of the spec not explicitly
1347 ## states that the next input character is to be reconsumed,
1348 ## it will be included to the |data| of the comment token
1349 ## generated from the bogus end tag, as defined in the
1350 ## "bogus comment state" entry.
1351 redo A;
1352 }
1353 } elsif ($self->{state} == CDATA_RCDATA_CLOSE_TAG_STATE) {
1354 my $ch = substr $self->{last_stag_name}, length $self->{s_kwd}, 1;
1355 if (length $ch) {
1356 my $CH = $ch;
1357 $ch =~ tr/a-z/A-Z/;
1358 my $nch = chr $self->{nc};
1359 if ($nch eq $ch or $nch eq $CH) {
1360 !!!cp (24);
1361 ## Stay in the state.
1362 $self->{s_kwd} .= $nch;
1363 !!!next-input-character;
1364 redo A;
1365 } else {
1366 !!!cp (25);
1367 $self->{state} = DATA_STATE;
1368 ## Reconsume.
1369 !!!emit ({type => CHARACTER_TOKEN,
1370 data => '</' . $self->{s_kwd},
1371 line => $self->{line_prev},
1372 column => $self->{column_prev} - 1 - length $self->{s_kwd},
1373 });
1374 redo A;
1375 }
1376 } else { # after "<{tag-name}"
1377 unless ($is_space->{$self->{nc}} or
1378 {
1379 0x003E => 1, # >
1380 0x002F => 1, # /
1381 -1 => 1, # EOF
1382 }->{$self->{nc}}) {
1383 !!!cp (26);
1384 ## Reconsume.
1385 $self->{state} = DATA_STATE;
1386 !!!emit ({type => CHARACTER_TOKEN,
1387 data => '</' . $self->{s_kwd},
1388 line => $self->{line_prev},
1389 column => $self->{column_prev} - 1 - length $self->{s_kwd},
1390 });
1391 redo A;
1392 } else {
1393 !!!cp (27);
1394 $self->{ct}
1395 = {type => END_TAG_TOKEN,
1396 tag_name => $self->{last_stag_name},
1397 line => $self->{line_prev},
1398 column => $self->{column_prev} - 1 - length $self->{s_kwd}};
1399 $self->{state} = TAG_NAME_STATE;
1400 ## Reconsume.
1401 redo A;
1402 }
1403 }
1404 } elsif ($self->{state} == TAG_NAME_STATE) {
1405 if ($is_space->{$self->{nc}}) {
1406 !!!cp (34);
1407 $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1408 !!!next-input-character;
1409 redo A;
1410 } elsif ($self->{nc} == 0x003E) { # >
1411 if ($self->{ct}->{type} == START_TAG_TOKEN) {
1412 !!!cp (35);
1413 $self->{last_stag_name} = $self->{ct}->{tag_name};
1414 } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1415 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1416 #if ($self->{ct}->{attributes}) {
1417 # ## NOTE: This should never be reached.
1418 # !!! cp (36);
1419 # !!! parse-error (type => 'end tag attribute');
1420 #} else {
1421 !!!cp (37);
1422 #}
1423 } else {
1424 die "$0: $self->{ct}->{type}: Unknown token type";
1425 }
1426 $self->{state} = DATA_STATE;
1427 !!!next-input-character;
1428
1429 !!!emit ($self->{ct}); # start tag or end tag
1430
1431 redo A;
1432 } elsif (0x0041 <= $self->{nc} and
1433 $self->{nc} <= 0x005A) { # A..Z
1434 !!!cp (38);
1435 $self->{ct}->{tag_name} .= chr ($self->{nc} + 0x0020);
1436 # start tag or end tag
1437 ## Stay in this state
1438 !!!next-input-character;
1439 redo A;
1440 } elsif ($self->{nc} == -1) {
1441 !!!parse-error (type => 'unclosed tag');
1442 if ($self->{ct}->{type} == START_TAG_TOKEN) {
1443 !!!cp (39);
1444 $self->{last_stag_name} = $self->{ct}->{tag_name};
1445 } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1446 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1447 #if ($self->{ct}->{attributes}) {
1448 # ## NOTE: This state should never be reached.
1449 # !!! cp (40);
1450 # !!! parse-error (type => 'end tag attribute');
1451 #} else {
1452 !!!cp (41);
1453 #}
1454 } else {
1455 die "$0: $self->{ct}->{type}: Unknown token type";
1456 }
1457 $self->{state} = DATA_STATE;
1458 # reconsume
1459
1460 !!!emit ($self->{ct}); # start tag or end tag
1461
1462 redo A;
1463 } elsif ($self->{nc} == 0x002F) { # /
1464 !!!cp (42);
1465 $self->{state} = SELF_CLOSING_START_TAG_STATE;
1466 !!!next-input-character;
1467 redo A;
1468 } else {
1469 !!!cp (44);
1470 $self->{ct}->{tag_name} .= chr $self->{nc};
1471 # start tag or end tag
1472 ## Stay in the state
1473 !!!next-input-character;
1474 redo A;
1475 }
1476 } elsif ($self->{state} == BEFORE_ATTRIBUTE_NAME_STATE) {
1477 if ($is_space->{$self->{nc}}) {
1478 !!!cp (45);
1479 ## Stay in the state
1480 !!!next-input-character;
1481 redo A;
1482 } elsif ($self->{nc} == 0x003E) { # >
1483 if ($self->{ct}->{type} == START_TAG_TOKEN) {
1484 !!!cp (46);
1485 $self->{last_stag_name} = $self->{ct}->{tag_name};
1486 } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1487 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1488 if ($self->{ct}->{attributes}) {
1489 !!!cp (47);
1490 !!!parse-error (type => 'end tag attribute');
1491 } else {
1492 !!!cp (48);
1493 }
1494 } else {
1495 die "$0: $self->{ct}->{type}: Unknown token type";
1496 }
1497 $self->{state} = DATA_STATE;
1498 !!!next-input-character;
1499
1500 !!!emit ($self->{ct}); # start tag or end tag
1501
1502 redo A;
1503 } elsif (0x0041 <= $self->{nc} and
1504 $self->{nc} <= 0x005A) { # A..Z
1505 !!!cp (49);
1506 $self->{ca}
1507 = {name => chr ($self->{nc} + 0x0020),
1508 value => '',
1509 line => $self->{line}, column => $self->{column}};
1510 $self->{state} = ATTRIBUTE_NAME_STATE;
1511 !!!next-input-character;
1512 redo A;
1513 } elsif ($self->{nc} == 0x002F) { # /
1514 !!!cp (50);
1515 $self->{state} = SELF_CLOSING_START_TAG_STATE;
1516 !!!next-input-character;
1517 redo A;
1518 } elsif ($self->{nc} == -1) {
1519 !!!parse-error (type => 'unclosed tag');
1520 if ($self->{ct}->{type} == START_TAG_TOKEN) {
1521 !!!cp (52);
1522 $self->{last_stag_name} = $self->{ct}->{tag_name};
1523 } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1524 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1525 if ($self->{ct}->{attributes}) {
1526 !!!cp (53);
1527 !!!parse-error (type => 'end tag attribute');
1528 } else {
1529 !!!cp (54);
1530 }
1531 } else {
1532 die "$0: $self->{ct}->{type}: Unknown token type";
1533 }
1534 $self->{state} = DATA_STATE;
1535 # reconsume
1536
1537 !!!emit ($self->{ct}); # start tag or end tag
1538
1539 redo A;
1540 } else {
1541 if ({
1542 0x0022 => 1, # "
1543 0x0027 => 1, # '
1544 0x003D => 1, # =
1545 }->{$self->{nc}}) {
1546 !!!cp (55);
1547 !!!parse-error (type => 'bad attribute name');
1548 } else {
1549 !!!cp (56);
1550 }
1551 $self->{ca}
1552 = {name => chr ($self->{nc}),
1553 value => '',
1554 line => $self->{line}, column => $self->{column}};
1555 $self->{state} = ATTRIBUTE_NAME_STATE;
1556 !!!next-input-character;
1557 redo A;
1558 }
1559 } elsif ($self->{state} == ATTRIBUTE_NAME_STATE) {
1560 my $before_leave = sub {
1561 if (exists $self->{ct}->{attributes} # start tag or end tag
1562 ->{$self->{ca}->{name}}) { # MUST
1563 !!!cp (57);
1564 !!!parse-error (type => 'duplicate attribute', text => $self->{ca}->{name}, line => $self->{ca}->{line}, column => $self->{ca}->{column});
1565 ## Discard $self->{ca} # MUST
1566 } else {
1567 !!!cp (58);
1568 $self->{ct}->{attributes}->{$self->{ca}->{name}}
1569 = $self->{ca};
1570 }
1571 }; # $before_leave
1572
1573 if ($is_space->{$self->{nc}}) {
1574 !!!cp (59);
1575 $before_leave->();
1576 $self->{state} = AFTER_ATTRIBUTE_NAME_STATE;
1577 !!!next-input-character;
1578 redo A;
1579 } elsif ($self->{nc} == 0x003D) { # =
1580 !!!cp (60);
1581 $before_leave->();
1582 $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;
1583 !!!next-input-character;
1584 redo A;
1585 } elsif ($self->{nc} == 0x003E) { # >
1586 $before_leave->();
1587 if ($self->{ct}->{type} == START_TAG_TOKEN) {
1588 !!!cp (61);
1589 $self->{last_stag_name} = $self->{ct}->{tag_name};
1590 } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1591 !!!cp (62);
1592 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1593 if ($self->{ct}->{attributes}) {
1594 !!!parse-error (type => 'end tag attribute');
1595 }
1596 } else {
1597 die "$0: $self->{ct}->{type}: Unknown token type";
1598 }
1599 $self->{state} = DATA_STATE;
1600 !!!next-input-character;
1601
1602 !!!emit ($self->{ct}); # start tag or end tag
1603
1604 redo A;
1605 } elsif (0x0041 <= $self->{nc} and
1606 $self->{nc} <= 0x005A) { # A..Z
1607 !!!cp (63);
1608 $self->{ca}->{name} .= chr ($self->{nc} + 0x0020);
1609 ## Stay in the state
1610 !!!next-input-character;
1611 redo A;
1612 } elsif ($self->{nc} == 0x002F) { # /
1613 !!!cp (64);
1614 $before_leave->();
1615 $self->{state} = SELF_CLOSING_START_TAG_STATE;
1616 !!!next-input-character;
1617 redo A;
1618 } elsif ($self->{nc} == -1) {
1619 !!!parse-error (type => 'unclosed tag');
1620 $before_leave->();
1621 if ($self->{ct}->{type} == START_TAG_TOKEN) {
1622 !!!cp (66);
1623 $self->{last_stag_name} = $self->{ct}->{tag_name};
1624 } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1625 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1626 if ($self->{ct}->{attributes}) {
1627 !!!cp (67);
1628 !!!parse-error (type => 'end tag attribute');
1629 } else {
1630 ## NOTE: This state should never be reached.
1631 !!!cp (68);
1632 }
1633 } else {
1634 die "$0: $self->{ct}->{type}: Unknown token type";
1635 }
1636 $self->{state} = DATA_STATE;
1637 # reconsume
1638
1639 !!!emit ($self->{ct}); # start tag or end tag
1640
1641 redo A;
1642 } else {
1643 if ($self->{nc} == 0x0022 or # "
1644 $self->{nc} == 0x0027) { # '
1645 !!!cp (69);
1646 !!!parse-error (type => 'bad attribute name');
1647 } else {
1648 !!!cp (70);
1649 }
1650 $self->{ca}->{name} .= chr ($self->{nc});
1651 ## Stay in the state
1652 !!!next-input-character;
1653 redo A;
1654 }
1655 } elsif ($self->{state} == AFTER_ATTRIBUTE_NAME_STATE) {
1656 if ($is_space->{$self->{nc}}) {
1657 !!!cp (71);
1658 ## Stay in the state
1659 !!!next-input-character;
1660 redo A;
1661 } elsif ($self->{nc} == 0x003D) { # =
1662 !!!cp (72);
1663 $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;
1664 !!!next-input-character;
1665 redo A;
1666 } elsif ($self->{nc} == 0x003E) { # >
1667 if ($self->{ct}->{type} == START_TAG_TOKEN) {
1668 !!!cp (73);
1669 $self->{last_stag_name} = $self->{ct}->{tag_name};
1670 } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1671 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1672 if ($self->{ct}->{attributes}) {
1673 !!!cp (74);
1674 !!!parse-error (type => 'end tag attribute');
1675 } else {
1676 ## NOTE: This state should never be reached.
1677 !!!cp (75);
1678 }
1679 } else {
1680 die "$0: $self->{ct}->{type}: Unknown token type";
1681 }
1682 $self->{state} = DATA_STATE;
1683 !!!next-input-character;
1684
1685 !!!emit ($self->{ct}); # start tag or end tag
1686
1687 redo A;
1688 } elsif (0x0041 <= $self->{nc} and
1689 $self->{nc} <= 0x005A) { # A..Z
1690 !!!cp (76);
1691 $self->{ca}
1692 = {name => chr ($self->{nc} + 0x0020),
1693 value => '',
1694 line => $self->{line}, column => $self->{column}};
1695 $self->{state} = ATTRIBUTE_NAME_STATE;
1696 !!!next-input-character;
1697 redo A;
1698 } elsif ($self->{nc} == 0x002F) { # /
1699 !!!cp (77);
1700 $self->{state} = SELF_CLOSING_START_TAG_STATE;
1701 !!!next-input-character;
1702 redo A;
1703 } elsif ($self->{nc} == -1) {
1704 !!!parse-error (type => 'unclosed tag');
1705 if ($self->{ct}->{type} == START_TAG_TOKEN) {
1706 !!!cp (79);
1707 $self->{last_stag_name} = $self->{ct}->{tag_name};
1708 } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1709 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1710 if ($self->{ct}->{attributes}) {
1711 !!!cp (80);
1712 !!!parse-error (type => 'end tag attribute');
1713 } else {
1714 ## NOTE: This state should never be reached.
1715 !!!cp (81);
1716 }
1717 } else {
1718 die "$0: $self->{ct}->{type}: Unknown token type";
1719 }
1720 $self->{state} = DATA_STATE;
1721 # reconsume
1722
1723 !!!emit ($self->{ct}); # start tag or end tag
1724
1725 redo A;
1726 } else {
1727 if ($self->{nc} == 0x0022 or # "
1728 $self->{nc} == 0x0027) { # '
1729 !!!cp (78);
1730 !!!parse-error (type => 'bad attribute name');
1731 } else {
1732 !!!cp (82);
1733 }
1734 $self->{ca}
1735 = {name => chr ($self->{nc}),
1736 value => '',
1737 line => $self->{line}, column => $self->{column}};
1738 $self->{state} = ATTRIBUTE_NAME_STATE;
1739 !!!next-input-character;
1740 redo A;
1741 }
1742 } elsif ($self->{state} == BEFORE_ATTRIBUTE_VALUE_STATE) {
1743 if ($is_space->{$self->{nc}}) {
1744 !!!cp (83);
1745 ## Stay in the state
1746 !!!next-input-character;
1747 redo A;
1748 } elsif ($self->{nc} == 0x0022) { # "
1749 !!!cp (84);
1750 $self->{state} = ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE;
1751 !!!next-input-character;
1752 redo A;
1753 } elsif ($self->{nc} == 0x0026) { # &
1754 !!!cp (85);
1755 $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;
1756 ## reconsume
1757 redo A;
1758 } elsif ($self->{nc} == 0x0027) { # '
1759 !!!cp (86);
1760 $self->{state} = ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE;
1761 !!!next-input-character;
1762 redo A;
1763 } elsif ($self->{nc} == 0x003E) { # >
1764 !!!parse-error (type => 'empty unquoted attribute value');
1765 if ($self->{ct}->{type} == START_TAG_TOKEN) {
1766 !!!cp (87);
1767 $self->{last_stag_name} = $self->{ct}->{tag_name};
1768 } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1769 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1770 if ($self->{ct}->{attributes}) {
1771 !!!cp (88);
1772 !!!parse-error (type => 'end tag attribute');
1773 } else {
1774 ## NOTE: This state should never be reached.
1775 !!!cp (89);
1776 }
1777 } else {
1778 die "$0: $self->{ct}->{type}: Unknown token type";
1779 }
1780 $self->{state} = DATA_STATE;
1781 !!!next-input-character;
1782
1783 !!!emit ($self->{ct}); # start tag or end tag
1784
1785 redo A;
1786 } elsif ($self->{nc} == -1) {
1787 !!!parse-error (type => 'unclosed tag');
1788 if ($self->{ct}->{type} == START_TAG_TOKEN) {
1789 !!!cp (90);
1790 $self->{last_stag_name} = $self->{ct}->{tag_name};
1791 } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1792 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1793 if ($self->{ct}->{attributes}) {
1794 !!!cp (91);
1795 !!!parse-error (type => 'end tag attribute');
1796 } else {
1797 ## NOTE: This state should never be reached.
1798 !!!cp (92);
1799 }
1800 } else {
1801 die "$0: $self->{ct}->{type}: Unknown token type";
1802 }
1803 $self->{state} = DATA_STATE;
1804 ## reconsume
1805
1806 !!!emit ($self->{ct}); # start tag or end tag
1807
1808 redo A;
1809 } else {
1810 if ($self->{nc} == 0x003D) { # =
1811 !!!cp (93);
1812 !!!parse-error (type => 'bad attribute value');
1813 } else {
1814 !!!cp (94);
1815 }
1816 $self->{ca}->{value} .= chr ($self->{nc});
1817 $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;
1818 !!!next-input-character;
1819 redo A;
1820 }
1821 } elsif ($self->{state} == ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE) {
1822 if ($self->{nc} == 0x0022) { # "
1823 !!!cp (95);
1824 $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;
1825 !!!next-input-character;
1826 redo A;
1827 } elsif ($self->{nc} == 0x0026) { # &
1828 !!!cp (96);
1829 ## NOTE: In the spec, the tokenizer is switched to the
1830 ## "entity in attribute value state". In this implementation, the
1831 ## tokenizer is switched to the |ENTITY_STATE|, which is an
1832 ## implementation of the "consume a character reference" algorithm.
1833 $self->{prev_state} = $self->{state};
1834 $self->{entity_add} = 0x0022; # "
1835 $self->{state} = ENTITY_STATE;
1836 !!!next-input-character;
1837 redo A;
1838 } elsif ($self->{nc} == -1) {
1839 !!!parse-error (type => 'unclosed attribute value');
1840 if ($self->{ct}->{type} == START_TAG_TOKEN) {
1841 !!!cp (97);
1842 $self->{last_stag_name} = $self->{ct}->{tag_name};
1843 } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1844 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1845 if ($self->{ct}->{attributes}) {
1846 !!!cp (98);
1847 !!!parse-error (type => 'end tag attribute');
1848 } else {
1849 ## NOTE: This state should never be reached.
1850 !!!cp (99);
1851 }
1852 } else {
1853 die "$0: $self->{ct}->{type}: Unknown token type";
1854 }
1855 $self->{state} = DATA_STATE;
1856 ## reconsume
1857
1858 !!!emit ($self->{ct}); # start tag or end tag
1859
1860 redo A;
1861 } else {
1862 !!!cp (100);
1863 $self->{ca}->{value} .= chr ($self->{nc});
1864 $self->{read_until}->($self->{ca}->{value},
1865 q["&],
1866 length $self->{ca}->{value});
1867
1868 ## Stay in the state
1869 !!!next-input-character;
1870 redo A;
1871 }
1872 } elsif ($self->{state} == ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE) {
1873 if ($self->{nc} == 0x0027) { # '
1874 !!!cp (101);
1875 $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;
1876 !!!next-input-character;
1877 redo A;
1878 } elsif ($self->{nc} == 0x0026) { # &
1879 !!!cp (102);
1880 ## NOTE: In the spec, the tokenizer is switched to the
1881 ## "entity in attribute value state". In this implementation, the
1882 ## tokenizer is switched to the |ENTITY_STATE|, which is an
1883 ## implementation of the "consume a character reference" algorithm.
1884 $self->{entity_add} = 0x0027; # '
1885 $self->{prev_state} = $self->{state};
1886 $self->{state} = ENTITY_STATE;
1887 !!!next-input-character;
1888 redo A;
1889 } elsif ($self->{nc} == -1) {
1890 !!!parse-error (type => 'unclosed attribute value');
1891 if ($self->{ct}->{type} == START_TAG_TOKEN) {
1892 !!!cp (103);
1893 $self->{last_stag_name} = $self->{ct}->{tag_name};
1894 } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1895 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1896 if ($self->{ct}->{attributes}) {
1897 !!!cp (104);
1898 !!!parse-error (type => 'end tag attribute');
1899 } else {
1900 ## NOTE: This state should never be reached.
1901 !!!cp (105);
1902 }
1903 } else {
1904 die "$0: $self->{ct}->{type}: Unknown token type";
1905 }
1906 $self->{state} = DATA_STATE;
1907 ## reconsume
1908
1909 !!!emit ($self->{ct}); # start tag or end tag
1910
1911 redo A;
1912 } else {
1913 !!!cp (106);
1914 $self->{ca}->{value} .= chr ($self->{nc});
1915 $self->{read_until}->($self->{ca}->{value},
1916 q['&],
1917 length $self->{ca}->{value});
1918
1919 ## Stay in the state
1920 !!!next-input-character;
1921 redo A;
1922 }
1923 } elsif ($self->{state} == ATTRIBUTE_VALUE_UNQUOTED_STATE) {
1924 if ($is_space->{$self->{nc}}) {
1925 !!!cp (107);
1926 $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1927 !!!next-input-character;
1928 redo A;
1929 } elsif ($self->{nc} == 0x0026) { # &
1930 !!!cp (108);
1931 ## NOTE: In the spec, the tokenizer is switched to the
1932 ## "entity in attribute value state". In this implementation, the
1933 ## tokenizer is switched to the |ENTITY_STATE|, which is an
1934 ## implementation of the "consume a character reference" algorithm.
1935 $self->{entity_add} = -1;
1936 $self->{prev_state} = $self->{state};
1937 $self->{state} = ENTITY_STATE;
1938 !!!next-input-character;
1939 redo A;
1940 } elsif ($self->{nc} == 0x003E) { # >
1941 if ($self->{ct}->{type} == START_TAG_TOKEN) {
1942 !!!cp (109);
1943 $self->{last_stag_name} = $self->{ct}->{tag_name};
1944 } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1945 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1946 if ($self->{ct}->{attributes}) {
1947 !!!cp (110);
1948 !!!parse-error (type => 'end tag attribute');
1949 } else {
1950 ## NOTE: This state should never be reached.
1951 !!!cp (111);
1952 }
1953 } else {
1954 die "$0: $self->{ct}->{type}: Unknown token type";
1955 }
1956 $self->{state} = DATA_STATE;
1957 !!!next-input-character;
1958
1959 !!!emit ($self->{ct}); # start tag or end tag
1960
1961 redo A;
1962 } elsif ($self->{nc} == -1) {
1963 !!!parse-error (type => 'unclosed tag');
1964 if ($self->{ct}->{type} == START_TAG_TOKEN) {
1965 !!!cp (112);
1966 $self->{last_stag_name} = $self->{ct}->{tag_name};
1967 } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1968 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1969 if ($self->{ct}->{attributes}) {
1970 !!!cp (113);
1971 !!!parse-error (type => 'end tag attribute');
1972 } else {
1973 ## NOTE: This state should never be reached.
1974 !!!cp (114);
1975 }
1976 } else {
1977 die "$0: $self->{ct}->{type}: Unknown token type";
1978 }
1979 $self->{state} = DATA_STATE;
1980 ## reconsume
1981
1982 !!!emit ($self->{ct}); # start tag or end tag
1983
1984 redo A;
1985 } else {
1986 if ({
1987 0x0022 => 1, # "
1988 0x0027 => 1, # '
1989 0x003D => 1, # =
1990 }->{$self->{nc}}) {
1991 !!!cp (115);
1992 !!!parse-error (type => 'bad attribute value');
1993 } else {
1994 !!!cp (116);
1995 }
1996 $self->{ca}->{value} .= chr ($self->{nc});
1997 $self->{read_until}->($self->{ca}->{value},
1998 q["'=& >],
1999 length $self->{ca}->{value});
2000
2001 ## Stay in the state
2002 !!!next-input-character;
2003 redo A;
2004 }
2005 } elsif ($self->{state} == AFTER_ATTRIBUTE_VALUE_QUOTED_STATE) {
2006 if ($is_space->{$self->{nc}}) {
2007 !!!cp (118);
2008 $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
2009 !!!next-input-character;
2010 redo A;
2011 } elsif ($self->{nc} == 0x003E) { # >
2012 if ($self->{ct}->{type} == START_TAG_TOKEN) {
2013 !!!cp (119);
2014 $self->{last_stag_name} = $self->{ct}->{tag_name};
2015 } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
2016 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
2017 if ($self->{ct}->{attributes}) {
2018 !!!cp (120);
2019 !!!parse-error (type => 'end tag attribute');
2020 } else {
2021 ## NOTE: This state should never be reached.
2022 !!!cp (121);
2023 }
2024 } else {
2025 die "$0: $self->{ct}->{type}: Unknown token type";
2026 }
2027 $self->{state} = DATA_STATE;
2028 !!!next-input-character;
2029
2030 !!!emit ($self->{ct}); # start tag or end tag
2031
2032 redo A;
2033 } elsif ($self->{nc} == 0x002F) { # /
2034 !!!cp (122);
2035 $self->{state} = SELF_CLOSING_START_TAG_STATE;
2036 !!!next-input-character;
2037 redo A;
2038 } elsif ($self->{nc} == -1) {
2039 !!!parse-error (type => 'unclosed tag');
2040 if ($self->{ct}->{type} == START_TAG_TOKEN) {
2041 !!!cp (122.3);
2042 $self->{last_stag_name} = $self->{ct}->{tag_name};
2043 } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
2044 if ($self->{ct}->{attributes}) {
2045 !!!cp (122.1);
2046 !!!parse-error (type => 'end tag attribute');
2047 } else {
2048 ## NOTE: This state should never be reached.
2049 !!!cp (122.2);
2050 }
2051 } else {
2052 die "$0: $self->{ct}->{type}: Unknown token type";
2053 }
2054 $self->{state} = DATA_STATE;
2055 ## Reconsume.
2056 !!!emit ($self->{ct}); # start tag or end tag
2057 redo A;
2058 } else {
2059 !!!cp ('124.1');
2060 !!!parse-error (type => 'no space between attributes');
2061 $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
2062 ## reconsume
2063 redo A;
2064 }
2065 } elsif ($self->{state} == SELF_CLOSING_START_TAG_STATE) {
2066 if ($self->{nc} == 0x003E) { # >
2067 if ($self->{ct}->{type} == END_TAG_TOKEN) {
2068 !!!cp ('124.2');
2069 !!!parse-error (type => 'nestc', token => $self->{ct});
2070 ## TODO: Different type than slash in start tag
2071 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
2072 if ($self->{ct}->{attributes}) {
2073 !!!cp ('124.4');
2074 !!!parse-error (type => 'end tag attribute');
2075 } else {
2076 !!!cp ('124.5');
2077 }
2078 ## TODO: Test |<title></title/>|
2079 } else {
2080 !!!cp ('124.3');
2081 $self->{self_closing} = 1;
2082 }
2083
2084 $self->{state} = DATA_STATE;
2085 !!!next-input-character;
2086
2087 !!!emit ($self->{ct}); # start tag or end tag
2088
2089 redo A;
2090 } elsif ($self->{nc} == -1) {
2091 !!!parse-error (type => 'unclosed tag');
2092 if ($self->{ct}->{type} == START_TAG_TOKEN) {
2093 !!!cp (124.7);
2094 $self->{last_stag_name} = $self->{ct}->{tag_name};
2095 } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
2096 if ($self->{ct}->{attributes}) {
2097 !!!cp (124.5);
2098 !!!parse-error (type => 'end tag attribute');
2099 } else {
2100 ## NOTE: This state should never be reached.
2101 !!!cp (124.6);
2102 }
2103 } else {
2104 die "$0: $self->{ct}->{type}: Unknown token type";
2105 }
2106 $self->{state} = DATA_STATE;
2107 ## Reconsume.
2108 !!!emit ($self->{ct}); # start tag or end tag
2109 redo A;
2110 } else {
2111 !!!cp ('124.4');
2112 !!!parse-error (type => 'nestc');
2113 ## TODO: This error type is wrong.
2114 $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
2115 ## Reconsume.
2116 redo A;
2117 }
2118 } elsif ($self->{state} == BOGUS_COMMENT_STATE) {
2119 ## (only happen if PCDATA state)
2120
2121 ## NOTE: Unlike spec's "bogus comment state", this implementation
2122 ## consumes characters one-by-one basis.
2123
2124 if ($self->{nc} == 0x003E) { # >
2125 !!!cp (124);
2126 $self->{state} = DATA_STATE;
2127 !!!next-input-character;
2128
2129 !!!emit ($self->{ct}); # comment
2130 redo A;
2131 } elsif ($self->{nc} == -1) {
2132 !!!cp (125);
2133 $self->{state} = DATA_STATE;
2134 ## reconsume
2135
2136 !!!emit ($self->{ct}); # comment
2137 redo A;
2138 } else {
2139 !!!cp (126);
2140 $self->{ct}->{data} .= chr ($self->{nc}); # comment
2141 $self->{read_until}->($self->{ct}->{data},
2142 q[>],
2143 length $self->{ct}->{data});
2144
2145 ## Stay in the state.
2146 !!!next-input-character;
2147 redo A;
2148 }
2149 } elsif ($self->{state} == MARKUP_DECLARATION_OPEN_STATE) {
2150 ## (only happen if PCDATA state)
2151
2152 if ($self->{nc} == 0x002D) { # -
2153 !!!cp (133);
2154 $self->{state} = MD_HYPHEN_STATE;
2155 !!!next-input-character;
2156 redo A;
2157 } elsif ($self->{nc} == 0x0044 or # D
2158 $self->{nc} == 0x0064) { # d
2159 ## ASCII case-insensitive.
2160 !!!cp (130);
2161 $self->{state} = MD_DOCTYPE_STATE;
2162 $self->{s_kwd} = chr $self->{nc};
2163 !!!next-input-character;
2164 redo A;
2165 } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM and
2166 $self->{open_elements}->[-1]->[1] & FOREIGN_EL and
2167 $self->{nc} == 0x005B) { # [
2168 !!!cp (135.4);
2169 $self->{state} = MD_CDATA_STATE;
2170 $self->{s_kwd} = '[';
2171 !!!next-input-character;
2172 redo A;
2173 } else {
2174 !!!cp (136);
2175 }
2176
2177 !!!parse-error (type => 'bogus comment',
2178 line => $self->{line_prev},
2179 column => $self->{column_prev} - 1);
2180 ## Reconsume.
2181 $self->{state} = BOGUS_COMMENT_STATE;
2182 $self->{ct} = {type => COMMENT_TOKEN, data => '',
2183 line => $self->{line_prev},
2184 column => $self->{column_prev} - 1,
2185 };
2186 redo A;
2187 } elsif ($self->{state} == MD_HYPHEN_STATE) {
2188 if ($self->{nc} == 0x002D) { # -
2189 !!!cp (127);
2190 $self->{ct} = {type => COMMENT_TOKEN, data => '',
2191 line => $self->{line_prev},
2192 column => $self->{column_prev} - 2,
2193 };
2194 $self->{state} = COMMENT_START_STATE;
2195 !!!next-input-character;
2196 redo A;
2197 } else {
2198 !!!cp (128);
2199 !!!parse-error (type => 'bogus comment',
2200 line => $self->{line_prev},
2201 column => $self->{column_prev} - 2);
2202 $self->{state} = BOGUS_COMMENT_STATE;
2203 ## Reconsume.
2204 $self->{ct} = {type => COMMENT_TOKEN,
2205 data => '-',
2206 line => $self->{line_prev},
2207 column => $self->{column_prev} - 2,
2208 };
2209 redo A;
2210 }
2211 } elsif ($self->{state} == MD_DOCTYPE_STATE) {
2212 ## ASCII case-insensitive.
2213 if ($self->{nc} == [
2214 undef,
2215 0x004F, # O
2216 0x0043, # C
2217 0x0054, # T
2218 0x0059, # Y
2219 0x0050, # P
2220 ]->[length $self->{s_kwd}] or
2221 $self->{nc} == [
2222 undef,
2223 0x006F, # o
2224 0x0063, # c
2225 0x0074, # t
2226 0x0079, # y
2227 0x0070, # p
2228 ]->[length $self->{s_kwd}]) {
2229 !!!cp (131);
2230 ## Stay in the state.
2231 $self->{s_kwd} .= chr $self->{nc};
2232 !!!next-input-character;
2233 redo A;
2234 } elsif ((length $self->{s_kwd}) == 6 and
2235 ($self->{nc} == 0x0045 or # E
2236 $self->{nc} == 0x0065)) { # e
2237 !!!cp (129);
2238 $self->{state} = DOCTYPE_STATE;
2239 $self->{ct} = {type => DOCTYPE_TOKEN,
2240 quirks => 1,
2241 line => $self->{line_prev},
2242 column => $self->{column_prev} - 7,
2243 };
2244 !!!next-input-character;
2245 redo A;
2246 } else {
2247 !!!cp (132);
2248 !!!parse-error (type => 'bogus comment',
2249 line => $self->{line_prev},
2250 column => $self->{column_prev} - 1 - length $self->{s_kwd});
2251 $self->{state} = BOGUS_COMMENT_STATE;
2252 ## Reconsume.
2253 $self->{ct} = {type => COMMENT_TOKEN,
2254 data => $self->{s_kwd},
2255 line => $self->{line_prev},
2256 column => $self->{column_prev} - 1 - length $self->{s_kwd},
2257 };
2258 redo A;
2259 }
2260 } elsif ($self->{state} == MD_CDATA_STATE) {
2261 if ($self->{nc} == {
2262 '[' => 0x0043, # C
2263 '[C' => 0x0044, # D
2264 '[CD' => 0x0041, # A
2265 '[CDA' => 0x0054, # T
2266 '[CDAT' => 0x0041, # A
2267 }->{$self->{s_kwd}}) {
2268 !!!cp (135.1);
2269 ## Stay in the state.
2270 $self->{s_kwd} .= chr $self->{nc};
2271 !!!next-input-character;
2272 redo A;
2273 } elsif ($self->{s_kwd} eq '[CDATA' and
2274 $self->{nc} == 0x005B) { # [
2275 !!!cp (135.2);
2276 $self->{ct} = {type => CHARACTER_TOKEN,
2277 data => '',
2278 line => $self->{line_prev},
2279 column => $self->{column_prev} - 7};
2280 $self->{state} = CDATA_SECTION_STATE;
2281 !!!next-input-character;
2282 redo A;
2283 } else {
2284 !!!cp (135.3);
2285 !!!parse-error (type => 'bogus comment',
2286 line => $self->{line_prev},
2287 column => $self->{column_prev} - 1 - length $self->{s_kwd});
2288 $self->{state} = BOGUS_COMMENT_STATE;
2289 ## Reconsume.
2290 $self->{ct} = {type => COMMENT_TOKEN,
2291 data => $self->{s_kwd},
2292 line => $self->{line_prev},
2293 column => $self->{column_prev} - 1 - length $self->{s_kwd},
2294 };
2295 redo A;
2296 }
2297 } elsif ($self->{state} == COMMENT_START_STATE) {
2298 if ($self->{nc} == 0x002D) { # -
2299 !!!cp (137);
2300 $self->{state} = COMMENT_START_DASH_STATE;
2301 !!!next-input-character;
2302 redo A;
2303 } elsif ($self->{nc} == 0x003E) { # >
2304 !!!cp (138);
2305 !!!parse-error (type => 'bogus comment');
2306 $self->{state} = DATA_STATE;
2307 !!!next-input-character;
2308
2309 !!!emit ($self->{ct}); # comment
2310
2311 redo A;
2312 } elsif ($self->{nc} == -1) {
2313 !!!cp (139);
2314 !!!parse-error (type => 'unclosed comment');
2315 $self->{state} = DATA_STATE;
2316 ## reconsume
2317
2318 !!!emit ($self->{ct}); # comment
2319
2320 redo A;
2321 } else {
2322 !!!cp (140);
2323 $self->{ct}->{data} # comment
2324 .= chr ($self->{nc});
2325 $self->{state} = COMMENT_STATE;
2326 !!!next-input-character;
2327 redo A;
2328 }
2329 } elsif ($self->{state} == COMMENT_START_DASH_STATE) {
2330 if ($self->{nc} == 0x002D) { # -
2331 !!!cp (141);
2332 $self->{state} = COMMENT_END_STATE;
2333 !!!next-input-character;
2334 redo A;
2335 } elsif ($self->{nc} == 0x003E) { # >
2336 !!!cp (142);
2337 !!!parse-error (type => 'bogus comment');
2338 $self->{state} = DATA_STATE;
2339 !!!next-input-character;
2340
2341 !!!emit ($self->{ct}); # comment
2342
2343 redo A;
2344 } elsif ($self->{nc} == -1) {
2345 !!!cp (143);
2346 !!!parse-error (type => 'unclosed comment');
2347 $self->{state} = DATA_STATE;
2348 ## reconsume
2349
2350 !!!emit ($self->{ct}); # comment
2351
2352 redo A;
2353 } else {
2354 !!!cp (144);
2355 $self->{ct}->{data} # comment
2356 .= '-' . chr ($self->{nc});
2357 $self->{state} = COMMENT_STATE;
2358 !!!next-input-character;
2359 redo A;
2360 }
2361 } elsif ($self->{state} == COMMENT_STATE) {
2362 if ($self->{nc} == 0x002D) { # -
2363 !!!cp (145);
2364 $self->{state} = COMMENT_END_DASH_STATE;
2365 !!!next-input-character;
2366 redo A;
2367 } elsif ($self->{nc} == -1) {
2368 !!!cp (146);
2369 !!!parse-error (type => 'unclosed comment');
2370 $self->{state} = DATA_STATE;
2371 ## reconsume
2372
2373 !!!emit ($self->{ct}); # comment
2374
2375 redo A;
2376 } else {
2377 !!!cp (147);
2378 $self->{ct}->{data} .= chr ($self->{nc}); # comment
2379 $self->{read_until}->($self->{ct}->{data},
2380 q[-],
2381 length $self->{ct}->{data});
2382
2383 ## Stay in the state
2384 !!!next-input-character;
2385 redo A;
2386 }
2387 } elsif ($self->{state} == COMMENT_END_DASH_STATE) {
2388 if ($self->{nc} == 0x002D) { # -
2389 !!!cp (148);
2390 $self->{state} = COMMENT_END_STATE;
2391 !!!next-input-character;
2392 redo A;
2393 } elsif ($self->{nc} == -1) {
2394 !!!cp (149);
2395 !!!parse-error (type => 'unclosed comment');
2396 $self->{state} = DATA_STATE;
2397 ## reconsume
2398
2399 !!!emit ($self->{ct}); # comment
2400
2401 redo A;
2402 } else {
2403 !!!cp (150);
2404 $self->{ct}->{data} .= '-' . chr ($self->{nc}); # comment
2405 $self->{state} = COMMENT_STATE;
2406 !!!next-input-character;
2407 redo A;
2408 }
2409 } elsif ($self->{state} == COMMENT_END_STATE) {
2410 if ($self->{nc} == 0x003E) { # >
2411 !!!cp (151);
2412 $self->{state} = DATA_STATE;
2413 !!!next-input-character;
2414
2415 !!!emit ($self->{ct}); # comment
2416
2417 redo A;
2418 } elsif ($self->{nc} == 0x002D) { # -
2419 !!!cp (152);
2420 !!!parse-error (type => 'dash in comment',
2421 line => $self->{line_prev},
2422 column => $self->{column_prev});
2423 $self->{ct}->{data} .= '-'; # comment
2424 ## Stay in the state
2425 !!!next-input-character;
2426 redo A;
2427 } elsif ($self->{nc} == -1) {
2428 !!!cp (153);
2429 !!!parse-error (type => 'unclosed comment');
2430 $self->{state} = DATA_STATE;
2431 ## reconsume
2432
2433 !!!emit ($self->{ct}); # comment
2434
2435 redo A;
2436 } else {
2437 !!!cp (154);
2438 !!!parse-error (type => 'dash in comment',
2439 line => $self->{line_prev},
2440 column => $self->{column_prev});
2441 $self->{ct}->{data} .= '--' . chr ($self->{nc}); # comment
2442 $self->{state} = COMMENT_STATE;
2443 !!!next-input-character;
2444 redo A;
2445 }
2446 } elsif ($self->{state} == DOCTYPE_STATE) {
2447 if ($is_space->{$self->{nc}}) {
2448 !!!cp (155);
2449 $self->{state} = BEFORE_DOCTYPE_NAME_STATE;
2450 !!!next-input-character;
2451 redo A;
2452 } else {
2453 !!!cp (156);
2454 !!!parse-error (type => 'no space before DOCTYPE name');
2455 $self->{state} = BEFORE_DOCTYPE_NAME_STATE;
2456 ## reconsume
2457 redo A;
2458 }
2459 } elsif ($self->{state} == BEFORE_DOCTYPE_NAME_STATE) {
2460 if ($is_space->{$self->{nc}}) {
2461 !!!cp (157);
2462 ## Stay in the state
2463 !!!next-input-character;
2464 redo A;
2465 } elsif ($self->{nc} == 0x003E) { # >
2466 !!!cp (158);
2467 !!!parse-error (type => 'no DOCTYPE name');
2468 $self->{state} = DATA_STATE;
2469 !!!next-input-character;
2470
2471 !!!emit ($self->{ct}); # DOCTYPE (quirks)
2472
2473 redo A;
2474 } elsif ($self->{nc} == -1) {
2475 !!!cp (159);
2476 !!!parse-error (type => 'no DOCTYPE name');
2477 $self->{state} = DATA_STATE;
2478 ## reconsume
2479
2480 !!!emit ($self->{ct}); # DOCTYPE (quirks)
2481
2482 redo A;
2483 } else {
2484 !!!cp (160);
2485 $self->{ct}->{name} = chr $self->{nc};
2486 delete $self->{ct}->{quirks};
2487 $self->{state} = DOCTYPE_NAME_STATE;
2488 !!!next-input-character;
2489 redo A;
2490 }
2491 } elsif ($self->{state} == DOCTYPE_NAME_STATE) {
2492 ## ISSUE: Redundant "First," in the spec.
2493 if ($is_space->{$self->{nc}}) {
2494 !!!cp (161);
2495 $self->{state} = AFTER_DOCTYPE_NAME_STATE;
2496 !!!next-input-character;
2497 redo A;
2498 } elsif ($self->{nc} == 0x003E) { # >
2499 !!!cp (162);
2500 $self->{state} = DATA_STATE;
2501 !!!next-input-character;
2502
2503 !!!emit ($self->{ct}); # DOCTYPE
2504
2505 redo A;
2506 } elsif ($self->{nc} == -1) {
2507 !!!cp (163);
2508 !!!parse-error (type => 'unclosed DOCTYPE');
2509 $self->{state} = DATA_STATE;
2510 ## reconsume
2511
2512 $self->{ct}->{quirks} = 1;
2513 !!!emit ($self->{ct}); # DOCTYPE
2514
2515 redo A;
2516 } else {
2517 !!!cp (164);
2518 $self->{ct}->{name}
2519 .= chr ($self->{nc}); # DOCTYPE
2520 ## Stay in the state
2521 !!!next-input-character;
2522 redo A;
2523 }
2524 } elsif ($self->{state} == AFTER_DOCTYPE_NAME_STATE) {
2525 if ($is_space->{$self->{nc}}) {
2526 !!!cp (165);
2527 ## Stay in the state
2528 !!!next-input-character;
2529 redo A;
2530 } elsif ($self->{nc} == 0x003E) { # >
2531 !!!cp (166);
2532 $self->{state} = DATA_STATE;
2533 !!!next-input-character;
2534
2535 !!!emit ($self->{ct}); # DOCTYPE
2536
2537 redo A;
2538 } elsif ($self->{nc} == -1) {
2539 !!!cp (167);
2540 !!!parse-error (type => 'unclosed DOCTYPE');
2541 $self->{state} = DATA_STATE;
2542 ## reconsume
2543
2544 $self->{ct}->{quirks} = 1;
2545 !!!emit ($self->{ct}); # DOCTYPE
2546
2547 redo A;
2548 } elsif ($self->{nc} == 0x0050 or # P
2549 $self->{nc} == 0x0070) { # p
2550 $self->{state} = PUBLIC_STATE;
2551 $self->{s_kwd} = chr $self->{nc};
2552 !!!next-input-character;
2553 redo A;
2554 } elsif ($self->{nc} == 0x0053 or # S
2555 $self->{nc} == 0x0073) { # s
2556 $self->{state} = SYSTEM_STATE;
2557 $self->{s_kwd} = chr $self->{nc};
2558 !!!next-input-character;
2559 redo A;
2560 } else {
2561 !!!cp (180);
2562 !!!parse-error (type => 'string after DOCTYPE name');
2563 $self->{ct}->{quirks} = 1;
2564
2565 $self->{state} = BOGUS_DOCTYPE_STATE;
2566 !!!next-input-character;
2567 redo A;
2568 }
2569 } elsif ($self->{state} == PUBLIC_STATE) {
2570 ## ASCII case-insensitive
2571 if ($self->{nc} == [
2572 undef,
2573 0x0055, # U
2574 0x0042, # B
2575 0x004C, # L
2576 0x0049, # I
2577 ]->[length $self->{s_kwd}] or
2578 $self->{nc} == [
2579 undef,
2580 0x0075, # u
2581 0x0062, # b
2582 0x006C, # l
2583 0x0069, # i
2584 ]->[length $self->{s_kwd}]) {
2585 !!!cp (175);
2586 ## Stay in the state.
2587 $self->{s_kwd} .= chr $self->{nc};
2588 !!!next-input-character;
2589 redo A;
2590 } elsif ((length $self->{s_kwd}) == 5 and
2591 ($self->{nc} == 0x0043 or # C
2592 $self->{nc} == 0x0063)) { # c
2593 !!!cp (168);
2594 $self->{state} = BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2595 !!!next-input-character;
2596 redo A;
2597 } else {
2598 !!!cp (169);
2599 !!!parse-error (type => 'string after DOCTYPE name',
2600 line => $self->{line_prev},
2601 column => $self->{column_prev} + 1 - length $self->{s_kwd});
2602 $self->{ct}->{quirks} = 1;
2603
2604 $self->{state} = BOGUS_DOCTYPE_STATE;
2605 ## Reconsume.
2606 redo A;
2607 }
2608 } elsif ($self->{state} == SYSTEM_STATE) {
2609 ## ASCII case-insensitive
2610 if ($self->{nc} == [
2611 undef,
2612 0x0059, # Y
2613 0x0053, # S
2614 0x0054, # T
2615 0x0045, # E
2616 ]->[length $self->{s_kwd}] or
2617 $self->{nc} == [
2618 undef,
2619 0x0079, # y
2620 0x0073, # s
2621 0x0074, # t
2622 0x0065, # e
2623 ]->[length $self->{s_kwd}]) {
2624 !!!cp (170);
2625 ## Stay in the state.
2626 $self->{s_kwd} .= chr $self->{nc};
2627 !!!next-input-character;
2628 redo A;
2629 } elsif ((length $self->{s_kwd}) == 5 and
2630 ($self->{nc} == 0x004D or # M
2631 $self->{nc} == 0x006D)) { # m
2632 !!!cp (171);
2633 $self->{state} = BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2634 !!!next-input-character;
2635 redo A;
2636 } else {
2637 !!!cp (172);
2638 !!!parse-error (type => 'string after DOCTYPE name',
2639 line => $self->{line_prev},
2640 column => $self->{column_prev} + 1 - length $self->{s_kwd});
2641 $self->{ct}->{quirks} = 1;
2642
2643 $self->{state} = BOGUS_DOCTYPE_STATE;
2644 ## Reconsume.
2645 redo A;
2646 }
2647 } elsif ($self->{state} == BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {
2648 if ($is_space->{$self->{nc}}) {
2649 !!!cp (181);
2650 ## Stay in the state
2651 !!!next-input-character;
2652 redo A;
2653 } elsif ($self->{nc} eq 0x0022) { # "
2654 !!!cp (182);
2655 $self->{ct}->{pubid} = ''; # DOCTYPE
2656 $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE;
2657 !!!next-input-character;
2658 redo A;
2659 } elsif ($self->{nc} eq 0x0027) { # '
2660 !!!cp (183);
2661 $self->{ct}->{pubid} = ''; # DOCTYPE
2662 $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE;
2663 !!!next-input-character;
2664 redo A;
2665 } elsif ($self->{nc} eq 0x003E) { # >
2666 !!!cp (184);
2667 !!!parse-error (type => 'no PUBLIC literal');
2668
2669 $self->{state} = DATA_STATE;
2670 !!!next-input-character;
2671
2672 $self->{ct}->{quirks} = 1;
2673 !!!emit ($self->{ct}); # DOCTYPE
2674
2675 redo A;
2676 } elsif ($self->{nc} == -1) {
2677 !!!cp (185);
2678 !!!parse-error (type => 'unclosed DOCTYPE');
2679
2680 $self->{state} = DATA_STATE;
2681 ## reconsume
2682
2683 $self->{ct}->{quirks} = 1;
2684 !!!emit ($self->{ct}); # DOCTYPE
2685
2686 redo A;
2687 } else {
2688 !!!cp (186);
2689 !!!parse-error (type => 'string after PUBLIC');
2690 $self->{ct}->{quirks} = 1;
2691
2692 $self->{state} = BOGUS_DOCTYPE_STATE;
2693 !!!next-input-character;
2694 redo A;
2695 }
2696 } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE) {
2697 if ($self->{nc} == 0x0022) { # "
2698 !!!cp (187);
2699 $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2700 !!!next-input-character;
2701 redo A;
2702 } elsif ($self->{nc} == 0x003E) { # >
2703 !!!cp (188);
2704 !!!parse-error (type => 'unclosed PUBLIC literal');
2705
2706 $self->{state} = DATA_STATE;
2707 !!!next-input-character;
2708
2709 $self->{ct}->{quirks} = 1;
2710 !!!emit ($self->{ct}); # DOCTYPE
2711
2712 redo A;
2713 } elsif ($self->{nc} == -1) {
2714 !!!cp (189);
2715 !!!parse-error (type => 'unclosed PUBLIC literal');
2716
2717 $self->{state} = DATA_STATE;
2718 ## reconsume
2719
2720 $self->{ct}->{quirks} = 1;
2721 !!!emit ($self->{ct}); # DOCTYPE
2722
2723 redo A;
2724 } else {
2725 !!!cp (190);
2726 $self->{ct}->{pubid} # DOCTYPE
2727 .= chr $self->{nc};
2728 $self->{read_until}->($self->{ct}->{pubid}, q[">],
2729 length $self->{ct}->{pubid});
2730
2731 ## Stay in the state
2732 !!!next-input-character;
2733 redo A;
2734 }
2735 } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE) {
2736 if ($self->{nc} == 0x0027) { # '
2737 !!!cp (191);
2738 $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2739 !!!next-input-character;
2740 redo A;
2741 } elsif ($self->{nc} == 0x003E) { # >
2742 !!!cp (192);
2743 !!!parse-error (type => 'unclosed PUBLIC literal');
2744
2745 $self->{state} = DATA_STATE;
2746 !!!next-input-character;
2747
2748 $self->{ct}->{quirks} = 1;
2749 !!!emit ($self->{ct}); # DOCTYPE
2750
2751 redo A;
2752 } elsif ($self->{nc} == -1) {
2753 !!!cp (193);
2754 !!!parse-error (type => 'unclosed PUBLIC literal');
2755
2756 $self->{state} = DATA_STATE;
2757 ## reconsume
2758
2759 $self->{ct}->{quirks} = 1;
2760 !!!emit ($self->{ct}); # DOCTYPE
2761
2762 redo A;
2763 } else {
2764 !!!cp (194);
2765 $self->{ct}->{pubid} # DOCTYPE
2766 .= chr $self->{nc};
2767 $self->{read_until}->($self->{ct}->{pubid}, q['>],
2768 length $self->{ct}->{pubid});
2769
2770 ## Stay in the state
2771 !!!next-input-character;
2772 redo A;
2773 }
2774 } elsif ($self->{state} == AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {
2775 if ($is_space->{$self->{nc}}) {
2776 !!!cp (195);
2777 ## Stay in the state
2778 !!!next-input-character;
2779 redo A;
2780 } elsif ($self->{nc} == 0x0022) { # "
2781 !!!cp (196);
2782 $self->{ct}->{sysid} = ''; # DOCTYPE
2783 $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;
2784 !!!next-input-character;
2785 redo A;
2786 } elsif ($self->{nc} == 0x0027) { # '
2787 !!!cp (197);
2788 $self->{ct}->{sysid} = ''; # DOCTYPE
2789 $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;
2790 !!!next-input-character;
2791 redo A;
2792 } elsif ($self->{nc} == 0x003E) { # >
2793 !!!cp (198);
2794 $self->{state} = DATA_STATE;
2795 !!!next-input-character;
2796
2797 !!!emit ($self->{ct}); # DOCTYPE
2798
2799 redo A;
2800 } elsif ($self->{nc} == -1) {
2801 !!!cp (199);
2802 !!!parse-error (type => 'unclosed DOCTYPE');
2803
2804 $self->{state} = DATA_STATE;
2805 ## reconsume
2806
2807 $self->{ct}->{quirks} = 1;
2808 !!!emit ($self->{ct}); # DOCTYPE
2809
2810 redo A;
2811 } else {
2812 !!!cp (200);
2813 !!!parse-error (type => 'string after PUBLIC literal');
2814 $self->{ct}->{quirks} = 1;
2815
2816 $self->{state} = BOGUS_DOCTYPE_STATE;
2817 !!!next-input-character;
2818 redo A;
2819 }
2820 } elsif ($self->{state} == BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {
2821 if ($is_space->{$self->{nc}}) {
2822 !!!cp (201);
2823 ## Stay in the state
2824 !!!next-input-character;
2825 redo A;
2826 } elsif ($self->{nc} == 0x0022) { # "
2827 !!!cp (202);
2828 $self->{ct}->{sysid} = ''; # DOCTYPE
2829 $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;
2830 !!!next-input-character;
2831 redo A;
2832 } elsif ($self->{nc} == 0x0027) { # '
2833 !!!cp (203);
2834 $self->{ct}->{sysid} = ''; # DOCTYPE
2835 $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;
2836 !!!next-input-character;
2837 redo A;
2838 } elsif ($self->{nc} == 0x003E) { # >
2839 !!!cp (204);
2840 !!!parse-error (type => 'no SYSTEM literal');
2841 $self->{state} = DATA_STATE;
2842 !!!next-input-character;
2843
2844 $self->{ct}->{quirks} = 1;
2845 !!!emit ($self->{ct}); # DOCTYPE
2846
2847 redo A;
2848 } elsif ($self->{nc} == -1) {
2849 !!!cp (205);
2850 !!!parse-error (type => 'unclosed DOCTYPE');
2851
2852 $self->{state} = DATA_STATE;
2853 ## reconsume
2854
2855 $self->{ct}->{quirks} = 1;
2856 !!!emit ($self->{ct}); # DOCTYPE
2857
2858 redo A;
2859 } else {
2860 !!!cp (206);
2861 !!!parse-error (type => 'string after SYSTEM');
2862 $self->{ct}->{quirks} = 1;
2863
2864 $self->{state} = BOGUS_DOCTYPE_STATE;
2865 !!!next-input-character;
2866 redo A;
2867 }
2868 } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE) {
2869 if ($self->{nc} == 0x0022) { # "
2870 !!!cp (207);
2871 $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2872 !!!next-input-character;
2873 redo A;
2874 } elsif ($self->{nc} == 0x003E) { # >
2875 !!!cp (208);
2876 !!!parse-error (type => 'unclosed SYSTEM literal');
2877
2878 $self->{state} = DATA_STATE;
2879 !!!next-input-character;
2880
2881 $self->{ct}->{quirks} = 1;
2882 !!!emit ($self->{ct}); # DOCTYPE
2883
2884 redo A;
2885 } elsif ($self->{nc} == -1) {
2886 !!!cp (209);
2887 !!!parse-error (type => 'unclosed SYSTEM literal');
2888
2889 $self->{state} = DATA_STATE;
2890 ## reconsume
2891
2892 $self->{ct}->{quirks} = 1;
2893 !!!emit ($self->{ct}); # DOCTYPE
2894
2895 redo A;
2896 } else {
2897 !!!cp (210);
2898 $self->{ct}->{sysid} # DOCTYPE
2899 .= chr $self->{nc};
2900 $self->{read_until}->($self->{ct}->{sysid}, q[">],
2901 length $self->{ct}->{sysid});
2902
2903 ## Stay in the state
2904 !!!next-input-character;
2905 redo A;
2906 }
2907 } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE) {
2908 if ($self->{nc} == 0x0027) { # '
2909 !!!cp (211);
2910 $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2911 !!!next-input-character;
2912 redo A;
2913 } elsif ($self->{nc} == 0x003E) { # >
2914 !!!cp (212);
2915 !!!parse-error (type => 'unclosed SYSTEM literal');
2916
2917 $self->{state} = DATA_STATE;
2918 !!!next-input-character;
2919
2920 $self->{ct}->{quirks} = 1;
2921 !!!emit ($self->{ct}); # DOCTYPE
2922
2923 redo A;
2924 } elsif ($self->{nc} == -1) {
2925 !!!cp (213);
2926 !!!parse-error (type => 'unclosed SYSTEM literal');
2927
2928 $self->{state} = DATA_STATE;
2929 ## reconsume
2930
2931 $self->{ct}->{quirks} = 1;
2932 !!!emit ($self->{ct}); # DOCTYPE
2933
2934 redo A;
2935 } else {
2936 !!!cp (214);
2937 $self->{ct}->{sysid} # DOCTYPE
2938 .= chr $self->{nc};
2939 $self->{read_until}->($self->{ct}->{sysid}, q['>],
2940 length $self->{ct}->{sysid});
2941
2942 ## Stay in the state
2943 !!!next-input-character;
2944 redo A;
2945 }
2946 } elsif ($self->{state} == AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {
2947 if ($is_space->{$self->{nc}}) {
2948 !!!cp (215);
2949 ## Stay in the state
2950 !!!next-input-character;
2951 redo A;
2952 } elsif ($self->{nc} == 0x003E) { # >
2953 !!!cp (216);
2954 $self->{state} = DATA_STATE;
2955 !!!next-input-character;
2956
2957 !!!emit ($self->{ct}); # DOCTYPE
2958
2959 redo A;
2960 } elsif ($self->{nc} == -1) {
2961 !!!cp (217);
2962 !!!parse-error (type => 'unclosed DOCTYPE');
2963 $self->{state} = DATA_STATE;
2964 ## reconsume
2965
2966 $self->{ct}->{quirks} = 1;
2967 !!!emit ($self->{ct}); # DOCTYPE
2968
2969 redo A;
2970 } else {
2971 !!!cp (218);
2972 !!!parse-error (type => 'string after SYSTEM literal');
2973 #$self->{ct}->{quirks} = 1;
2974
2975 $self->{state} = BOGUS_DOCTYPE_STATE;
2976 !!!next-input-character;
2977 redo A;
2978 }
2979 } elsif ($self->{state} == BOGUS_DOCTYPE_STATE) {
2980 if ($self->{nc} == 0x003E) { # >
2981 !!!cp (219);
2982 $self->{state} = DATA_STATE;
2983 !!!next-input-character;
2984
2985 !!!emit ($self->{ct}); # DOCTYPE
2986
2987 redo A;
2988 } elsif ($self->{nc} == -1) {
2989 !!!cp (220);
2990 $self->{state} = DATA_STATE;
2991 ## reconsume
2992
2993 !!!emit ($self->{ct}); # DOCTYPE
2994
2995 redo A;
2996 } else {
2997 !!!cp (221);
2998 my $s = '';
2999 $self->{read_until}->($s, q[>], 0);
3000
3001 ## Stay in the state
3002 !!!next-input-character;
3003 redo A;
3004 }
3005 } elsif ($self->{state} == CDATA_SECTION_STATE) {
3006 ## NOTE: "CDATA section state" in the state is jointly implemented
3007 ## by three states, |CDATA_SECTION_STATE|, |CDATA_SECTION_MSE1_STATE|,
3008 ## and |CDATA_SECTION_MSE2_STATE|.
3009
3010 if ($self->{nc} == 0x005D) { # ]
3011 !!!cp (221.1);
3012 $self->{state} = CDATA_SECTION_MSE1_STATE;
3013 !!!next-input-character;
3014 redo A;
3015 } elsif ($self->{nc} == -1) {
3016 $self->{state} = DATA_STATE;
3017 !!!next-input-character;
3018 if (length $self->{ct}->{data}) { # character
3019 !!!cp (221.2);
3020 !!!emit ($self->{ct}); # character
3021 } else {
3022 !!!cp (221.3);
3023 ## No token to emit. $self->{ct} is discarded.
3024 }
3025 redo A;
3026 } else {
3027 !!!cp (221.4);
3028 $self->{ct}->{data} .= chr $self->{nc};
3029 $self->{read_until}->($self->{ct}->{data},
3030 q<]>,
3031 length $self->{ct}->{data});
3032
3033 ## Stay in the state.
3034 !!!next-input-character;
3035 redo A;
3036 }
3037
3038 ## ISSUE: "text tokens" in spec.
3039 } elsif ($self->{state} == CDATA_SECTION_MSE1_STATE) {
3040 if ($self->{nc} == 0x005D) { # ]
3041 !!!cp (221.5);
3042 $self->{state} = CDATA_SECTION_MSE2_STATE;
3043 !!!next-input-character;
3044 redo A;
3045 } else {
3046 !!!cp (221.6);
3047 $self->{ct}->{data} .= ']';
3048 $self->{state} = CDATA_SECTION_STATE;
3049 ## Reconsume.
3050 redo A;
3051 }
3052 } elsif ($self->{state} == CDATA_SECTION_MSE2_STATE) {
3053 if ($self->{nc} == 0x003E) { # >
3054 $self->{state} = DATA_STATE;
3055 !!!next-input-character;
3056 if (length $self->{ct}->{data}) { # character
3057 !!!cp (221.7);
3058 !!!emit ($self->{ct}); # character
3059 } else {
3060 !!!cp (221.8);
3061 ## No token to emit. $self->{ct} is discarded.
3062 }
3063 redo A;
3064 } elsif ($self->{nc} == 0x005D) { # ]
3065 !!!cp (221.9); # character
3066 $self->{ct}->{data} .= ']'; ## Add first "]" of "]]]".
3067 ## Stay in the state.
3068 !!!next-input-character;
3069 redo A;
3070 } else {
3071 !!!cp (221.11);
3072 $self->{ct}->{data} .= ']]'; # character
3073 $self->{state} = CDATA_SECTION_STATE;
3074 ## Reconsume.
3075 redo A;
3076 }
3077 } elsif ($self->{state} == ENTITY_STATE) {
3078 if ($is_space->{$self->{nc}} or
3079 {
3080 0x003C => 1, 0x0026 => 1, -1 => 1, # <, &
3081 $self->{entity_add} => 1,
3082 }->{$self->{nc}}) {
3083 !!!cp (1001);
3084 ## Don't consume
3085 ## No error
3086 ## Return nothing.
3087 #
3088 } elsif ($self->{nc} == 0x0023) { # #
3089 !!!cp (999);
3090 $self->{state} = ENTITY_HASH_STATE;
3091 $self->{s_kwd} = '#';
3092 !!!next-input-character;
3093 redo A;
3094 } elsif ((0x0041 <= $self->{nc} and
3095 $self->{nc} <= 0x005A) or # A..Z
3096 (0x0061 <= $self->{nc} and
3097 $self->{nc} <= 0x007A)) { # a..z
3098 !!!cp (998);
3099 require Whatpm::_NamedEntityList;
3100 $self->{state} = ENTITY_NAME_STATE;
3101 $self->{s_kwd} = chr $self->{nc};
3102 $self->{entity__value} = $self->{s_kwd};
3103 $self->{entity__match} = 0;
3104 !!!next-input-character;
3105 redo A;
3106 } else {
3107 !!!cp (1027);
3108 !!!parse-error (type => 'bare ero');
3109 ## Return nothing.
3110 #
3111 }
3112
3113 ## NOTE: No character is consumed by the "consume a character
3114 ## reference" algorithm. In other word, there is an "&" character
3115 ## that does not introduce a character reference, which would be
3116 ## appended to the parent element or the attribute value in later
3117 ## process of the tokenizer.
3118
3119 if ($self->{prev_state} == DATA_STATE) {
3120 !!!cp (997);
3121 $self->{state} = $self->{prev_state};
3122 ## Reconsume.
3123 !!!emit ({type => CHARACTER_TOKEN, data => '&',
3124 line => $self->{line_prev},
3125 column => $self->{column_prev},
3126 });
3127 redo A;
3128 } else {
3129 !!!cp (996);
3130 $self->{ca}->{value} .= '&';
3131 $self->{state} = $self->{prev_state};
3132 ## Reconsume.
3133 redo A;
3134 }
3135 } elsif ($self->{state} == ENTITY_HASH_STATE) {
3136 if ($self->{nc} == 0x0078 or # x
3137 $self->{nc} == 0x0058) { # X
3138 !!!cp (995);
3139 $self->{state} = HEXREF_X_STATE;
3140 $self->{s_kwd} .= chr $self->{nc};
3141 !!!next-input-character;
3142 redo A;
3143 } elsif (0x0030 <= $self->{nc} and
3144 $self->{nc} <= 0x0039) { # 0..9
3145 !!!cp (994);
3146 $self->{state} = NCR_NUM_STATE;
3147 $self->{s_kwd} = $self->{nc} - 0x0030;
3148 !!!next-input-character;
3149 redo A;
3150 } else {
3151 !!!parse-error (type => 'bare nero',
3152 line => $self->{line_prev},
3153 column => $self->{column_prev} - 1);
3154
3155 ## NOTE: According to the spec algorithm, nothing is returned,
3156 ## and then "&#" is appended to the parent element or the attribute
3157 ## value in the later processing.
3158
3159 if ($self->{prev_state} == DATA_STATE) {
3160 !!!cp (1019);
3161 $self->{state} = $self->{prev_state};
3162 ## Reconsume.
3163 !!!emit ({type => CHARACTER_TOKEN,
3164 data => '&#',
3165 line => $self->{line_prev},
3166 column => $self->{column_prev} - 1,
3167 });
3168 redo A;
3169 } else {
3170 !!!cp (993);
3171 $self->{ca}->{value} .= '&#';
3172 $self->{state} = $self->{prev_state};
3173 ## Reconsume.
3174 redo A;
3175 }
3176 }
3177 } elsif ($self->{state} == NCR_NUM_STATE) {
3178 if (0x0030 <= $self->{nc} and
3179 $self->{nc} <= 0x0039) { # 0..9
3180 !!!cp (1012);
3181 $self->{s_kwd} *= 10;
3182 $self->{s_kwd} += $self->{nc} - 0x0030;
3183
3184 ## Stay in the state.
3185 !!!next-input-character;
3186 redo A;
3187 } elsif ($self->{nc} == 0x003B) { # ;
3188 !!!cp (1013);
3189 !!!next-input-character;
3190 #
3191 } else {
3192 !!!cp (1014);
3193 !!!parse-error (type => 'no refc');
3194 ## Reconsume.
3195 #
3196 }
3197
3198 my $code = $self->{s_kwd};
3199 my $l = $self->{line_prev};
3200 my $c = $self->{column_prev};
3201 if ($charref_map->{$code}) {
3202 !!!cp (1015);
3203 !!!parse-error (type => 'invalid character reference',
3204 text => (sprintf 'U+%04X', $code),
3205 line => $l, column => $c);
3206 $code = $charref_map->{$code};
3207 } elsif ($code > 0x10FFFF) {
3208 !!!cp (1016);
3209 !!!parse-error (type => 'invalid character reference',
3210 text => (sprintf 'U-%08X', $code),
3211 line => $l, column => $c);
3212 $code = 0xFFFD;
3213 }
3214
3215 if ($self->{prev_state} == DATA_STATE) {
3216 !!!cp (992);
3217 $self->{state} = $self->{prev_state};
3218 ## Reconsume.
3219 !!!emit ({type => CHARACTER_TOKEN, data => chr $code,
3220 line => $l, column => $c,
3221 });
3222 redo A;
3223 } else {
3224 !!!cp (991);
3225 $self->{ca}->{value} .= chr $code;
3226 $self->{ca}->{has_reference} = 1;
3227 $self->{state} = $self->{prev_state};
3228 ## Reconsume.
3229 redo A;
3230 }
3231 } elsif ($self->{state} == HEXREF_X_STATE) {
3232 if ((0x0030 <= $self->{nc} and $self->{nc} <= 0x0039) or
3233 (0x0041 <= $self->{nc} and $self->{nc} <= 0x0046) or
3234 (0x0061 <= $self->{nc} and $self->{nc} <= 0x0066)) {
3235 # 0..9, A..F, a..f
3236 !!!cp (990);
3237 $self->{state} = HEXREF_HEX_STATE;
3238 $self->{s_kwd} = 0;
3239 ## Reconsume.
3240 redo A;
3241 } else {
3242 !!!parse-error (type => 'bare hcro',
3243 line => $self->{line_prev},
3244 column => $self->{column_prev} - 2);
3245
3246 ## NOTE: According to the spec algorithm, nothing is returned,
3247 ## and then "&#" followed by "X" or "x" is appended to the parent
3248 ## element or the attribute value in the later processing.
3249
3250 if ($self->{prev_state} == DATA_STATE) {
3251 !!!cp (1005);
3252 $self->{state} = $self->{prev_state};
3253 ## Reconsume.
3254 !!!emit ({type => CHARACTER_TOKEN,
3255 data => '&' . $self->{s_kwd},
3256 line => $self->{line_prev},
3257 column => $self->{column_prev} - length $self->{s_kwd},
3258 });
3259 redo A;
3260 } else {
3261 !!!cp (989);
3262 $self->{ca}->{value} .= '&' . $self->{s_kwd};
3263 $self->{state} = $self->{prev_state};
3264 ## Reconsume.
3265 redo A;
3266 }
3267 }
3268 } elsif ($self->{state} == HEXREF_HEX_STATE) {
3269 if (0x0030 <= $self->{nc} and $self->{nc} <= 0x0039) {
3270 # 0..9
3271 !!!cp (1002);
3272 $self->{s_kwd} *= 0x10;
3273 $self->{s_kwd} += $self->{nc} - 0x0030;
3274 ## Stay in the state.
3275 !!!next-input-character;
3276 redo A;
3277 } elsif (0x0061 <= $self->{nc} and
3278 $self->{nc} <= 0x0066) { # a..f
3279 !!!cp (1003);
3280 $self->{s_kwd} *= 0x10;
3281 $self->{s_kwd} += $self->{nc} - 0x0060 + 9;
3282 ## Stay in the state.
3283 !!!next-input-character;
3284 redo A;
3285 } elsif (0x0041 <= $self->{nc} and
3286 $self->{nc} <= 0x0046) { # A..F
3287 !!!cp (1004);
3288 $self->{s_kwd} *= 0x10;
3289 $self->{s_kwd} += $self->{nc} - 0x0040 + 9;
3290 ## Stay in the state.
3291 !!!next-input-character;
3292 redo A;
3293 } elsif ($self->{nc} == 0x003B) { # ;
3294 !!!cp (1006);
3295 !!!next-input-character;
3296 #
3297 } else {
3298 !!!cp (1007);
3299 !!!parse-error (type => 'no refc',
3300 line => $self->{line},
3301 column => $self->{column});
3302 ## Reconsume.
3303 #
3304 }
3305
3306 my $code = $self->{s_kwd};
3307 my $l = $self->{line_prev};
3308 my $c = $self->{column_prev};
3309 if ($charref_map->{$code}) {
3310 !!!cp (1008);
3311 !!!parse-error (type => 'invalid character reference',
3312 text => (sprintf 'U+%04X', $code),
3313 line => $l, column => $c);
3314 $code = $charref_map->{$code};
3315 } elsif ($code > 0x10FFFF) {
3316 !!!cp (1009);
3317 !!!parse-error (type => 'invalid character reference',
3318 text => (sprintf 'U-%08X', $code),
3319 line => $l, column => $c);
3320 $code = 0xFFFD;
3321 }
3322
3323 if ($self->{prev_state} == DATA_STATE) {
3324 !!!cp (988);
3325 $self->{state} = $self->{prev_state};
3326 ## Reconsume.
3327 !!!emit ({type => CHARACTER_TOKEN, data => chr $code,
3328 line => $l, column => $c,
3329 });
3330 redo A;
3331 } else {
3332 !!!cp (987);
3333 $self->{ca}->{value} .= chr $code;
3334 $self->{ca}->{has_reference} = 1;
3335 $self->{state} = $self->{prev_state};
3336 ## Reconsume.
3337 redo A;
3338 }
3339 } elsif ($self->{state} == ENTITY_NAME_STATE) {
3340 if (length $self->{s_kwd} < 30 and
3341 ## NOTE: Some number greater than the maximum length of entity name
3342 ((0x0041 <= $self->{nc} and # a
3343 $self->{nc} <= 0x005A) or # x
3344 (0x0061 <= $self->{nc} and # a
3345 $self->{nc} <= 0x007A) or # z
3346 (0x0030 <= $self->{nc} and # 0
3347 $self->{nc} <= 0x0039) or # 9
3348 $self->{nc} == 0x003B)) { # ;
3349 our $EntityChar;
3350 $self->{s_kwd} .= chr $self->{nc};
3351 if (defined $EntityChar->{$self->{s_kwd}}) {
3352 if ($self->{nc} == 0x003B) { # ;
3353 !!!cp (1020);
3354 $self->{entity__value} = $EntityChar->{$self->{s_kwd}};
3355 $self->{entity__match} = 1;
3356 !!!next-input-character;
3357 #
3358 } else {
3359 !!!cp (1021);
3360 $self->{entity__value} = $EntityChar->{$self->{s_kwd}};
3361 $self->{entity__match} = -1;
3362 ## Stay in the state.
3363 !!!next-input-character;
3364 redo A;
3365 }
3366 } else {
3367 !!!cp (1022);
3368 $self->{entity__value} .= chr $self->{nc};
3369 $self->{entity__match} *= 2;
3370 ## Stay in the state.
3371 !!!next-input-character;
3372 redo A;
3373 }
3374 }
3375
3376 my $data;
3377 my $has_ref;
3378 if ($self->{entity__match} > 0) {
3379 !!!cp (1023);
3380 $data = $self->{entity__value};
3381 $has_ref = 1;
3382 #
3383 } elsif ($self->{entity__match} < 0) {
3384 !!!parse-error (type => 'no refc');
3385 if ($self->{prev_state} != DATA_STATE and # in attribute
3386 $self->{entity__match} < -1) {
3387 !!!cp (1024);
3388 $data = '&' . $self->{s_kwd};
3389 #
3390 } else {
3391 !!!cp (1025);
3392 $data = $self->{entity__value};
3393 $has_ref = 1;
3394 #
3395 }
3396 } else {
3397 !!!cp (1026);
3398 !!!parse-error (type => 'bare ero',
3399 line => $self->{line_prev},
3400 column => $self->{column_prev} - length $self->{s_kwd});
3401 $data = '&' . $self->{s_kwd};
3402 #
3403 }
3404
3405 ## NOTE: In these cases, when a character reference is found,
3406 ## it is consumed and a character token is returned, or, otherwise,
3407 ## nothing is consumed and returned, according to the spec algorithm.
3408 ## In this implementation, anything that has been examined by the
3409 ## tokenizer is appended to the parent element or the attribute value
3410 ## as string, either literal string when no character reference or
3411 ## entity-replaced string otherwise, in this stage, since any characters
3412 ## that would not be consumed are appended in the data state or in an
3413 ## appropriate attribute value state anyway.
3414
3415 if ($self->{prev_state} == DATA_STATE) {
3416 !!!cp (986);
3417 $self->{state} = $self->{prev_state};
3418 ## Reconsume.
3419 !!!emit ({type => CHARACTER_TOKEN,
3420 data => $data,
3421 line => $self->{line_prev},
3422 column => $self->{column_prev} + 1 - length $self->{s_kwd},
3423 });
3424 redo A;
3425 } else {
3426 !!!cp (985);
3427 $self->{ca}->{value} .= $data;
3428 $self->{ca}->{has_reference} = 1 if $has_ref;
3429 $self->{state} = $self->{prev_state};
3430 ## Reconsume.
3431 redo A;
3432 }
3433 } else {
3434 die "$0: $self->{state}: Unknown state";
3435 }
3436 } # A
3437
3438 die "$0: _get_next_token: unexpected case";
3439 } # _get_next_token
3440
3441 sub _initialize_tree_constructor ($) {
3442 my $self = shift;
3443 ## NOTE: $self->{document} MUST be specified before this method is called
3444 $self->{document}->strict_error_checking (0);
3445 ## TODO: Turn mutation events off # MUST
3446 ## TODO: Turn loose Document option (manakai extension) on
3447 $self->{document}->manakai_is_html (1); # MUST
3448 $self->{document}->set_user_data (manakai_source_line => 1);
3449 $self->{document}->set_user_data (manakai_source_column => 1);
3450 } # _initialize_tree_constructor
3451
3452 sub _terminate_tree_constructor ($) {
3453 my $self = shift;
3454 $self->{document}->strict_error_checking (1);
3455 ## TODO: Turn mutation events on
3456 } # _terminate_tree_constructor
3457
3458 ## ISSUE: Should append_child (for example) in script executed in tree construction stage fire mutation events?
3459
3460 { # tree construction stage
3461 my $token;
3462
3463 sub _construct_tree ($) {
3464 my ($self) = @_;
3465
3466 ## When an interactive UA render the $self->{document} available
3467 ## to the user, or when it begin accepting user input, are
3468 ## not defined.
3469
3470 ## Append a character: collect it and all subsequent consecutive
3471 ## characters and insert one Text node whose data is concatenation
3472 ## of all those characters. # MUST
3473
3474 !!!next-token;
3475
3476 undef $self->{form_element};
3477 undef $self->{head_element};
3478 undef $self->{head_element_inserted};
3479 $self->{open_elements} = [];
3480 undef $self->{inner_html_node};
3481
3482 ## NOTE: The "initial" insertion mode.
3483 $self->_tree_construction_initial; # MUST
3484
3485 ## NOTE: The "before html" insertion mode.
3486 $self->_tree_construction_root_element;
3487 $self->{insertion_mode} = BEFORE_HEAD_IM;
3488
3489 ## NOTE: The "before head" insertion mode and so on.
3490 $self->_tree_construction_main;
3491 } # _construct_tree
3492
3493 sub _tree_construction_initial ($) {
3494 my $self = shift;
3495
3496 ## NOTE: "initial" insertion mode
3497
3498 INITIAL: {
3499 if ($token->{type} == DOCTYPE_TOKEN) {
3500 ## NOTE: Conformance checkers MAY, instead of reporting "not HTML5"
3501 ## error, switch to a conformance checking mode for another
3502 ## language.
3503 my $doctype_name = $token->{name};
3504 $doctype_name = '' unless defined $doctype_name;
3505 $doctype_name =~ tr/a-z/A-Z/; # ASCII case-insensitive
3506 if (not defined $token->{name} or # <!DOCTYPE>
3507 defined $token->{sysid}) {
3508 !!!cp ('t1');
3509 !!!parse-error (type => 'not HTML5', token => $token);
3510 } elsif ($doctype_name ne 'HTML') {
3511 !!!cp ('t2');
3512 !!!parse-error (type => 'not HTML5', token => $token);
3513 } elsif (defined $token->{pubid}) {
3514 if ($token->{pubid} eq 'XSLT-compat') {
3515 !!!cp ('t1.2');
3516 !!!parse-error (type => 'XSLT-compat', token => $token,
3517 level => $self->{level}->{should});
3518 } else {
3519 !!!parse-error (type => 'not HTML5', token => $token);
3520 }
3521 } else {
3522 !!!cp ('t3');
3523 #
3524 }
3525
3526 my $doctype = $self->{document}->create_document_type_definition
3527 ($token->{name}); ## ISSUE: If name is missing (e.g. <!DOCTYPE>)?
3528 ## NOTE: Default value for both |public_id| and |system_id| attributes
3529 ## are empty strings, so that we don't set any value in missing cases.
3530 $doctype->public_id ($token->{pubid}) if defined $token->{pubid};
3531 $doctype->system_id ($token->{sysid}) if defined $token->{sysid};
3532 ## NOTE: Other DocumentType attributes are null or empty lists.
3533 ## ISSUE: internalSubset = null??
3534 $self->{document}->append_child ($doctype);
3535
3536 if ($token->{quirks} or $doctype_name ne 'HTML') {
3537 !!!cp ('t4');
3538 $self->{document}->manakai_compat_mode ('quirks');
3539 } elsif (defined $token->{pubid}) {
3540 my $pubid = $token->{pubid};
3541 $pubid =~ tr/a-z/A-z/;
3542 my $prefix = [
3543 "+//SILMARIL//DTD HTML PRO V0R11 19970101//",
3544 "-//ADVASOFT LTD//DTD HTML 3.0 ASWEDIT + EXTENSIONS//",
3545 "-//AS//DTD HTML 3.0 ASWEDIT + EXTENSIONS//",
3546 "-//IETF//DTD HTML 2.0 LEVEL 1//",
3547 "-//IETF//DTD HTML 2.0 LEVEL 2//",
3548 "-//IETF//DTD HTML 2.0 STRICT LEVEL 1//",
3549 "-//IETF//DTD HTML 2.0 STRICT LEVEL 2//",
3550 "-//IETF//DTD HTML 2.0 STRICT//",
3551 "-//IETF//DTD HTML 2.0//",
3552 "-//IETF//DTD HTML 2.1E//",
3553 "-//IETF//DTD HTML 3.0//",
3554 "-//IETF//DTD HTML 3.2 FINAL//",
3555 "-//IETF//DTD HTML 3.2//",
3556 "-//IETF//DTD HTML 3//",
3557 "-//IETF//DTD HTML LEVEL 0//",
3558 "-//IETF//DTD HTML LEVEL 1//",
3559 "-//IETF//DTD HTML LEVEL 2//",
3560 "-//IETF//DTD HTML LEVEL 3//",
3561 "-//IETF//DTD HTML STRICT LEVEL 0//",
3562 "-//IETF//DTD HTML STRICT LEVEL 1//",
3563 "-//IETF//DTD HTML STRICT LEVEL 2//",
3564 "-//IETF//DTD HTML STRICT LEVEL 3//",
3565 "-//IETF//DTD HTML STRICT//",
3566 "-//IETF//DTD HTML//",
3567 "-//METRIUS//DTD METRIUS PRESENTATIONAL//",
3568 "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 HTML STRICT//",
3569 "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 HTML//",
3570 "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 TABLES//",
3571 "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 HTML STRICT//",
3572 "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 HTML//",
3573 "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 TABLES//",
3574 "-//NETSCAPE COMM. CORP.//DTD HTML//",
3575 "-//NETSCAPE COMM. CORP.//DTD STRICT HTML//",
3576 "-//O'REILLY AND ASSOCIATES//DTD HTML 2.0//",
3577 "-//O'REILLY AND ASSOCIATES//DTD HTML EXTENDED 1.0//",
3578 "-//O'REILLY AND ASSOCIATES//DTD HTML EXTENDED RELAXED 1.0//",
3579 "-//SOFTQUAD SOFTWARE//DTD HOTMETAL PRO 6.0::19990601::EXTENSIONS TO HTML 4.0//",
3580 "-//SOFTQUAD//DTD HOTMETAL PRO 4.0::19971010::EXTENSIONS TO HTML 4.0//",
3581 "-//SPYGLASS//DTD HTML 2.0 EXTENDED//",
3582 "-//SQ//DTD HTML 2.0 HOTMETAL + EXTENSIONS//",
3583 "-//SUN MICROSYSTEMS CORP.//DTD HOTJAVA HTML//",
3584 "-//SUN MICROSYSTEMS CORP.//DTD HOTJAVA STRICT HTML//",
3585 "-//W3C//DTD HTML 3 1995-03-24//",
3586 "-//W3C//DTD HTML 3.2 DRAFT//",
3587 "-//W3C//DTD HTML 3.2 FINAL//",
3588 "-//W3C//DTD HTML 3.2//",
3589 "-//W3C//DTD HTML 3.2S DRAFT//",
3590 "-//W3C//DTD HTML 4.0 FRAMESET//",
3591 "-//W3C//DTD HTML 4.0 TRANSITIONAL//",
3592 "-//W3C//DTD HTML EXPERIMETNAL 19960712//",
3593 "-//W3C//DTD HTML EXPERIMENTAL 970421//",
3594 "-//W3C//DTD W3 HTML//",
3595 "-//W3O//DTD W3 HTML 3.0//",
3596 "-//WEBTECHS//DTD MOZILLA HTML 2.0//",
3597 "-//WEBTECHS//DTD MOZILLA HTML//",
3598 ]; # $prefix
3599 my $match;
3600 for (@$prefix) {
3601 if (substr ($prefix, 0, length $_) eq $_) {
3602 $match = 1;
3603 last;
3604 }
3605 }
3606 if ($match or
3607 $pubid eq "-//W3O//DTD W3 HTML STRICT 3.0//EN//" or
3608 $pubid eq "-/W3C/DTD HTML 4.0 TRANSITIONAL/EN" or
3609 $pubid eq "HTML") {
3610 !!!cp ('t5');
3611 $self->{document}->manakai_compat_mode ('quirks');
3612 } elsif ($pubid =~ m[^-//W3C//DTD HTML 4.01 FRAMESET//] or
3613 $pubid =~ m[^-//W3C//DTD HTML 4.01 TRANSITIONAL//]) {
3614 if (defined $token->{sysid}) {
3615 !!!cp ('t6');
3616 $self->{document}->manakai_compat_mode ('quirks');
3617 } else {
3618 !!!cp ('t7');
3619 $self->{document}->manakai_compat_mode ('limited quirks');
3620 }
3621 } elsif ($pubid =~ m[^-//W3C//DTD XHTML 1.0 FRAMESET//] or
3622 $pubid =~ m[^-//W3C//DTD XHTML 1.0 TRANSITIONAL//]) {
3623 !!!cp ('t8');
3624 $self->{document}->manakai_compat_mode ('limited quirks');
3625 } else {
3626 !!!cp ('t9');
3627 }
3628 } else {
3629 !!!cp ('t10');
3630 }
3631 if (defined $token->{sysid}) {
3632 my $sysid = $token->{sysid};
3633 $sysid =~ tr/A-Z/a-z/;
3634 if ($sysid eq "http://www.ibm.com/data/dtd/v11/ibmxhtml1-transitional.dtd") {
3635 ## NOTE: Ensure that |PUBLIC "(limited quirks)" "(quirks)"| is
3636 ## marked as quirks.
3637 $self->{document}->manakai_compat_mode ('quirks');
3638 !!!cp ('t11');
3639 } else {
3640 !!!cp ('t12');
3641 }
3642 } else {
3643 !!!cp ('t13');
3644 }
3645
3646 ## Go to the "before html" insertion mode.
3647 !!!next-token;
3648 return;
3649 } elsif ({
3650 START_TAG_TOKEN, 1,
3651 END_TAG_TOKEN, 1,
3652 END_OF_FILE_TOKEN, 1,
3653 }->{$token->{type}}) {
3654 !!!cp ('t14');
3655 !!!parse-error (type => 'no DOCTYPE', token => $token);
3656 $self->{document}->manakai_compat_mode ('quirks');
3657 ## Go to the "before html" insertion mode.
3658 ## reprocess
3659 !!!ack-later;
3660 return;
3661 } elsif ($token->{type} == CHARACTER_TOKEN) {
3662 if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
3663 ## Ignore the token
3664
3665 unless (length $token->{data}) {
3666 !!!cp ('t15');
3667 ## Stay in the insertion mode.
3668 !!!next-token;
3669 redo INITIAL;
3670 } else {
3671 !!!cp ('t16');
3672 }
3673 } else {
3674 !!!cp ('t17');
3675 }
3676
3677 !!!parse-error (type => 'no DOCTYPE', token => $token);
3678 $self->{document}->manakai_compat_mode ('quirks');
3679 ## Go to the "before html" insertion mode.
3680 ## reprocess
3681 return;
3682 } elsif ($token->{type} == COMMENT_TOKEN) {
3683 !!!cp ('t18');
3684 my $comment = $self->{document}->create_comment ($token->{data});
3685 $self->{document}->append_child ($comment);
3686
3687 ## Stay in the insertion mode.
3688 !!!next-token;
3689 redo INITIAL;
3690 } else {
3691 die "$0: $token->{type}: Unknown token type";
3692 }
3693 } # INITIAL
3694
3695 die "$0: _tree_construction_initial: This should be never reached";
3696 } # _tree_construction_initial
3697
3698 sub _tree_construction_root_element ($) {
3699 my $self = shift;
3700
3701 ## NOTE: "before html" insertion mode.
3702
3703 B: {
3704 if ($token->{type} == DOCTYPE_TOKEN) {
3705 !!!cp ('t19');
3706 !!!parse-error (type => 'in html:#DOCTYPE', token => $token);
3707 ## Ignore the token
3708 ## Stay in the insertion mode.
3709 !!!next-token;
3710 redo B;
3711 } elsif ($token->{type} == COMMENT_TOKEN) {
3712 !!!cp ('t20');
3713 my $comment = $self->{document}->create_comment ($token->{data});
3714 $self->{document}->append_child ($comment);
3715 ## Stay in the insertion mode.
3716 !!!next-token;
3717 redo B;
3718 } elsif ($token->{type} == CHARACTER_TOKEN) {
3719 if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
3720 ## Ignore the token.
3721
3722 unless (length $token->{data}) {
3723 !!!cp ('t21');
3724 ## Stay in the insertion mode.
3725 !!!next-token;
3726 redo B;
3727 } else {
3728 !!!cp ('t22');
3729 }
3730 } else {
3731 !!!cp ('t23');
3732 }
3733
3734 $self->{application_cache_selection}->(undef);
3735
3736 #
3737 } elsif ($token->{type} == START_TAG_TOKEN) {
3738 if ($token->{tag_name} eq 'html') {
3739 my $root_element;
3740 !!!create-element ($root_element, $HTML_NS, $token->{tag_name}, $token->{attributes}, $token);
3741 $self->{document}->append_child ($root_element);
3742 push @{$self->{open_elements}},
3743 [$root_element, $el_category->{html}];
3744
3745 if ($token->{attributes}->{manifest}) {
3746 !!!cp ('t24');
3747 $self->{application_cache_selection}
3748 ->($token->{attributes}->{manifest}->{value});
3749 ## ISSUE: Spec is unclear on relative references.
3750 ## According to Hixie (#whatwg 2008-03-19), it should be
3751 ## resolved against the base URI of the document in HTML
3752 ## or xml:base of the element in XHTML.
3753 } else {
3754 !!!cp ('t25');
3755 $self->{application_cache_selection}->(undef);
3756 }
3757
3758 !!!nack ('t25c');
3759
3760 !!!next-token;
3761 return; ## Go to the "before head" insertion mode.
3762 } else {
3763 !!!cp ('t25.1');
3764 #
3765 }
3766 } elsif ({
3767 END_TAG_TOKEN, 1,
3768 END_OF_FILE_TOKEN, 1,
3769 }->{$token->{type}}) {
3770 !!!cp ('t26');
3771 #
3772 } else {
3773 die "$0: $token->{type}: Unknown token type";
3774 }
3775
3776 my $root_element;
3777 !!!create-element ($root_element, $HTML_NS, 'html',, $token);
3778 $self->{document}->append_child ($root_element);
3779 push @{$self->{open_elements}}, [$root_element, $el_category->{html}];
3780
3781 $self->{application_cache_selection}->(undef);
3782
3783 ## NOTE: Reprocess the token.
3784 !!!ack-later;
3785 return; ## Go to the "before head" insertion mode.
3786 } # B
3787
3788 die "$0: _tree_construction_root_element: This should never be reached";
3789 } # _tree_construction_root_element
3790
3791 sub _reset_insertion_mode ($) {
3792 my $self = shift;
3793
3794 ## Step 1
3795 my $last;
3796
3797 ## Step 2
3798 my $i = -1;
3799 my $node = $self->{open_elements}->[$i];
3800
3801 ## Step 3
3802 S3: {
3803 if ($self->{open_elements}->[0]->[0] eq $node->[0]) {
3804 $last = 1;
3805 if (defined $self->{inner_html_node}) {
3806 !!!cp ('t28');
3807 $node = $self->{inner_html_node};
3808 } else {
3809 die "_reset_insertion_mode: t27";
3810 }
3811 }
3812
3813 ## Step 4..14
3814 my $new_mode;
3815 if ($node->[1] & FOREIGN_EL) {
3816 !!!cp ('t28.1');
3817 ## NOTE: Strictly spaking, the line below only applies to MathML and
3818 ## SVG elements. Currently the HTML syntax supports only MathML and
3819 ## SVG elements as foreigners.
3820 $new_mode = IN_BODY_IM | IN_FOREIGN_CONTENT_IM;
3821 } elsif ($node->[1] & TABLE_CELL_EL) {
3822 if ($last) {
3823 !!!cp ('t28.2');
3824 #
3825 } else {
3826 !!!cp ('t28.3');
3827 $new_mode = IN_CELL_IM;
3828 }
3829 } else {
3830 !!!cp ('t28.4');
3831 $new_mode = {
3832 select => IN_SELECT_IM,
3833 ## NOTE: |option| and |optgroup| do not set
3834 ## insertion mode to "in select" by themselves.
3835 tr => IN_ROW_IM,
3836 tbody => IN_TABLE_BODY_IM,
3837 thead => IN_TABLE_BODY_IM,
3838 tfoot => IN_TABLE_BODY_IM,
3839 caption => IN_CAPTION_IM,
3840 colgroup => IN_COLUMN_GROUP_IM,
3841 table => IN_TABLE_IM,
3842 head => IN_BODY_IM, # not in head!
3843 body => IN_BODY_IM,
3844 frameset => IN_FRAMESET_IM,
3845 }->{$node->[0]->manakai_local_name};
3846 }
3847 $self->{insertion_mode} = $new_mode and return if defined $new_mode;
3848
3849 ## Step 15
3850 if ($node->[1] & HTML_EL) {
3851 unless (defined $self->{head_element}) {
3852 !!!cp ('t29');
3853 $self->{insertion_mode} = BEFORE_HEAD_IM;
3854 } else {
3855 ## ISSUE: Can this state be reached?
3856 !!!cp ('t30');
3857 $self->{insertion_mode} = AFTER_HEAD_IM;
3858 }
3859 return;
3860 } else {
3861 !!!cp ('t31');
3862 }
3863
3864 ## Step 16
3865 $self->{insertion_mode} = IN_BODY_IM and return if $last;
3866
3867 ## Step 17
3868 $i--;
3869 $node = $self->{open_elements}->[$i];
3870
3871 ## Step 18
3872 redo S3;
3873 } # S3
3874
3875 die "$0: _reset_insertion_mode: This line should never be reached";
3876 } # _reset_insertion_mode
3877
3878 sub _tree_construction_main ($) {
3879 my $self = shift;
3880
3881 my $active_formatting_elements = [];
3882
3883 my $reconstruct_active_formatting_elements = sub { # MUST
3884 my $insert = shift;
3885
3886 ## Step 1
3887 return unless @$active_formatting_elements;
3888
3889 ## Step 3
3890 my $i = -1;
3891 my $entry = $active_formatting_elements->[$i];
3892
3893 ## Step 2
3894 return if $entry->[0] eq '#marker';
3895 for (@{$self->{open_elements}}) {
3896 if ($entry->[0] eq $_->[0]) {
3897 !!!cp ('t32');
3898 return;
3899 }
3900 }
3901
3902 S4: {
3903 ## Step 4
3904 last S4 if $active_formatting_elements->[0]->[0] eq $entry->[0];
3905
3906 ## Step 5
3907 $i--;
3908 $entry = $active_formatting_elements->[$i];
3909
3910 ## Step 6
3911 if ($entry->[0] eq '#marker') {
3912 !!!cp ('t33_1');
3913 #
3914 } else {
3915 my $in_open_elements;
3916 OE: for (@{$self->{open_elements}}) {
3917 if ($entry->[0] eq $_->[0]) {
3918 !!!cp ('t33');
3919 $in_open_elements = 1;
3920 last OE;
3921 }
3922 }
3923 if ($in_open_elements) {
3924 !!!cp ('t34');
3925 #
3926 } else {
3927 ## NOTE: <!DOCTYPE HTML><p><b><i><u></p> <p>X
3928 !!!cp ('t35');
3929 redo S4;
3930 }
3931 }
3932
3933 ## Step 7
3934 $i++;
3935 $entry = $active_formatting_elements->[$i];
3936 } # S4
3937
3938 S7: {
3939 ## Step 8
3940 my $clone = [$entry->[0]->clone_node (0), $entry->[1]];
3941
3942 ## Step 9
3943 $insert->($clone->[0]);
3944 push @{$self->{open_elements}}, $clone;
3945
3946 ## Step 10
3947 $active_formatting_elements->[$i] = $self->{open_elements}->[-1];
3948
3949 ## Step 11
3950 unless ($clone->[0] eq $active_formatting_elements->[-1]->[0]) {
3951 !!!cp ('t36');
3952 ## Step 7'
3953 $i++;
3954 $entry = $active_formatting_elements->[$i];
3955
3956 redo S7;
3957 }
3958
3959 !!!cp ('t37');
3960 } # S7
3961 }; # $reconstruct_active_formatting_elements
3962
3963 my $clear_up_to_marker = sub {
3964 for (reverse 0..$#$active_formatting_elements) {
3965 if ($active_formatting_elements->[$_]->[0] eq '#marker') {
3966 !!!cp ('t38');
3967 splice @$active_formatting_elements, $_;
3968 return;
3969 }
3970 }
3971
3972 !!!cp ('t39');
3973 }; # $clear_up_to_marker
3974
3975 my $insert;
3976
3977 my $parse_rcdata = sub ($) {
3978 my ($content_model_flag) = @_;
3979
3980 ## Step 1
3981 my $start_tag_name = $token->{tag_name};
3982 my $el;
3983 !!!create-element ($el, $HTML_NS, $start_tag_name, $token->{attributes}, $token);
3984
3985 ## Step 2
3986 $insert->($el);
3987
3988 ## Step 3
3989 $self->{content_model} = $content_model_flag; # CDATA or RCDATA
3990 delete $self->{escape}; # MUST
3991
3992 ## Step 4
3993 my $text = '';
3994 !!!nack ('t40.1');
3995 !!!next-token;
3996 while ($token->{type} == CHARACTER_TOKEN) { # or until stop tokenizing
3997 !!!cp ('t40');
3998 $text .= $token->{data};
3999 !!!next-token;
4000 }
4001
4002 ## Step 5
4003 if (length $text) {
4004 !!!cp ('t41');
4005 my $text = $self->{document}->create_text_node ($text);
4006 $el->append_child ($text);
4007 }
4008
4009 ## Step 6
4010 $self->{content_model} = PCDATA_CONTENT_MODEL;
4011
4012 ## Step 7
4013 if ($token->{type} == END_TAG_TOKEN and
4014 $token->{tag_name} eq $start_tag_name) {
4015 !!!cp ('t42');
4016 ## Ignore the token
4017 } else {
4018 ## NOTE: An end-of-file token.
4019 if ($content_model_flag == CDATA_CONTENT_MODEL) {
4020 !!!cp ('t43');
4021 !!!parse-error (type => 'in CDATA:#eof', token => $token);
4022 } elsif ($content_model_flag == RCDATA_CONTENT_MODEL) {
4023 !!!cp ('t44');
4024 !!!parse-error (type => 'in RCDATA:#eof', token => $token);
4025 } else {
4026 die "$0: $content_model_flag in parse_rcdata";
4027 }
4028 }
4029 !!!next-token;
4030 }; # $parse_rcdata
4031
4032 my $script_start_tag = sub () {
4033 my $script_el;
4034 !!!create-element ($script_el, $HTML_NS, 'script', $token->{attributes}, $token);
4035 ## TODO: mark as "parser-inserted"
4036
4037 $self->{content_model} = CDATA_CONTENT_MODEL;
4038 delete $self->{escape}; # MUST
4039
4040 my $text = '';
4041 !!!nack ('t45.1');
4042 !!!next-token;
4043 while ($token->{type} == CHARACTER_TOKEN) {
4044 !!!cp ('t45');
4045 $text .= $token->{data};
4046 !!!next-token;
4047 } # stop if non-character token or tokenizer stops tokenising
4048 if (length $text) {
4049 !!!cp ('t46');
4050 $script_el->manakai_append_text ($text);
4051 }
4052
4053 $self->{content_model} = PCDATA_CONTENT_MODEL;
4054
4055 if ($token->{type} == END_TAG_TOKEN and
4056 $token->{tag_name} eq 'script') {
4057 !!!cp ('t47');
4058 ## Ignore the token
4059 } else {
4060 !!!cp ('t48');
4061 !!!parse-error (type => 'in CDATA:#eof', token => $token);
4062 ## ISSUE: And ignore?
4063 ## TODO: mark as "already executed"
4064 }
4065
4066 if (defined $self->{inner_html_node}) {
4067 !!!cp ('t49');
4068 ## TODO: mark as "already executed"
4069 } else {
4070 !!!cp ('t50');
4071 ## TODO: $old_insertion_point = current insertion point
4072 ## TODO: insertion point = just before the next input character
4073
4074 $insert->($script_el);
4075
4076 ## TODO: insertion point = $old_insertion_point (might be "undefined")
4077
4078 ## TODO: if there is a script that will execute as soon as the parser resume, then...
4079 }
4080
4081 !!!next-token;
4082 }; # $script_start_tag
4083
4084 ## NOTE: $open_tables->[-1]->[0] is the "current table" element node.
4085 ## NOTE: $open_tables->[-1]->[1] is the "tainted" flag.
4086 ## NOTE: $open_tables->[-1]->[2] is set false when non-Text node inserted.
4087 my $open_tables = [[$self->{open_elements}->[0]->[0]]];
4088
4089 my $formatting_end_tag = sub {
4090 my $end_tag_token = shift;
4091 my $tag_name = $end_tag_token->{tag_name};
4092
4093 ## NOTE: The adoption agency algorithm (AAA).
4094
4095 FET: {
4096 ## Step 1
4097 my $formatting_element;
4098 my $formatting_element_i_in_active;
4099 AFE: for (reverse 0..$#$active_formatting_elements) {
4100 if ($active_formatting_elements->[$_]->[0] eq '#marker') {
4101 !!!cp ('t52');
4102 last AFE;
4103 } elsif ($active_formatting_elements->[$_]->[0]->manakai_local_name
4104 eq $tag_name) {
4105 !!!cp ('t51');
4106 $formatting_element = $active_formatting_elements->[$_];
4107 $formatting_element_i_in_active = $_;
4108 last AFE;
4109 }
4110 } # AFE
4111 unless (defined $formatting_element) {
4112 !!!cp ('t53');
4113 !!!parse-error (type => 'unmatched end tag', text => $tag_name, token => $end_tag_token);
4114 ## Ignore the token
4115 !!!next-token;
4116 return;
4117 }
4118 ## has an element in scope
4119 my $in_scope = 1;
4120 my $formatting_element_i_in_open;
4121 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
4122 my $node = $self->{open_elements}->[$_];
4123 if ($node->[0] eq $formatting_element->[0]) {
4124 if ($in_scope) {
4125 !!!cp ('t54');
4126 $formatting_element_i_in_open = $_;
4127 last INSCOPE;
4128 } else { # in open elements but not in scope
4129 !!!cp ('t55');
4130 !!!parse-error (type => 'unmatched end tag',
4131 text => $token->{tag_name},
4132 token => $end_tag_token);
4133 ## Ignore the token
4134 !!!next-token;
4135 return;
4136 }
4137 } elsif ($node->[1] & SCOPING_EL) {
4138 !!!cp ('t56');
4139 $in_scope = 0;
4140 }
4141 } # INSCOPE
4142 unless (defined $formatting_element_i_in_open) {
4143 !!!cp ('t57');
4144 !!!parse-error (type => 'unmatched end tag',
4145 text => $token->{tag_name},
4146 token => $end_tag_token);
4147 pop @$active_formatting_elements; # $formatting_element
4148 !!!next-token; ## TODO: ok?
4149 return;
4150 }
4151 if (not $self->{open_elements}->[-1]->[0] eq $formatting_element->[0]) {
4152 !!!cp ('t58');
4153 !!!parse-error (type => 'not closed',
4154 text => $self->{open_elements}->[-1]->[0]
4155 ->manakai_local_name,
4156 token => $end_tag_token);
4157 }
4158
4159 ## Step 2
4160 my $furthest_block;
4161 my $furthest_block_i_in_open;
4162 OE: for (reverse 0..$#{$self->{open_elements}}) {
4163 my $node = $self->{open_elements}->[$_];
4164 if (not ($node->[1] & FORMATTING_EL) and
4165 #not $phrasing_category->{$node->[1]} and
4166 ($node->[1] & SPECIAL_EL or
4167 $node->[1] & SCOPING_EL)) { ## Scoping is redundant, maybe
4168 !!!cp ('t59');
4169 $furthest_block = $node;
4170 $furthest_block_i_in_open = $_;
4171 ## NOTE: The topmost (eldest) node.
4172 } elsif ($node->[0] eq $formatting_element->[0]) {
4173 !!!cp ('t60');
4174 last OE;
4175 }
4176 } # OE
4177
4178 ## Step 3
4179 unless (defined $furthest_block) { # MUST
4180 !!!cp ('t61');
4181 splice @{$self->{open_elements}}, $formatting_element_i_in_open;
4182 splice @$active_formatting_elements, $formatting_element_i_in_active, 1;
4183 !!!next-token;
4184 return;
4185 }
4186
4187 ## Step 4
4188 my $common_ancestor_node = $self->{open_elements}->[$formatting_element_i_in_open - 1];
4189
4190 ## Step 5
4191 my $furthest_block_parent = $furthest_block->[0]->parent_node;
4192 if (defined $furthest_block_parent) {
4193 !!!cp ('t62');
4194 $furthest_block_parent->remove_child ($furthest_block->[0]);
4195 }
4196
4197 ## Step 6
4198 my $bookmark_prev_el
4199 = $active_formatting_elements->[$formatting_element_i_in_active - 1]
4200 ->[0];
4201
4202 ## Step 7
4203 my $node = $furthest_block;
4204 my $node_i_in_open = $furthest_block_i_in_open;
4205 my $last_node = $furthest_block;
4206 S7: {
4207 ## Step 1
4208 $node_i_in_open--;
4209 $node = $self->{open_elements}->[$node_i_in_open];
4210
4211 ## Step 2
4212 my $node_i_in_active;
4213 S7S2: {
4214 for (reverse 0..$#$active_formatting_elements) {
4215 if ($active_formatting_elements->[$_]->[0] eq $node->[0]) {
4216 !!!cp ('t63');
4217 $node_i_in_active = $_;
4218 last S7S2;
4219 }
4220 }
4221 splice @{$self->{open_elements}}, $node_i_in_open, 1;
4222 redo S7;
4223 } # S7S2
4224
4225 ## Step 3
4226 last S7 if $node->[0] eq $formatting_element->[0];
4227
4228 ## Step 4
4229 if ($last_node->[0] eq $furthest_block->[0]) {
4230 !!!cp ('t64');
4231 $bookmark_prev_el = $node->[0];
4232 }
4233
4234 ## Step 5
4235 if ($node->[0]->has_child_nodes ()) {
4236 !!!cp ('t65');
4237 my $clone = [$node->[0]->clone_node (0), $node->[1]];
4238 $active_formatting_elements->[$node_i_in_active] = $clone;
4239 $self->{open_elements}->[$node_i_in_open] = $clone;
4240 $node = $clone;
4241 }
4242
4243 ## Step 6
4244 $node->[0]->append_child ($last_node->[0]);
4245
4246 ## Step 7
4247 $last_node = $node;
4248
4249 ## Step 8
4250 redo S7;
4251 } # S7
4252
4253 ## Step 8
4254 if ($common_ancestor_node->[1] & TABLE_ROWS_EL) {
4255 my $foster_parent_element;
4256 my $next_sibling;
4257 OE: for (reverse 0..$#{$self->{open_elements}}) {
4258 if ($self->{open_elements}->[$_]->[1] & TABLE_EL) {
4259 my $parent = $self->{open_elements}->[$_]->[0]->parent_node;
4260 if (defined $parent and $parent->node_type == 1) {
4261 !!!cp ('t65.1');
4262 $foster_parent_element = $parent;
4263 $next_sibling = $self->{open_elements}->[$_]->[0];
4264 } else {
4265 !!!cp ('t65.2');
4266 $foster_parent_element
4267 = $self->{open_elements}->[$_ - 1]->[0];
4268 }
4269 last OE;
4270 }
4271 } # OE
4272 $foster_parent_element = $self->{open_elements}->[0]->[0]
4273 unless defined $foster_parent_element;
4274 $foster_parent_element->insert_before ($last_node->[0], $next_sibling);
4275 $open_tables->[-1]->[1] = 1; # tainted
4276 } else {
4277 !!!cp ('t65.3');
4278 $common_ancestor_node->[0]->append_child ($last_node->[0]);
4279 }
4280
4281 ## Step 9
4282 my $clone = [$formatting_element->[0]->clone_node (0),
4283 $formatting_element->[1]];
4284
4285 ## Step 10
4286 my @cn = @{$furthest_block->[0]->child_nodes};
4287 $clone->[0]->append_child ($_) for @cn;
4288
4289 ## Step 11
4290 $furthest_block->[0]->append_child ($clone->[0]);
4291
4292 ## Step 12
4293 my $i;
4294 AFE: for (reverse 0..$#$active_formatting_elements) {
4295 if ($active_formatting_elements->[$_]->[0] eq $formatting_element->[0]) {
4296 !!!cp ('t66');
4297 splice @$active_formatting_elements, $_, 1;
4298 $i-- and last AFE if defined $i;
4299 } elsif ($active_formatting_elements->[$_]->[0] eq $bookmark_prev_el) {
4300 !!!cp ('t67');
4301 $i = $_;
4302 }
4303 } # AFE
4304 splice @$active_formatting_elements, $i + 1, 0, $clone;
4305
4306 ## Step 13
4307 undef $i;
4308 OE: for (reverse 0..$#{$self->{open_elements}}) {
4309 if ($self->{open_elements}->[$_]->[0] eq $formatting_element->[0]) {
4310 !!!cp ('t68');
4311 splice @{$self->{open_elements}}, $_, 1;
4312 $i-- and last OE if defined $i;
4313 } elsif ($self->{open_elements}->[$_]->[0] eq $furthest_block->[0]) {
4314 !!!cp ('t69');
4315 $i = $_;
4316 }
4317 } # OE
4318 splice @{$self->{open_elements}}, $i + 1, 0, $clone;
4319
4320 ## Step 14
4321 redo FET;
4322 } # FET
4323 }; # $formatting_end_tag
4324
4325 $insert = my $insert_to_current = sub {
4326 $self->{open_elements}->[-1]->[0]->append_child ($_[0]);
4327 }; # $insert_to_current
4328
4329 my $insert_to_foster = sub {
4330 my $child = shift;
4331 if ($self->{open_elements}->[-1]->[1] & TABLE_ROWS_EL) {
4332 # MUST
4333 my $foster_parent_element;
4334 my $next_sibling;
4335 OE: for (reverse 0..$#{$self->{open_elements}}) {
4336 if ($self->{open_elements}->[$_]->[1] & TABLE_EL) {
4337 my $parent = $self->{open_elements}->[$_]->[0]->parent_node;
4338 if (defined $parent and $parent->node_type == 1) {
4339 !!!cp ('t70');
4340 $foster_parent_element = $parent;
4341 $next_sibling = $self->{open_elements}->[$_]->[0];
4342 } else {
4343 !!!cp ('t71');
4344 $foster_parent_element
4345 = $self->{open_elements}->[$_ - 1]->[0];
4346 }
4347 last OE;
4348 }
4349 } # OE
4350 $foster_parent_element = $self->{open_elements}->[0]->[0]
4351 unless defined $foster_parent_element;
4352 $foster_parent_element->insert_before
4353 ($child, $next_sibling);
4354 $open_tables->[-1]->[1] = 1; # tainted
4355 } else {
4356 !!!cp ('t72');
4357 $self->{open_elements}->[-1]->[0]->append_child ($child);
4358 }
4359 }; # $insert_to_foster
4360
4361 ## NOTE: When a character is inserted, if the last node that was
4362 ## inserted by the parser is a Text node and the character has to be
4363 ## inserted after that node, then the character is appended to the
4364 ## Text node. However, if any other node is inserted by the parser,
4365 ## then a new Text node is created and the character is appended as
4366 ## that Text node. If I'm not wrong, there are only two cases where
4367 ## this occurs. One is the case where an element node is inserted
4368 ## to the |head| element. This is covered by using the
4369 ## |$self->{head_element_inserted}| flag. Another is the case where
4370 ## an element or comment is inserted into the |table| subtree while
4371 ## foster parenting happens. This is covered by using the [2] flag
4372 ## of the |$open_tables| structure. All other cases are handled
4373 ## simply by calling |manakai_append_text| method.
4374
4375 B: while (1) {
4376 if ($token->{type} == DOCTYPE_TOKEN) {
4377 !!!cp ('t73');
4378 !!!parse-error (type => 'in html:#DOCTYPE', token => $token);
4379 ## Ignore the token
4380 ## Stay in the phase
4381 !!!next-token;
4382 next B;
4383 } elsif ($token->{type} == START_TAG_TOKEN and
4384 $token->{tag_name} eq 'html') {
4385 if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
4386 !!!cp ('t79');
4387 !!!parse-error (type => 'after html', text => 'html', token => $token);
4388 $self->{insertion_mode} = AFTER_BODY_IM;
4389 } elsif ($self->{insertion_mode} == AFTER_HTML_FRAMESET_IM) {
4390 !!!cp ('t80');
4391 !!!parse-error (type => 'after html', text => 'html', token => $token);
4392 $self->{insertion_mode} = AFTER_FRAMESET_IM;
4393 } else {
4394 !!!cp ('t81');
4395 }
4396
4397 !!!cp ('t82');
4398 !!!parse-error (type => 'not first start tag', token => $token);
4399 my $top_el = $self->{open_elements}->[0]->[0];
4400 for my $attr_name (keys %{$token->{attributes}}) {
4401 unless ($top_el->has_attribute_ns (undef, $attr_name)) {
4402 !!!cp ('t84');
4403 $top_el->set_attribute_ns
4404 (undef, [undef, $attr_name],
4405 $token->{attributes}->{$attr_name}->{value});
4406 }
4407 }
4408 !!!nack ('t84.1');
4409 !!!next-token;
4410 next B;
4411 } elsif ($token->{type} == COMMENT_TOKEN) {
4412 my $comment = $self->{document}->create_comment ($token->{data});
4413 if ($self->{insertion_mode} & AFTER_HTML_IMS) {
4414 !!!cp ('t85');
4415 $self->{document}->append_child ($comment);
4416 } elsif ($self->{insertion_mode} == AFTER_BODY_IM) {
4417 !!!cp ('t86');
4418 $self->{open_elements}->[0]->[0]->append_child ($comment);
4419 } else {
4420 !!!cp ('t87');
4421 $self->{open_elements}->[-1]->[0]->append_child ($comment);
4422 $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
4423 }
4424 !!!next-token;
4425 next B;
4426 } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM) {
4427 if ($token->{type} == CHARACTER_TOKEN) {
4428 !!!cp ('t87.1');
4429 $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});
4430 !!!next-token;
4431 next B;
4432 } elsif ($token->{type} == START_TAG_TOKEN) {
4433 if ((not {mglyph => 1, malignmark => 1}->{$token->{tag_name}} and
4434 $self->{open_elements}->[-1]->[1] & FOREIGN_FLOW_CONTENT_EL) or
4435 not ($self->{open_elements}->[-1]->[1] & FOREIGN_EL) or
4436 ($token->{tag_name} eq 'svg' and
4437 $self->{open_elements}->[-1]->[1] & MML_AXML_EL)) {
4438 ## NOTE: "using the rules for secondary insertion mode"then"continue"
4439 !!!cp ('t87.2');
4440 #
4441 } elsif ({
4442 b => 1, big => 1, blockquote => 1, body => 1, br => 1,
4443 center => 1, code => 1, dd => 1, div => 1, dl => 1, dt => 1,
4444 em => 1, embed => 1, font => 1, h1 => 1, h2 => 1, h3 => 1,
4445 h4 => 1, h5 => 1, h6 => 1, head => 1, hr => 1, i => 1,
4446 img => 1, li => 1, listing => 1, menu => 1, meta => 1,
4447 nobr => 1, ol => 1, p => 1, pre => 1, ruby => 1, s => 1,
4448 small => 1, span => 1, strong => 1, strike => 1, sub => 1,
4449 sup => 1, table => 1, tt => 1, u => 1, ul => 1, var => 1,
4450 }->{$token->{tag_name}}) {
4451 !!!cp ('t87.2');
4452 !!!parse-error (type => 'not closed',
4453 text => $self->{open_elements}->[-1]->[0]
4454 ->manakai_local_name,
4455 token => $token);
4456
4457 pop @{$self->{open_elements}}
4458 while $self->{open_elements}->[-1]->[1] & FOREIGN_EL;
4459
4460 $self->{insertion_mode} &= ~ IN_FOREIGN_CONTENT_IM;
4461 ## Reprocess.
4462 next B;
4463 } else {
4464 my $nsuri = $self->{open_elements}->[-1]->[0]->namespace_uri;
4465 my $tag_name = $token->{tag_name};
4466 if ($nsuri eq $SVG_NS) {
4467 $tag_name = {
4468 altglyph => 'altGlyph',
4469 altglyphdef => 'altGlyphDef',
4470 altglyphitem => 'altGlyphItem',
4471 animatecolor => 'animateColor',
4472 animatemotion => 'animateMotion',
4473 animatetransform => 'animateTransform',
4474 clippath => 'clipPath',
4475 feblend => 'feBlend',
4476 fecolormatrix => 'feColorMatrix',
4477 fecomponenttransfer => 'feComponentTransfer',
4478 fecomposite => 'feComposite',
4479 feconvolvematrix => 'feConvolveMatrix',
4480 fediffuselighting => 'feDiffuseLighting',
4481 fedisplacementmap => 'feDisplacementMap',
4482 fedistantlight => 'feDistantLight',
4483 feflood => 'feFlood',
4484 fefunca => 'feFuncA',
4485 fefuncb => 'feFuncB',
4486 fefuncg => 'feFuncG',
4487 fefuncr => 'feFuncR',
4488 fegaussianblur => 'feGaussianBlur',
4489 feimage => 'feImage',
4490 femerge => 'feMerge',
4491 femergenode => 'feMergeNode',
4492 femorphology => 'feMorphology',
4493 feoffset => 'feOffset',
4494 fepointlight => 'fePointLight',
4495 fespecularlighting => 'feSpecularLighting',
4496 fespotlight => 'feSpotLight',
4497 fetile => 'feTile',
4498 feturbulence => 'feTurbulence',
4499 foreignobject => 'foreignObject',
4500 glyphref => 'glyphRef',
4501 lineargradient => 'linearGradient',
4502 radialgradient => 'radialGradient',
4503 #solidcolor => 'solidColor', ## NOTE: Commented in spec (SVG1.2)
4504 textpath => 'textPath',
4505 }->{$tag_name} || $tag_name;
4506 }
4507
4508 ## "adjust SVG attributes" (SVG only) - done in insert-element-f
4509
4510 ## "adjust foreign attributes" - done in insert-element-f
4511
4512 !!!insert-element-f ($nsuri, $tag_name, $token->{attributes}, $token);
4513
4514 if ($self->{self_closing}) {
4515 pop @{$self->{open_elements}};
4516 !!!ack ('t87.3');
4517 } else {
4518 !!!cp ('t87.4');
4519 }
4520
4521 !!!next-token;
4522 next B;
4523 }
4524 } elsif ($token->{type} == END_TAG_TOKEN) {
4525 ## NOTE: "using the rules for secondary insertion mode" then "continue"
4526 !!!cp ('t87.5');
4527 #
4528 } elsif ($token->{type} == END_OF_FILE_TOKEN) {
4529 !!!cp ('t87.6');
4530 !!!parse-error (type => 'not closed',
4531 text => $self->{open_elements}->[-1]->[0]
4532 ->manakai_local_name,
4533 token => $token);
4534
4535 pop @{$self->{open_elements}}
4536 while $self->{open_elements}->[-1]->[1] & FOREIGN_EL;
4537
4538 ## NOTE: |<span><svg>| ... two parse errors, |<svg>| ... a parse error.
4539
4540 $self->{insertion_mode} &= ~ IN_FOREIGN_CONTENT_IM;
4541 ## Reprocess.
4542 next B;
4543 } else {
4544 die "$0: $token->{type}: Unknown token type";
4545 }
4546 }
4547
4548 if ($self->{insertion_mode} & HEAD_IMS) {
4549 if ($token->{type} == CHARACTER_TOKEN) {
4550 if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
4551 unless ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4552 if ($self->{head_element_inserted}) {
4553 !!!cp ('t88.3');
4554 $self->{open_elements}->[-1]->[0]->append_child
4555 ($self->{document}->create_text_node ($1));
4556 delete $self->{head_element_inserted};
4557 ## NOTE: |</head> <link> |
4558 #
4559 } else {
4560 !!!cp ('t88.2');
4561 $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
4562 ## NOTE: |</head> &#x20;|
4563 #
4564 }
4565 } else {
4566 !!!cp ('t88.1');
4567 ## Ignore the token.
4568 #
4569 }
4570 unless (length $token->{data}) {
4571 !!!cp ('t88');
4572 !!!next-token;
4573 next B;
4574 }
4575 ## TODO: set $token->{column} appropriately
4576 }
4577
4578 if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4579 !!!cp ('t89');
4580 ## As if <head>
4581 !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);
4582 $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
4583 push @{$self->{open_elements}},
4584 [$self->{head_element}, $el_category->{head}];
4585
4586 ## Reprocess in the "in head" insertion mode...
4587 pop @{$self->{open_elements}};
4588
4589 ## Reprocess in the "after head" insertion mode...
4590 } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4591 !!!cp ('t90');
4592 ## As if </noscript>
4593 pop @{$self->{open_elements}};
4594 !!!parse-error (type => 'in noscript:#text', token => $token);
4595
4596 ## Reprocess in the "in head" insertion mode...
4597 ## As if </head>
4598 pop @{$self->{open_elements}};
4599
4600 ## Reprocess in the "after head" insertion mode...
4601 } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
4602 !!!cp ('t91');
4603 pop @{$self->{open_elements}};
4604
4605 ## Reprocess in the "after head" insertion mode...
4606 } else {
4607 !!!cp ('t92');
4608 }
4609
4610 ## "after head" insertion mode
4611 ## As if <body>
4612 !!!insert-element ('body',, $token);
4613 $self->{insertion_mode} = IN_BODY_IM;
4614 ## reprocess
4615 next B;
4616 } elsif ($token->{type} == START_TAG_TOKEN) {
4617 if ($token->{tag_name} eq 'head') {
4618 if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4619 !!!cp ('t93');
4620 !!!create-element ($self->{head_element}, $HTML_NS, $token->{tag_name}, $token->{attributes}, $token);
4621 $self->{open_elements}->[-1]->[0]->append_child
4622 ($self->{head_element});
4623 push @{$self->{open_elements}},
4624 [$self->{head_element}, $el_category->{head}];
4625 $self->{insertion_mode} = IN_HEAD_IM;
4626 !!!nack ('t93.1');
4627 !!!next-token;
4628 next B;
4629 } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4630 !!!cp ('t93.2');
4631 !!!parse-error (type => 'after head', text => 'head',
4632 token => $token);
4633 ## Ignore the token
4634 !!!nack ('t93.3');
4635 !!!next-token;
4636 next B;
4637 } else {
4638 !!!cp ('t95');
4639 !!!parse-error (type => 'in head:head',
4640 token => $token); # or in head noscript
4641 ## Ignore the token
4642 !!!nack ('t95.1');
4643 !!!next-token;
4644 next B;
4645 }
4646 } elsif ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4647 !!!cp ('t96');
4648 ## As if <head>
4649 !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);
4650 $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
4651 push @{$self->{open_elements}},
4652 [$self->{head_element}, $el_category->{head}];
4653
4654 $self->{insertion_mode} = IN_HEAD_IM;
4655 ## Reprocess in the "in head" insertion mode...
4656 } else {
4657 !!!cp ('t97');
4658 }
4659
4660 if ($token->{tag_name} eq 'base') {
4661 if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4662 !!!cp ('t98');
4663 ## As if </noscript>
4664 pop @{$self->{open_elements}};
4665 !!!parse-error (type => 'in noscript', text => 'base',
4666 token => $token);
4667
4668 $self->{insertion_mode} = IN_HEAD_IM;
4669 ## Reprocess in the "in head" insertion mode...
4670 } else {
4671 !!!cp ('t99');
4672 }
4673
4674 ## NOTE: There is a "as if in head" code clone.
4675 if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4676 !!!cp ('t100');
4677 !!!parse-error (type => 'after head',
4678 text => $token->{tag_name}, token => $token);
4679 push @{$self->{open_elements}},
4680 [$self->{head_element}, $el_category->{head}];
4681 $self->{head_element_inserted} = 1;
4682 } else {
4683 !!!cp ('t101');
4684 }
4685 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4686 pop @{$self->{open_elements}};
4687 pop @{$self->{open_elements}} # <head>
4688 if $self->{insertion_mode} == AFTER_HEAD_IM;
4689 !!!nack ('t101.1');
4690 !!!next-token;
4691 next B;
4692 } elsif ($token->{tag_name} eq 'link') {
4693 ## NOTE: There is a "as if in head" code clone.
4694 if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4695 !!!cp ('t102');
4696 !!!parse-error (type => 'after head',
4697 text => $token->{tag_name}, token => $token);
4698 push @{$self->{open_elements}},
4699 [$self->{head_element}, $el_category->{head}];
4700 $self->{head_element_inserted} = 1;
4701 } else {
4702 !!!cp ('t103');
4703 }
4704 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4705 pop @{$self->{open_elements}};
4706 pop @{$self->{open_elements}} # <head>
4707 if $self->{insertion_mode} == AFTER_HEAD_IM;
4708 !!!ack ('t103.1');
4709 !!!next-token;
4710 next B;
4711 } elsif ($token->{tag_name} eq 'command' or
4712 $token->{tag_name} eq 'eventsource') {
4713 if ($self->{insertion_mode} == IN_HEAD_IM) {
4714 ## NOTE: If the insertion mode at the time of the emission
4715 ## of the token was "before head", $self->{insertion_mode}
4716 ## is already changed to |IN_HEAD_IM|.
4717
4718 ## NOTE: There is a "as if in head" code clone.
4719 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4720 pop @{$self->{open_elements}};
4721 pop @{$self->{open_elements}} # <head>
4722 if $self->{insertion_mode} == AFTER_HEAD_IM;
4723 !!!ack ('t103.2');
4724 !!!next-token;
4725 next B;
4726 } else {
4727 ## NOTE: "in head noscript" or "after head" insertion mode
4728 ## - in these cases, these tags are treated as same as
4729 ## normal in-body tags.
4730 !!!cp ('t103.3');
4731 #
4732 }
4733 } elsif ($token->{tag_name} eq 'meta') {
4734 ## NOTE: There is a "as if in head" code clone.
4735 if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4736 !!!cp ('t104');
4737 !!!parse-error (type => 'after head',
4738 text => $token->{tag_name}, token => $token);
4739 push @{$self->{open_elements}},
4740 [$self->{head_element}, $el_category->{head}];
4741 $self->{head_element_inserted} = 1;
4742 } else {
4743 !!!cp ('t105');
4744 }
4745 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4746 my $meta_el = pop @{$self->{open_elements}};
4747
4748 unless ($self->{confident}) {
4749 if ($token->{attributes}->{charset}) {
4750 !!!cp ('t106');
4751 ## NOTE: Whether the encoding is supported or not is handled
4752 ## in the {change_encoding} callback.
4753 $self->{change_encoding}
4754 ->($self, $token->{attributes}->{charset}->{value},
4755 $token);
4756
4757 $meta_el->[0]->get_attribute_node_ns (undef, 'charset')
4758 ->set_user_data (manakai_has_reference =>
4759 $token->{attributes}->{charset}
4760 ->{has_reference});
4761 } elsif ($token->{attributes}->{content}) {
4762 if ($token->{attributes}->{content}->{value}
4763 =~ /[Cc][Hh][Aa][Rr][Ss][Ee][Tt]
4764 [\x09\x0A\x0C\x0D\x20]*=
4765 [\x09\x0A\x0C\x0D\x20]*(?>"([^"]*)"|'([^']*)'|
4766 ([^"'\x09\x0A\x0C\x0D\x20]
4767 [^\x09\x0A\x0C\x0D\x20\x3B]*))/x) {
4768 !!!cp ('t107');
4769 ## NOTE: Whether the encoding is supported or not is handled
4770 ## in the {change_encoding} callback.
4771 $self->{change_encoding}
4772 ->($self, defined $1 ? $1 : defined $2 ? $2 : $3,
4773 $token);
4774 $meta_el->[0]->get_attribute_node_ns (undef, 'content')
4775 ->set_user_data (manakai_has_reference =>
4776 $token->{attributes}->{content}
4777 ->{has_reference});
4778 } else {
4779 !!!cp ('t108');
4780 }
4781 }
4782 } else {
4783 if ($token->{attributes}->{charset}) {
4784 !!!cp ('t109');
4785 $meta_el->[0]->get_attribute_node_ns (undef, 'charset')
4786 ->set_user_data (manakai_has_reference =>
4787 $token->{attributes}->{charset}
4788 ->{has_reference});
4789 }
4790 if ($token->{attributes}->{content}) {
4791 !!!cp ('t110');
4792 $meta_el->[0]->get_attribute_node_ns (undef, 'content')
4793 ->set_user_data (manakai_has_reference =>
4794 $token->{attributes}->{content}
4795 ->{has_reference});
4796 }
4797 }
4798
4799 pop @{$self->{open_elements}} # <head>
4800 if $self->{insertion_mode} == AFTER_HEAD_IM;
4801 !!!ack ('t110.1');
4802 !!!next-token;
4803 next B;
4804 } elsif ($token->{tag_name} eq 'title') {
4805 if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4806 !!!cp ('t111');
4807 ## As if </noscript>
4808 pop @{$self->{open_elements}};
4809 !!!parse-error (type => 'in noscript', text => 'title',
4810 token => $token);
4811
4812 $self->{insertion_mode} = IN_HEAD_IM;
4813 ## Reprocess in the "in head" insertion mode...
4814 } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4815 !!!cp ('t112');
4816 !!!parse-error (type => 'after head',
4817 text => $token->{tag_name}, token => $token);
4818 push @{$self->{open_elements}},
4819 [$self->{head_element}, $el_category->{head}];
4820 $self->{head_element_inserted} = 1;
4821 } else {
4822 !!!cp ('t113');
4823 }
4824
4825 ## NOTE: There is a "as if in head" code clone.
4826 $parse_rcdata->(RCDATA_CONTENT_MODEL);
4827 pop @{$self->{open_elements}} # <head>
4828 if $self->{insertion_mode} == AFTER_HEAD_IM;
4829 next B;
4830 } elsif ($token->{tag_name} eq 'style' or
4831 $token->{tag_name} eq 'noframes') {
4832 ## NOTE: Or (scripting is enabled and tag_name eq 'noscript' and
4833 ## insertion mode IN_HEAD_IM)
4834 ## NOTE: There is a "as if in head" code clone.
4835 if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4836 !!!cp ('t114');
4837 !!!parse-error (type => 'after head',
4838 text => $token->{tag_name}, token => $token);
4839 push @{$self->{open_elements}},
4840 [$self->{head_element}, $el_category->{head}];
4841 $self->{head_element_inserted} = 1;
4842 } else {
4843 !!!cp ('t115');
4844 }
4845 $parse_rcdata->(CDATA_CONTENT_MODEL);
4846 pop @{$self->{open_elements}} # <head>
4847 if $self->{insertion_mode} == AFTER_HEAD_IM;
4848 next B;
4849 } elsif ($token->{tag_name} eq 'noscript') {
4850 if ($self->{insertion_mode} == IN_HEAD_IM) {
4851 !!!cp ('t116');
4852 ## NOTE: and scripting is disalbed
4853 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4854 $self->{insertion_mode} = IN_HEAD_NOSCRIPT_IM;
4855 !!!nack ('t116.1');
4856 !!!next-token;
4857 next B;
4858 } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4859 !!!cp ('t117');
4860 !!!parse-error (type => 'in noscript', text => 'noscript',
4861 token => $token);
4862 ## Ignore the token
4863 !!!nack ('t117.1');
4864 !!!next-token;
4865 next B;
4866 } else {
4867 !!!cp ('t118');
4868 #
4869 }
4870 } elsif ($token->{tag_name} eq 'script') {
4871 if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4872 !!!cp ('t119');
4873 ## As if </noscript>
4874 pop @{$self->{open_elements}};
4875 !!!parse-error (type => 'in noscript', text => 'script',
4876 token => $token);
4877
4878 $self->{insertion_mode} = IN_HEAD_IM;
4879 ## Reprocess in the "in head" insertion mode...
4880 } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4881 !!!cp ('t120');
4882 !!!parse-error (type => 'after head',
4883 text => $token->{tag_name}, token => $token);
4884 push @{$self->{open_elements}},
4885 [$self->{head_element}, $el_category->{head}];
4886 $self->{head_element_inserted} = 1;
4887 } else {
4888 !!!cp ('t121');
4889 }
4890
4891 ## NOTE: There is a "as if in head" code clone.
4892 $script_start_tag->();
4893 pop @{$self->{open_elements}} # <head>
4894 if $self->{insertion_mode} == AFTER_HEAD_IM;
4895 next B;
4896 } elsif ($token->{tag_name} eq 'body' or
4897 $token->{tag_name} eq 'frameset') {
4898 if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4899 !!!cp ('t122');
4900 ## As if </noscript>
4901 pop @{$self->{open_elements}};
4902 !!!parse-error (type => 'in noscript',
4903 text => $token->{tag_name}, token => $token);
4904
4905 ## Reprocess in the "in head" insertion mode...
4906 ## As if </head>
4907 pop @{$self->{open_elements}};
4908
4909 ## Reprocess in the "after head" insertion mode...
4910 } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
4911 !!!cp ('t124');
4912 pop @{$self->{open_elements}};
4913
4914 ## Reprocess in the "after head" insertion mode...
4915 } else {
4916 !!!cp ('t125');
4917 }
4918
4919 ## "after head" insertion mode
4920 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4921 if ($token->{tag_name} eq 'body') {
4922 !!!cp ('t126');
4923 $self->{insertion_mode} = IN_BODY_IM;
4924 } elsif ($token->{tag_name} eq 'frameset') {
4925 !!!cp ('t127');
4926 $self->{insertion_mode} = IN_FRAMESET_IM;
4927 } else {
4928 die "$0: tag name: $self->{tag_name}";
4929 }
4930 !!!nack ('t127.1');
4931 !!!next-token;
4932 next B;
4933 } else {
4934 !!!cp ('t128');
4935 #
4936 }
4937
4938 if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4939 !!!cp ('t129');
4940 ## As if </noscript>
4941 pop @{$self->{open_elements}};
4942 !!!parse-error (type => 'in noscript:/',
4943 text => $token->{tag_name}, token => $token);
4944
4945 ## Reprocess in the "in head" insertion mode...
4946 ## As if </head>
4947 pop @{$self->{open_elements}};
4948
4949 ## Reprocess in the "after head" insertion mode...
4950 } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
4951 !!!cp ('t130');
4952 ## As if </head>
4953 pop @{$self->{open_elements}};
4954
4955 ## Reprocess in the "after head" insertion mode...
4956 } else {
4957 !!!cp ('t131');
4958 }
4959
4960 ## "after head" insertion mode
4961 ## As if <body>
4962 !!!insert-element ('body',, $token);
4963 $self->{insertion_mode} = IN_BODY_IM;
4964 ## reprocess
4965 !!!ack-later;
4966 next B;
4967 } elsif ($token->{type} == END_TAG_TOKEN) {
4968 if ($token->{tag_name} eq 'head') {
4969 if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4970 !!!cp ('t132');
4971 ## As if <head>
4972 !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);
4973 $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
4974 push @{$self->{open_elements}},
4975 [$self->{head_element}, $el_category->{head}];
4976
4977 ## Reprocess in the "in head" insertion mode...
4978 pop @{$self->{open_elements}};
4979 $self->{insertion_mode} = AFTER_HEAD_IM;
4980 !!!next-token;
4981 next B;
4982 } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4983 !!!cp ('t133');
4984 ## As if </noscript>
4985 pop @{$self->{open_elements}};
4986 !!!parse-error (type => 'in noscript:/',
4987 text => 'head', token => $token);
4988
4989 ## Reprocess in the "in head" insertion mode...
4990 pop @{$self->{open_elements}};
4991 $self->{insertion_mode} = AFTER_HEAD_IM;
4992 !!!next-token;
4993 next B;
4994 } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
4995 !!!cp ('t134');
4996 pop @{$self->{open_elements}};
4997 $self->{insertion_mode} = AFTER_HEAD_IM;
4998 !!!next-token;
4999 next B;
5000 } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
5001 !!!cp ('t134.1');
5002 !!!parse-error (type => 'unmatched end tag', text => 'head',
5003 token => $token);
5004 ## Ignore the token
5005 !!!next-token;
5006 next B;
5007 } else {
5008 die "$0: $self->{insertion_mode}: Unknown insertion mode";
5009 }
5010 } elsif ($token->{tag_name} eq 'noscript') {
5011 if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
5012 !!!cp ('t136');
5013 pop @{$self->{open_elements}};
5014 $self->{insertion_mode} = IN_HEAD_IM;
5015 !!!next-token;
5016 next B;
5017 } elsif ($self->{insertion_mode} == BEFORE_HEAD_IM or
5018 $self->{insertion_mode} == AFTER_HEAD_IM) {
5019 !!!cp ('t137');
5020 !!!parse-error (type => 'unmatched end tag',
5021 text => 'noscript', token => $token);
5022 ## Ignore the token ## ISSUE: An issue in the spec.
5023 !!!next-token;
5024 next B;
5025 } else {
5026 !!!cp ('t138');
5027 #
5028 }
5029 } elsif ({
5030 body => 1, html => 1,
5031 }->{$token->{tag_name}}) {
5032 ## TODO: This branch is entirely redundant.
5033 if ($self->{insertion_mode} == BEFORE_HEAD_IM or
5034 $self->{insertion_mode} == IN_HEAD_IM or
5035 $self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
5036 !!!cp ('t140');
5037 !!!parse-error (type => 'unmatched end tag',
5038 text => $token->{tag_name}, token => $token);
5039 ## Ignore the token
5040 !!!next-token;
5041 next B;
5042 } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
5043 !!!cp ('t140.1');
5044 !!!parse-error (type => 'unmatched end tag',
5045 text => $token->{tag_name}, token => $token);
5046 ## Ignore the token
5047 !!!next-token;
5048 next B;
5049 } else {
5050 die "$0: $self->{insertion_mode}: Unknown insertion mode";
5051 }
5052 } elsif ($token->{tag_name} eq 'p') {
5053 !!!cp ('t142');
5054 !!!parse-error (type => 'unmatched end tag',
5055 text => $token->{tag_name}, token => $token);
5056 ## Ignore the token
5057 !!!next-token;
5058 next B;
5059 } elsif ($token->{tag_name} eq 'br') {
5060 if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
5061 !!!cp ('t142.2');
5062 ## (before head) as if <head>, (in head) as if </head>
5063 !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);
5064 $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
5065 $self->{insertion_mode} = AFTER_HEAD_IM;
5066
5067 ## Reprocess in the "after head" insertion mode...
5068 } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
5069 !!!cp ('t143.2');
5070 ## As if </head>
5071 pop @{$self->{open_elements}};
5072 $self->{insertion_mode} = AFTER_HEAD_IM;
5073
5074 ## Reprocess in the "after head" insertion mode...
5075 } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
5076 !!!cp ('t143.3');
5077 ## ISSUE: Two parse errors for <head><noscript></br>
5078 !!!parse-error (type => 'unmatched end tag',
5079 text => 'br', token => $token);
5080 ## As if </noscript>
5081 pop @{$self->{open_elements}};
5082 $self->{insertion_mode} = IN_HEAD_IM;
5083
5084 ## Reprocess in the "in head" insertion mode...
5085 ## As if </head>
5086 pop @{$self->{open_elements}};
5087 $self->{insertion_mode} = AFTER_HEAD_IM;
5088
5089 ## Reprocess in the "after head" insertion mode...
5090 } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
5091 !!!cp ('t143.4');
5092 #
5093 } else {
5094 die "$0: $self->{insertion_mode}: Unknown insertion mode";
5095 }
5096
5097 ## ISSUE: does not agree with IE7 - it doesn't ignore </br>.
5098 !!!parse-error (type => 'unmatched end tag',
5099 text => 'br', token => $token);
5100 ## Ignore the token
5101 !!!next-token;
5102 next B;
5103 } else {
5104 !!!cp ('t145');
5105 !!!parse-error (type => 'unmatched end tag',
5106 text => $token->{tag_name}, token => $token);
5107 ## Ignore the token
5108 !!!next-token;
5109 next B;
5110 }
5111
5112 if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
5113 !!!cp ('t146');
5114 ## As if </noscript>
5115 pop @{$self->{open_elements}};
5116 !!!parse-error (type => 'in noscript:/',
5117 text => $token->{tag_name}, token => $token);
5118
5119 ## Reprocess in the "in head" insertion mode...
5120 ## As if </head>
5121 pop @{$self->{open_elements}};
5122
5123 ## Reprocess in the "after head" insertion mode...
5124 } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
5125 !!!cp ('t147');
5126 ## As if </head>
5127 pop @{$self->{open_elements}};
5128
5129 ## Reprocess in the "after head" insertion mode...
5130 } elsif ($self->{insertion_mode} == BEFORE_HEAD_IM) {
5131 ## ISSUE: This case cannot be reached?
5132 !!!cp ('t148');
5133 !!!parse-error (type => 'unmatched end tag',
5134 text => $token->{tag_name}, token => $token);
5135 ## Ignore the token ## ISSUE: An issue in the spec.
5136 !!!next-token;
5137 next B;
5138 } else {
5139 !!!cp ('t149');
5140 }
5141
5142 ## "after head" insertion mode
5143 ## As if <body>
5144 !!!insert-element ('body',, $token);
5145 $self->{insertion_mode} = IN_BODY_IM;
5146 ## reprocess
5147 next B;
5148 } elsif ($token->{type} == END_OF_FILE_TOKEN) {
5149 if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
5150 !!!cp ('t149.1');
5151
5152 ## NOTE: As if <head>
5153 !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);
5154 $self->{open_elements}->[-1]->[0]->append_child
5155 ($self->{head_element});
5156 #push @{$self->{open_elements}},
5157 # [$self->{head_element}, $el_category->{head}];
5158 #$self->{insertion_mode} = IN_HEAD_IM;
5159 ## NOTE: Reprocess.
5160
5161 ## NOTE: As if </head>
5162 #pop @{$self->{open_elements}};
5163 #$self->{insertion_mode} = IN_AFTER_HEAD_IM;
5164 ## NOTE: Reprocess.
5165
5166 #
5167 } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
5168 !!!cp ('t149.2');
5169
5170 ## NOTE: As if </head>
5171 pop @{$self->{open_elements}};
5172 #$self->{insertion_mode} = IN_AFTER_HEAD_IM;
5173 ## NOTE: Reprocess.
5174
5175 #
5176 } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
5177 !!!cp ('t149.3');
5178
5179 !!!parse-error (type => 'in noscript:#eof', token => $token);
5180
5181 ## As if </noscript>
5182 pop @{$self->{open_elements}};
5183 #$self->{insertion_mode} = IN_HEAD_IM;
5184 ## NOTE: Reprocess.
5185
5186 ## NOTE: As if </head>
5187 pop @{$self->{open_elements}};
5188 #$self->{insertion_mode} = IN_AFTER_HEAD_IM;
5189 ## NOTE: Reprocess.
5190
5191 #
5192 } else {
5193 !!!cp ('t149.4');
5194 #
5195 }
5196
5197 ## NOTE: As if <body>
5198 !!!insert-element ('body',, $token);
5199 $self->{insertion_mode} = IN_BODY_IM;
5200 ## NOTE: Reprocess.
5201 next B;
5202 } else {
5203 die "$0: $token->{type}: Unknown token type";
5204 }
5205 } elsif ($self->{insertion_mode} & BODY_IMS) {
5206 if ($token->{type} == CHARACTER_TOKEN) {
5207 !!!cp ('t150');
5208 ## NOTE: There is a code clone of "character in body".
5209 $reconstruct_active_formatting_elements->($insert_to_current);
5210
5211 $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});
5212
5213 !!!next-token;
5214 next B;
5215 } elsif ($token->{type} == START_TAG_TOKEN) {
5216 if ({
5217 caption => 1, col => 1, colgroup => 1, tbody => 1,
5218 td => 1, tfoot => 1, th => 1, thead => 1, tr => 1,
5219 }->{$token->{tag_name}}) {
5220 if ($self->{insertion_mode} == IN_CELL_IM) {
5221 ## have an element in table scope
5222 for (reverse 0..$#{$self->{open_elements}}) {
5223 my $node = $self->{open_elements}->[$_];
5224 if ($node->[1] & TABLE_CELL_EL) {
5225 !!!cp ('t151');
5226
5227 ## Close the cell
5228 !!!back-token; # <x>
5229 $token = {type => END_TAG_TOKEN,
5230 tag_name => $node->[0]->manakai_local_name,
5231 line => $token->{line},
5232 column => $token->{column}};
5233 next B;
5234 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5235 !!!cp ('t152');
5236 ## ISSUE: This case can never be reached, maybe.
5237 last;
5238 }
5239 }
5240
5241 !!!cp ('t153');
5242 !!!parse-error (type => 'start tag not allowed',
5243 text => $token->{tag_name}, token => $token);
5244 ## Ignore the token
5245 !!!nack ('t153.1');
5246 !!!next-token;
5247 next B;
5248 } elsif ($self->{insertion_mode} == IN_CAPTION_IM) {
5249 !!!parse-error (type => 'not closed', text => 'caption',
5250 token => $token);
5251
5252 ## NOTE: As if </caption>.
5253 ## have a table element in table scope
5254 my $i;
5255 INSCOPE: {
5256 for (reverse 0..$#{$self->{open_elements}}) {
5257 my $node = $self->{open_elements}->[$_];
5258 if ($node->[1] & CAPTION_EL) {
5259 !!!cp ('t155');
5260 $i = $_;
5261 last INSCOPE;
5262 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5263 !!!cp ('t156');
5264 last;
5265 }
5266 }
5267
5268 !!!cp ('t157');
5269 !!!parse-error (type => 'start tag not allowed',
5270 text => $token->{tag_name}, token => $token);
5271 ## Ignore the token
5272 !!!nack ('t157.1');
5273 !!!next-token;
5274 next B;
5275 } # INSCOPE
5276
5277 ## generate implied end tags
5278 while ($self->{open_elements}->[-1]->[1]
5279 & END_TAG_OPTIONAL_EL) {
5280 !!!cp ('t158');
5281 pop @{$self->{open_elements}};
5282 }
5283
5284 unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {
5285 !!!cp ('t159');
5286 !!!parse-error (type => 'not closed',
5287 text => $self->{open_elements}->[-1]->[0]
5288 ->manakai_local_name,
5289 token => $token);
5290 } else {
5291 !!!cp ('t160');
5292 }
5293
5294 splice @{$self->{open_elements}}, $i;
5295
5296 $clear_up_to_marker->();
5297
5298 $self->{insertion_mode} = IN_TABLE_IM;
5299
5300 ## reprocess
5301 !!!ack-later;
5302 next B;
5303 } else {
5304 !!!cp ('t161');
5305 #
5306 }
5307 } else {
5308 !!!cp ('t162');
5309 #
5310 }
5311 } elsif ($token->{type} == END_TAG_TOKEN) {
5312 if ($token->{tag_name} eq 'td' or $token->{tag_name} eq 'th') {
5313 if ($self->{insertion_mode} == IN_CELL_IM) {
5314 ## have an element in table scope
5315 my $i;
5316 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5317 my $node = $self->{open_elements}->[$_];
5318 if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
5319 !!!cp ('t163');
5320 $i = $_;
5321 last INSCOPE;
5322 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5323 !!!cp ('t164');
5324 last INSCOPE;
5325 }
5326 } # INSCOPE
5327 unless (defined $i) {
5328 !!!cp ('t165');
5329 !!!parse-error (type => 'unmatched end tag',
5330 text => $token->{tag_name},
5331 token => $token);
5332 ## Ignore the token
5333 !!!next-token;
5334 next B;
5335 }
5336
5337 ## generate implied end tags
5338 while ($self->{open_elements}->[-1]->[1]
5339 & END_TAG_OPTIONAL_EL) {
5340 !!!cp ('t166');
5341 pop @{$self->{open_elements}};
5342 }
5343
5344 if ($self->{open_elements}->[-1]->[0]->manakai_local_name
5345 ne $token->{tag_name}) {
5346 !!!cp ('t167');
5347 !!!parse-error (type => 'not closed',
5348 text => $self->{open_elements}->[-1]->[0]
5349 ->manakai_local_name,
5350 token => $token);
5351 } else {
5352 !!!cp ('t168');
5353 }
5354
5355 splice @{$self->{open_elements}}, $i;
5356
5357 $clear_up_to_marker->();
5358
5359 $self->{insertion_mode} = IN_ROW_IM;
5360
5361 !!!next-token;
5362 next B;
5363 } elsif ($self->{insertion_mode} == IN_CAPTION_IM) {
5364 !!!cp ('t169');
5365 !!!parse-error (type => 'unmatched end tag',
5366 text => $token->{tag_name}, token => $token);
5367 ## Ignore the token
5368 !!!next-token;
5369 next B;
5370 } else {
5371 !!!cp ('t170');
5372 #
5373 }
5374 } elsif ($token->{tag_name} eq 'caption') {
5375 if ($self->{insertion_mode} == IN_CAPTION_IM) {
5376 ## have a table element in table scope
5377 my $i;
5378 INSCOPE: {
5379 for (reverse 0..$#{$self->{open_elements}}) {
5380 my $node = $self->{open_elements}->[$_];
5381 if ($node->[1] & CAPTION_EL) {
5382 !!!cp ('t171');
5383 $i = $_;
5384 last INSCOPE;
5385 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5386 !!!cp ('t172');
5387 last;
5388 }
5389 }
5390
5391 !!!cp ('t173');
5392 !!!parse-error (type => 'unmatched end tag',
5393 text => $token->{tag_name}, token => $token);
5394 ## Ignore the token
5395 !!!next-token;
5396 next B;
5397 } # INSCOPE
5398
5399 ## generate implied end tags
5400 while ($self->{open_elements}->[-1]->[1]
5401 & END_TAG_OPTIONAL_EL) {
5402 !!!cp ('t174');
5403 pop @{$self->{open_elements}};
5404 }
5405
5406 unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {
5407 !!!cp ('t175');
5408 !!!parse-error (type => 'not closed',
5409 text => $self->{open_elements}->[-1]->[0]
5410 ->manakai_local_name,
5411 token => $token);
5412 } else {
5413 !!!cp ('t176');
5414 }
5415
5416 splice @{$self->{open_elements}}, $i;
5417
5418 $clear_up_to_marker->();
5419
5420 $self->{insertion_mode} = IN_TABLE_IM;
5421
5422 !!!next-token;
5423 next B;
5424 } elsif ($self->{insertion_mode} == IN_CELL_IM) {
5425 !!!cp ('t177');
5426 !!!parse-error (type => 'unmatched end tag',
5427 text => $token->{tag_name}, token => $token);
5428 ## Ignore the token
5429 !!!next-token;
5430 next B;
5431 } else {
5432 !!!cp ('t178');
5433 #
5434 }
5435 } elsif ({
5436 table => 1, tbody => 1, tfoot => 1,
5437 thead => 1, tr => 1,
5438 }->{$token->{tag_name}} and
5439 $self->{insertion_mode} == IN_CELL_IM) {
5440 ## have an element in table scope
5441 my $i;
5442 my $tn;
5443 INSCOPE: {
5444 for (reverse 0..$#{$self->{open_elements}}) {
5445 my $node = $self->{open_elements}->[$_];
5446 if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
5447 !!!cp ('t179');
5448 $i = $_;
5449
5450 ## Close the cell
5451 !!!back-token; # </x>
5452 $token = {type => END_TAG_TOKEN, tag_name => $tn,
5453 line => $token->{line},
5454 column => $token->{column}};
5455 next B;
5456 } elsif ($node->[1] & TABLE_CELL_EL) {
5457 !!!cp ('t180');
5458 $tn = $node->[0]->manakai_local_name;
5459 ## NOTE: There is exactly one |td| or |th| element
5460 ## in scope in the stack of open elements by definition.
5461 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5462 ## ISSUE: Can this be reached?
5463 !!!cp ('t181');
5464 last;
5465 }
5466 }
5467
5468 !!!cp ('t182');
5469 !!!parse-error (type => 'unmatched end tag',
5470 text => $token->{tag_name}, token => $token);
5471 ## Ignore the token
5472 !!!next-token;
5473 next B;
5474 } # INSCOPE
5475 } elsif ($token->{tag_name} eq 'table' and
5476 $self->{insertion_mode} == IN_CAPTION_IM) {
5477 !!!parse-error (type => 'not closed', text => 'caption',
5478 token => $token);
5479
5480 ## As if </caption>
5481 ## have a table element in table scope
5482 my $i;
5483 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5484 my $node = $self->{open_elements}->[$_];
5485 if ($node->[1] & CAPTION_EL) {
5486 !!!cp ('t184');
5487 $i = $_;
5488 last INSCOPE;
5489 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5490 !!!cp ('t185');
5491 last INSCOPE;
5492 }
5493 } # INSCOPE
5494 unless (defined $i) {
5495 !!!cp ('t186');
5496 !!!parse-error (type => 'unmatched end tag',
5497 text => 'caption', token => $token);
5498 ## Ignore the token
5499 !!!next-token;
5500 next B;
5501 }
5502
5503 ## generate implied end tags
5504 while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
5505 !!!cp ('t187');
5506 pop @{$self->{open_elements}};
5507 }
5508
5509 unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {
5510 !!!cp ('t188');
5511 !!!parse-error (type => 'not closed',
5512 text => $self->{open_elements}->[-1]->[0]
5513 ->manakai_local_name,
5514 token => $token);
5515 } else {
5516 !!!cp ('t189');
5517 }
5518
5519 splice @{$self->{open_elements}}, $i;
5520
5521 $clear_up_to_marker->();
5522
5523 $self->{insertion_mode} = IN_TABLE_IM;
5524
5525 ## reprocess
5526 next B;
5527 } elsif ({
5528 body => 1, col => 1, colgroup => 1, html => 1,
5529 }->{$token->{tag_name}}) {
5530 if ($self->{insertion_mode} & BODY_TABLE_IMS) {
5531 !!!cp ('t190');
5532 !!!parse-error (type => 'unmatched end tag',
5533 text => $token->{tag_name}, token => $token);
5534 ## Ignore the token
5535 !!!next-token;
5536 next B;
5537 } else {
5538 !!!cp ('t191');
5539 #
5540 }
5541 } elsif ({
5542 tbody => 1, tfoot => 1,
5543 thead => 1, tr => 1,
5544 }->{$token->{tag_name}} and
5545 $self->{insertion_mode} == IN_CAPTION_IM) {
5546 !!!cp ('t192');
5547 !!!parse-error (type => 'unmatched end tag',
5548 text => $token->{tag_name}, token => $token);
5549 ## Ignore the token
5550 !!!next-token;
5551 next B;
5552 } else {
5553 !!!cp ('t193');
5554 #
5555 }
5556 } elsif ($token->{type} == END_OF_FILE_TOKEN) {
5557 for my $entry (@{$self->{open_elements}}) {
5558 unless ($entry->[1] & ALL_END_TAG_OPTIONAL_EL) {
5559 !!!cp ('t75');
5560 !!!parse-error (type => 'in body:#eof', token => $token);
5561 last;
5562 }
5563 }
5564
5565 ## Stop parsing.
5566 last B;
5567 } else {
5568 die "$0: $token->{type}: Unknown token type";
5569 }
5570
5571 $insert = $insert_to_current;
5572 #
5573 } elsif ($self->{insertion_mode} & TABLE_IMS) {
5574 if ($token->{type} == CHARACTER_TOKEN) {
5575 if (not $open_tables->[-1]->[1] and # tainted
5576 $token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
5577 $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
5578
5579 unless (length $token->{data}) {
5580 !!!cp ('t194');
5581 !!!next-token;
5582 next B;
5583 } else {
5584 !!!cp ('t195');
5585 }
5586 }
5587
5588 !!!parse-error (type => 'in table:#text', token => $token);
5589
5590 ## NOTE: As if in body, but insert into the foster parent element.
5591 $reconstruct_active_formatting_elements->($insert_to_foster);
5592
5593 if ($self->{open_elements}->[-1]->[1] & TABLE_ROWS_EL) {
5594 # MUST
5595 my $foster_parent_element;
5596 my $next_sibling;
5597 my $prev_sibling;
5598 OE: for (reverse 0..$#{$self->{open_elements}}) {
5599 if ($self->{open_elements}->[$_]->[1] & TABLE_EL) {
5600 my $parent = $self->{open_elements}->[$_]->[0]->parent_node;
5601 if (defined $parent and $parent->node_type == 1) {
5602 $foster_parent_element = $parent;
5603 !!!cp ('t196');
5604 $next_sibling = $self->{open_elements}->[$_]->[0];
5605 $prev_sibling = $next_sibling->previous_sibling;
5606 #
5607 } else {
5608 !!!cp ('t197');
5609 $foster_parent_element = $self->{open_elements}->[$_ - 1]->[0];
5610 $prev_sibling = $foster_parent_element->last_child;
5611 #
5612 }
5613 last OE;
5614 }
5615 } # OE
5616 $foster_parent_element = $self->{open_elements}->[0]->[0] and
5617 $prev_sibling = $foster_parent_element->last_child
5618 unless defined $foster_parent_element;
5619 undef $prev_sibling unless $open_tables->[-1]->[2]; # ~node inserted
5620 if (defined $prev_sibling and
5621 $prev_sibling->node_type == 3) {
5622 !!!cp ('t198');
5623 $prev_sibling->manakai_append_text ($token->{data});
5624 } else {
5625 !!!cp ('t199');
5626 $foster_parent_element->insert_before
5627 ($self->{document}->create_text_node ($token->{data}),
5628 $next_sibling);
5629 }
5630 $open_tables->[-1]->[1] = 1; # tainted
5631 $open_tables->[-1]->[2] = 1; # ~node inserted
5632 } else {
5633 ## NOTE: Fragment case or in a foster parent'ed element
5634 ## (e.g. |<table><span>a|). In fragment case, whether the
5635 ## character is appended to existing node or a new node is
5636 ## created is irrelevant, since the foster parent'ed nodes
5637 ## are discarded and fragment parsing does not invoke any
5638 ## script.
5639 !!!cp ('t200');
5640 $self->{open_elements}->[-1]->[0]->manakai_append_text
5641 ($token->{data});
5642 }
5643
5644 !!!next-token;
5645 next B;
5646 } elsif ($token->{type} == START_TAG_TOKEN) {
5647 if ({
5648 tr => ($self->{insertion_mode} != IN_ROW_IM),
5649 th => 1, td => 1,
5650 }->{$token->{tag_name}}) {
5651 if ($self->{insertion_mode} == IN_TABLE_IM) {
5652 ## Clear back to table context
5653 while (not ($self->{open_elements}->[-1]->[1]
5654 & TABLE_SCOPING_EL)) {
5655 !!!cp ('t201');
5656 pop @{$self->{open_elements}};
5657 }
5658
5659 !!!insert-element ('tbody',, $token);
5660 $self->{insertion_mode} = IN_TABLE_BODY_IM;
5661 ## reprocess in the "in table body" insertion mode...
5662 }
5663
5664 if ($self->{insertion_mode} == IN_TABLE_BODY_IM) {
5665 unless ($token->{tag_name} eq 'tr') {
5666 !!!cp ('t202');
5667 !!!parse-error (type => 'missing start tag:tr', token => $token);
5668 }
5669
5670 ## Clear back to table body context
5671 while (not ($self->{open_elements}->[-1]->[1]
5672 & TABLE_ROWS_SCOPING_EL)) {
5673 !!!cp ('t203');
5674 ## ISSUE: Can this case be reached?
5675 pop @{$self->{open_elements}};
5676 }
5677
5678 $self->{insertion_mode} = IN_ROW_IM;
5679 if ($token->{tag_name} eq 'tr') {
5680 !!!cp ('t204');
5681 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5682 $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5683 !!!nack ('t204');
5684 !!!next-token;
5685 next B;
5686 } else {
5687 !!!cp ('t205');
5688 !!!insert-element ('tr',, $token);
5689 ## reprocess in the "in row" insertion mode
5690 }
5691 } else {
5692 !!!cp ('t206');
5693 }
5694
5695 ## Clear back to table row context
5696 while (not ($self->{open_elements}->[-1]->[1]
5697 & TABLE_ROW_SCOPING_EL)) {
5698 !!!cp ('t207');
5699 pop @{$self->{open_elements}};
5700 }
5701
5702 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5703 $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5704 $self->{insertion_mode} = IN_CELL_IM;
5705
5706 push @$active_formatting_elements, ['#marker', ''];
5707
5708 !!!nack ('t207.1');
5709 !!!next-token;
5710 next B;
5711 } elsif ({
5712 caption => 1, col => 1, colgroup => 1,
5713 tbody => 1, tfoot => 1, thead => 1,
5714 tr => 1, # $self->{insertion_mode} == IN_ROW_IM
5715 }->{$token->{tag_name}}) {
5716 if ($self->{insertion_mode} == IN_ROW_IM) {
5717 ## As if </tr>
5718 ## have an element in table scope
5719 my $i;
5720 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5721 my $node = $self->{open_elements}->[$_];
5722 if ($node->[1] & TABLE_ROW_EL) {
5723 !!!cp ('t208');
5724 $i = $_;
5725 last INSCOPE;
5726 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5727 !!!cp ('t209');
5728 last INSCOPE;
5729 }
5730 } # INSCOPE
5731 unless (defined $i) {
5732 !!!cp ('t210');
5733 ## TODO: This type is wrong.
5734 !!!parse-error (type => 'unmacthed end tag',
5735 text => $token->{tag_name}, token => $token);
5736 ## Ignore the token
5737 !!!nack ('t210.1');
5738 !!!next-token;
5739 next B;
5740 }
5741
5742 ## Clear back to table row context
5743 while (not ($self->{open_elements}->[-1]->[1]
5744 & TABLE_ROW_SCOPING_EL)) {
5745 !!!cp ('t211');
5746 ## ISSUE: Can this case be reached?
5747 pop @{$self->{open_elements}};
5748 }
5749
5750 pop @{$self->{open_elements}}; # tr
5751 $self->{insertion_mode} = IN_TABLE_BODY_IM;
5752 if ($token->{tag_name} eq 'tr') {
5753 !!!cp ('t212');
5754 ## reprocess
5755 !!!ack-later;
5756 next B;
5757 } else {
5758 !!!cp ('t213');
5759 ## reprocess in the "in table body" insertion mode...
5760 }
5761 }
5762
5763 if ($self->{insertion_mode} == IN_TABLE_BODY_IM) {
5764 ## have an element in table scope
5765 my $i;
5766 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5767 my $node = $self->{open_elements}->[$_];
5768 if ($node->[1] & TABLE_ROW_GROUP_EL) {
5769 !!!cp ('t214');
5770 $i = $_;
5771 last INSCOPE;
5772 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5773 !!!cp ('t215');
5774 last INSCOPE;
5775 }
5776 } # INSCOPE
5777 unless (defined $i) {
5778 !!!cp ('t216');
5779 ## TODO: This erorr type is wrong.
5780 !!!parse-error (type => 'unmatched end tag',
5781 text => $token->{tag_name}, token => $token);
5782 ## Ignore the token
5783 !!!nack ('t216.1');
5784 !!!next-token;
5785 next B;
5786 }
5787
5788 ## Clear back to table body context
5789 while (not ($self->{open_elements}->[-1]->[1]
5790 & TABLE_ROWS_SCOPING_EL)) {
5791 !!!cp ('t217');
5792 ## ISSUE: Can this state be reached?
5793 pop @{$self->{open_elements}};
5794 }
5795
5796 ## As if <{current node}>
5797 ## have an element in table scope
5798 ## true by definition
5799
5800 ## Clear back to table body context
5801 ## nop by definition
5802
5803 pop @{$self->{open_elements}};
5804 $self->{insertion_mode} = IN_TABLE_IM;
5805 ## reprocess in "in table" insertion mode...
5806 } else {
5807 !!!cp ('t218');
5808 }
5809
5810 if ($token->{tag_name} eq 'col') {
5811 ## Clear back to table context
5812 while (not ($self->{open_elements}->[-1]->[1]
5813 & TABLE_SCOPING_EL)) {
5814 !!!cp ('t219');
5815 ## ISSUE: Can this state be reached?
5816 pop @{$self->{open_elements}};
5817 }
5818
5819 !!!insert-element ('colgroup',, $token);
5820 $self->{insertion_mode} = IN_COLUMN_GROUP_IM;
5821 ## reprocess
5822 $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5823 !!!ack-later;
5824 next B;
5825 } elsif ({
5826 caption => 1,
5827 colgroup => 1,
5828 tbody => 1, tfoot => 1, thead => 1,
5829 }->{$token->{tag_name}}) {
5830 ## Clear back to table context
5831 while (not ($self->{open_elements}->[-1]->[1]
5832 & TABLE_SCOPING_EL)) {
5833 !!!cp ('t220');
5834 ## ISSUE: Can this state be reached?
5835 pop @{$self->{open_elements}};
5836 }
5837
5838 push @$active_formatting_elements, ['#marker', '']
5839 if $token->{tag_name} eq 'caption';
5840
5841 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5842 $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5843 $self->{insertion_mode} = {
5844 caption => IN_CAPTION_IM,
5845 colgroup => IN_COLUMN_GROUP_IM,
5846 tbody => IN_TABLE_BODY_IM,
5847 tfoot => IN_TABLE_BODY_IM,
5848 thead => IN_TABLE_BODY_IM,
5849 }->{$token->{tag_name}};
5850 !!!next-token;
5851 !!!nack ('t220.1');
5852 next B;
5853 } else {
5854 die "$0: in table: <>: $token->{tag_name}";
5855 }
5856 } elsif ($token->{tag_name} eq 'table') {
5857 !!!parse-error (type => 'not closed',
5858 text => $self->{open_elements}->[-1]->[0]
5859 ->manakai_local_name,
5860 token => $token);
5861
5862 ## As if </table>
5863 ## have a table element in table scope
5864 my $i;
5865 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5866 my $node = $self->{open_elements}->[$_];
5867 if ($node->[1] & TABLE_EL) {
5868 !!!cp ('t221');
5869 $i = $_;
5870 last INSCOPE;
5871 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5872 !!!cp ('t222');
5873 last INSCOPE;
5874 }
5875 } # INSCOPE
5876 unless (defined $i) {
5877 !!!cp ('t223');
5878 ## TODO: The following is wrong, maybe.
5879 !!!parse-error (type => 'unmatched end tag', text => 'table',
5880 token => $token);
5881 ## Ignore tokens </table><table>
5882 !!!nack ('t223.1');
5883 !!!next-token;
5884 next B;
5885 }
5886
5887 ## TODO: Followings are removed from the latest spec.
5888 ## generate implied end tags
5889 while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
5890 !!!cp ('t224');
5891 pop @{$self->{open_elements}};
5892 }
5893
5894 unless ($self->{open_elements}->[-1]->[1] & TABLE_EL) {
5895 !!!cp ('t225');
5896 ## NOTE: |<table><tr><table>|
5897 !!!parse-error (type => 'not closed',
5898 text => $self->{open_elements}->[-1]->[0]
5899 ->manakai_local_name,
5900 token => $token);
5901 } else {
5902 !!!cp ('t226');
5903 }
5904
5905 splice @{$self->{open_elements}}, $i;
5906 pop @{$open_tables};
5907
5908 $self->_reset_insertion_mode;
5909
5910 ## reprocess
5911 !!!ack-later;
5912 next B;
5913 } elsif ($token->{tag_name} eq 'style') {
5914 if (not $open_tables->[-1]->[1]) { # tainted
5915 !!!cp ('t227.8');
5916 ## NOTE: This is a "as if in head" code clone.
5917 $parse_rcdata->(CDATA_CONTENT_MODEL);
5918 $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5919 next B;
5920 } else {
5921 !!!cp ('t227.7');
5922 #
5923 }
5924 } elsif ($token->{tag_name} eq 'script') {
5925 if (not $open_tables->[-1]->[1]) { # tainted
5926 !!!cp ('t227.6');
5927 ## NOTE: This is a "as if in head" code clone.
5928 $script_start_tag->();
5929 $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5930 next B;
5931 } else {
5932 !!!cp ('t227.5');
5933 #
5934 }
5935 } elsif ($token->{tag_name} eq 'input') {
5936 if (not $open_tables->[-1]->[1]) { # tainted
5937 if ($token->{attributes}->{type}) { ## TODO: case
5938 my $type = lc $token->{attributes}->{type}->{value};
5939 if ($type eq 'hidden') {
5940 !!!cp ('t227.3');
5941 !!!parse-error (type => 'in table',
5942 text => $token->{tag_name}, token => $token);
5943
5944 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5945 $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5946
5947 ## TODO: form element pointer
5948
5949 pop @{$self->{open_elements}};
5950
5951 !!!next-token;
5952 !!!ack ('t227.2.1');
5953 next B;
5954 } else {
5955 !!!cp ('t227.2');
5956 #
5957 }
5958 } else {
5959 !!!cp ('t227.1');
5960 #
5961 }
5962 } else {
5963 !!!cp ('t227.4');
5964 #
5965 }
5966 } else {
5967 !!!cp ('t227');
5968 #
5969 }
5970
5971 !!!parse-error (type => 'in table', text => $token->{tag_name},
5972 token => $token);
5973
5974 $insert = $insert_to_foster;
5975 #
5976 } elsif ($token->{type} == END_TAG_TOKEN) {
5977 if ($token->{tag_name} eq 'tr' and
5978 $self->{insertion_mode} == IN_ROW_IM) {
5979 ## have an element in table scope
5980 my $i;
5981 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5982 my $node = $self->{open_elements}->[$_];
5983 if ($node->[1] & TABLE_ROW_EL) {
5984 !!!cp ('t228');
5985 $i = $_;
5986 last INSCOPE;
5987 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5988 !!!cp ('t229');
5989 last INSCOPE;
5990 }
5991 } # INSCOPE
5992 unless (defined $i) {
5993 !!!cp ('t230');
5994 !!!parse-error (type => 'unmatched end tag',
5995 text => $token->{tag_name}, token => $token);
5996 ## Ignore the token
5997 !!!nack ('t230.1');
5998 !!!next-token;
5999 next B;
6000 } else {
6001 !!!cp ('t232');
6002 }
6003
6004 ## Clear back to table row context
6005 while (not ($self->{open_elements}->[-1]->[1]
6006 & TABLE_ROW_SCOPING_EL)) {
6007 !!!cp ('t231');
6008 ## ISSUE: Can this state be reached?
6009 pop @{$self->{open_elements}};
6010 }
6011
6012 pop @{$self->{open_elements}}; # tr
6013 $self->{insertion_mode} = IN_TABLE_BODY_IM;
6014 !!!next-token;
6015 !!!nack ('t231.1');
6016 next B;
6017 } elsif ($token->{tag_name} eq 'table') {
6018 if ($self->{insertion_mode} == IN_ROW_IM) {
6019 ## As if </tr>
6020 ## have an element in table scope
6021 my $i;
6022 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6023 my $node = $self->{open_elements}->[$_];
6024 if ($node->[1] & TABLE_ROW_EL) {
6025 !!!cp ('t233');
6026 $i = $_;
6027 last INSCOPE;
6028 } elsif ($node->[1] & TABLE_SCOPING_EL) {
6029 !!!cp ('t234');
6030 last INSCOPE;
6031 }
6032 } # INSCOPE
6033 unless (defined $i) {
6034 !!!cp ('t235');
6035 ## TODO: The following is wrong.
6036 !!!parse-error (type => 'unmatched end tag',
6037 text => $token->{type}, token => $token);
6038 ## Ignore the token
6039 !!!nack ('t236.1');
6040 !!!next-token;
6041 next B;
6042 }
6043
6044 ## Clear back to table row context
6045 while (not ($self->{open_elements}->[-1]->[1]
6046 & TABLE_ROW_SCOPING_EL)) {
6047 !!!cp ('t236');
6048 ## ISSUE: Can this state be reached?
6049 pop @{$self->{open_elements}};
6050 }
6051
6052 pop @{$self->{open_elements}}; # tr
6053 $self->{insertion_mode} = IN_TABLE_BODY_IM;
6054 ## reprocess in the "in table body" insertion mode...
6055 }
6056
6057 if ($self->{insertion_mode} == IN_TABLE_BODY_IM) {
6058 ## have an element in table scope
6059 my $i;
6060 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6061 my $node = $self->{open_elements}->[$_];
6062 if ($node->[1] & TABLE_ROW_GROUP_EL) {
6063 !!!cp ('t237');
6064 $i = $_;
6065 last INSCOPE;
6066 } elsif ($node->[1] & TABLE_SCOPING_EL) {
6067 !!!cp ('t238');
6068 last INSCOPE;
6069 }
6070 } # INSCOPE
6071 unless (defined $i) {
6072 !!!cp ('t239');
6073 !!!parse-error (type => 'unmatched end tag',
6074 text => $token->{tag_name}, token => $token);
6075 ## Ignore the token
6076 !!!nack ('t239.1');
6077 !!!next-token;
6078 next B;
6079 }
6080
6081 ## Clear back to table body context
6082 while (not ($self->{open_elements}->[-1]->[1]
6083 & TABLE_ROWS_SCOPING_EL)) {
6084 !!!cp ('t240');
6085 pop @{$self->{open_elements}};
6086 }
6087
6088 ## As if <{current node}>
6089 ## have an element in table scope
6090 ## true by definition
6091
6092 ## Clear back to table body context
6093 ## nop by definition
6094
6095 pop @{$self->{open_elements}};
6096 $self->{insertion_mode} = IN_TABLE_IM;
6097 ## reprocess in the "in table" insertion mode...
6098 }
6099
6100 ## NOTE: </table> in the "in table" insertion mode.
6101 ## When you edit the code fragment below, please ensure that
6102 ## the code for <table> in the "in table" insertion mode
6103 ## is synced with it.
6104
6105 ## have a table element in table scope
6106 my $i;
6107 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6108 my $node = $self->{open_elements}->[$_];
6109 if ($node->[1] & TABLE_EL) {
6110 !!!cp ('t241');
6111 $i = $_;
6112 last INSCOPE;
6113 } elsif ($node->[1] & TABLE_SCOPING_EL) {
6114 !!!cp ('t242');
6115 last INSCOPE;
6116 }
6117 } # INSCOPE
6118 unless (defined $i) {
6119 !!!cp ('t243');
6120 !!!parse-error (type => 'unmatched end tag',
6121 text => $token->{tag_name}, token => $token);
6122 ## Ignore the token
6123 !!!nack ('t243.1');
6124 !!!next-token;
6125 next B;
6126 }
6127
6128 splice @{$self->{open_elements}}, $i;
6129 pop @{$open_tables};
6130
6131 $self->_reset_insertion_mode;
6132
6133 !!!next-token;
6134 next B;
6135 } elsif ({
6136 tbody => 1, tfoot => 1, thead => 1,
6137 }->{$token->{tag_name}} and
6138 $self->{insertion_mode} & ROW_IMS) {
6139 if ($self->{insertion_mode} == IN_ROW_IM) {
6140 ## have an element in table scope
6141 my $i;
6142 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6143 my $node = $self->{open_elements}->[$_];
6144 if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
6145 !!!cp ('t247');
6146 $i = $_;
6147 last INSCOPE;
6148 } elsif ($node->[1] & TABLE_SCOPING_EL) {
6149 !!!cp ('t248');
6150 last INSCOPE;
6151 }
6152 } # INSCOPE
6153 unless (defined $i) {
6154 !!!cp ('t249');
6155 !!!parse-error (type => 'unmatched end tag',
6156 text => $token->{tag_name}, token => $token);
6157 ## Ignore the token
6158 !!!nack ('t249.1');
6159 !!!next-token;
6160 next B;
6161 }
6162
6163 ## As if </tr>
6164 ## have an element in table scope
6165 my $i;
6166 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6167 my $node = $self->{open_elements}->[$_];
6168 if ($node->[1] & TABLE_ROW_EL) {
6169 !!!cp ('t250');
6170 $i = $_;
6171 last INSCOPE;
6172 } elsif ($node->[1] & TABLE_SCOPING_EL) {
6173 !!!cp ('t251');
6174 last INSCOPE;
6175 }
6176 } # INSCOPE
6177 unless (defined $i) {
6178 !!!cp ('t252');
6179 !!!parse-error (type => 'unmatched end tag',
6180 text => 'tr', token => $token);
6181 ## Ignore the token
6182 !!!nack ('t252.1');
6183 !!!next-token;
6184 next B;
6185 }
6186
6187 ## Clear back to table row context
6188 while (not ($self->{open_elements}->[-1]->[1]
6189 & TABLE_ROW_SCOPING_EL)) {
6190 !!!cp ('t253');
6191 ## ISSUE: Can this case be reached?
6192 pop @{$self->{open_elements}};
6193 }
6194
6195 pop @{$self->{open_elements}}; # tr
6196 $self->{insertion_mode} = IN_TABLE_BODY_IM;
6197 ## reprocess in the "in table body" insertion mode...
6198 }
6199
6200 ## have an element in table scope
6201 my $i;
6202 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6203 my $node = $self->{open_elements}->[$_];
6204 if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
6205 !!!cp ('t254');
6206 $i = $_;
6207 last INSCOPE;
6208 } elsif ($node->[1] & TABLE_SCOPING_EL) {
6209 !!!cp ('t255');
6210 last INSCOPE;
6211 }
6212 } # INSCOPE
6213 unless (defined $i) {
6214 !!!cp ('t256');
6215 !!!parse-error (type => 'unmatched end tag',
6216 text => $token->{tag_name}, token => $token);
6217 ## Ignore the token
6218 !!!nack ('t256.1');
6219 !!!next-token;
6220 next B;
6221 }
6222
6223 ## Clear back to table body context
6224 while (not ($self->{open_elements}->[-1]->[1]
6225 & TABLE_ROWS_SCOPING_EL)) {
6226 !!!cp ('t257');
6227 ## ISSUE: Can this case be reached?
6228 pop @{$self->{open_elements}};
6229 }
6230
6231 pop @{$self->{open_elements}};
6232 $self->{insertion_mode} = IN_TABLE_IM;
6233 !!!nack ('t257.1');
6234 !!!next-token;
6235 next B;
6236 } elsif ({
6237 body => 1, caption => 1, col => 1, colgroup => 1,
6238 html => 1, td => 1, th => 1,
6239 tr => 1, # $self->{insertion_mode} == IN_ROW_IM
6240 tbody => 1, tfoot => 1, thead => 1, # $self->{insertion_mode} == IN_TABLE_IM
6241 }->{$token->{tag_name}}) {
6242 !!!cp ('t258');
6243 !!!parse-error (type => 'unmatched end tag',
6244 text => $token->{tag_name}, token => $token);
6245 ## Ignore the token
6246 !!!nack ('t258.1');
6247 !!!next-token;
6248 next B;
6249 } else {
6250 !!!cp ('t259');
6251 !!!parse-error (type => 'in table:/',
6252 text => $token->{tag_name}, token => $token);
6253
6254 $insert = $insert_to_foster;
6255 #
6256 }
6257 } elsif ($token->{type} == END_OF_FILE_TOKEN) {
6258 unless ($self->{open_elements}->[-1]->[1] & HTML_EL and
6259 @{$self->{open_elements}} == 1) { # redundant, maybe
6260 !!!parse-error (type => 'in body:#eof', token => $token);
6261 !!!cp ('t259.1');
6262 #
6263 } else {
6264 !!!cp ('t259.2');
6265 #
6266 }
6267
6268 ## Stop parsing
6269 last B;
6270 } else {
6271 die "$0: $token->{type}: Unknown token type";
6272 }
6273 } elsif ($self->{insertion_mode} == IN_COLUMN_GROUP_IM) {
6274 if ($token->{type} == CHARACTER_TOKEN) {
6275 if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
6276 $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
6277 unless (length $token->{data}) {
6278 !!!cp ('t260');
6279 !!!next-token;
6280 next B;
6281 }
6282 }
6283
6284 !!!cp ('t261');
6285 #
6286 } elsif ($token->{type} == START_TAG_TOKEN) {
6287 if ($token->{tag_name} eq 'col') {
6288 !!!cp ('t262');
6289 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
6290 pop @{$self->{open_elements}};
6291 !!!ack ('t262.1');
6292 !!!next-token;
6293 next B;
6294 } else {
6295 !!!cp ('t263');
6296 #
6297 }
6298 } elsif ($token->{type} == END_TAG_TOKEN) {
6299 if ($token->{tag_name} eq 'colgroup') {
6300 if ($self->{open_elements}->[-1]->[1] & HTML_EL) {
6301 !!!cp ('t264');
6302 !!!parse-error (type => 'unmatched end tag',
6303 text => 'colgroup', token => $token);
6304 ## Ignore the token
6305 !!!next-token;
6306 next B;
6307 } else {
6308 !!!cp ('t265');
6309 pop @{$self->{open_elements}}; # colgroup
6310 $self->{insertion_mode} = IN_TABLE_IM;
6311 !!!next-token;
6312 next B;
6313 }
6314 } elsif ($token->{tag_name} eq 'col') {
6315 !!!cp ('t266');
6316 !!!parse-error (type => 'unmatched end tag',
6317 text => 'col', token => $token);
6318 ## Ignore the token
6319 !!!next-token;
6320 next B;
6321 } else {
6322 !!!cp ('t267');
6323 #
6324 }
6325 } elsif ($token->{type} == END_OF_FILE_TOKEN) {
6326 if ($self->{open_elements}->[-1]->[1] & HTML_EL and
6327 @{$self->{open_elements}} == 1) { # redundant, maybe
6328 !!!cp ('t270.2');
6329 ## Stop parsing.
6330 last B;
6331 } else {
6332 ## NOTE: As if </colgroup>.
6333 !!!cp ('t270.1');
6334 pop @{$self->{open_elements}}; # colgroup
6335 $self->{insertion_mode} = IN_TABLE_IM;
6336 ## Reprocess.
6337 next B;
6338 }
6339 } else {
6340 die "$0: $token->{type}: Unknown token type";
6341 }
6342
6343 ## As if </colgroup>
6344 if ($self->{open_elements}->[-1]->[1] & HTML_EL) {
6345 !!!cp ('t269');
6346 ## TODO: Wrong error type?
6347 !!!parse-error (type => 'unmatched end tag',
6348 text => 'colgroup', token => $token);
6349 ## Ignore the token
6350 !!!nack ('t269.1');
6351 !!!next-token;
6352 next B;
6353 } else {
6354 !!!cp ('t270');
6355 pop @{$self->{open_elements}}; # colgroup
6356 $self->{insertion_mode} = IN_TABLE_IM;
6357 !!!ack-later;
6358 ## reprocess
6359 next B;
6360 }
6361 } elsif ($self->{insertion_mode} & SELECT_IMS) {
6362 if ($token->{type} == CHARACTER_TOKEN) {
6363 !!!cp ('t271');
6364 $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});
6365 !!!next-token;
6366 next B;
6367 } elsif ($token->{type} == START_TAG_TOKEN) {
6368 if ($token->{tag_name} eq 'option') {
6369 if ($self->{open_elements}->[-1]->[1] & OPTION_EL) {
6370 !!!cp ('t272');
6371 ## As if </option>
6372 pop @{$self->{open_elements}};
6373 } else {
6374 !!!cp ('t273');
6375 }
6376
6377 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
6378 !!!nack ('t273.1');
6379 !!!next-token;
6380 next B;
6381 } elsif ($token->{tag_name} eq 'optgroup') {
6382 if ($self->{open_elements}->[-1]->[1] & OPTION_EL) {
6383 !!!cp ('t274');
6384 ## As if </option>
6385 pop @{$self->{open_elements}};
6386 } else {
6387 !!!cp ('t275');
6388 }
6389
6390 if ($self->{open_elements}->[-1]->[1] & OPTGROUP_EL) {
6391 !!!cp ('t276');
6392 ## As if </optgroup>
6393 pop @{$self->{open_elements}};
6394 } else {
6395 !!!cp ('t277');
6396 }
6397
6398 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
6399 !!!nack ('t277.1');
6400 !!!next-token;
6401 next B;
6402 } elsif ({
6403 select => 1, input => 1, textarea => 1,
6404 }->{$token->{tag_name}} or
6405 ($self->{insertion_mode} == IN_SELECT_IN_TABLE_IM and
6406 {
6407 caption => 1, table => 1,
6408 tbody => 1, tfoot => 1, thead => 1,
6409 tr => 1, td => 1, th => 1,
6410 }->{$token->{tag_name}})) {
6411 ## TODO: The type below is not good - <select> is replaced by </select>
6412 !!!parse-error (type => 'not closed', text => 'select',
6413 token => $token);
6414 ## NOTE: As if the token were </select> (<select> case) or
6415 ## as if there were </select> (otherwise).
6416 ## have an element in table scope
6417 my $i;
6418 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6419 my $node = $self->{open_elements}->[$_];
6420 if ($node->[1] & SELECT_EL) {
6421 !!!cp ('t278');
6422 $i = $_;
6423 last INSCOPE;
6424 } elsif ($node->[1] & TABLE_SCOPING_EL) {
6425 !!!cp ('t279');
6426 last INSCOPE;
6427 }
6428 } # INSCOPE
6429 unless (defined $i) {
6430 !!!cp ('t280');
6431 !!!parse-error (type => 'unmatched end tag',
6432 text => 'select', token => $token);
6433 ## Ignore the token
6434 !!!nack ('t280.1');
6435 !!!next-token;
6436 next B;
6437 }
6438
6439 !!!cp ('t281');
6440 splice @{$self->{open_elements}}, $i;
6441
6442 $self->_reset_insertion_mode;
6443
6444 if ($token->{tag_name} eq 'select') {
6445 !!!nack ('t281.2');
6446 !!!next-token;
6447 next B;
6448 } else {
6449 !!!cp ('t281.1');
6450 !!!ack-later;
6451 ## Reprocess the token.
6452 next B;
6453 }
6454 } else {
6455 !!!cp ('t282');
6456 !!!parse-error (type => 'in select',
6457 text => $token->{tag_name}, token => $token);
6458 ## Ignore the token
6459 !!!nack ('t282.1');
6460 !!!next-token;
6461 next B;
6462 }
6463 } elsif ($token->{type} == END_TAG_TOKEN) {
6464 if ($token->{tag_name} eq 'optgroup') {
6465 if ($self->{open_elements}->[-1]->[1] & OPTION_EL and
6466 $self->{open_elements}->[-2]->[1] & OPTGROUP_EL) {
6467 !!!cp ('t283');
6468 ## As if </option>
6469 splice @{$self->{open_elements}}, -2;
6470 } elsif ($self->{open_elements}->[-1]->[1] & OPTGROUP_EL) {
6471 !!!cp ('t284');
6472 pop @{$self->{open_elements}};
6473 } else {
6474 !!!cp ('t285');
6475 !!!parse-error (type => 'unmatched end tag',
6476 text => $token->{tag_name}, token => $token);
6477 ## Ignore the token
6478 }
6479 !!!nack ('t285.1');
6480 !!!next-token;
6481 next B;
6482 } elsif ($token->{tag_name} eq 'option') {
6483 if ($self->{open_elements}->[-1]->[1] & OPTION_EL) {
6484 !!!cp ('t286');
6485 pop @{$self->{open_elements}};
6486 } else {
6487 !!!cp ('t287');
6488 !!!parse-error (type => 'unmatched end tag',
6489 text => $token->{tag_name}, token => $token);
6490 ## Ignore the token
6491 }
6492 !!!nack ('t287.1');
6493 !!!next-token;
6494 next B;
6495 } elsif ($token->{tag_name} eq 'select') {
6496 ## have an element in table scope
6497 my $i;
6498 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6499 my $node = $self->{open_elements}->[$_];
6500 if ($node->[1] & SELECT_EL) {
6501 !!!cp ('t288');
6502 $i = $_;
6503 last INSCOPE;
6504 } elsif ($node->[1] & TABLE_SCOPING_EL) {
6505 !!!cp ('t289');
6506 last INSCOPE;
6507 }
6508 } # INSCOPE
6509 unless (defined $i) {
6510 !!!cp ('t290');
6511 !!!parse-error (type => 'unmatched end tag',
6512 text => $token->{tag_name}, token => $token);
6513 ## Ignore the token
6514 !!!nack ('t290.1');
6515 !!!next-token;
6516 next B;
6517 }
6518
6519 !!!cp ('t291');
6520 splice @{$self->{open_elements}}, $i;
6521
6522 $self->_reset_insertion_mode;
6523
6524 !!!nack ('t291.1');
6525 !!!next-token;
6526 next B;
6527 } elsif ($self->{insertion_mode} == IN_SELECT_IN_TABLE_IM and
6528 {
6529 caption => 1, table => 1, tbody => 1,
6530 tfoot => 1, thead => 1, tr => 1, td => 1, th => 1,
6531 }->{$token->{tag_name}}) {
6532 ## TODO: The following is wrong?
6533 !!!parse-error (type => 'unmatched end tag',
6534 text => $token->{tag_name}, token => $token);
6535
6536 ## have an element in table scope
6537 my $i;
6538 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6539 my $node = $self->{open_elements}->[$_];
6540 if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
6541 !!!cp ('t292');
6542 $i = $_;
6543 last INSCOPE;
6544 } elsif ($node->[1] & TABLE_SCOPING_EL) {
6545 !!!cp ('t293');
6546 last INSCOPE;
6547 }
6548 } # INSCOPE
6549 unless (defined $i) {
6550 !!!cp ('t294');
6551 ## Ignore the token
6552 !!!nack ('t294.1');
6553 !!!next-token;
6554 next B;
6555 }
6556
6557 ## As if </select>
6558 ## have an element in table scope
6559 undef $i;
6560 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6561 my $node = $self->{open_elements}->[$_];
6562 if ($node->[1] & SELECT_EL) {
6563 !!!cp ('t295');
6564 $i = $_;
6565 last INSCOPE;
6566 } elsif ($node->[1] & TABLE_SCOPING_EL) {
6567 ## ISSUE: Can this state be reached?
6568 !!!cp ('t296');
6569 last INSCOPE;
6570 }
6571 } # INSCOPE
6572 unless (defined $i) {
6573 !!!cp ('t297');
6574 ## TODO: The following error type is correct?
6575 !!!parse-error (type => 'unmatched end tag',
6576 text => 'select', token => $token);
6577 ## Ignore the </select> token
6578 !!!nack ('t297.1');
6579 !!!next-token; ## TODO: ok?
6580 next B;
6581 }
6582
6583 !!!cp ('t298');
6584 splice @{$self->{open_elements}}, $i;
6585
6586 $self->_reset_insertion_mode;
6587
6588 !!!ack-later;
6589 ## reprocess
6590 next B;
6591 } else {
6592 !!!cp ('t299');
6593 !!!parse-error (type => 'in select:/',
6594 text => $token->{tag_name}, token => $token);
6595 ## Ignore the token
6596 !!!nack ('t299.3');
6597 !!!next-token;
6598 next B;
6599 }
6600 } elsif ($token->{type} == END_OF_FILE_TOKEN) {
6601 unless ($self->{open_elements}->[-1]->[1] & HTML_EL and
6602 @{$self->{open_elements}} == 1) { # redundant, maybe
6603 !!!cp ('t299.1');
6604 !!!parse-error (type => 'in body:#eof', token => $token);
6605 } else {
6606 !!!cp ('t299.2');
6607 }
6608
6609 ## Stop parsing.
6610 last B;
6611 } else {
6612 die "$0: $token->{type}: Unknown token type";
6613 }
6614 } elsif ($self->{insertion_mode} & BODY_AFTER_IMS) {
6615 if ($token->{type} == CHARACTER_TOKEN) {
6616 if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
6617 my $data = $1;
6618 ## As if in body
6619 $reconstruct_active_formatting_elements->($insert_to_current);
6620
6621 $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
6622
6623 unless (length $token->{data}) {
6624 !!!cp ('t300');
6625 !!!next-token;
6626 next B;
6627 }
6628 }
6629
6630 if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
6631 !!!cp ('t301');
6632 !!!parse-error (type => 'after html:#text', token => $token);
6633 #
6634 } else {
6635 !!!cp ('t302');
6636 ## "after body" insertion mode
6637 !!!parse-error (type => 'after body:#text', token => $token);
6638 #
6639 }
6640
6641 $self->{insertion_mode} = IN_BODY_IM;
6642 ## reprocess
6643 next B;
6644 } elsif ($token->{type} == START_TAG_TOKEN) {
6645 if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
6646 !!!cp ('t303');
6647 !!!parse-error (type => 'after html',
6648 text => $token->{tag_name}, token => $token);
6649 #
6650 } else {
6651 !!!cp ('t304');
6652 ## "after body" insertion mode
6653 !!!parse-error (type => 'after body',
6654 text => $token->{tag_name}, token => $token);
6655 #
6656 }
6657
6658 $self->{insertion_mode} = IN_BODY_IM;
6659 !!!ack-later;
6660 ## reprocess
6661 next B;
6662 } elsif ($token->{type} == END_TAG_TOKEN) {
6663 if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
6664 !!!cp ('t305');
6665 !!!parse-error (type => 'after html:/',
6666 text => $token->{tag_name}, token => $token);
6667
6668 $self->{insertion_mode} = IN_BODY_IM;
6669 ## Reprocess.
6670 next B;
6671 } else {
6672 !!!cp ('t306');
6673 }
6674
6675 ## "after body" insertion mode
6676 if ($token->{tag_name} eq 'html') {
6677 if (defined $self->{inner_html_node}) {
6678 !!!cp ('t307');
6679 !!!parse-error (type => 'unmatched end tag',
6680 text => 'html', token => $token);
6681 ## Ignore the token
6682 !!!next-token;
6683 next B;
6684 } else {
6685 !!!cp ('t308');
6686 $self->{insertion_mode} = AFTER_HTML_BODY_IM;
6687 !!!next-token;
6688 next B;
6689 }
6690 } else {
6691 !!!cp ('t309');
6692 !!!parse-error (type => 'after body:/',
6693 text => $token->{tag_name}, token => $token);
6694
6695 $self->{insertion_mode} = IN_BODY_IM;
6696 ## reprocess
6697 next B;
6698 }
6699 } elsif ($token->{type} == END_OF_FILE_TOKEN) {
6700 !!!cp ('t309.2');
6701 ## Stop parsing
6702 last B;
6703 } else {
6704 die "$0: $token->{type}: Unknown token type";
6705 }
6706 } elsif ($self->{insertion_mode} & FRAME_IMS) {
6707 if ($token->{type} == CHARACTER_TOKEN) {
6708 if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
6709 $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
6710
6711 unless (length $token->{data}) {
6712 !!!cp ('t310');
6713 !!!next-token;
6714 next B;
6715 }
6716 }
6717
6718 if ($token->{data} =~ s/^[^\x09\x0A\x0C\x20]+//) {
6719 if ($self->{insertion_mode} == IN_FRAMESET_IM) {
6720 !!!cp ('t311');
6721 !!!parse-error (type => 'in frameset:#text', token => $token);
6722 } elsif ($self->{insertion_mode} == AFTER_FRAMESET_IM) {
6723 !!!cp ('t312');
6724 !!!parse-error (type => 'after frameset:#text', token => $token);
6725 } else { # "after after frameset"
6726 !!!cp ('t313');
6727 !!!parse-error (type => 'after html:#text', token => $token);
6728 }
6729
6730 ## Ignore the token.
6731 if (length $token->{data}) {
6732 !!!cp ('t314');
6733 ## reprocess the rest of characters
6734 } else {
6735 !!!cp ('t315');
6736 !!!next-token;
6737 }
6738 next B;
6739 }
6740
6741 die qq[$0: Character "$token->{data}"];
6742 } elsif ($token->{type} == START_TAG_TOKEN) {
6743 if ($token->{tag_name} eq 'frameset' and
6744 $self->{insertion_mode} == IN_FRAMESET_IM) {
6745 !!!cp ('t318');
6746 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
6747 !!!nack ('t318.1');
6748 !!!next-token;
6749 next B;
6750 } elsif ($token->{tag_name} eq 'frame' and
6751 $self->{insertion_mode} == IN_FRAMESET_IM) {
6752 !!!cp ('t319');
6753 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
6754 pop @{$self->{open_elements}};
6755 !!!ack ('t319.1');
6756 !!!next-token;
6757 next B;
6758 } elsif ($token->{tag_name} eq 'noframes') {
6759 !!!cp ('t320');
6760 ## NOTE: As if in head.
6761 $parse_rcdata->(CDATA_CONTENT_MODEL);
6762 next B;
6763
6764 ## NOTE: |<!DOCTYPE HTML><frameset></frameset></html><noframes></noframes>|
6765 ## has no parse error.
6766 } else {
6767 if ($self->{insertion_mode} == IN_FRAMESET_IM) {
6768 !!!cp ('t321');
6769 !!!parse-error (type => 'in frameset',
6770 text => $token->{tag_name}, token => $token);
6771 } elsif ($self->{insertion_mode} == AFTER_FRAMESET_IM) {
6772 !!!cp ('t322');
6773 !!!parse-error (type => 'after frameset',
6774 text => $token->{tag_name}, token => $token);
6775 } else { # "after after frameset"
6776 !!!cp ('t322.2');
6777 !!!parse-error (type => 'after after frameset',
6778 text => $token->{tag_name}, token => $token);
6779 }
6780 ## Ignore the token
6781 !!!nack ('t322.1');
6782 !!!next-token;
6783 next B;
6784 }
6785 } elsif ($token->{type} == END_TAG_TOKEN) {
6786 if ($token->{tag_name} eq 'frameset' and
6787 $self->{insertion_mode} == IN_FRAMESET_IM) {
6788 if ($self->{open_elements}->[-1]->[1] & HTML_EL and
6789 @{$self->{open_elements}} == 1) {
6790 !!!cp ('t325');
6791 !!!parse-error (type => 'unmatched end tag',
6792 text => $token->{tag_name}, token => $token);
6793 ## Ignore the token
6794 !!!next-token;
6795 } else {
6796 !!!cp ('t326');
6797 pop @{$self->{open_elements}};
6798 !!!next-token;
6799 }
6800
6801 if (not defined $self->{inner_html_node} and
6802 not ($self->{open_elements}->[-1]->[1] & FRAMESET_EL)) {
6803 !!!cp ('t327');
6804 $self->{insertion_mode} = AFTER_FRAMESET_IM;
6805 } else {
6806 !!!cp ('t328');
6807 }
6808 next B;
6809 } elsif ($token->{tag_name} eq 'html' and
6810 $self->{insertion_mode} == AFTER_FRAMESET_IM) {
6811 !!!cp ('t329');
6812 $self->{insertion_mode} = AFTER_HTML_FRAMESET_IM;
6813 !!!next-token;
6814 next B;
6815 } else {
6816 if ($self->{insertion_mode} == IN_FRAMESET_IM) {
6817 !!!cp ('t330');
6818 !!!parse-error (type => 'in frameset:/',
6819 text => $token->{tag_name}, token => $token);
6820 } elsif ($self->{insertion_mode} == AFTER_FRAMESET_IM) {
6821 !!!cp ('t330.1');
6822 !!!parse-error (type => 'after frameset:/',
6823 text => $token->{tag_name}, token => $token);
6824 } else { # "after after html"
6825 !!!cp ('t331');
6826 !!!parse-error (type => 'after after frameset:/',
6827 text => $token->{tag_name}, token => $token);
6828 }
6829 ## Ignore the token
6830 !!!next-token;
6831 next B;
6832 }
6833 } elsif ($token->{type} == END_OF_FILE_TOKEN) {
6834 unless ($self->{open_elements}->[-1]->[1] & HTML_EL and
6835 @{$self->{open_elements}} == 1) { # redundant, maybe
6836 !!!cp ('t331.1');
6837 !!!parse-error (type => 'in body:#eof', token => $token);
6838 } else {
6839 !!!cp ('t331.2');
6840 }
6841
6842 ## Stop parsing
6843 last B;
6844 } else {
6845 die "$0: $token->{type}: Unknown token type";
6846 }
6847 } else {
6848 die "$0: $self->{insertion_mode}: Unknown insertion mode";
6849 }
6850
6851 ## "in body" insertion mode
6852 if ($token->{type} == START_TAG_TOKEN) {
6853 if ($token->{tag_name} eq 'script') {
6854 !!!cp ('t332');
6855 ## NOTE: This is an "as if in head" code clone
6856 $script_start_tag->();
6857 next B;
6858 } elsif ($token->{tag_name} eq 'style') {
6859 !!!cp ('t333');
6860 ## NOTE: This is an "as if in head" code clone
6861 $parse_rcdata->(CDATA_CONTENT_MODEL);
6862 next B;
6863 } elsif ({
6864 base => 1, command => 1, eventsource => 1, link => 1,
6865 }->{$token->{tag_name}}) {
6866 !!!cp ('t334');
6867 ## NOTE: This is an "as if in head" code clone, only "-t" differs
6868 !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
6869 pop @{$self->{open_elements}};
6870 !!!ack ('t334.1');
6871 !!!next-token;
6872 next B;
6873 } elsif ($token->{tag_name} eq 'meta') {
6874 ## NOTE: This is an "as if in head" code clone, only "-t" differs
6875 !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
6876 my $meta_el = pop @{$self->{open_elements}};
6877
6878 unless ($self->{confident}) {
6879 if ($token->{attributes}->{charset}) {
6880 !!!cp ('t335');
6881 ## NOTE: Whether the encoding is supported or not is handled
6882 ## in the {change_encoding} callback.
6883 $self->{change_encoding}
6884 ->($self, $token->{attributes}->{charset}->{value}, $token);
6885
6886 $meta_el->[0]->get_attribute_node_ns (undef, 'charset')
6887 ->set_user_data (manakai_has_reference =>
6888 $token->{attributes}->{charset}
6889 ->{has_reference});
6890 } elsif ($token->{attributes}->{content}) {
6891 if ($token->{attributes}->{content}->{value}
6892 =~ /[Cc][Hh][Aa][Rr][Ss][Ee][Tt]
6893 [\x09\x0A\x0C\x0D\x20]*=
6894 [\x09\x0A\x0C\x0D\x20]*(?>"([^"]*)"|'([^']*)'|
6895 ([^"'\x09\x0A\x0C\x0D\x20][^\x09\x0A\x0C\x0D\x20\x3B]*))
6896 /x) {
6897 !!!cp ('t336');
6898 ## NOTE: Whether the encoding is supported or not is handled
6899 ## in the {change_encoding} callback.
6900 $self->{change_encoding}
6901 ->($self, defined $1 ? $1 : defined $2 ? $2 : $3, $token);
6902 $meta_el->[0]->get_attribute_node_ns (undef, 'content')
6903 ->set_user_data (manakai_has_reference =>
6904 $token->{attributes}->{content}
6905 ->{has_reference});
6906 }
6907 }
6908 } else {
6909 if ($token->{attributes}->{charset}) {
6910 !!!cp ('t337');
6911 $meta_el->[0]->get_attribute_node_ns (undef, 'charset')
6912 ->set_user_data (manakai_has_reference =>
6913 $token->{attributes}->{charset}
6914 ->{has_reference});
6915 }
6916 if ($token->{attributes}->{content}) {
6917 !!!cp ('t338');
6918 $meta_el->[0]->get_attribute_node_ns (undef, 'content')
6919 ->set_user_data (manakai_has_reference =>
6920 $token->{attributes}->{content}
6921 ->{has_reference});
6922 }
6923 }
6924
6925 !!!ack ('t338.1');
6926 !!!next-token;
6927 next B;
6928 } elsif ($token->{tag_name} eq 'title') {
6929 !!!cp ('t341');
6930 ## NOTE: This is an "as if in head" code clone
6931 $parse_rcdata->(RCDATA_CONTENT_MODEL);
6932 next B;
6933 } elsif ($token->{tag_name} eq 'body') {
6934 !!!parse-error (type => 'in body', text => 'body', token => $token);
6935
6936 if (@{$self->{open_elements}} == 1 or
6937 not ($self->{open_elements}->[1]->[1] & BODY_EL)) {
6938 !!!cp ('t342');
6939 ## Ignore the token
6940 } else {
6941 my $body_el = $self->{open_elements}->[1]->[0];
6942 for my $attr_name (keys %{$token->{attributes}}) {
6943 unless ($body_el->has_attribute_ns (undef, $attr_name)) {
6944 !!!cp ('t343');
6945 $body_el->set_attribute_ns
6946 (undef, [undef, $attr_name],
6947 $token->{attributes}->{$attr_name}->{value});
6948 }
6949 }
6950 }
6951 !!!nack ('t343.1');
6952 !!!next-token;
6953 next B;
6954 } elsif ({
6955 ## NOTE: Start tags for non-phrasing flow content elements
6956
6957 ## NOTE: The normal one
6958 address => 1, article => 1, aside => 1, blockquote => 1,
6959 center => 1, datagrid => 1, details => 1, dialog => 1,
6960 dir => 1, div => 1, dl => 1, fieldset => 1, figure => 1,
6961 footer => 1, h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1,
6962 h6 => 1, header => 1, menu => 1, nav => 1, ol => 1, p => 1,
6963 section => 1, ul => 1,
6964 ## NOTE: As normal, but drops leading newline
6965 pre => 1, listing => 1,
6966 ## NOTE: As normal, but interacts with the form element pointer
6967 form => 1,
6968
6969 table => 1,
6970 hr => 1,
6971 }->{$token->{tag_name}}) {
6972 if ($token->{tag_name} eq 'form' and defined $self->{form_element}) {
6973 !!!cp ('t350');
6974 !!!parse-error (type => 'in form:form', token => $token);
6975 ## Ignore the token
6976 !!!nack ('t350.1');
6977 !!!next-token;
6978 next B;
6979 }
6980
6981 ## has a p element in scope
6982 INSCOPE: for (reverse @{$self->{open_elements}}) {
6983 if ($_->[1] & P_EL) {
6984 !!!cp ('t344');
6985 !!!back-token; # <form>
6986 $token = {type => END_TAG_TOKEN, tag_name => 'p',
6987 line => $token->{line}, column => $token->{column}};
6988 next B;
6989 } elsif ($_->[1] & SCOPING_EL) {
6990 !!!cp ('t345');
6991 last INSCOPE;
6992 }
6993 } # INSCOPE
6994
6995 !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
6996 if ($token->{tag_name} eq 'pre' or $token->{tag_name} eq 'listing') {
6997 !!!nack ('t346.1');
6998 !!!next-token;
6999 if ($token->{type} == CHARACTER_TOKEN) {
7000 $token->{data} =~ s/^\x0A//;
7001 unless (length $token->{data}) {
7002 !!!cp ('t346');
7003 !!!next-token;
7004 } else {
7005 !!!cp ('t349');
7006 }
7007 } else {
7008 !!!cp ('t348');
7009 }
7010 } elsif ($token->{tag_name} eq 'form') {
7011 !!!cp ('t347.1');
7012 $self->{form_element} = $self->{open_elements}->[-1]->[0];
7013
7014 !!!nack ('t347.2');
7015 !!!next-token;
7016 } elsif ($token->{tag_name} eq 'table') {
7017 !!!cp ('t382');
7018 push @{$open_tables}, [$self->{open_elements}->[-1]->[0]];
7019
7020 $self->{insertion_mode} = IN_TABLE_IM;
7021
7022 !!!nack ('t382.1');
7023 !!!next-token;
7024 } elsif ($token->{tag_name} eq 'hr') {
7025 !!!cp ('t386');
7026 pop @{$self->{open_elements}};
7027
7028 !!!nack ('t386.1');
7029 !!!next-token;
7030 } else {
7031 !!!nack ('t347.1');
7032 !!!next-token;
7033 }
7034 next B;
7035 } elsif ($token->{tag_name} eq 'li') {
7036 ## NOTE: As normal, but imply </li> when there's another <li> ...
7037
7038 ## NOTE: Special, Scope (<li><foo><li> == <li><foo><li/></foo></li>)
7039 ## Interpreted as <li><foo/></li><li/> (non-conforming)
7040 ## blockquote (O9.27), center (O), dd (Fx3, O, S3.1.2, IE7),
7041 ## dt (Fx, O, S, IE), dl (O), fieldset (O, S, IE), form (Fx, O, S),
7042 ## hn (O), pre (O), applet (O, S), button (O, S), marquee (Fx, O, S),
7043 ## object (Fx)
7044 ## Generate non-tree (non-conforming)
7045 ## basefont (IE7 (where basefont is non-void)), center (IE),
7046 ## form (IE), hn (IE)
7047 ## address, div, p (<li><foo><li> == <li><foo/></li><li/>)
7048 ## Interpreted as <li><foo><li/></foo></li> (non-conforming)
7049 ## div (Fx, S)
7050
7051 my $non_optional;
7052 my $i = -1;
7053
7054 ## 1.
7055 for my $node (reverse @{$self->{open_elements}}) {
7056 if ($node->[1] & LI_EL) {
7057 ## 2. (a) As if </li>
7058 {
7059 ## If no </li> - not applied
7060 #
7061
7062 ## Otherwise
7063
7064 ## 1. generate implied end tags, except for </li>
7065 #
7066
7067 ## 2. If current node != "li", parse error
7068 if ($non_optional) {
7069 !!!parse-error (type => 'not closed',
7070 text => $non_optional->[0]->manakai_local_name,
7071 token => $token);
7072 !!!cp ('t355');
7073 } else {
7074 !!!cp ('t356');
7075 }
7076
7077 ## 3. Pop
7078 splice @{$self->{open_elements}}, $i;
7079 }
7080
7081 last; ## 2. (b) goto 5.
7082 } elsif (
7083 ## NOTE: not "formatting" and not "phrasing"
7084 ($node->[1] & SPECIAL_EL or
7085 $node->[1] & SCOPING_EL) and
7086 ## NOTE: "li", "dt", and "dd" are in |SPECIAL_EL|.
7087
7088 (not $node->[1] & ADDRESS_EL) &
7089 (not $node->[1] & DIV_EL) &
7090 (not $node->[1] & P_EL)) {
7091 ## 3.
7092 !!!cp ('t357');
7093 last; ## goto 5.
7094 } elsif ($node->[1] & END_TAG_OPTIONAL_EL) {
7095 !!!cp ('t358');
7096 #
7097 } else {
7098 !!!cp ('t359');
7099 $non_optional ||= $node;
7100 #
7101 }
7102 ## 4.
7103 ## goto 2.
7104 $i--;
7105 }
7106
7107 ## 5. (a) has a |p| element in scope
7108 INSCOPE: for (reverse @{$self->{open_elements}}) {
7109 if ($_->[1] & P_EL) {
7110 !!!cp ('t353');
7111
7112 ## NOTE: |<p><li>|, for example.
7113
7114 !!!back-token; # <x>
7115 $token = {type => END_TAG_TOKEN, tag_name => 'p',
7116 line => $token->{line}, column => $token->{column}};
7117 next B;
7118 } elsif ($_->[1] & SCOPING_EL) {
7119 !!!cp ('t354');
7120 last INSCOPE;
7121 }
7122 } # INSCOPE
7123
7124 ## 5. (b) insert
7125 !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7126 !!!nack ('t359.1');
7127 !!!next-token;
7128 next B;
7129 } elsif ($token->{tag_name} eq 'dt' or
7130 $token->{tag_name} eq 'dd') {
7131 ## NOTE: As normal, but imply </dt> or </dd> when ...
7132
7133 my $non_optional;
7134 my $i = -1;
7135
7136 ## 1.
7137 for my $node (reverse @{$self->{open_elements}}) {
7138 if ($node->[1] & DT_EL or $node->[1] & DD_EL) {
7139 ## 2. (a) As if </li>
7140 {
7141 ## If no </li> - not applied
7142 #
7143
7144 ## Otherwise
7145
7146 ## 1. generate implied end tags, except for </dt> or </dd>
7147 #
7148
7149 ## 2. If current node != "dt"|"dd", parse error
7150 if ($non_optional) {
7151 !!!parse-error (type => 'not closed',
7152 text => $non_optional->[0]->manakai_local_name,
7153 token => $token);
7154 !!!cp ('t355.1');
7155 } else {
7156 !!!cp ('t356.1');
7157 }
7158
7159 ## 3. Pop
7160 splice @{$self->{open_elements}}, $i;
7161 }
7162
7163 last; ## 2. (b) goto 5.
7164 } elsif (
7165 ## NOTE: not "formatting" and not "phrasing"
7166 ($node->[1] & SPECIAL_EL or
7167 $node->[1] & SCOPING_EL) and
7168 ## NOTE: "li", "dt", and "dd" are in |SPECIAL_EL|.
7169
7170 (not $node->[1] & ADDRESS_EL) &
7171 (not $node->[1] & DIV_EL) &
7172 (not $node->[1] & P_EL)) {
7173 ## 3.
7174 !!!cp ('t357.1');
7175 last; ## goto 5.
7176 } elsif ($node->[1] & END_TAG_OPTIONAL_EL) {
7177 !!!cp ('t358.1');
7178 #
7179 } else {
7180 !!!cp ('t359.1');
7181 $non_optional ||= $node;
7182 #
7183 }
7184 ## 4.
7185 ## goto 2.
7186 $i--;
7187 }
7188
7189 ## 5. (a) has a |p| element in scope
7190 INSCOPE: for (reverse @{$self->{open_elements}}) {
7191 if ($_->[1] & P_EL) {
7192 !!!cp ('t353.1');
7193 !!!back-token; # <x>
7194 $token = {type => END_TAG_TOKEN, tag_name => 'p',
7195 line => $token->{line}, column => $token->{column}};
7196 next B;
7197 } elsif ($_->[1] & SCOPING_EL) {
7198 !!!cp ('t354.1');
7199 last INSCOPE;
7200 }
7201 } # INSCOPE
7202
7203 ## 5. (b) insert
7204 !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7205 !!!nack ('t359.2');
7206 !!!next-token;
7207 next B;
7208 } elsif ($token->{tag_name} eq 'plaintext') {
7209 ## NOTE: As normal, but effectively ends parsing
7210
7211 ## has a p element in scope
7212 INSCOPE: for (reverse @{$self->{open_elements}}) {
7213 if ($_->[1] & P_EL) {
7214 !!!cp ('t367');
7215 !!!back-token; # <plaintext>
7216 $token = {type => END_TAG_TOKEN, tag_name => 'p',
7217 line => $token->{line}, column => $token->{column}};
7218 next B;
7219 } elsif ($_->[1] & SCOPING_EL) {
7220 !!!cp ('t368');
7221 last INSCOPE;
7222 }
7223 } # INSCOPE
7224
7225 !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7226
7227 $self->{content_model} = PLAINTEXT_CONTENT_MODEL;
7228
7229 !!!nack ('t368.1');
7230 !!!next-token;
7231 next B;
7232 } elsif ($token->{tag_name} eq 'a') {
7233 AFE: for my $i (reverse 0..$#$active_formatting_elements) {
7234 my $node = $active_formatting_elements->[$i];
7235 if ($node->[1] & A_EL) {
7236 !!!cp ('t371');
7237 !!!parse-error (type => 'in a:a', token => $token);
7238
7239 !!!back-token; # <a>
7240 $token = {type => END_TAG_TOKEN, tag_name => 'a',
7241 line => $token->{line}, column => $token->{column}};
7242 $formatting_end_tag->($token);
7243
7244 AFE2: for (reverse 0..$#$active_formatting_elements) {
7245 if ($active_formatting_elements->[$_]->[0] eq $node->[0]) {
7246 !!!cp ('t372');
7247 splice @$active_formatting_elements, $_, 1;
7248 last AFE2;
7249 }
7250 } # AFE2
7251 OE: for (reverse 0..$#{$self->{open_elements}}) {
7252 if ($self->{open_elements}->[$_]->[0] eq $node->[0]) {
7253 !!!cp ('t373');
7254 splice @{$self->{open_elements}}, $_, 1;
7255 last OE;
7256 }
7257 } # OE
7258 last AFE;
7259 } elsif ($node->[0] eq '#marker') {
7260 !!!cp ('t374');
7261 last AFE;
7262 }
7263 } # AFE
7264
7265 $reconstruct_active_formatting_elements->($insert_to_current);
7266
7267 !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7268 push @$active_formatting_elements, $self->{open_elements}->[-1];
7269
7270 !!!nack ('t374.1');
7271 !!!next-token;
7272 next B;
7273 } elsif ($token->{tag_name} eq 'nobr') {
7274 $reconstruct_active_formatting_elements->($insert_to_current);
7275
7276 ## has a |nobr| element in scope
7277 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7278 my $node = $self->{open_elements}->[$_];
7279 if ($node->[1] & NOBR_EL) {
7280 !!!cp ('t376');
7281 !!!parse-error (type => 'in nobr:nobr', token => $token);
7282 !!!back-token; # <nobr>
7283 $token = {type => END_TAG_TOKEN, tag_name => 'nobr',
7284 line => $token->{line}, column => $token->{column}};
7285 next B;
7286 } elsif ($node->[1] & SCOPING_EL) {
7287 !!!cp ('t377');
7288 last INSCOPE;
7289 }
7290 } # INSCOPE
7291
7292 !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7293 push @$active_formatting_elements, $self->{open_elements}->[-1];
7294
7295 !!!nack ('t377.1');
7296 !!!next-token;
7297 next B;
7298 } elsif ($token->{tag_name} eq 'button') {
7299 ## has a button element in scope
7300 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7301 my $node = $self->{open_elements}->[$_];
7302 if ($node->[1] & BUTTON_EL) {
7303 !!!cp ('t378');
7304 !!!parse-error (type => 'in button:button', token => $token);
7305 !!!back-token; # <button>
7306 $token = {type => END_TAG_TOKEN, tag_name => 'button',
7307 line => $token->{line}, column => $token->{column}};
7308 next B;
7309 } elsif ($node->[1] & SCOPING_EL) {
7310 !!!cp ('t379');
7311 last INSCOPE;
7312 }
7313 } # INSCOPE
7314
7315 $reconstruct_active_formatting_elements->($insert_to_current);
7316
7317 !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7318
7319 ## TODO: associate with $self->{form_element} if defined
7320
7321 push @$active_formatting_elements, ['#marker', ''];
7322
7323 !!!nack ('t379.1');
7324 !!!next-token;
7325 next B;
7326 } elsif ({
7327 xmp => 1,
7328 iframe => 1,
7329 noembed => 1,
7330 noframes => 1, ## NOTE: This is an "as if in head" code clone.
7331 noscript => 0, ## TODO: 1 if scripting is enabled
7332 }->{$token->{tag_name}}) {
7333 if ($token->{tag_name} eq 'xmp') {
7334 !!!cp ('t381');
7335 $reconstruct_active_formatting_elements->($insert_to_current);
7336 } else {
7337 !!!cp ('t399');
7338 }
7339 ## NOTE: There is an "as if in body" code clone.
7340 $parse_rcdata->(CDATA_CONTENT_MODEL);
7341 next B;
7342 } elsif ($token->{tag_name} eq 'isindex') {
7343 !!!parse-error (type => 'isindex', token => $token);
7344
7345 if (defined $self->{form_element}) {
7346 !!!cp ('t389');
7347 ## Ignore the token
7348 !!!nack ('t389'); ## NOTE: Not acknowledged.
7349 !!!next-token;
7350 next B;
7351 } else {
7352 !!!ack ('t391.1');
7353
7354 my $at = $token->{attributes};
7355 my $form_attrs;
7356 $form_attrs->{action} = $at->{action} if $at->{action};
7357 my $prompt_attr = $at->{prompt};
7358 $at->{name} = {name => 'name', value => 'isindex'};
7359 delete $at->{action};
7360 delete $at->{prompt};
7361 my @tokens = (
7362 {type => START_TAG_TOKEN, tag_name => 'form',
7363 attributes => $form_attrs,
7364 line => $token->{line}, column => $token->{column}},
7365 {type => START_TAG_TOKEN, tag_name => 'hr',
7366 line => $token->{line}, column => $token->{column}},
7367 {type => START_TAG_TOKEN, tag_name => 'p',
7368 line => $token->{line}, column => $token->{column}},
7369 {type => START_TAG_TOKEN, tag_name => 'label',
7370 line => $token->{line}, column => $token->{column}},
7371 );
7372 if ($prompt_attr) {
7373 !!!cp ('t390');
7374 push @tokens, {type => CHARACTER_TOKEN, data => $prompt_attr->{value},
7375 #line => $token->{line}, column => $token->{column},
7376 };
7377 } else {
7378 !!!cp ('t391');
7379 push @tokens, {type => CHARACTER_TOKEN,
7380 data => 'This is a searchable index. Insert your search keywords here: ',
7381 #line => $token->{line}, column => $token->{column},
7382 }; # SHOULD
7383 ## TODO: make this configurable
7384 }
7385 push @tokens,
7386 {type => START_TAG_TOKEN, tag_name => 'input', attributes => $at,
7387 line => $token->{line}, column => $token->{column}},
7388 #{type => CHARACTER_TOKEN, data => ''}, # SHOULD
7389 {type => END_TAG_TOKEN, tag_name => 'label',
7390 line => $token->{line}, column => $token->{column}},
7391 {type => END_TAG_TOKEN, tag_name => 'p',
7392 line => $token->{line}, column => $token->{column}},
7393 {type => START_TAG_TOKEN, tag_name => 'hr',
7394 line => $token->{line}, column => $token->{column}},
7395 {type => END_TAG_TOKEN, tag_name => 'form',
7396 line => $token->{line}, column => $token->{column}};
7397 !!!back-token (@tokens);
7398 !!!next-token;
7399 next B;
7400 }
7401 } elsif ($token->{tag_name} eq 'textarea') {
7402 my $tag_name = $token->{tag_name};
7403 my $el;
7404 !!!create-element ($el, $HTML_NS, $token->{tag_name}, $token->{attributes}, $token);
7405
7406 ## TODO: $self->{form_element} if defined
7407 $self->{content_model} = RCDATA_CONTENT_MODEL;
7408 delete $self->{escape}; # MUST
7409
7410 $insert->($el);
7411
7412 my $text = '';
7413 !!!nack ('t392.1');
7414 !!!next-token;
7415 if ($token->{type} == CHARACTER_TOKEN) {
7416 $token->{data} =~ s/^\x0A//;
7417 unless (length $token->{data}) {
7418 !!!cp ('t392');
7419 !!!next-token;
7420 } else {
7421 !!!cp ('t393');
7422 }
7423 } else {
7424 !!!cp ('t394');
7425 }
7426 while ($token->{type} == CHARACTER_TOKEN) {
7427 !!!cp ('t395');
7428 $text .= $token->{data};
7429 !!!next-token;
7430 }
7431 if (length $text) {
7432 !!!cp ('t396');
7433 $el->manakai_append_text ($text);
7434 }
7435
7436 $self->{content_model} = PCDATA_CONTENT_MODEL;
7437
7438 if ($token->{type} == END_TAG_TOKEN and
7439 $token->{tag_name} eq $tag_name) {
7440 !!!cp ('t397');
7441 ## Ignore the token
7442 } else {
7443 !!!cp ('t398');
7444 !!!parse-error (type => 'in RCDATA:#eof', token => $token);
7445 }
7446 !!!next-token;
7447 next B;
7448 } elsif ($token->{tag_name} eq 'optgroup' or
7449 $token->{tag_name} eq 'option') {
7450 ## has an |option| element in scope
7451 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7452 my $node = $self->{open_elements}->[$_];
7453 if ($node->[1] & OPTION_EL) {
7454 !!!cp ('t397.1');
7455 ## NOTE: As if </option>
7456 !!!back-token; # <option> or <optgroup>
7457 $token = {type => END_TAG_TOKEN, tag_name => 'option',
7458 line => $token->{line}, column => $token->{column}};
7459 next B;
7460 } elsif ($node->[1] & SCOPING_EL) {
7461 !!!cp ('t397.2');
7462 last INSCOPE;
7463 }
7464 } # INSCOPE
7465
7466 $reconstruct_active_formatting_elements->($insert_to_current);
7467
7468 !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7469
7470 !!!nack ('t397.3');
7471 !!!next-token;
7472 redo B;
7473 } elsif ($token->{tag_name} eq 'rt' or
7474 $token->{tag_name} eq 'rp') {
7475 ## has a |ruby| element in scope
7476 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7477 my $node = $self->{open_elements}->[$_];
7478 if ($node->[1] & RUBY_EL) {
7479 !!!cp ('t398.1');
7480 ## generate implied end tags
7481 while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
7482 !!!cp ('t398.2');
7483 pop @{$self->{open_elements}};
7484 }
7485 unless ($self->{open_elements}->[-1]->[1] & RUBY_EL) {
7486 !!!cp ('t398.3');
7487 !!!parse-error (type => 'not closed',
7488 text => $self->{open_elements}->[-1]->[0]
7489 ->manakai_local_name,
7490 token => $token);
7491 pop @{$self->{open_elements}}
7492 while not $self->{open_elements}->[-1]->[1] & RUBY_EL;
7493 }
7494 last INSCOPE;
7495 } elsif ($node->[1] & SCOPING_EL) {
7496 !!!cp ('t398.4');
7497 last INSCOPE;
7498 }
7499 } # INSCOPE
7500
7501 !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7502
7503 !!!nack ('t398.5');
7504 !!!next-token;
7505 redo B;
7506 } elsif ($token->{tag_name} eq 'math' or
7507 $token->{tag_name} eq 'svg') {
7508 $reconstruct_active_formatting_elements->($insert_to_current);
7509
7510 ## "Adjust MathML attributes" ('math' only) - done in insert-element-f
7511
7512 ## "adjust SVG attributes" ('svg' only) - done in insert-element-f
7513
7514 ## "adjust foreign attributes" - done in insert-element-f
7515
7516 !!!insert-element-f ($token->{tag_name} eq 'math' ? $MML_NS : $SVG_NS, $token->{tag_name}, $token->{attributes}, $token);
7517
7518 if ($self->{self_closing}) {
7519 pop @{$self->{open_elements}};
7520 !!!ack ('t398.6');
7521 } else {
7522 !!!cp ('t398.7');
7523 $self->{insertion_mode} |= IN_FOREIGN_CONTENT_IM;
7524 ## NOTE: |<body><math><mi><svg>| -> "in foreign content" insertion
7525 ## mode, "in body" (not "in foreign content") secondary insertion
7526 ## mode, maybe.
7527 }
7528
7529 !!!next-token;
7530 next B;
7531 } elsif ({
7532 caption => 1, col => 1, colgroup => 1, frame => 1,
7533 frameset => 1, head => 1,
7534 tbody => 1, td => 1, tfoot => 1, th => 1,
7535 thead => 1, tr => 1,
7536 }->{$token->{tag_name}}) {
7537 !!!cp ('t401');
7538 !!!parse-error (type => 'in body',
7539 text => $token->{tag_name}, token => $token);
7540 ## Ignore the token
7541 !!!nack ('t401.1'); ## NOTE: |<col/>| or |<frame/>| here is an error.
7542 !!!next-token;
7543 next B;
7544 } elsif ($token->{tag_name} eq 'param' or
7545 $token->{tag_name} eq 'source') {
7546 !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7547 pop @{$self->{open_elements}};
7548
7549 !!!ack ('t398.5');
7550 !!!next-token;
7551 redo B;
7552 } else {
7553 if ($token->{tag_name} eq 'image') {
7554 !!!cp ('t384');
7555 !!!parse-error (type => 'image', token => $token);
7556 $token->{tag_name} = 'img';
7557 } else {
7558 !!!cp ('t385');
7559 }
7560
7561 ## NOTE: There is an "as if <br>" code clone.
7562 $reconstruct_active_formatting_elements->($insert_to_current);
7563
7564 !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7565
7566 if ({
7567 applet => 1, marquee => 1, object => 1,
7568 }->{$token->{tag_name}}) {
7569 !!!cp ('t380');
7570 push @$active_formatting_elements, ['#marker', ''];
7571 !!!nack ('t380.1');
7572 } elsif ({
7573 b => 1, big => 1, em => 1, font => 1, i => 1,
7574 s => 1, small => 1, strike => 1,
7575 strong => 1, tt => 1, u => 1,
7576 }->{$token->{tag_name}}) {
7577 !!!cp ('t375');
7578 push @$active_formatting_elements, $self->{open_elements}->[-1];
7579 !!!nack ('t375.1');
7580 } elsif ($token->{tag_name} eq 'input') {
7581 !!!cp ('t388');
7582 ## TODO: associate with $self->{form_element} if defined
7583 pop @{$self->{open_elements}};
7584 !!!ack ('t388.2');
7585 } elsif ({
7586 area => 1, basefont => 1, bgsound => 1, br => 1,
7587 embed => 1, img => 1, spacer => 1, wbr => 1,
7588 }->{$token->{tag_name}}) {
7589 !!!cp ('t388.1');
7590 pop @{$self->{open_elements}};
7591 !!!ack ('t388.3');
7592 } elsif ($token->{tag_name} eq 'select') {
7593 ## TODO: associate with $self->{form_element} if defined
7594
7595 if ($self->{insertion_mode} & TABLE_IMS or
7596 $self->{insertion_mode} & BODY_TABLE_IMS or
7597 $self->{insertion_mode} == IN_COLUMN_GROUP_IM) {
7598 !!!cp ('t400.1');
7599 $self->{insertion_mode} = IN_SELECT_IN_TABLE_IM;
7600 } else {
7601 !!!cp ('t400.2');
7602 $self->{insertion_mode} = IN_SELECT_IM;
7603 }
7604 !!!nack ('t400.3');
7605 } else {
7606 !!!nack ('t402');
7607 }
7608
7609 !!!next-token;
7610 next B;
7611 }
7612 } elsif ($token->{type} == END_TAG_TOKEN) {
7613 if ($token->{tag_name} eq 'body') {
7614 ## has a |body| element in scope
7615 my $i;
7616 INSCOPE: {
7617 for (reverse @{$self->{open_elements}}) {
7618 if ($_->[1] & BODY_EL) {
7619 !!!cp ('t405');
7620 $i = $_;
7621 last INSCOPE;
7622 } elsif ($_->[1] & SCOPING_EL) {
7623 !!!cp ('t405.1');
7624 last;
7625 }
7626 }
7627
7628 ## NOTE: |<marquee></body>|, |<svg><foreignobject></body>|
7629
7630 !!!parse-error (type => 'unmatched end tag',
7631 text => $token->{tag_name}, token => $token);
7632 ## NOTE: Ignore the token.
7633 !!!next-token;
7634 next B;
7635 } # INSCOPE
7636
7637 for (@{$self->{open_elements}}) {
7638 unless ($_->[1] & ALL_END_TAG_OPTIONAL_EL) {
7639 !!!cp ('t403');
7640 !!!parse-error (type => 'not closed',
7641 text => $_->[0]->manakai_local_name,
7642 token => $token);
7643 last;
7644 } else {
7645 !!!cp ('t404');
7646 }
7647 }
7648
7649 $self->{insertion_mode} = AFTER_BODY_IM;
7650 !!!next-token;
7651 next B;
7652 } elsif ($token->{tag_name} eq 'html') {
7653 ## TODO: Update this code. It seems that the code below is not
7654 ## up-to-date, though it has same effect as speced.
7655 if (@{$self->{open_elements}} > 1 and
7656 $self->{open_elements}->[1]->[1] & BODY_EL) {
7657 unless ($self->{open_elements}->[-1]->[1] & BODY_EL) {
7658 !!!cp ('t406');
7659 !!!parse-error (type => 'not closed',
7660 text => $self->{open_elements}->[1]->[0]
7661 ->manakai_local_name,
7662 token => $token);
7663 } else {
7664 !!!cp ('t407');
7665 }
7666 $self->{insertion_mode} = AFTER_BODY_IM;
7667 ## reprocess
7668 next B;
7669 } else {
7670 !!!cp ('t408');
7671 !!!parse-error (type => 'unmatched end tag',
7672 text => $token->{tag_name}, token => $token);
7673 ## Ignore the token
7674 !!!next-token;
7675 next B;
7676 }
7677 } elsif ({
7678 ## NOTE: End tags for non-phrasing flow content elements
7679
7680 ## NOTE: The normal ones
7681 address => 1, article => 1, aside => 1, blockquote => 1,
7682 center => 1, datagrid => 1, details => 1, dialog => 1,
7683 dir => 1, div => 1, dl => 1, fieldset => 1, figure => 1,
7684 footer => 1, header => 1, listing => 1, menu => 1, nav => 1,
7685 ol => 1, pre => 1, section => 1, ul => 1,
7686
7687 ## NOTE: As normal, but ... optional tags
7688 dd => 1, dt => 1, li => 1,
7689
7690 applet => 1, button => 1, marquee => 1, object => 1,
7691 }->{$token->{tag_name}}) {
7692 ## NOTE: Code for <li> start tags includes "as if </li>" code.
7693 ## Code for <dt> or <dd> start tags includes "as if </dt> or
7694 ## </dd>" code.
7695
7696 ## has an element in scope
7697 my $i;
7698 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7699 my $node = $self->{open_elements}->[$_];
7700 if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
7701 !!!cp ('t410');
7702 $i = $_;
7703 last INSCOPE;
7704 } elsif ($node->[1] & SCOPING_EL) {
7705 !!!cp ('t411');
7706 last INSCOPE;
7707 }
7708 } # INSCOPE
7709
7710 unless (defined $i) { # has an element in scope
7711 !!!cp ('t413');
7712 !!!parse-error (type => 'unmatched end tag',
7713 text => $token->{tag_name}, token => $token);
7714 ## NOTE: Ignore the token.
7715 } else {
7716 ## Step 1. generate implied end tags
7717 while ({
7718 ## END_TAG_OPTIONAL_EL
7719 dd => ($token->{tag_name} ne 'dd'),
7720 dt => ($token->{tag_name} ne 'dt'),
7721 li => ($token->{tag_name} ne 'li'),
7722 option => 1,
7723 optgroup => 1,
7724 p => 1,
7725 rt => 1,
7726 rp => 1,
7727 }->{$self->{open_elements}->[-1]->[0]->manakai_local_name}) {
7728 !!!cp ('t409');
7729 pop @{$self->{open_elements}};
7730 }
7731
7732 ## Step 2.
7733 if ($self->{open_elements}->[-1]->[0]->manakai_local_name
7734 ne $token->{tag_name}) {
7735 !!!cp ('t412');
7736 !!!parse-error (type => 'not closed',
7737 text => $self->{open_elements}->[-1]->[0]
7738 ->manakai_local_name,
7739 token => $token);
7740 } else {
7741 !!!cp ('t414');
7742 }
7743
7744 ## Step 3.
7745 splice @{$self->{open_elements}}, $i;
7746
7747 ## Step 4.
7748 $clear_up_to_marker->()
7749 if {
7750 applet => 1, button => 1, marquee => 1, object => 1,
7751 }->{$token->{tag_name}};
7752 }
7753 !!!next-token;
7754 next B;
7755 } elsif ($token->{tag_name} eq 'form') {
7756 ## NOTE: As normal, but interacts with the form element pointer
7757
7758 undef $self->{form_element};
7759
7760 ## has an element in scope
7761 my $i;
7762 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7763 my $node = $self->{open_elements}->[$_];
7764 if ($node->[1] & FORM_EL) {
7765 !!!cp ('t418');
7766 $i = $_;
7767 last INSCOPE;
7768 } elsif ($node->[1] & SCOPING_EL) {
7769 !!!cp ('t419');
7770 last INSCOPE;
7771 }
7772 } # INSCOPE
7773
7774 unless (defined $i) { # has an element in scope
7775 !!!cp ('t421');
7776 !!!parse-error (type => 'unmatched end tag',
7777 text => $token->{tag_name}, token => $token);
7778 ## NOTE: Ignore the token.
7779 } else {
7780 ## Step 1. generate implied end tags
7781 while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
7782 !!!cp ('t417');
7783 pop @{$self->{open_elements}};
7784 }
7785
7786 ## Step 2.
7787 if ($self->{open_elements}->[-1]->[0]->manakai_local_name
7788 ne $token->{tag_name}) {
7789 !!!cp ('t417.1');
7790 !!!parse-error (type => 'not closed',
7791 text => $self->{open_elements}->[-1]->[0]
7792 ->manakai_local_name,
7793 token => $token);
7794 } else {
7795 !!!cp ('t420');
7796 }
7797
7798 ## Step 3.
7799 splice @{$self->{open_elements}}, $i;
7800 }
7801
7802 !!!next-token;
7803 next B;
7804 } elsif ({
7805 ## NOTE: As normal, except acts as a closer for any ...
7806 h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,
7807 }->{$token->{tag_name}}) {
7808 ## has an element in scope
7809 my $i;
7810 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7811 my $node = $self->{open_elements}->[$_];
7812 if ($node->[1] & HEADING_EL) {
7813 !!!cp ('t423');
7814 $i = $_;
7815 last INSCOPE;
7816 } elsif ($node->[1] & SCOPING_EL) {
7817 !!!cp ('t424');
7818 last INSCOPE;
7819 }
7820 } # INSCOPE
7821
7822 unless (defined $i) { # has an element in scope
7823 !!!cp ('t425.1');
7824 !!!parse-error (type => 'unmatched end tag',
7825 text => $token->{tag_name}, token => $token);
7826 ## NOTE: Ignore the token.
7827 } else {
7828 ## Step 1. generate implied end tags
7829 while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
7830 !!!cp ('t422');
7831 pop @{$self->{open_elements}};
7832 }
7833
7834 ## Step 2.
7835 if ($self->{open_elements}->[-1]->[0]->manakai_local_name
7836 ne $token->{tag_name}) {
7837 !!!cp ('t425');
7838 !!!parse-error (type => 'unmatched end tag',
7839 text => $token->{tag_name}, token => $token);
7840 } else {
7841 !!!cp ('t426');
7842 }
7843
7844 ## Step 3.
7845 splice @{$self->{open_elements}}, $i;
7846 }
7847
7848 !!!next-token;
7849 next B;
7850 } elsif ($token->{tag_name} eq 'p') {
7851 ## NOTE: As normal, except </p> implies <p> and ...
7852
7853 ## has an element in scope
7854 my $non_optional;
7855 my $i;
7856 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7857 my $node = $self->{open_elements}->[$_];
7858 if ($node->[1] & P_EL) {
7859 !!!cp ('t410.1');
7860 $i = $_;
7861 last INSCOPE;
7862 } elsif ($node->[1] & SCOPING_EL) {
7863 !!!cp ('t411.1');
7864 last INSCOPE;
7865 } elsif ($node->[1] & END_TAG_OPTIONAL_EL) {
7866 ## NOTE: |END_TAG_OPTIONAL_EL| includes "p"
7867 !!!cp ('t411.2');
7868 #
7869 } else {
7870 !!!cp ('t411.3');
7871 $non_optional ||= $node;
7872 #
7873 }
7874 } # INSCOPE
7875
7876 if (defined $i) {
7877 ## 1. Generate implied end tags
7878 #
7879
7880 ## 2. If current node != "p", parse error
7881 if ($non_optional) {
7882 !!!cp ('t412.1');
7883 !!!parse-error (type => 'not closed',
7884 text => $non_optional->[0]->manakai_local_name,
7885 token => $token);
7886 } else {
7887 !!!cp ('t414.1');
7888 }
7889
7890 ## 3. Pop
7891 splice @{$self->{open_elements}}, $i;
7892 } else {
7893 !!!cp ('t413.1');
7894 !!!parse-error (type => 'unmatched end tag',
7895 text => $token->{tag_name}, token => $token);
7896
7897 !!!cp ('t415.1');
7898 ## As if <p>, then reprocess the current token
7899 my $el;
7900 !!!create-element ($el, $HTML_NS, 'p',, $token);
7901 $insert->($el);
7902 ## NOTE: Not inserted into |$self->{open_elements}|.
7903 }
7904
7905 !!!next-token;
7906 next B;
7907 } elsif ({
7908 a => 1,
7909 b => 1, big => 1, em => 1, font => 1, i => 1,
7910 nobr => 1, s => 1, small => 1, strike => 1,
7911 strong => 1, tt => 1, u => 1,
7912 }->{$token->{tag_name}}) {
7913 !!!cp ('t427');
7914 $formatting_end_tag->($token);
7915 next B;
7916 } elsif ($token->{tag_name} eq 'br') {
7917 !!!cp ('t428');
7918 !!!parse-error (type => 'unmatched end tag',
7919 text => 'br', token => $token);
7920
7921 ## As if <br>
7922 $reconstruct_active_formatting_elements->($insert_to_current);
7923
7924 my $el;
7925 !!!create-element ($el, $HTML_NS, 'br',, $token);
7926 $insert->($el);
7927
7928 ## Ignore the token.
7929 !!!next-token;
7930 next B;
7931 } else {
7932 if ($token->{tag_name} eq 'sarcasm') {
7933 sleep 0.001; # take a deep breath
7934 }
7935
7936 ## Step 1
7937 my $node_i = -1;
7938 my $node = $self->{open_elements}->[$node_i];
7939
7940 ## Step 2
7941 S2: {
7942 my $node_tag_name = $node->[0]->manakai_local_name;
7943 $node_tag_name =~ tr/A-Z/a-z/; # for SVG camelCase tag names
7944 if ($node_tag_name eq $token->{tag_name}) {
7945 ## Step 1
7946 ## generate implied end tags
7947 while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
7948 !!!cp ('t430');
7949 ## NOTE: |<ruby><rt></ruby>|.
7950 ## ISSUE: <ruby><rt></rt> will also take this code path,
7951 ## which seems wrong.
7952 pop @{$self->{open_elements}};
7953 $node_i++;
7954 }
7955
7956 ## Step 2
7957 my $current_tag_name
7958 = $self->{open_elements}->[-1]->[0]->manakai_local_name;
7959 $current_tag_name =~ tr/A-Z/a-z/;
7960 if ($current_tag_name ne $token->{tag_name}) {
7961 !!!cp ('t431');
7962 ## NOTE: <x><y></x>
7963 !!!parse-error (type => 'not closed',
7964 text => $self->{open_elements}->[-1]->[0]
7965 ->manakai_local_name,
7966 token => $token);
7967 } else {
7968 !!!cp ('t432');
7969 }
7970
7971 ## Step 3
7972 splice @{$self->{open_elements}}, $node_i if $node_i < 0;
7973
7974 !!!next-token;
7975 last S2;
7976 } else {
7977 ## Step 3
7978 if (not ($node->[1] & FORMATTING_EL) and
7979 #not $phrasing_category->{$node->[1]} and
7980 ($node->[1] & SPECIAL_EL or
7981 $node->[1] & SCOPING_EL)) {
7982 !!!cp ('t433');
7983 !!!parse-error (type => 'unmatched end tag',
7984 text => $token->{tag_name}, token => $token);
7985 ## Ignore the token
7986 !!!next-token;
7987 last S2;
7988
7989 ## NOTE: |<span><dd></span>a|: In Safari 3.1.2 and Opera
7990 ## 9.27, "a" is a child of <dd> (conforming). In
7991 ## Firefox 3.0.2, "a" is a child of <body>. In WinIE 7,
7992 ## "a" is a child of both <body> and <dd>.
7993 }
7994
7995 !!!cp ('t434');
7996 }
7997
7998 ## Step 4
7999 $node_i--;
8000 $node = $self->{open_elements}->[$node_i];
8001
8002 ## Step 5;
8003 redo S2;
8004 } # S2
8005 next B;
8006 }
8007 }
8008 next B;
8009 } continue { # B
8010 if ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM) {
8011 ## NOTE: The code below is executed in cases where it does not have
8012 ## to be, but it it is harmless even in those cases.
8013 ## has an element in scope
8014 INSCOPE: {
8015 for (reverse 0..$#{$self->{open_elements}}) {
8016 my $node = $self->{open_elements}->[$_];
8017 if ($node->[1] & FOREIGN_EL) {
8018 last INSCOPE;
8019 } elsif ($node->[1] & SCOPING_EL) {
8020 last;
8021 }
8022 }
8023
8024 ## NOTE: No foreign element in scope.
8025 $self->{insertion_mode} &= ~ IN_FOREIGN_CONTENT_IM;
8026 } # INSCOPE
8027 }
8028 } # B
8029
8030 ## Stop parsing # MUST
8031
8032 ## TODO: script stuffs
8033 } # _tree_construct_main
8034
8035 sub set_inner_html ($$$$;$) {
8036 my $class = shift;
8037 my $node = shift;
8038 #my $s = \$_[0];
8039 my $onerror = $_[1];
8040 my $get_wrapper = $_[2] || sub ($) { return $_[0] };
8041
8042 ## ISSUE: Should {confident} be true?
8043
8044 my $nt = $node->node_type;
8045 if ($nt == 9) {
8046 # MUST
8047
8048 ## Step 1 # MUST
8049 ## TODO: If the document has an active parser, ...
8050 ## ISSUE: There is an issue in the spec.
8051
8052 ## Step 2 # MUST
8053 my @cn = @{$node->child_nodes};
8054 for (@cn) {
8055 $node->remove_child ($_);
8056 }
8057
8058 ## Step 3, 4, 5 # MUST
8059 $class->parse_char_string ($_[0] => $node, $onerror, $get_wrapper);
8060 } elsif ($nt == 1) {
8061 ## TODO: If non-html element
8062
8063 ## NOTE: Most of this code is copied from |parse_string|
8064
8065 ## TODO: Support for $get_wrapper
8066
8067 ## Step 1 # MUST
8068 my $this_doc = $node->owner_document;
8069 my $doc = $this_doc->implementation->create_document;
8070 $doc->manakai_is_html (1);
8071 my $p = $class->new;
8072 $p->{document} = $doc;
8073
8074 ## Step 8 # MUST
8075 my $i = 0;
8076 $p->{line_prev} = $p->{line} = 1;
8077 $p->{column_prev} = $p->{column} = 0;
8078 require Whatpm::Charset::DecodeHandle;
8079 my $input = Whatpm::Charset::DecodeHandle::CharString->new (\($_[0]));
8080 $input = $get_wrapper->($input);
8081 $p->{set_nc} = sub {
8082 my $self = shift;
8083
8084 my $char = '';
8085 if (defined $self->{next_nc}) {
8086 $char = $self->{next_nc};
8087 delete $self->{next_nc};
8088 $self->{nc} = ord $char;
8089 } else {
8090 $self->{char_buffer} = '';
8091 $self->{char_buffer_pos} = 0;
8092
8093 my $count = $input->manakai_read_until
8094 ($self->{char_buffer}, qr/[^\x00\x0A\x0D]/,
8095 $self->{char_buffer_pos});
8096 if ($count) {
8097 $self->{line_prev} = $self->{line};
8098 $self->{column_prev} = $self->{column};
8099 $self->{column}++;
8100 $self->{nc}
8101 = ord substr ($self->{char_buffer},
8102 $self->{char_buffer_pos}++, 1);
8103 return;
8104 }
8105
8106 if ($input->read ($char, 1)) {
8107 $self->{nc} = ord $char;
8108 } else {
8109 $self->{nc} = -1;
8110 return;
8111 }
8112 }
8113
8114 ($p->{line_prev}, $p->{column_prev}) = ($p->{line}, $p->{column});
8115 $p->{column}++;
8116
8117 if ($self->{nc} == 0x000A) { # LF
8118 $p->{line}++;
8119 $p->{column} = 0;
8120 !!!cp ('i1');
8121 } elsif ($self->{nc} == 0x000D) { # CR
8122 ## TODO: support for abort/streaming
8123 my $next = '';
8124 if ($input->read ($next, 1) and $next ne "\x0A") {
8125 $self->{next_nc} = $next;
8126 }
8127 $self->{nc} = 0x000A; # LF # MUST
8128 $p->{line}++;
8129 $p->{column} = 0;
8130 !!!cp ('i2');
8131 } elsif ($self->{nc} == 0x0000) { # NULL
8132 !!!cp ('i4');
8133 !!!parse-error (type => 'NULL');
8134 $self->{nc} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
8135 }
8136 };
8137
8138 $p->{read_until} = sub {
8139 #my ($scalar, $specials_range, $offset) = @_;
8140 return 0 if defined $p->{next_nc};
8141
8142 my $pattern = qr/[^$_[1]\x00\x0A\x0D]/;
8143 my $offset = $_[2] || 0;
8144
8145 if ($p->{char_buffer_pos} < length $p->{char_buffer}) {
8146 pos ($p->{char_buffer}) = $p->{char_buffer_pos};
8147 if ($p->{char_buffer} =~ /\G(?>$pattern)+/) {
8148 substr ($_[0], $offset)
8149 = substr ($p->{char_buffer}, $-[0], $+[0] - $-[0]);
8150 my $count = $+[0] - $-[0];
8151 if ($count) {
8152 $p->{column} += $count;
8153 $p->{char_buffer_pos} += $count;
8154 $p->{line_prev} = $p->{line};
8155 $p->{column_prev} = $p->{column} - 1;
8156 $p->{nc} = -1;
8157 }
8158 return $count;
8159 } else {
8160 return 0;
8161 }
8162 } else {
8163 my $count = $input->manakai_read_until ($_[0], $pattern, $_[2]);
8164 if ($count) {
8165 $p->{column} += $count;
8166 $p->{column_prev} += $count;
8167 $p->{nc} = -1;
8168 }
8169 return $count;
8170 }
8171 }; # $p->{read_until}
8172
8173 my $ponerror = $onerror || sub {
8174 my (%opt) = @_;
8175 my $line = $opt{line};
8176 my $column = $opt{column};
8177 if (defined $opt{token} and defined $opt{token}->{line}) {
8178 $line = $opt{token}->{line};
8179 $column = $opt{token}->{column};
8180 }
8181 warn "Parse error ($opt{type}) at line $line column $column\n";
8182 };
8183 $p->{parse_error} = sub {
8184 $ponerror->(line => $p->{line}, column => $p->{column}, @_);
8185 };
8186
8187 my $char_onerror = sub {
8188 my (undef, $type, %opt) = @_;
8189 $ponerror->(layer => 'encode',
8190 line => $p->{line}, column => $p->{column} + 1,
8191 %opt, type => $type);
8192 }; # $char_onerror
8193 $input->onerror ($char_onerror);
8194
8195 $p->_initialize_tokenizer;
8196 $p->_initialize_tree_constructor;
8197
8198 ## Step 2
8199 my $node_ln = $node->manakai_local_name;
8200 $p->{content_model} = {
8201 title => RCDATA_CONTENT_MODEL,
8202 textarea => RCDATA_CONTENT_MODEL,
8203 style => CDATA_CONTENT_MODEL,
8204 script => CDATA_CONTENT_MODEL,
8205 xmp => CDATA_CONTENT_MODEL,
8206 iframe => CDATA_CONTENT_MODEL,
8207 noembed => CDATA_CONTENT_MODEL,
8208 noframes => CDATA_CONTENT_MODEL,
8209 noscript => CDATA_CONTENT_MODEL,
8210 plaintext => PLAINTEXT_CONTENT_MODEL,
8211 }->{$node_ln};
8212 $p->{content_model} = PCDATA_CONTENT_MODEL
8213 unless defined $p->{content_model};
8214 ## ISSUE: What is "the name of the element"? local name?
8215
8216 $p->{inner_html_node} = [$node, $el_category->{$node_ln}];
8217 ## TODO: Foreign element OK?
8218
8219 ## Step 3
8220 my $root = $doc->create_element_ns
8221 ('http://www.w3.org/1999/xhtml', [undef, 'html']);
8222
8223 ## Step 4 # MUST
8224 $doc->append_child ($root);
8225
8226 ## Step 5 # MUST
8227 push @{$p->{open_elements}}, [$root, $el_category->{html}];
8228
8229 undef $p->{head_element};
8230 undef $p->{head_element_inserted};
8231
8232 ## Step 6 # MUST
8233 $p->_reset_insertion_mode;
8234
8235 ## Step 7 # MUST
8236 my $anode = $node;
8237 AN: while (defined $anode) {
8238 if ($anode->node_type == 1) {
8239 my $nsuri = $anode->namespace_uri;
8240 if (defined $nsuri and $nsuri eq 'http://www.w3.org/1999/xhtml') {
8241 if ($anode->manakai_local_name eq 'form') {
8242 !!!cp ('i5');
8243 $p->{form_element} = $anode;
8244 last AN;
8245 }
8246 }
8247 }
8248 $anode = $anode->parent_node;
8249 } # AN
8250
8251 ## Step 9 # MUST
8252 {
8253 my $self = $p;
8254 !!!next-token;
8255 }
8256 $p->_tree_construction_main;
8257
8258 ## Step 10 # MUST
8259 my @cn = @{$node->child_nodes};
8260 for (@cn) {
8261 $node->remove_child ($_);
8262 }
8263 ## ISSUE: mutation events? read-only?
8264
8265 ## Step 11 # MUST
8266 @cn = @{$root->child_nodes};
8267 for (@cn) {
8268 $this_doc->adopt_node ($_);
8269 $node->append_child ($_);
8270 }
8271 ## ISSUE: mutation events?
8272
8273 $p->_terminate_tree_constructor;
8274
8275 delete $p->{parse_error}; # delete loop
8276 } else {
8277 die "$0: |set_inner_html| is not defined for node of type $nt";
8278 }
8279 } # set_inner_html
8280
8281 } # tree construction stage
8282
8283 package Whatpm::HTML::RestartParser;
8284 push our @ISA, 'Error';
8285
8286 1;
8287 # $Date: 2008/10/04 14:31:28 $

[email protected]
ViewVC Help
Powered by ViewVC 1.1.24