/[suikacvs]/markup/html/whatpm/Whatpm/HTML.pm.src
Suika

Contents of /markup/html/whatpm/Whatpm/HTML.pm.src

Parent Directory Parent Directory | Revision Log Revision Log


Revision 1.174 - (show annotations) (download) (as text)
Sun Sep 14 06:32:49 2008 UTC (18 years ago) by wakaba
Branch: MAIN
Changes since 1.173: +7 -5 lines
File MIME type: application/x-wais-source
++ whatpm/Whatpm/ChangeLog	14 Sep 2008 06:32:02 -0000
	* HTML.pm.src ($char_onerror): Have character decoder's |line|
	and |column| a higher priority than the one set by the
	tokenizer's input handler.
	($self->{read_until}): Exclude U+FFFD (but this might
	not be necessary, since now we do line/column fixup in
	the character decode handle).

2008-09-14  Wakaba  <wakaba@suika.fam.cx>

++ whatpm/Whatpm/Charset/ChangeLog	14 Sep 2008 06:32:40 -0000
	* DecodeHandle.pm: EUCJP class reimplemented using |read|-centric
	model.

2008-09-14  Wakaba  <wakaba@suika.fam.cx>

1 package Whatpm::HTML;
2 use strict;
3 our $VERSION=do{my @r=(q$Revision: 1.173 $=~/\d+/g);sprintf "%d."."%02d" x $#r,@r};
4 use Error qw(:try);
5
6 ## ISSUE:
7 ## var doc = implementation.createDocument (null, null, null);
8 ## doc.write ('');
9 ## alert (doc.compatMode);
10
11 require IO::Handle;
12
13 my $HTML_NS = q<http://www.w3.org/1999/xhtml>;
14 my $MML_NS = q<http://www.w3.org/1998/Math/MathML>;
15 my $SVG_NS = q<http://www.w3.org/2000/svg>;
16 my $XLINK_NS = q<http://www.w3.org/1999/xlink>;
17 my $XML_NS = q<http://www.w3.org/XML/1998/namespace>;
18 my $XMLNS_NS = q<http://www.w3.org/2000/xmlns/>;
19
20 sub A_EL () { 0b1 }
21 sub ADDRESS_EL () { 0b10 }
22 sub BODY_EL () { 0b100 }
23 sub BUTTON_EL () { 0b1000 }
24 sub CAPTION_EL () { 0b10000 }
25 sub DD_EL () { 0b100000 }
26 sub DIV_EL () { 0b1000000 }
27 sub DT_EL () { 0b10000000 }
28 sub FORM_EL () { 0b100000000 }
29 sub FORMATTING_EL () { 0b1000000000 }
30 sub FRAMESET_EL () { 0b10000000000 }
31 sub HEADING_EL () { 0b100000000000 }
32 sub HTML_EL () { 0b1000000000000 }
33 sub LI_EL () { 0b10000000000000 }
34 sub NOBR_EL () { 0b100000000000000 }
35 sub OPTION_EL () { 0b1000000000000000 }
36 sub OPTGROUP_EL () { 0b10000000000000000 }
37 sub P_EL () { 0b100000000000000000 }
38 sub SELECT_EL () { 0b1000000000000000000 }
39 sub TABLE_EL () { 0b10000000000000000000 }
40 sub TABLE_CELL_EL () { 0b100000000000000000000 }
41 sub TABLE_ROW_EL () { 0b1000000000000000000000 }
42 sub TABLE_ROW_GROUP_EL () { 0b10000000000000000000000 }
43 sub MISC_SCOPING_EL () { 0b100000000000000000000000 }
44 sub MISC_SPECIAL_EL () { 0b1000000000000000000000000 }
45 sub FOREIGN_EL () { 0b10000000000000000000000000 }
46 sub FOREIGN_FLOW_CONTENT_EL () { 0b100000000000000000000000000 }
47 sub MML_AXML_EL () { 0b1000000000000000000000000000 }
48 sub RUBY_EL () { 0b10000000000000000000000000000 }
49 sub RUBY_COMPONENT_EL () { 0b100000000000000000000000000000 }
50
51 sub TABLE_ROWS_EL () {
52 TABLE_EL |
53 TABLE_ROW_EL |
54 TABLE_ROW_GROUP_EL
55 }
56
57 ## NOTE: Used in "generate implied end tags" algorithm.
58 ## NOTE: There is a code where a modified version of END_TAG_OPTIONAL_EL
59 ## is used in "generate implied end tags" implementation (search for the
60 ## function mae).
61 sub END_TAG_OPTIONAL_EL () {
62 DD_EL |
63 DT_EL |
64 LI_EL |
65 P_EL |
66 RUBY_COMPONENT_EL
67 }
68
69 ## NOTE: Used in </body> and EOF algorithms.
70 sub ALL_END_TAG_OPTIONAL_EL () {
71 DD_EL |
72 DT_EL |
73 LI_EL |
74 P_EL |
75
76 BODY_EL |
77 HTML_EL |
78 TABLE_CELL_EL |
79 TABLE_ROW_EL |
80 TABLE_ROW_GROUP_EL
81 }
82
83 sub SCOPING_EL () {
84 BUTTON_EL |
85 CAPTION_EL |
86 HTML_EL |
87 TABLE_EL |
88 TABLE_CELL_EL |
89 MISC_SCOPING_EL
90 }
91
92 sub TABLE_SCOPING_EL () {
93 HTML_EL |
94 TABLE_EL
95 }
96
97 sub TABLE_ROWS_SCOPING_EL () {
98 HTML_EL |
99 TABLE_ROW_GROUP_EL
100 }
101
102 sub TABLE_ROW_SCOPING_EL () {
103 HTML_EL |
104 TABLE_ROW_EL
105 }
106
107 sub SPECIAL_EL () {
108 ADDRESS_EL |
109 BODY_EL |
110 DIV_EL |
111
112 DD_EL |
113 DT_EL |
114 LI_EL |
115 P_EL |
116
117 FORM_EL |
118 FRAMESET_EL |
119 HEADING_EL |
120 OPTION_EL |
121 OPTGROUP_EL |
122 SELECT_EL |
123 TABLE_ROW_EL |
124 TABLE_ROW_GROUP_EL |
125 MISC_SPECIAL_EL
126 }
127
128 my $el_category = {
129 a => A_EL | FORMATTING_EL,
130 address => ADDRESS_EL,
131 applet => MISC_SCOPING_EL,
132 area => MISC_SPECIAL_EL,
133 b => FORMATTING_EL,
134 base => MISC_SPECIAL_EL,
135 basefont => MISC_SPECIAL_EL,
136 bgsound => MISC_SPECIAL_EL,
137 big => FORMATTING_EL,
138 blockquote => MISC_SPECIAL_EL,
139 body => BODY_EL,
140 br => MISC_SPECIAL_EL,
141 button => BUTTON_EL,
142 caption => CAPTION_EL,
143 center => MISC_SPECIAL_EL,
144 col => MISC_SPECIAL_EL,
145 colgroup => MISC_SPECIAL_EL,
146 dd => DD_EL,
147 dir => MISC_SPECIAL_EL,
148 div => DIV_EL,
149 dl => MISC_SPECIAL_EL,
150 dt => DT_EL,
151 em => FORMATTING_EL,
152 embed => MISC_SPECIAL_EL,
153 fieldset => MISC_SPECIAL_EL,
154 font => FORMATTING_EL,
155 form => FORM_EL,
156 frame => MISC_SPECIAL_EL,
157 frameset => FRAMESET_EL,
158 h1 => HEADING_EL,
159 h2 => HEADING_EL,
160 h3 => HEADING_EL,
161 h4 => HEADING_EL,
162 h5 => HEADING_EL,
163 h6 => HEADING_EL,
164 head => MISC_SPECIAL_EL,
165 hr => MISC_SPECIAL_EL,
166 html => HTML_EL,
167 i => FORMATTING_EL,
168 iframe => MISC_SPECIAL_EL,
169 img => MISC_SPECIAL_EL,
170 input => MISC_SPECIAL_EL,
171 isindex => MISC_SPECIAL_EL,
172 li => LI_EL,
173 link => MISC_SPECIAL_EL,
174 listing => MISC_SPECIAL_EL,
175 marquee => MISC_SCOPING_EL,
176 menu => MISC_SPECIAL_EL,
177 meta => MISC_SPECIAL_EL,
178 nobr => NOBR_EL | FORMATTING_EL,
179 noembed => MISC_SPECIAL_EL,
180 noframes => MISC_SPECIAL_EL,
181 noscript => MISC_SPECIAL_EL,
182 object => MISC_SCOPING_EL,
183 ol => MISC_SPECIAL_EL,
184 optgroup => OPTGROUP_EL,
185 option => OPTION_EL,
186 p => P_EL,
187 param => MISC_SPECIAL_EL,
188 plaintext => MISC_SPECIAL_EL,
189 pre => MISC_SPECIAL_EL,
190 rp => RUBY_COMPONENT_EL,
191 rt => RUBY_COMPONENT_EL,
192 ruby => RUBY_EL,
193 s => FORMATTING_EL,
194 script => MISC_SPECIAL_EL,
195 select => SELECT_EL,
196 small => FORMATTING_EL,
197 spacer => MISC_SPECIAL_EL,
198 strike => FORMATTING_EL,
199 strong => FORMATTING_EL,
200 style => MISC_SPECIAL_EL,
201 table => TABLE_EL,
202 tbody => TABLE_ROW_GROUP_EL,
203 td => TABLE_CELL_EL,
204 textarea => MISC_SPECIAL_EL,
205 tfoot => TABLE_ROW_GROUP_EL,
206 th => TABLE_CELL_EL,
207 thead => TABLE_ROW_GROUP_EL,
208 title => MISC_SPECIAL_EL,
209 tr => TABLE_ROW_EL,
210 tt => FORMATTING_EL,
211 u => FORMATTING_EL,
212 ul => MISC_SPECIAL_EL,
213 wbr => MISC_SPECIAL_EL,
214 };
215
216 my $el_category_f = {
217 $MML_NS => {
218 'annotation-xml' => MML_AXML_EL,
219 mi => FOREIGN_FLOW_CONTENT_EL,
220 mo => FOREIGN_FLOW_CONTENT_EL,
221 mn => FOREIGN_FLOW_CONTENT_EL,
222 ms => FOREIGN_FLOW_CONTENT_EL,
223 mtext => FOREIGN_FLOW_CONTENT_EL,
224 },
225 $SVG_NS => {
226 foreignObject => FOREIGN_FLOW_CONTENT_EL,
227 desc => FOREIGN_FLOW_CONTENT_EL,
228 title => FOREIGN_FLOW_CONTENT_EL,
229 },
230 ## NOTE: In addition, FOREIGN_EL is set to non-HTML elements.
231 };
232
233 my $svg_attr_name = {
234 attributename => 'attributeName',
235 attributetype => 'attributeType',
236 basefrequency => 'baseFrequency',
237 baseprofile => 'baseProfile',
238 calcmode => 'calcMode',
239 clippathunits => 'clipPathUnits',
240 contentscripttype => 'contentScriptType',
241 contentstyletype => 'contentStyleType',
242 diffuseconstant => 'diffuseConstant',
243 edgemode => 'edgeMode',
244 externalresourcesrequired => 'externalResourcesRequired',
245 filterres => 'filterRes',
246 filterunits => 'filterUnits',
247 glyphref => 'glyphRef',
248 gradienttransform => 'gradientTransform',
249 gradientunits => 'gradientUnits',
250 kernelmatrix => 'kernelMatrix',
251 kernelunitlength => 'kernelUnitLength',
252 keypoints => 'keyPoints',
253 keysplines => 'keySplines',
254 keytimes => 'keyTimes',
255 lengthadjust => 'lengthAdjust',
256 limitingconeangle => 'limitingConeAngle',
257 markerheight => 'markerHeight',
258 markerunits => 'markerUnits',
259 markerwidth => 'markerWidth',
260 maskcontentunits => 'maskContentUnits',
261 maskunits => 'maskUnits',
262 numoctaves => 'numOctaves',
263 pathlength => 'pathLength',
264 patterncontentunits => 'patternContentUnits',
265 patterntransform => 'patternTransform',
266 patternunits => 'patternUnits',
267 pointsatx => 'pointsAtX',
268 pointsaty => 'pointsAtY',
269 pointsatz => 'pointsAtZ',
270 preservealpha => 'preserveAlpha',
271 preserveaspectratio => 'preserveAspectRatio',
272 primitiveunits => 'primitiveUnits',
273 refx => 'refX',
274 refy => 'refY',
275 repeatcount => 'repeatCount',
276 repeatdur => 'repeatDur',
277 requiredextensions => 'requiredExtensions',
278 requiredfeatures => 'requiredFeatures',
279 specularconstant => 'specularConstant',
280 specularexponent => 'specularExponent',
281 spreadmethod => 'spreadMethod',
282 startoffset => 'startOffset',
283 stddeviation => 'stdDeviation',
284 stitchtiles => 'stitchTiles',
285 surfacescale => 'surfaceScale',
286 systemlanguage => 'systemLanguage',
287 tablevalues => 'tableValues',
288 targetx => 'targetX',
289 targety => 'targetY',
290 textlength => 'textLength',
291 viewbox => 'viewBox',
292 viewtarget => 'viewTarget',
293 xchannelselector => 'xChannelSelector',
294 ychannelselector => 'yChannelSelector',
295 zoomandpan => 'zoomAndPan',
296 };
297
298 my $foreign_attr_xname = {
299 'xlink:actuate' => [$XLINK_NS, ['xlink', 'actuate']],
300 'xlink:arcrole' => [$XLINK_NS, ['xlink', 'arcrole']],
301 'xlink:href' => [$XLINK_NS, ['xlink', 'href']],
302 'xlink:role' => [$XLINK_NS, ['xlink', 'role']],
303 'xlink:show' => [$XLINK_NS, ['xlink', 'show']],
304 'xlink:title' => [$XLINK_NS, ['xlink', 'title']],
305 'xlink:type' => [$XLINK_NS, ['xlink', 'type']],
306 'xml:base' => [$XML_NS, ['xml', 'base']],
307 'xml:lang' => [$XML_NS, ['xml', 'lang']],
308 'xml:space' => [$XML_NS, ['xml', 'space']],
309 'xmlns' => [$XMLNS_NS, [undef, 'xmlns']],
310 'xmlns:xlink' => [$XMLNS_NS, ['xmlns', 'xlink']],
311 };
312
313 ## ISSUE: xmlns:xlink="non-xlink-ns" is not an error.
314
315 my $c1_entity_char = {
316 0x80 => 0x20AC,
317 0x81 => 0xFFFD,
318 0x82 => 0x201A,
319 0x83 => 0x0192,
320 0x84 => 0x201E,
321 0x85 => 0x2026,
322 0x86 => 0x2020,
323 0x87 => 0x2021,
324 0x88 => 0x02C6,
325 0x89 => 0x2030,
326 0x8A => 0x0160,
327 0x8B => 0x2039,
328 0x8C => 0x0152,
329 0x8D => 0xFFFD,
330 0x8E => 0x017D,
331 0x8F => 0xFFFD,
332 0x90 => 0xFFFD,
333 0x91 => 0x2018,
334 0x92 => 0x2019,
335 0x93 => 0x201C,
336 0x94 => 0x201D,
337 0x95 => 0x2022,
338 0x96 => 0x2013,
339 0x97 => 0x2014,
340 0x98 => 0x02DC,
341 0x99 => 0x2122,
342 0x9A => 0x0161,
343 0x9B => 0x203A,
344 0x9C => 0x0153,
345 0x9D => 0xFFFD,
346 0x9E => 0x017E,
347 0x9F => 0x0178,
348 }; # $c1_entity_char
349
350 sub parse_byte_string ($$$$;$) {
351 my $self = shift;
352 my $charset_name = shift;
353 open my $input, '<', ref $_[0] ? $_[0] : \($_[0]);
354 return $self->parse_byte_stream ($charset_name, $input, @_[1..$#_]);
355 } # parse_byte_string
356
357 sub parse_byte_stream ($$$$;$$) {
358 # my ($self, $charset_name, $byte_stream, $doc, $onerror, $get_wrapper) = @_;
359 my $self = ref $_[0] ? shift : shift->new;
360 my $charset_name = shift;
361 my $byte_stream = $_[0];
362
363 my $onerror = $_[2] || sub {
364 my (%opt) = @_;
365 warn "Parse error ($opt{type})\n";
366 };
367 $self->{parse_error} = $onerror; # updated later by parse_char_string
368
369 my $get_wrapper = $_[3] || sub ($) {
370 return $_[0]; # $_[0] = byte stream handle, returned = arg to char handle
371 };
372
373 ## HTML5 encoding sniffing algorithm
374 require Message::Charset::Info;
375 my $charset;
376 my $buffer;
377 my ($char_stream, $e_status);
378
379 SNIFFING: {
380 ## NOTE: By setting |allow_fallback| option true when the
381 ## |get_decode_handle| method is invoked, we ignore what the HTML5
382 ## spec requires, i.e. unsupported encoding should be ignored.
383 ## TODO: We should not do this unless the parser is invoked
384 ## in the conformance checking mode, in which this behavior
385 ## would be useful.
386
387 ## Step 1
388 if (defined $charset_name) {
389 $charset = Message::Charset::Info->get_by_html_name ($charset_name);
390 ## TODO: Is this ok? Transfer protocol's parameter should be
391 ## interpreted in its semantics?
392
393 ## ISSUE: Unsupported encoding is not ignored according to the spec.
394 ($char_stream, $e_status) = $charset->get_decode_handle
395 ($byte_stream, allow_error_reporting => 1,
396 allow_fallback => 1);
397 if ($char_stream) {
398 $self->{confident} = 1;
399 last SNIFFING;
400 } else {
401 ## TODO: unsupported error
402 }
403 }
404
405 ## Step 2
406 my $byte_buffer = '';
407 for (1..1024) {
408 my $char = $byte_stream->getc;
409 last unless defined $char;
410 $byte_buffer .= $char;
411 } ## TODO: timeout
412
413 ## Step 3
414 if ($byte_buffer =~ /^\xFE\xFF/) {
415 $charset = Message::Charset::Info->get_by_html_name ('utf-16be');
416 ($char_stream, $e_status) = $charset->get_decode_handle
417 ($byte_stream, allow_error_reporting => 1,
418 allow_fallback => 1, byte_buffer => \$byte_buffer);
419 $self->{confident} = 1;
420 last SNIFFING;
421 } elsif ($byte_buffer =~ /^\xFF\xFE/) {
422 $charset = Message::Charset::Info->get_by_html_name ('utf-16le');
423 ($char_stream, $e_status) = $charset->get_decode_handle
424 ($byte_stream, allow_error_reporting => 1,
425 allow_fallback => 1, byte_buffer => \$byte_buffer);
426 $self->{confident} = 1;
427 last SNIFFING;
428 } elsif ($byte_buffer =~ /^\xEF\xBB\xBF/) {
429 $charset = Message::Charset::Info->get_by_html_name ('utf-8');
430 ($char_stream, $e_status) = $charset->get_decode_handle
431 ($byte_stream, allow_error_reporting => 1,
432 allow_fallback => 1, byte_buffer => \$byte_buffer);
433 $self->{confident} = 1;
434 last SNIFFING;
435 }
436
437 ## Step 4
438 ## TODO: <meta charset>
439
440 ## Step 5
441 ## TODO: from history
442
443 ## Step 6
444 require Whatpm::Charset::UniversalCharDet;
445 $charset_name = Whatpm::Charset::UniversalCharDet->detect_byte_string
446 ($byte_buffer);
447 if (defined $charset_name) {
448 $charset = Message::Charset::Info->get_by_html_name ($charset_name);
449
450 ## ISSUE: Unsupported encoding is not ignored according to the spec.
451 require Whatpm::Charset::DecodeHandle;
452 $buffer = Whatpm::Charset::DecodeHandle::ByteBuffer->new
453 ($byte_stream);
454 ($char_stream, $e_status) = $charset->get_decode_handle
455 ($buffer, allow_error_reporting => 1,
456 allow_fallback => 1, byte_buffer => \$byte_buffer);
457 if ($char_stream) {
458 $buffer->{buffer} = $byte_buffer;
459 !!!parse-error (type => 'sniffing:chardet',
460 text => $charset_name,
461 level => $self->{level}->{info},
462 layer => 'encode',
463 line => 1, column => 1);
464 $self->{confident} = 0;
465 last SNIFFING;
466 }
467 }
468
469 ## Step 7: default
470 ## TODO: Make this configurable.
471 $charset = Message::Charset::Info->get_by_html_name ('windows-1252');
472 ## NOTE: We choose |windows-1252| here, since |utf-8| should be
473 ## detectable in the step 6.
474 require Whatpm::Charset::DecodeHandle;
475 $buffer = Whatpm::Charset::DecodeHandle::ByteBuffer->new
476 ($byte_stream);
477 ($char_stream, $e_status)
478 = $charset->get_decode_handle ($buffer,
479 allow_error_reporting => 1,
480 allow_fallback => 1,
481 byte_buffer => \$byte_buffer);
482 $buffer->{buffer} = $byte_buffer;
483 !!!parse-error (type => 'sniffing:default',
484 text => 'windows-1252',
485 level => $self->{level}->{info},
486 line => 1, column => 1,
487 layer => 'encode');
488 $self->{confident} = 0;
489 } # SNIFFING
490
491 if ($e_status & Message::Charset::Info::FALLBACK_ENCODING_IMPL ()) {
492 $self->{input_encoding} = $charset->get_iana_name; ## TODO: Should we set actual charset decoder's encoding name?
493 !!!parse-error (type => 'chardecode:fallback',
494 #text => $self->{input_encoding},
495 level => $self->{level}->{uncertain},
496 line => 1, column => 1,
497 layer => 'encode');
498 } elsif (not ($e_status &
499 Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL())) {
500 $self->{input_encoding} = $charset->get_iana_name;
501 !!!parse-error (type => 'chardecode:no error',
502 text => $self->{input_encoding},
503 level => $self->{level}->{uncertain},
504 line => 1, column => 1,
505 layer => 'encode');
506 } else {
507 $self->{input_encoding} = $charset->get_iana_name;
508 }
509
510 $self->{change_encoding} = sub {
511 my $self = shift;
512 $charset_name = shift;
513 my $token = shift;
514
515 $charset = Message::Charset::Info->get_by_html_name ($charset_name);
516 ($char_stream, $e_status) = $charset->get_decode_handle
517 ($byte_stream, allow_error_reporting => 1, allow_fallback => 1,
518 byte_buffer => \ $buffer->{buffer});
519
520 if ($char_stream) { # if supported
521 ## "Change the encoding" algorithm:
522
523 ## Step 1
524 if ($charset->{category} &
525 Message::Charset::Info::CHARSET_CATEGORY_UTF16 ()) {
526 $charset = Message::Charset::Info->get_by_html_name ('utf-8');
527 ($char_stream, $e_status) = $charset->get_decode_handle
528 ($byte_stream,
529 byte_buffer => \ $buffer->{buffer});
530 }
531 $charset_name = $charset->get_iana_name;
532
533 ## Step 2
534 if (defined $self->{input_encoding} and
535 $self->{input_encoding} eq $charset_name) {
536 !!!parse-error (type => 'charset label:matching',
537 text => $charset_name,
538 level => $self->{level}->{info});
539 $self->{confident} = 1;
540 return;
541 }
542
543 !!!parse-error (type => 'charset label detected',
544 text => $self->{input_encoding},
545 value => $charset_name,
546 level => $self->{level}->{warn},
547 token => $token);
548
549 ## Step 3
550 # if (can) {
551 ## change the encoding on the fly.
552 #$self->{confident} = 1;
553 #return;
554 # }
555
556 ## Step 4
557 throw Whatpm::HTML::RestartParser ();
558 }
559 }; # $self->{change_encoding}
560
561 my $char_onerror = sub {
562 my (undef, $type, %opt) = @_;
563 !!!parse-error (layer => 'encode',
564 line => $self->{line}, column => $self->{column} + 1,
565 %opt, type => $type);
566 if ($opt{octets}) {
567 ${$opt{octets}} = "\x{FFFD}"; # relacement character
568 }
569 };
570
571 my $wrapped_char_stream = $get_wrapper->($char_stream);
572 $wrapped_char_stream->onerror ($char_onerror);
573
574 my @args = @_; shift @args; # $s
575 my $return;
576 try {
577 $return = $self->parse_char_stream ($wrapped_char_stream, @args);
578 } catch Whatpm::HTML::RestartParser with {
579 ## NOTE: Invoked after {change_encoding}.
580
581 if ($e_status & Message::Charset::Info::FALLBACK_ENCODING_IMPL ()) {
582 $self->{input_encoding} = $charset->get_iana_name; ## TODO: Should we set actual charset decoder's encoding name?
583 !!!parse-error (type => 'chardecode:fallback',
584 level => $self->{level}->{uncertain},
585 #text => $self->{input_encoding},
586 line => 1, column => 1,
587 layer => 'encode');
588 } elsif (not ($e_status &
589 Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL())) {
590 $self->{input_encoding} = $charset->get_iana_name;
591 !!!parse-error (type => 'chardecode:no error',
592 text => $self->{input_encoding},
593 level => $self->{level}->{uncertain},
594 line => 1, column => 1,
595 layer => 'encode');
596 } else {
597 $self->{input_encoding} = $charset->get_iana_name;
598 }
599 $self->{confident} = 1;
600
601 $wrapped_char_stream = $get_wrapper->($char_stream);
602 $wrapped_char_stream->onerror ($char_onerror);
603
604 $return = $self->parse_char_stream ($wrapped_char_stream, @args);
605 };
606 return $return;
607 } # parse_byte_stream
608
609 ## NOTE: HTML5 spec says that the encoding layer MUST NOT strip BOM
610 ## and the HTML layer MUST ignore it. However, we does strip BOM in
611 ## the encoding layer and the HTML layer does not ignore any U+FEFF,
612 ## because the core part of our HTML parser expects a string of character,
613 ## not a string of bytes or code units or anything which might contain a BOM.
614 ## Therefore, any parser interface that accepts a string of bytes,
615 ## such as |parse_byte_string| in this module, must ensure that it does
616 ## strip the BOM and never strip any ZWNBSP.
617
618 sub parse_char_string ($$$;$$) {
619 #my ($self, $s, $doc, $onerror, $get_wrapper) = @_;
620 my $self = shift;
621 my $s = ref $_[0] ? $_[0] : \($_[0]);
622 require Whatpm::Charset::DecodeHandle;
623 my $input = Whatpm::Charset::DecodeHandle::CharString->new ($s);
624 if ($_[3]) {
625 $input = $_[3]->($input);
626 }
627 return $self->parse_char_stream ($input, @_[1..$#_]);
628 } # parse_char_string
629 *parse_string = \&parse_char_string; ## NOTE: Alias for backward compatibility.
630
631 sub parse_char_stream ($$$;$) {
632 my $self = ref $_[0] ? shift : shift->new;
633 my $input = $_[0];
634 $self->{document} = $_[1];
635 @{$self->{document}->child_nodes} = ();
636
637 ## NOTE: |set_inner_html| copies most of this method's code
638
639 $self->{confident} = 1 unless exists $self->{confident};
640 $self->{document}->input_encoding ($self->{input_encoding})
641 if defined $self->{input_encoding};
642
643 my $i = 0;
644 $self->{line_prev} = $self->{line} = 1;
645 $self->{column_prev} = $self->{column} = 0;
646 $self->{set_next_char} = sub {
647 my $self = shift;
648
649 pop @{$self->{prev_char}};
650 unshift @{$self->{prev_char}}, $self->{next_char};
651
652 my $char;
653 if (defined $self->{next_next_char}) {
654 $char = $self->{next_next_char};
655 delete $self->{next_next_char};
656 } else {
657 $char = $input->getc;
658 }
659 $self->{next_char} = -1 and return unless defined $char;
660 $self->{next_char} = ord $char;
661
662 ($self->{line_prev}, $self->{column_prev})
663 = ($self->{line}, $self->{column});
664 $self->{column}++;
665
666 if ($self->{next_char} == 0x000A) { # LF
667 !!!cp ('j1');
668 $self->{line}++;
669 $self->{column} = 0;
670 } elsif ($self->{next_char} == 0x000D) { # CR
671 !!!cp ('j2');
672 ## TODO: support for abort/streaming
673 my $next = $input->getc;
674 if (defined $next and $next ne "\x0A") {
675 $self->{next_next_char} = $next;
676 }
677 $self->{next_char} = 0x000A; # LF # MUST
678 $self->{line}++;
679 $self->{column} = 0;
680 } elsif ($self->{next_char} > 0x10FFFF) {
681 !!!cp ('j3');
682 $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
683 } elsif ($self->{next_char} == 0x0000) { # NULL
684 !!!cp ('j4');
685 !!!parse-error (type => 'NULL');
686 $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
687 } elsif ($self->{next_char} <= 0x0008 or
688 (0x000E <= $self->{next_char} and $self->{next_char} <= 0x001F) or
689 (0x007F <= $self->{next_char} and $self->{next_char} <= 0x009F) or
690 (0xD800 <= $self->{next_char} and $self->{next_char} <= 0xDFFF) or
691 (0xFDD0 <= $self->{next_char} and $self->{next_char} <= 0xFDDF) or
692 ## ISSUE: U+FDE0-U+FDEF are not excluded
693 {
694 0xFFFE => 1, 0xFFFF => 1, 0x1FFFE => 1, 0x1FFFF => 1,
695 0x2FFFE => 1, 0x2FFFF => 1, 0x3FFFE => 1, 0x3FFFF => 1,
696 0x4FFFE => 1, 0x4FFFF => 1, 0x5FFFE => 1, 0x5FFFF => 1,
697 0x6FFFE => 1, 0x6FFFF => 1, 0x7FFFE => 1, 0x7FFFF => 1,
698 0x8FFFE => 1, 0x8FFFF => 1, 0x9FFFE => 1, 0x9FFFF => 1,
699 0xAFFFE => 1, 0xAFFFF => 1, 0xBFFFE => 1, 0xBFFFF => 1,
700 0xCFFFE => 1, 0xCFFFF => 1, 0xDFFFE => 1, 0xDFFFF => 1,
701 0xEFFFE => 1, 0xEFFFF => 1, 0xFFFFE => 1, 0xFFFFF => 1,
702 0x10FFFE => 1, 0x10FFFF => 1,
703 }->{$self->{next_char}}) {
704 !!!cp ('j5');
705 if ($self->{next_char} < 0x10000) {
706 !!!parse-error (type => 'control char',
707 text => (sprintf 'U+%04X', $self->{next_char}));
708 } else {
709 !!!parse-error (type => 'control char',
710 text => (sprintf 'U-%08X', $self->{next_char}));
711 }
712 }
713 };
714 $self->{prev_char} = [-1, -1, -1];
715 $self->{next_char} = -1;
716
717 $self->{read_until} = sub {
718 #my ($scalar, $specials_range, $offset) = @_;
719 my $specials_range = $_[1];
720 return 0 if defined $self->{next_next_char};
721 my $count = $input->manakai_read_until
722 ($_[0],
723 qr/(?![$specials_range\x{FDD0}-\x{FDDF}\x{FFFD}-\x{FFFF}\x{1FFFE}\x{1FFFF}\x{2FFFE}\x{2FFFF}\x{3FFFE}\x{3FFFF}\x{4FFFE}\x{4FFFF}\x{5FFFE}\x{5FFFF}\x{6FFFE}\x{6FFFF}\x{7FFFE}\x{7FFFF}\x{8FFFE}\x{8FFFF}\x{9FFFE}\x{9FFFF}\x{AFFFE}\x{AFFFF}\x{BFFFE}\x{BFFFF}\x{CFFFE}\x{CFFFF}\x{DFFFE}\x{DFFFF}\x{EFFFE}\x{EFFFF}\x{FFFFE}\x{FFFFF}])[\x20-\x7E\xA0-\x{D7FF}\x{E000}-\x{10FFFD}]/,
724 $_[2]);
725 ## NOTE: We need to exclude U+FFFD, otherwise reported line/column
726 ## of unassigned/illegal code point error would be wrong.
727 if ($count) {
728 $self->{column} += $count;
729 $self->{column_prev} += $count;
730 $self->{prev_char} = [-1, -1, -1];
731 $self->{next_char} = -1;
732 }
733 return $count;
734 }; # $self->{read_until}
735
736 my $onerror = $_[2] || sub {
737 my (%opt) = @_;
738 my $line = $opt{token} ? $opt{token}->{line} : $opt{line};
739 my $column = $opt{token} ? $opt{token}->{column} : $opt{column};
740 warn "Parse error ($opt{type}) at line $line column $column\n";
741 };
742 $self->{parse_error} = sub {
743 $onerror->(line => $self->{line}, column => $self->{column}, @_);
744 };
745
746 $self->_initialize_tokenizer;
747 $self->_initialize_tree_constructor;
748 $self->_construct_tree;
749 $self->_terminate_tree_constructor;
750
751 delete $self->{parse_error}; # remove loop
752
753 return $self->{document};
754 } # parse_char_stream
755
756 sub new ($) {
757 my $class = shift;
758 my $self = bless {
759 level => {must => 'm',
760 should => 's',
761 warn => 'w',
762 info => 'i',
763 uncertain => 'u'},
764 }, $class;
765 $self->{set_next_char} = sub {
766 $self->{next_char} = -1;
767 };
768 $self->{parse_error} = sub {
769 #
770 };
771 $self->{change_encoding} = sub {
772 # if ($_[0] is a supported encoding) {
773 # run "change the encoding" algorithm;
774 # throw Whatpm::HTML::RestartParser (charset => $new_encoding);
775 # }
776 };
777 $self->{application_cache_selection} = sub {
778 #
779 };
780 return $self;
781 } # new
782
783 sub CM_ENTITY () { 0b001 } # & markup in data
784 sub CM_LIMITED_MARKUP () { 0b010 } # < markup in data (limited)
785 sub CM_FULL_MARKUP () { 0b100 } # < markup in data (any)
786
787 sub PLAINTEXT_CONTENT_MODEL () { 0 }
788 sub CDATA_CONTENT_MODEL () { CM_LIMITED_MARKUP }
789 sub RCDATA_CONTENT_MODEL () { CM_ENTITY | CM_LIMITED_MARKUP }
790 sub PCDATA_CONTENT_MODEL () { CM_ENTITY | CM_FULL_MARKUP }
791
792 sub DATA_STATE () { 0 }
793 #sub ENTITY_DATA_STATE () { 1 }
794 sub TAG_OPEN_STATE () { 2 }
795 sub CLOSE_TAG_OPEN_STATE () { 3 }
796 sub TAG_NAME_STATE () { 4 }
797 sub BEFORE_ATTRIBUTE_NAME_STATE () { 5 }
798 sub ATTRIBUTE_NAME_STATE () { 6 }
799 sub AFTER_ATTRIBUTE_NAME_STATE () { 7 }
800 sub BEFORE_ATTRIBUTE_VALUE_STATE () { 8 }
801 sub ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE () { 9 }
802 sub ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE () { 10 }
803 sub ATTRIBUTE_VALUE_UNQUOTED_STATE () { 11 }
804 #sub ENTITY_IN_ATTRIBUTE_VALUE_STATE () { 12 }
805 sub MARKUP_DECLARATION_OPEN_STATE () { 13 }
806 sub COMMENT_START_STATE () { 14 }
807 sub COMMENT_START_DASH_STATE () { 15 }
808 sub COMMENT_STATE () { 16 }
809 sub COMMENT_END_STATE () { 17 }
810 sub COMMENT_END_DASH_STATE () { 18 }
811 sub BOGUS_COMMENT_STATE () { 19 }
812 sub DOCTYPE_STATE () { 20 }
813 sub BEFORE_DOCTYPE_NAME_STATE () { 21 }
814 sub DOCTYPE_NAME_STATE () { 22 }
815 sub AFTER_DOCTYPE_NAME_STATE () { 23 }
816 sub BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE () { 24 }
817 sub DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE () { 25 }
818 sub DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE () { 26 }
819 sub AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE () { 27 }
820 sub BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE () { 28 }
821 sub DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE () { 29 }
822 sub DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE () { 30 }
823 sub AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE () { 31 }
824 sub BOGUS_DOCTYPE_STATE () { 32 }
825 sub AFTER_ATTRIBUTE_VALUE_QUOTED_STATE () { 33 }
826 sub SELF_CLOSING_START_TAG_STATE () { 34 }
827 sub CDATA_SECTION_STATE () { 35 }
828 sub MD_HYPHEN_STATE () { 36 } # "markup declaration open state" in the spec
829 sub MD_DOCTYPE_STATE () { 37 } # "markup declaration open state" in the spec
830 sub MD_CDATA_STATE () { 38 } # "markup declaration open state" in the spec
831 sub CDATA_PCDATA_CLOSE_TAG_STATE () { 39 } # "close tag open state" in the spec
832 sub CDATA_SECTION_MSE1_STATE () { 40 } # "CDATA section state" in the spec
833 sub CDATA_SECTION_MSE2_STATE () { 41 } # "CDATA section state" in the spec
834 sub PUBLIC_STATE () { 42 } # "after DOCTYPE name state" in the spec
835 sub SYSTEM_STATE () { 43 } # "after DOCTYPE name state" in the spec
836 ## NOTE: "Entity data state", "entity in attribute value state", and
837 ## "consume a character reference" algorithm are jointly implemented
838 ## using the following six states:
839 sub ENTITY_STATE () { 44 }
840 sub ENTITY_HASH_STATE () { 45 }
841 sub NCR_NUM_STATE () { 46 }
842 sub HEXREF_X_STATE () { 47 }
843 sub HEXREF_HEX_STATE () { 48 }
844 sub ENTITY_NAME_STATE () { 49 }
845
846 sub DOCTYPE_TOKEN () { 1 }
847 sub COMMENT_TOKEN () { 2 }
848 sub START_TAG_TOKEN () { 3 }
849 sub END_TAG_TOKEN () { 4 }
850 sub END_OF_FILE_TOKEN () { 5 }
851 sub CHARACTER_TOKEN () { 6 }
852
853 sub AFTER_HTML_IMS () { 0b100 }
854 sub HEAD_IMS () { 0b1000 }
855 sub BODY_IMS () { 0b10000 }
856 sub BODY_TABLE_IMS () { 0b100000 }
857 sub TABLE_IMS () { 0b1000000 }
858 sub ROW_IMS () { 0b10000000 }
859 sub BODY_AFTER_IMS () { 0b100000000 }
860 sub FRAME_IMS () { 0b1000000000 }
861 sub SELECT_IMS () { 0b10000000000 }
862 sub IN_FOREIGN_CONTENT_IM () { 0b100000000000 }
863 ## NOTE: "in foreign content" insertion mode is special; it is combined
864 ## with the secondary insertion mode. In this parser, they are stored
865 ## together in the bit-or'ed form.
866
867 ## NOTE: "initial" and "before html" insertion modes have no constants.
868
869 ## NOTE: "after after body" insertion mode.
870 sub AFTER_HTML_BODY_IM () { AFTER_HTML_IMS | BODY_AFTER_IMS }
871
872 ## NOTE: "after after frameset" insertion mode.
873 sub AFTER_HTML_FRAMESET_IM () { AFTER_HTML_IMS | FRAME_IMS }
874
875 sub IN_HEAD_IM () { HEAD_IMS | 0b00 }
876 sub IN_HEAD_NOSCRIPT_IM () { HEAD_IMS | 0b01 }
877 sub AFTER_HEAD_IM () { HEAD_IMS | 0b10 }
878 sub BEFORE_HEAD_IM () { HEAD_IMS | 0b11 }
879 sub IN_BODY_IM () { BODY_IMS }
880 sub IN_CELL_IM () { BODY_IMS | BODY_TABLE_IMS | 0b01 }
881 sub IN_CAPTION_IM () { BODY_IMS | BODY_TABLE_IMS | 0b10 }
882 sub IN_ROW_IM () { TABLE_IMS | ROW_IMS | 0b01 }
883 sub IN_TABLE_BODY_IM () { TABLE_IMS | ROW_IMS | 0b10 }
884 sub IN_TABLE_IM () { TABLE_IMS }
885 sub AFTER_BODY_IM () { BODY_AFTER_IMS }
886 sub IN_FRAMESET_IM () { FRAME_IMS | 0b01 }
887 sub AFTER_FRAMESET_IM () { FRAME_IMS | 0b10 }
888 sub IN_SELECT_IM () { SELECT_IMS | 0b01 }
889 sub IN_SELECT_IN_TABLE_IM () { SELECT_IMS | 0b10 }
890 sub IN_COLUMN_GROUP_IM () { 0b10 }
891
892 ## Implementations MUST act as if state machine in the spec
893
894 sub _initialize_tokenizer ($) {
895 my $self = shift;
896 $self->{state} = DATA_STATE; # MUST
897 #$self->{state_keyword}; # initialized when used
898 #$self->{entity__value}; # initialized when used
899 #$self->{entity__match}; # initialized when used
900 $self->{content_model} = PCDATA_CONTENT_MODEL; # be
901 undef $self->{current_token};
902 undef $self->{current_attribute};
903 undef $self->{last_emitted_start_tag_name};
904 #$self->{prev_state}; # initialized when used
905 delete $self->{self_closing};
906 # $self->{next_char}
907 !!!next-input-character;
908 $self->{token} = [];
909 # $self->{escape}
910 } # _initialize_tokenizer
911
912 ## A token has:
913 ## ->{type} == DOCTYPE_TOKEN, START_TAG_TOKEN, END_TAG_TOKEN, COMMENT_TOKEN,
914 ## CHARACTER_TOKEN, or END_OF_FILE_TOKEN
915 ## ->{name} (DOCTYPE_TOKEN)
916 ## ->{tag_name} (START_TAG_TOKEN, END_TAG_TOKEN)
917 ## ->{public_identifier} (DOCTYPE_TOKEN)
918 ## ->{system_identifier} (DOCTYPE_TOKEN)
919 ## ->{quirks} == 1 or 0 (DOCTYPE_TOKEN): "force-quirks" flag
920 ## ->{attributes} isa HASH (START_TAG_TOKEN, END_TAG_TOKEN)
921 ## ->{name}
922 ## ->{value}
923 ## ->{has_reference} == 1 or 0
924 ## ->{data} (COMMENT_TOKEN, CHARACTER_TOKEN)
925 ## NOTE: The "self-closing flag" is hold as |$self->{self_closing}|.
926 ## |->{self_closing}| is used to save the value of |$self->{self_closing}|
927 ## while the token is pushed back to the stack.
928
929 ## Emitted token MUST immediately be handled by the tree construction state.
930
931 ## Before each step, UA MAY check to see if either one of the scripts in
932 ## "list of scripts that will execute as soon as possible" or the first
933 ## script in the "list of scripts that will execute asynchronously",
934 ## has completed loading. If one has, then it MUST be executed
935 ## and removed from the list.
936
937 ## TODO: Polytheistic slash SHOULD NOT be used. (Applied only to atheists.)
938 ## (This requirement was dropped from HTML5 spec, unfortunately.)
939
940 sub _get_next_token ($) {
941 my $self = shift;
942
943 if ($self->{self_closing}) {
944 !!!parse-error (type => 'nestc', token => $self->{current_token});
945 ## NOTE: The |self_closing| flag is only set by start tag token.
946 ## In addition, when a start tag token is emitted, it is always set to
947 ## |current_token|.
948 delete $self->{self_closing};
949 }
950
951 if (@{$self->{token}}) {
952 $self->{self_closing} = $self->{token}->[0]->{self_closing};
953 return shift @{$self->{token}};
954 }
955
956 A: {
957 if ($self->{state} == DATA_STATE) {
958 if ($self->{next_char} == 0x0026) { # &
959 if ($self->{content_model} & CM_ENTITY and # PCDATA | RCDATA
960 not $self->{escape}) {
961 !!!cp (1);
962 ## NOTE: In the spec, the tokenizer is switched to the
963 ## "entity data state". In this implementation, the tokenizer
964 ## is switched to the |ENTITY_STATE|, which is an implementation
965 ## of the "consume a character reference" algorithm.
966 $self->{entity_additional} = -1;
967 $self->{prev_state} = DATA_STATE;
968 $self->{state} = ENTITY_STATE;
969 !!!next-input-character;
970 redo A;
971 } else {
972 !!!cp (2);
973 #
974 }
975 } elsif ($self->{next_char} == 0x002D) { # -
976 if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
977 unless ($self->{escape}) {
978 if ($self->{prev_char}->[0] == 0x002D and # -
979 $self->{prev_char}->[1] == 0x0021 and # !
980 $self->{prev_char}->[2] == 0x003C) { # <
981 !!!cp (3);
982 $self->{escape} = 1;
983 } else {
984 !!!cp (4);
985 }
986 } else {
987 !!!cp (5);
988 }
989 }
990
991 #
992 } elsif ($self->{next_char} == 0x003C) { # <
993 if ($self->{content_model} & CM_FULL_MARKUP or # PCDATA
994 (($self->{content_model} & CM_LIMITED_MARKUP) and # CDATA | RCDATA
995 not $self->{escape})) {
996 !!!cp (6);
997 $self->{state} = TAG_OPEN_STATE;
998 !!!next-input-character;
999 redo A;
1000 } else {
1001 !!!cp (7);
1002 #
1003 }
1004 } elsif ($self->{next_char} == 0x003E) { # >
1005 if ($self->{escape} and
1006 ($self->{content_model} & CM_LIMITED_MARKUP)) { # RCDATA | CDATA
1007 if ($self->{prev_char}->[0] == 0x002D and # -
1008 $self->{prev_char}->[1] == 0x002D) { # -
1009 !!!cp (8);
1010 delete $self->{escape};
1011 } else {
1012 !!!cp (9);
1013 }
1014 } else {
1015 !!!cp (10);
1016 }
1017
1018 #
1019 } elsif ($self->{next_char} == -1) {
1020 !!!cp (11);
1021 !!!emit ({type => END_OF_FILE_TOKEN,
1022 line => $self->{line}, column => $self->{column}});
1023 last A; ## TODO: ok?
1024 } else {
1025 !!!cp (12);
1026 }
1027 # Anything else
1028 my $token = {type => CHARACTER_TOKEN,
1029 data => chr $self->{next_char},
1030 line => $self->{line}, column => $self->{column},
1031 };
1032 $self->{read_until}->($token->{data}, q[-!<>&], length $token->{data});
1033
1034 ## Stay in the data state
1035 !!!next-input-character;
1036
1037 !!!emit ($token);
1038
1039 redo A;
1040 } elsif ($self->{state} == TAG_OPEN_STATE) {
1041 if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
1042 if ($self->{next_char} == 0x002F) { # /
1043 !!!cp (15);
1044 !!!next-input-character;
1045 $self->{state} = CLOSE_TAG_OPEN_STATE;
1046 redo A;
1047 } else {
1048 !!!cp (16);
1049 ## reconsume
1050 $self->{state} = DATA_STATE;
1051
1052 !!!emit ({type => CHARACTER_TOKEN, data => '<',
1053 line => $self->{line_prev},
1054 column => $self->{column_prev},
1055 });
1056
1057 redo A;
1058 }
1059 } elsif ($self->{content_model} & CM_FULL_MARKUP) { # PCDATA
1060 if ($self->{next_char} == 0x0021) { # !
1061 !!!cp (17);
1062 $self->{state} = MARKUP_DECLARATION_OPEN_STATE;
1063 !!!next-input-character;
1064 redo A;
1065 } elsif ($self->{next_char} == 0x002F) { # /
1066 !!!cp (18);
1067 $self->{state} = CLOSE_TAG_OPEN_STATE;
1068 !!!next-input-character;
1069 redo A;
1070 } elsif (0x0041 <= $self->{next_char} and
1071 $self->{next_char} <= 0x005A) { # A..Z
1072 !!!cp (19);
1073 $self->{current_token}
1074 = {type => START_TAG_TOKEN,
1075 tag_name => chr ($self->{next_char} + 0x0020),
1076 line => $self->{line_prev},
1077 column => $self->{column_prev}};
1078 $self->{state} = TAG_NAME_STATE;
1079 !!!next-input-character;
1080 redo A;
1081 } elsif (0x0061 <= $self->{next_char} and
1082 $self->{next_char} <= 0x007A) { # a..z
1083 !!!cp (20);
1084 $self->{current_token} = {type => START_TAG_TOKEN,
1085 tag_name => chr ($self->{next_char}),
1086 line => $self->{line_prev},
1087 column => $self->{column_prev}};
1088 $self->{state} = TAG_NAME_STATE;
1089 !!!next-input-character;
1090 redo A;
1091 } elsif ($self->{next_char} == 0x003E) { # >
1092 !!!cp (21);
1093 !!!parse-error (type => 'empty start tag',
1094 line => $self->{line_prev},
1095 column => $self->{column_prev});
1096 $self->{state} = DATA_STATE;
1097 !!!next-input-character;
1098
1099 !!!emit ({type => CHARACTER_TOKEN, data => '<>',
1100 line => $self->{line_prev},
1101 column => $self->{column_prev},
1102 });
1103
1104 redo A;
1105 } elsif ($self->{next_char} == 0x003F) { # ?
1106 !!!cp (22);
1107 !!!parse-error (type => 'pio',
1108 line => $self->{line_prev},
1109 column => $self->{column_prev});
1110 $self->{state} = BOGUS_COMMENT_STATE;
1111 $self->{current_token} = {type => COMMENT_TOKEN, data => '',
1112 line => $self->{line_prev},
1113 column => $self->{column_prev},
1114 };
1115 ## $self->{next_char} is intentionally left as is
1116 redo A;
1117 } else {
1118 !!!cp (23);
1119 !!!parse-error (type => 'bare stago',
1120 line => $self->{line_prev},
1121 column => $self->{column_prev});
1122 $self->{state} = DATA_STATE;
1123 ## reconsume
1124
1125 !!!emit ({type => CHARACTER_TOKEN, data => '<',
1126 line => $self->{line_prev},
1127 column => $self->{column_prev},
1128 });
1129
1130 redo A;
1131 }
1132 } else {
1133 die "$0: $self->{content_model} in tag open";
1134 }
1135 } elsif ($self->{state} == CLOSE_TAG_OPEN_STATE) {
1136 ## NOTE: The "close tag open state" in the spec is implemented as
1137 ## |CLOSE_TAG_OPEN_STATE| and |CDATA_PCDATA_CLOSE_TAG_STATE|.
1138
1139 my ($l, $c) = ($self->{line_prev}, $self->{column_prev} - 1); # "<"of"</"
1140 if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
1141 if (defined $self->{last_emitted_start_tag_name}) {
1142 $self->{state} = CDATA_PCDATA_CLOSE_TAG_STATE;
1143 $self->{state_keyword} = '';
1144 ## Reconsume.
1145 redo A;
1146 } else {
1147 ## No start tag token has ever been emitted
1148 ## NOTE: See <http://krijnhoetmer.nl/irc-logs/whatwg/20070626#l-564>.
1149 !!!cp (28);
1150 $self->{state} = DATA_STATE;
1151 ## Reconsume.
1152 !!!emit ({type => CHARACTER_TOKEN, data => '</',
1153 line => $l, column => $c,
1154 });
1155 redo A;
1156 }
1157 }
1158
1159 if (0x0041 <= $self->{next_char} and
1160 $self->{next_char} <= 0x005A) { # A..Z
1161 !!!cp (29);
1162 $self->{current_token}
1163 = {type => END_TAG_TOKEN,
1164 tag_name => chr ($self->{next_char} + 0x0020),
1165 line => $l, column => $c};
1166 $self->{state} = TAG_NAME_STATE;
1167 !!!next-input-character;
1168 redo A;
1169 } elsif (0x0061 <= $self->{next_char} and
1170 $self->{next_char} <= 0x007A) { # a..z
1171 !!!cp (30);
1172 $self->{current_token} = {type => END_TAG_TOKEN,
1173 tag_name => chr ($self->{next_char}),
1174 line => $l, column => $c};
1175 $self->{state} = TAG_NAME_STATE;
1176 !!!next-input-character;
1177 redo A;
1178 } elsif ($self->{next_char} == 0x003E) { # >
1179 !!!cp (31);
1180 !!!parse-error (type => 'empty end tag',
1181 line => $self->{line_prev}, ## "<" in "</>"
1182 column => $self->{column_prev} - 1);
1183 $self->{state} = DATA_STATE;
1184 !!!next-input-character;
1185 redo A;
1186 } elsif ($self->{next_char} == -1) {
1187 !!!cp (32);
1188 !!!parse-error (type => 'bare etago');
1189 $self->{state} = DATA_STATE;
1190 # reconsume
1191
1192 !!!emit ({type => CHARACTER_TOKEN, data => '</',
1193 line => $l, column => $c,
1194 });
1195
1196 redo A;
1197 } else {
1198 !!!cp (33);
1199 !!!parse-error (type => 'bogus end tag');
1200 $self->{state} = BOGUS_COMMENT_STATE;
1201 $self->{current_token} = {type => COMMENT_TOKEN, data => '',
1202 line => $self->{line_prev}, # "<" of "</"
1203 column => $self->{column_prev} - 1,
1204 };
1205 ## NOTE: $self->{next_char} is intentionally left as is.
1206 ## Although the "anything else" case of the spec not explicitly
1207 ## states that the next input character is to be reconsumed,
1208 ## it will be included to the |data| of the comment token
1209 ## generated from the bogus end tag, as defined in the
1210 ## "bogus comment state" entry.
1211 redo A;
1212 }
1213 } elsif ($self->{state} == CDATA_PCDATA_CLOSE_TAG_STATE) {
1214 my $ch = substr $self->{last_emitted_start_tag_name}, length $self->{state_keyword}, 1;
1215 if (length $ch) {
1216 my $CH = $ch;
1217 $ch =~ tr/a-z/A-Z/;
1218 my $nch = chr $self->{next_char};
1219 if ($nch eq $ch or $nch eq $CH) {
1220 !!!cp (24);
1221 ## Stay in the state.
1222 $self->{state_keyword} .= $nch;
1223 !!!next-input-character;
1224 redo A;
1225 } else {
1226 !!!cp (25);
1227 $self->{state} = DATA_STATE;
1228 ## Reconsume.
1229 !!!emit ({type => CHARACTER_TOKEN,
1230 data => '</' . $self->{state_keyword},
1231 line => $self->{line_prev},
1232 column => $self->{column_prev} - 1 - length $self->{state_keyword},
1233 });
1234 redo A;
1235 }
1236 } else { # after "<{tag-name}"
1237 unless ({
1238 0x0009 => 1, # HT
1239 0x000A => 1, # LF
1240 0x000B => 1, # VT
1241 0x000C => 1, # FF
1242 0x0020 => 1, # SP
1243 0x003E => 1, # >
1244 0x002F => 1, # /
1245 -1 => 1, # EOF
1246 }->{$self->{next_char}}) {
1247 !!!cp (26);
1248 ## Reconsume.
1249 $self->{state} = DATA_STATE;
1250 !!!emit ({type => CHARACTER_TOKEN,
1251 data => '</' . $self->{state_keyword},
1252 line => $self->{line_prev},
1253 column => $self->{column_prev} - 1 - length $self->{state_keyword},
1254 });
1255 redo A;
1256 } else {
1257 !!!cp (27);
1258 $self->{current_token}
1259 = {type => END_TAG_TOKEN,
1260 tag_name => $self->{last_emitted_start_tag_name},
1261 line => $self->{line_prev},
1262 column => $self->{column_prev} - 1 - length $self->{state_keyword}};
1263 $self->{state} = TAG_NAME_STATE;
1264 ## Reconsume.
1265 redo A;
1266 }
1267 }
1268 } elsif ($self->{state} == TAG_NAME_STATE) {
1269 if ($self->{next_char} == 0x0009 or # HT
1270 $self->{next_char} == 0x000A or # LF
1271 $self->{next_char} == 0x000B or # VT
1272 $self->{next_char} == 0x000C or # FF
1273 $self->{next_char} == 0x0020) { # SP
1274 !!!cp (34);
1275 $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1276 !!!next-input-character;
1277 redo A;
1278 } elsif ($self->{next_char} == 0x003E) { # >
1279 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1280 !!!cp (35);
1281 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1282 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1283 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1284 #if ($self->{current_token}->{attributes}) {
1285 # ## NOTE: This should never be reached.
1286 # !!! cp (36);
1287 # !!! parse-error (type => 'end tag attribute');
1288 #} else {
1289 !!!cp (37);
1290 #}
1291 } else {
1292 die "$0: $self->{current_token}->{type}: Unknown token type";
1293 }
1294 $self->{state} = DATA_STATE;
1295 !!!next-input-character;
1296
1297 !!!emit ($self->{current_token}); # start tag or end tag
1298
1299 redo A;
1300 } elsif (0x0041 <= $self->{next_char} and
1301 $self->{next_char} <= 0x005A) { # A..Z
1302 !!!cp (38);
1303 $self->{current_token}->{tag_name} .= chr ($self->{next_char} + 0x0020);
1304 # start tag or end tag
1305 ## Stay in this state
1306 !!!next-input-character;
1307 redo A;
1308 } elsif ($self->{next_char} == -1) {
1309 !!!parse-error (type => 'unclosed tag');
1310 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1311 !!!cp (39);
1312 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1313 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1314 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1315 #if ($self->{current_token}->{attributes}) {
1316 # ## NOTE: This state should never be reached.
1317 # !!! cp (40);
1318 # !!! parse-error (type => 'end tag attribute');
1319 #} else {
1320 !!!cp (41);
1321 #}
1322 } else {
1323 die "$0: $self->{current_token}->{type}: Unknown token type";
1324 }
1325 $self->{state} = DATA_STATE;
1326 # reconsume
1327
1328 !!!emit ($self->{current_token}); # start tag or end tag
1329
1330 redo A;
1331 } elsif ($self->{next_char} == 0x002F) { # /
1332 !!!cp (42);
1333 $self->{state} = SELF_CLOSING_START_TAG_STATE;
1334 !!!next-input-character;
1335 redo A;
1336 } else {
1337 !!!cp (44);
1338 $self->{current_token}->{tag_name} .= chr $self->{next_char};
1339 # start tag or end tag
1340 ## Stay in the state
1341 !!!next-input-character;
1342 redo A;
1343 }
1344 } elsif ($self->{state} == BEFORE_ATTRIBUTE_NAME_STATE) {
1345 if ($self->{next_char} == 0x0009 or # HT
1346 $self->{next_char} == 0x000A or # LF
1347 $self->{next_char} == 0x000B or # VT
1348 $self->{next_char} == 0x000C or # FF
1349 $self->{next_char} == 0x0020) { # SP
1350 !!!cp (45);
1351 ## Stay in the state
1352 !!!next-input-character;
1353 redo A;
1354 } elsif ($self->{next_char} == 0x003E) { # >
1355 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1356 !!!cp (46);
1357 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1358 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1359 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1360 if ($self->{current_token}->{attributes}) {
1361 !!!cp (47);
1362 !!!parse-error (type => 'end tag attribute');
1363 } else {
1364 !!!cp (48);
1365 }
1366 } else {
1367 die "$0: $self->{current_token}->{type}: Unknown token type";
1368 }
1369 $self->{state} = DATA_STATE;
1370 !!!next-input-character;
1371
1372 !!!emit ($self->{current_token}); # start tag or end tag
1373
1374 redo A;
1375 } elsif (0x0041 <= $self->{next_char} and
1376 $self->{next_char} <= 0x005A) { # A..Z
1377 !!!cp (49);
1378 $self->{current_attribute}
1379 = {name => chr ($self->{next_char} + 0x0020),
1380 value => '',
1381 line => $self->{line}, column => $self->{column}};
1382 $self->{state} = ATTRIBUTE_NAME_STATE;
1383 !!!next-input-character;
1384 redo A;
1385 } elsif ($self->{next_char} == 0x002F) { # /
1386 !!!cp (50);
1387 $self->{state} = SELF_CLOSING_START_TAG_STATE;
1388 !!!next-input-character;
1389 redo A;
1390 } elsif ($self->{next_char} == -1) {
1391 !!!parse-error (type => 'unclosed tag');
1392 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1393 !!!cp (52);
1394 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1395 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1396 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1397 if ($self->{current_token}->{attributes}) {
1398 !!!cp (53);
1399 !!!parse-error (type => 'end tag attribute');
1400 } else {
1401 !!!cp (54);
1402 }
1403 } else {
1404 die "$0: $self->{current_token}->{type}: Unknown token type";
1405 }
1406 $self->{state} = DATA_STATE;
1407 # reconsume
1408
1409 !!!emit ($self->{current_token}); # start tag or end tag
1410
1411 redo A;
1412 } else {
1413 if ({
1414 0x0022 => 1, # "
1415 0x0027 => 1, # '
1416 0x003D => 1, # =
1417 }->{$self->{next_char}}) {
1418 !!!cp (55);
1419 !!!parse-error (type => 'bad attribute name');
1420 } else {
1421 !!!cp (56);
1422 }
1423 $self->{current_attribute}
1424 = {name => chr ($self->{next_char}),
1425 value => '',
1426 line => $self->{line}, column => $self->{column}};
1427 $self->{state} = ATTRIBUTE_NAME_STATE;
1428 !!!next-input-character;
1429 redo A;
1430 }
1431 } elsif ($self->{state} == ATTRIBUTE_NAME_STATE) {
1432 my $before_leave = sub {
1433 if (exists $self->{current_token}->{attributes} # start tag or end tag
1434 ->{$self->{current_attribute}->{name}}) { # MUST
1435 !!!cp (57);
1436 !!!parse-error (type => 'duplicate attribute', text => $self->{current_attribute}->{name}, line => $self->{current_attribute}->{line}, column => $self->{current_attribute}->{column});
1437 ## Discard $self->{current_attribute} # MUST
1438 } else {
1439 !!!cp (58);
1440 $self->{current_token}->{attributes}->{$self->{current_attribute}->{name}}
1441 = $self->{current_attribute};
1442 }
1443 }; # $before_leave
1444
1445 if ($self->{next_char} == 0x0009 or # HT
1446 $self->{next_char} == 0x000A or # LF
1447 $self->{next_char} == 0x000B or # VT
1448 $self->{next_char} == 0x000C or # FF
1449 $self->{next_char} == 0x0020) { # SP
1450 !!!cp (59);
1451 $before_leave->();
1452 $self->{state} = AFTER_ATTRIBUTE_NAME_STATE;
1453 !!!next-input-character;
1454 redo A;
1455 } elsif ($self->{next_char} == 0x003D) { # =
1456 !!!cp (60);
1457 $before_leave->();
1458 $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;
1459 !!!next-input-character;
1460 redo A;
1461 } elsif ($self->{next_char} == 0x003E) { # >
1462 $before_leave->();
1463 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1464 !!!cp (61);
1465 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1466 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1467 !!!cp (62);
1468 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1469 if ($self->{current_token}->{attributes}) {
1470 !!!parse-error (type => 'end tag attribute');
1471 }
1472 } else {
1473 die "$0: $self->{current_token}->{type}: Unknown token type";
1474 }
1475 $self->{state} = DATA_STATE;
1476 !!!next-input-character;
1477
1478 !!!emit ($self->{current_token}); # start tag or end tag
1479
1480 redo A;
1481 } elsif (0x0041 <= $self->{next_char} and
1482 $self->{next_char} <= 0x005A) { # A..Z
1483 !!!cp (63);
1484 $self->{current_attribute}->{name} .= chr ($self->{next_char} + 0x0020);
1485 ## Stay in the state
1486 !!!next-input-character;
1487 redo A;
1488 } elsif ($self->{next_char} == 0x002F) { # /
1489 !!!cp (64);
1490 $before_leave->();
1491 $self->{state} = SELF_CLOSING_START_TAG_STATE;
1492 !!!next-input-character;
1493 redo A;
1494 } elsif ($self->{next_char} == -1) {
1495 !!!parse-error (type => 'unclosed tag');
1496 $before_leave->();
1497 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1498 !!!cp (66);
1499 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1500 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1501 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1502 if ($self->{current_token}->{attributes}) {
1503 !!!cp (67);
1504 !!!parse-error (type => 'end tag attribute');
1505 } else {
1506 ## NOTE: This state should never be reached.
1507 !!!cp (68);
1508 }
1509 } else {
1510 die "$0: $self->{current_token}->{type}: Unknown token type";
1511 }
1512 $self->{state} = DATA_STATE;
1513 # reconsume
1514
1515 !!!emit ($self->{current_token}); # start tag or end tag
1516
1517 redo A;
1518 } else {
1519 if ($self->{next_char} == 0x0022 or # "
1520 $self->{next_char} == 0x0027) { # '
1521 !!!cp (69);
1522 !!!parse-error (type => 'bad attribute name');
1523 } else {
1524 !!!cp (70);
1525 }
1526 $self->{current_attribute}->{name} .= chr ($self->{next_char});
1527 ## Stay in the state
1528 !!!next-input-character;
1529 redo A;
1530 }
1531 } elsif ($self->{state} == AFTER_ATTRIBUTE_NAME_STATE) {
1532 if ($self->{next_char} == 0x0009 or # HT
1533 $self->{next_char} == 0x000A or # LF
1534 $self->{next_char} == 0x000B or # VT
1535 $self->{next_char} == 0x000C or # FF
1536 $self->{next_char} == 0x0020) { # SP
1537 !!!cp (71);
1538 ## Stay in the state
1539 !!!next-input-character;
1540 redo A;
1541 } elsif ($self->{next_char} == 0x003D) { # =
1542 !!!cp (72);
1543 $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;
1544 !!!next-input-character;
1545 redo A;
1546 } elsif ($self->{next_char} == 0x003E) { # >
1547 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1548 !!!cp (73);
1549 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1550 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1551 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1552 if ($self->{current_token}->{attributes}) {
1553 !!!cp (74);
1554 !!!parse-error (type => 'end tag attribute');
1555 } else {
1556 ## NOTE: This state should never be reached.
1557 !!!cp (75);
1558 }
1559 } else {
1560 die "$0: $self->{current_token}->{type}: Unknown token type";
1561 }
1562 $self->{state} = DATA_STATE;
1563 !!!next-input-character;
1564
1565 !!!emit ($self->{current_token}); # start tag or end tag
1566
1567 redo A;
1568 } elsif (0x0041 <= $self->{next_char} and
1569 $self->{next_char} <= 0x005A) { # A..Z
1570 !!!cp (76);
1571 $self->{current_attribute}
1572 = {name => chr ($self->{next_char} + 0x0020),
1573 value => '',
1574 line => $self->{line}, column => $self->{column}};
1575 $self->{state} = ATTRIBUTE_NAME_STATE;
1576 !!!next-input-character;
1577 redo A;
1578 } elsif ($self->{next_char} == 0x002F) { # /
1579 !!!cp (77);
1580 $self->{state} = SELF_CLOSING_START_TAG_STATE;
1581 !!!next-input-character;
1582 redo A;
1583 } elsif ($self->{next_char} == -1) {
1584 !!!parse-error (type => 'unclosed tag');
1585 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1586 !!!cp (79);
1587 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1588 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1589 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1590 if ($self->{current_token}->{attributes}) {
1591 !!!cp (80);
1592 !!!parse-error (type => 'end tag attribute');
1593 } else {
1594 ## NOTE: This state should never be reached.
1595 !!!cp (81);
1596 }
1597 } else {
1598 die "$0: $self->{current_token}->{type}: Unknown token type";
1599 }
1600 $self->{state} = DATA_STATE;
1601 # reconsume
1602
1603 !!!emit ($self->{current_token}); # start tag or end tag
1604
1605 redo A;
1606 } else {
1607 if ($self->{next_char} == 0x0022 or # "
1608 $self->{next_char} == 0x0027) { # '
1609 !!!cp (78);
1610 !!!parse-error (type => 'bad attribute name');
1611 } else {
1612 !!!cp (82);
1613 }
1614 $self->{current_attribute}
1615 = {name => chr ($self->{next_char}),
1616 value => '',
1617 line => $self->{line}, column => $self->{column}};
1618 $self->{state} = ATTRIBUTE_NAME_STATE;
1619 !!!next-input-character;
1620 redo A;
1621 }
1622 } elsif ($self->{state} == BEFORE_ATTRIBUTE_VALUE_STATE) {
1623 if ($self->{next_char} == 0x0009 or # HT
1624 $self->{next_char} == 0x000A or # LF
1625 $self->{next_char} == 0x000B or # VT
1626 $self->{next_char} == 0x000C or # FF
1627 $self->{next_char} == 0x0020) { # SP
1628 !!!cp (83);
1629 ## Stay in the state
1630 !!!next-input-character;
1631 redo A;
1632 } elsif ($self->{next_char} == 0x0022) { # "
1633 !!!cp (84);
1634 $self->{state} = ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE;
1635 !!!next-input-character;
1636 redo A;
1637 } elsif ($self->{next_char} == 0x0026) { # &
1638 !!!cp (85);
1639 $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;
1640 ## reconsume
1641 redo A;
1642 } elsif ($self->{next_char} == 0x0027) { # '
1643 !!!cp (86);
1644 $self->{state} = ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE;
1645 !!!next-input-character;
1646 redo A;
1647 } elsif ($self->{next_char} == 0x003E) { # >
1648 !!!parse-error (type => 'empty unquoted attribute value');
1649 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1650 !!!cp (87);
1651 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1652 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1653 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1654 if ($self->{current_token}->{attributes}) {
1655 !!!cp (88);
1656 !!!parse-error (type => 'end tag attribute');
1657 } else {
1658 ## NOTE: This state should never be reached.
1659 !!!cp (89);
1660 }
1661 } else {
1662 die "$0: $self->{current_token}->{type}: Unknown token type";
1663 }
1664 $self->{state} = DATA_STATE;
1665 !!!next-input-character;
1666
1667 !!!emit ($self->{current_token}); # start tag or end tag
1668
1669 redo A;
1670 } elsif ($self->{next_char} == -1) {
1671 !!!parse-error (type => 'unclosed tag');
1672 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1673 !!!cp (90);
1674 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1675 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1676 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1677 if ($self->{current_token}->{attributes}) {
1678 !!!cp (91);
1679 !!!parse-error (type => 'end tag attribute');
1680 } else {
1681 ## NOTE: This state should never be reached.
1682 !!!cp (92);
1683 }
1684 } else {
1685 die "$0: $self->{current_token}->{type}: Unknown token type";
1686 }
1687 $self->{state} = DATA_STATE;
1688 ## reconsume
1689
1690 !!!emit ($self->{current_token}); # start tag or end tag
1691
1692 redo A;
1693 } else {
1694 if ($self->{next_char} == 0x003D) { # =
1695 !!!cp (93);
1696 !!!parse-error (type => 'bad attribute value');
1697 } else {
1698 !!!cp (94);
1699 }
1700 $self->{current_attribute}->{value} .= chr ($self->{next_char});
1701 $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;
1702 !!!next-input-character;
1703 redo A;
1704 }
1705 } elsif ($self->{state} == ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE) {
1706 if ($self->{next_char} == 0x0022) { # "
1707 !!!cp (95);
1708 $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;
1709 !!!next-input-character;
1710 redo A;
1711 } elsif ($self->{next_char} == 0x0026) { # &
1712 !!!cp (96);
1713 ## NOTE: In the spec, the tokenizer is switched to the
1714 ## "entity in attribute value state". In this implementation, the
1715 ## tokenizer is switched to the |ENTITY_STATE|, which is an
1716 ## implementation of the "consume a character reference" algorithm.
1717 $self->{prev_state} = $self->{state};
1718 $self->{entity_additional} = 0x0022; # "
1719 $self->{state} = ENTITY_STATE;
1720 !!!next-input-character;
1721 redo A;
1722 } elsif ($self->{next_char} == -1) {
1723 !!!parse-error (type => 'unclosed attribute value');
1724 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1725 !!!cp (97);
1726 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1727 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1728 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1729 if ($self->{current_token}->{attributes}) {
1730 !!!cp (98);
1731 !!!parse-error (type => 'end tag attribute');
1732 } else {
1733 ## NOTE: This state should never be reached.
1734 !!!cp (99);
1735 }
1736 } else {
1737 die "$0: $self->{current_token}->{type}: Unknown token type";
1738 }
1739 $self->{state} = DATA_STATE;
1740 ## reconsume
1741
1742 !!!emit ($self->{current_token}); # start tag or end tag
1743
1744 redo A;
1745 } else {
1746 !!!cp (100);
1747 $self->{current_attribute}->{value} .= chr ($self->{next_char});
1748 $self->{read_until}->($self->{current_attribute}->{value},
1749 q["&],
1750 length $self->{current_attribute}->{value});
1751
1752 ## Stay in the state
1753 !!!next-input-character;
1754 redo A;
1755 }
1756 } elsif ($self->{state} == ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE) {
1757 if ($self->{next_char} == 0x0027) { # '
1758 !!!cp (101);
1759 $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;
1760 !!!next-input-character;
1761 redo A;
1762 } elsif ($self->{next_char} == 0x0026) { # &
1763 !!!cp (102);
1764 ## NOTE: In the spec, the tokenizer is switched to the
1765 ## "entity in attribute value state". In this implementation, the
1766 ## tokenizer is switched to the |ENTITY_STATE|, which is an
1767 ## implementation of the "consume a character reference" algorithm.
1768 $self->{entity_additional} = 0x0027; # '
1769 $self->{prev_state} = $self->{state};
1770 $self->{state} = ENTITY_STATE;
1771 !!!next-input-character;
1772 redo A;
1773 } elsif ($self->{next_char} == -1) {
1774 !!!parse-error (type => 'unclosed attribute value');
1775 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1776 !!!cp (103);
1777 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1778 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1779 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1780 if ($self->{current_token}->{attributes}) {
1781 !!!cp (104);
1782 !!!parse-error (type => 'end tag attribute');
1783 } else {
1784 ## NOTE: This state should never be reached.
1785 !!!cp (105);
1786 }
1787 } else {
1788 die "$0: $self->{current_token}->{type}: Unknown token type";
1789 }
1790 $self->{state} = DATA_STATE;
1791 ## reconsume
1792
1793 !!!emit ($self->{current_token}); # start tag or end tag
1794
1795 redo A;
1796 } else {
1797 !!!cp (106);
1798 $self->{current_attribute}->{value} .= chr ($self->{next_char});
1799 $self->{read_until}->($self->{current_attribute}->{value},
1800 q['&],
1801 length $self->{current_attribute}->{value});
1802
1803 ## Stay in the state
1804 !!!next-input-character;
1805 redo A;
1806 }
1807 } elsif ($self->{state} == ATTRIBUTE_VALUE_UNQUOTED_STATE) {
1808 if ($self->{next_char} == 0x0009 or # HT
1809 $self->{next_char} == 0x000A or # LF
1810 $self->{next_char} == 0x000B or # HT
1811 $self->{next_char} == 0x000C or # FF
1812 $self->{next_char} == 0x0020) { # SP
1813 !!!cp (107);
1814 $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1815 !!!next-input-character;
1816 redo A;
1817 } elsif ($self->{next_char} == 0x0026) { # &
1818 !!!cp (108);
1819 ## NOTE: In the spec, the tokenizer is switched to the
1820 ## "entity in attribute value state". In this implementation, the
1821 ## tokenizer is switched to the |ENTITY_STATE|, which is an
1822 ## implementation of the "consume a character reference" algorithm.
1823 $self->{entity_additional} = -1;
1824 $self->{prev_state} = $self->{state};
1825 $self->{state} = ENTITY_STATE;
1826 !!!next-input-character;
1827 redo A;
1828 } elsif ($self->{next_char} == 0x003E) { # >
1829 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1830 !!!cp (109);
1831 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1832 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1833 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1834 if ($self->{current_token}->{attributes}) {
1835 !!!cp (110);
1836 !!!parse-error (type => 'end tag attribute');
1837 } else {
1838 ## NOTE: This state should never be reached.
1839 !!!cp (111);
1840 }
1841 } else {
1842 die "$0: $self->{current_token}->{type}: Unknown token type";
1843 }
1844 $self->{state} = DATA_STATE;
1845 !!!next-input-character;
1846
1847 !!!emit ($self->{current_token}); # start tag or end tag
1848
1849 redo A;
1850 } elsif ($self->{next_char} == -1) {
1851 !!!parse-error (type => 'unclosed tag');
1852 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1853 !!!cp (112);
1854 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1855 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1856 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1857 if ($self->{current_token}->{attributes}) {
1858 !!!cp (113);
1859 !!!parse-error (type => 'end tag attribute');
1860 } else {
1861 ## NOTE: This state should never be reached.
1862 !!!cp (114);
1863 }
1864 } else {
1865 die "$0: $self->{current_token}->{type}: Unknown token type";
1866 }
1867 $self->{state} = DATA_STATE;
1868 ## reconsume
1869
1870 !!!emit ($self->{current_token}); # start tag or end tag
1871
1872 redo A;
1873 } else {
1874 if ({
1875 0x0022 => 1, # "
1876 0x0027 => 1, # '
1877 0x003D => 1, # =
1878 }->{$self->{next_char}}) {
1879 !!!cp (115);
1880 !!!parse-error (type => 'bad attribute value');
1881 } else {
1882 !!!cp (116);
1883 }
1884 $self->{current_attribute}->{value} .= chr ($self->{next_char});
1885 $self->{read_until}->($self->{current_attribute}->{value},
1886 q["'=& >],
1887 length $self->{current_attribute}->{value});
1888
1889 ## Stay in the state
1890 !!!next-input-character;
1891 redo A;
1892 }
1893 } elsif ($self->{state} == AFTER_ATTRIBUTE_VALUE_QUOTED_STATE) {
1894 if ($self->{next_char} == 0x0009 or # HT
1895 $self->{next_char} == 0x000A or # LF
1896 $self->{next_char} == 0x000B or # VT
1897 $self->{next_char} == 0x000C or # FF
1898 $self->{next_char} == 0x0020) { # SP
1899 !!!cp (118);
1900 $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1901 !!!next-input-character;
1902 redo A;
1903 } elsif ($self->{next_char} == 0x003E) { # >
1904 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1905 !!!cp (119);
1906 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1907 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1908 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1909 if ($self->{current_token}->{attributes}) {
1910 !!!cp (120);
1911 !!!parse-error (type => 'end tag attribute');
1912 } else {
1913 ## NOTE: This state should never be reached.
1914 !!!cp (121);
1915 }
1916 } else {
1917 die "$0: $self->{current_token}->{type}: Unknown token type";
1918 }
1919 $self->{state} = DATA_STATE;
1920 !!!next-input-character;
1921
1922 !!!emit ($self->{current_token}); # start tag or end tag
1923
1924 redo A;
1925 } elsif ($self->{next_char} == 0x002F) { # /
1926 !!!cp (122);
1927 $self->{state} = SELF_CLOSING_START_TAG_STATE;
1928 !!!next-input-character;
1929 redo A;
1930 } elsif ($self->{next_char} == -1) {
1931 !!!parse-error (type => 'unclosed tag');
1932 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1933 !!!cp (122.3);
1934 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1935 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1936 if ($self->{current_token}->{attributes}) {
1937 !!!cp (122.1);
1938 !!!parse-error (type => 'end tag attribute');
1939 } else {
1940 ## NOTE: This state should never be reached.
1941 !!!cp (122.2);
1942 }
1943 } else {
1944 die "$0: $self->{current_token}->{type}: Unknown token type";
1945 }
1946 $self->{state} = DATA_STATE;
1947 ## Reconsume.
1948 !!!emit ($self->{current_token}); # start tag or end tag
1949 redo A;
1950 } else {
1951 !!!cp ('124.1');
1952 !!!parse-error (type => 'no space between attributes');
1953 $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1954 ## reconsume
1955 redo A;
1956 }
1957 } elsif ($self->{state} == SELF_CLOSING_START_TAG_STATE) {
1958 if ($self->{next_char} == 0x003E) { # >
1959 if ($self->{current_token}->{type} == END_TAG_TOKEN) {
1960 !!!cp ('124.2');
1961 !!!parse-error (type => 'nestc', token => $self->{current_token});
1962 ## TODO: Different type than slash in start tag
1963 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1964 if ($self->{current_token}->{attributes}) {
1965 !!!cp ('124.4');
1966 !!!parse-error (type => 'end tag attribute');
1967 } else {
1968 !!!cp ('124.5');
1969 }
1970 ## TODO: Test |<title></title/>|
1971 } else {
1972 !!!cp ('124.3');
1973 $self->{self_closing} = 1;
1974 }
1975
1976 $self->{state} = DATA_STATE;
1977 !!!next-input-character;
1978
1979 !!!emit ($self->{current_token}); # start tag or end tag
1980
1981 redo A;
1982 } elsif ($self->{next_char} == -1) {
1983 !!!parse-error (type => 'unclosed tag');
1984 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1985 !!!cp (124.7);
1986 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1987 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1988 if ($self->{current_token}->{attributes}) {
1989 !!!cp (124.5);
1990 !!!parse-error (type => 'end tag attribute');
1991 } else {
1992 ## NOTE: This state should never be reached.
1993 !!!cp (124.6);
1994 }
1995 } else {
1996 die "$0: $self->{current_token}->{type}: Unknown token type";
1997 }
1998 $self->{state} = DATA_STATE;
1999 ## Reconsume.
2000 !!!emit ($self->{current_token}); # start tag or end tag
2001 redo A;
2002 } else {
2003 !!!cp ('124.4');
2004 !!!parse-error (type => 'nestc');
2005 ## TODO: This error type is wrong.
2006 $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
2007 ## Reconsume.
2008 redo A;
2009 }
2010 } elsif ($self->{state} == BOGUS_COMMENT_STATE) {
2011 ## (only happen if PCDATA state)
2012
2013 ## NOTE: Unlike spec's "bogus comment state", this implementation
2014 ## consumes characters one-by-one basis.
2015
2016 if ($self->{next_char} == 0x003E) { # >
2017 !!!cp (124);
2018 $self->{state} = DATA_STATE;
2019 !!!next-input-character;
2020
2021 !!!emit ($self->{current_token}); # comment
2022 redo A;
2023 } elsif ($self->{next_char} == -1) {
2024 !!!cp (125);
2025 $self->{state} = DATA_STATE;
2026 ## reconsume
2027
2028 !!!emit ($self->{current_token}); # comment
2029 redo A;
2030 } else {
2031 !!!cp (126);
2032 $self->{current_token}->{data} .= chr ($self->{next_char}); # comment
2033 $self->{read_until}->($self->{current_token}->{data},
2034 q[>],
2035 length $self->{current_token}->{data});
2036
2037 ## Stay in the state.
2038 !!!next-input-character;
2039 redo A;
2040 }
2041 } elsif ($self->{state} == MARKUP_DECLARATION_OPEN_STATE) {
2042 ## (only happen if PCDATA state)
2043
2044 if ($self->{next_char} == 0x002D) { # -
2045 !!!cp (133);
2046 $self->{state} = MD_HYPHEN_STATE;
2047 !!!next-input-character;
2048 redo A;
2049 } elsif ($self->{next_char} == 0x0044 or # D
2050 $self->{next_char} == 0x0064) { # d
2051 ## ASCII case-insensitive.
2052 !!!cp (130);
2053 $self->{state} = MD_DOCTYPE_STATE;
2054 $self->{state_keyword} = chr $self->{next_char};
2055 !!!next-input-character;
2056 redo A;
2057 } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM and
2058 $self->{open_elements}->[-1]->[1] & FOREIGN_EL and
2059 $self->{next_char} == 0x005B) { # [
2060 !!!cp (135.4);
2061 $self->{state} = MD_CDATA_STATE;
2062 $self->{state_keyword} = '[';
2063 !!!next-input-character;
2064 redo A;
2065 } else {
2066 !!!cp (136);
2067 }
2068
2069 !!!parse-error (type => 'bogus comment',
2070 line => $self->{line_prev},
2071 column => $self->{column_prev} - 1);
2072 ## Reconsume.
2073 $self->{state} = BOGUS_COMMENT_STATE;
2074 $self->{current_token} = {type => COMMENT_TOKEN, data => '',
2075 line => $self->{line_prev},
2076 column => $self->{column_prev} - 1,
2077 };
2078 redo A;
2079 } elsif ($self->{state} == MD_HYPHEN_STATE) {
2080 if ($self->{next_char} == 0x002D) { # -
2081 !!!cp (127);
2082 $self->{current_token} = {type => COMMENT_TOKEN, data => '',
2083 line => $self->{line_prev},
2084 column => $self->{column_prev} - 2,
2085 };
2086 $self->{state} = COMMENT_START_STATE;
2087 !!!next-input-character;
2088 redo A;
2089 } else {
2090 !!!cp (128);
2091 !!!parse-error (type => 'bogus comment',
2092 line => $self->{line_prev},
2093 column => $self->{column_prev} - 2);
2094 $self->{state} = BOGUS_COMMENT_STATE;
2095 ## Reconsume.
2096 $self->{current_token} = {type => COMMENT_TOKEN,
2097 data => '-',
2098 line => $self->{line_prev},
2099 column => $self->{column_prev} - 2,
2100 };
2101 redo A;
2102 }
2103 } elsif ($self->{state} == MD_DOCTYPE_STATE) {
2104 ## ASCII case-insensitive.
2105 if ($self->{next_char} == [
2106 undef,
2107 0x004F, # O
2108 0x0043, # C
2109 0x0054, # T
2110 0x0059, # Y
2111 0x0050, # P
2112 ]->[length $self->{state_keyword}] or
2113 $self->{next_char} == [
2114 undef,
2115 0x006F, # o
2116 0x0063, # c
2117 0x0074, # t
2118 0x0079, # y
2119 0x0070, # p
2120 ]->[length $self->{state_keyword}]) {
2121 !!!cp (131);
2122 ## Stay in the state.
2123 $self->{state_keyword} .= chr $self->{next_char};
2124 !!!next-input-character;
2125 redo A;
2126 } elsif ((length $self->{state_keyword}) == 6 and
2127 ($self->{next_char} == 0x0045 or # E
2128 $self->{next_char} == 0x0065)) { # e
2129 !!!cp (129);
2130 $self->{state} = DOCTYPE_STATE;
2131 $self->{current_token} = {type => DOCTYPE_TOKEN,
2132 quirks => 1,
2133 line => $self->{line_prev},
2134 column => $self->{column_prev} - 7,
2135 };
2136 !!!next-input-character;
2137 redo A;
2138 } else {
2139 !!!cp (132);
2140 !!!parse-error (type => 'bogus comment',
2141 line => $self->{line_prev},
2142 column => $self->{column_prev} - 1 - length $self->{state_keyword});
2143 $self->{state} = BOGUS_COMMENT_STATE;
2144 ## Reconsume.
2145 $self->{current_token} = {type => COMMENT_TOKEN,
2146 data => $self->{state_keyword},
2147 line => $self->{line_prev},
2148 column => $self->{column_prev} - 1 - length $self->{state_keyword},
2149 };
2150 redo A;
2151 }
2152 } elsif ($self->{state} == MD_CDATA_STATE) {
2153 if ($self->{next_char} == {
2154 '[' => 0x0043, # C
2155 '[C' => 0x0044, # D
2156 '[CD' => 0x0041, # A
2157 '[CDA' => 0x0054, # T
2158 '[CDAT' => 0x0041, # A
2159 }->{$self->{state_keyword}}) {
2160 !!!cp (135.1);
2161 ## Stay in the state.
2162 $self->{state_keyword} .= chr $self->{next_char};
2163 !!!next-input-character;
2164 redo A;
2165 } elsif ($self->{state_keyword} eq '[CDATA' and
2166 $self->{next_char} == 0x005B) { # [
2167 !!!cp (135.2);
2168 $self->{current_token} = {type => CHARACTER_TOKEN,
2169 data => '',
2170 line => $self->{line_prev},
2171 column => $self->{column_prev} - 7};
2172 $self->{state} = CDATA_SECTION_STATE;
2173 !!!next-input-character;
2174 redo A;
2175 } else {
2176 !!!cp (135.3);
2177 !!!parse-error (type => 'bogus comment',
2178 line => $self->{line_prev},
2179 column => $self->{column_prev} - 1 - length $self->{state_keyword});
2180 $self->{state} = BOGUS_COMMENT_STATE;
2181 ## Reconsume.
2182 $self->{current_token} = {type => COMMENT_TOKEN,
2183 data => $self->{state_keyword},
2184 line => $self->{line_prev},
2185 column => $self->{column_prev} - 1 - length $self->{state_keyword},
2186 };
2187 redo A;
2188 }
2189 } elsif ($self->{state} == COMMENT_START_STATE) {
2190 if ($self->{next_char} == 0x002D) { # -
2191 !!!cp (137);
2192 $self->{state} = COMMENT_START_DASH_STATE;
2193 !!!next-input-character;
2194 redo A;
2195 } elsif ($self->{next_char} == 0x003E) { # >
2196 !!!cp (138);
2197 !!!parse-error (type => 'bogus comment');
2198 $self->{state} = DATA_STATE;
2199 !!!next-input-character;
2200
2201 !!!emit ($self->{current_token}); # comment
2202
2203 redo A;
2204 } elsif ($self->{next_char} == -1) {
2205 !!!cp (139);
2206 !!!parse-error (type => 'unclosed comment');
2207 $self->{state} = DATA_STATE;
2208 ## reconsume
2209
2210 !!!emit ($self->{current_token}); # comment
2211
2212 redo A;
2213 } else {
2214 !!!cp (140);
2215 $self->{current_token}->{data} # comment
2216 .= chr ($self->{next_char});
2217 $self->{state} = COMMENT_STATE;
2218 !!!next-input-character;
2219 redo A;
2220 }
2221 } elsif ($self->{state} == COMMENT_START_DASH_STATE) {
2222 if ($self->{next_char} == 0x002D) { # -
2223 !!!cp (141);
2224 $self->{state} = COMMENT_END_STATE;
2225 !!!next-input-character;
2226 redo A;
2227 } elsif ($self->{next_char} == 0x003E) { # >
2228 !!!cp (142);
2229 !!!parse-error (type => 'bogus comment');
2230 $self->{state} = DATA_STATE;
2231 !!!next-input-character;
2232
2233 !!!emit ($self->{current_token}); # comment
2234
2235 redo A;
2236 } elsif ($self->{next_char} == -1) {
2237 !!!cp (143);
2238 !!!parse-error (type => 'unclosed comment');
2239 $self->{state} = DATA_STATE;
2240 ## reconsume
2241
2242 !!!emit ($self->{current_token}); # comment
2243
2244 redo A;
2245 } else {
2246 !!!cp (144);
2247 $self->{current_token}->{data} # comment
2248 .= '-' . chr ($self->{next_char});
2249 $self->{state} = COMMENT_STATE;
2250 !!!next-input-character;
2251 redo A;
2252 }
2253 } elsif ($self->{state} == COMMENT_STATE) {
2254 if ($self->{next_char} == 0x002D) { # -
2255 !!!cp (145);
2256 $self->{state} = COMMENT_END_DASH_STATE;
2257 !!!next-input-character;
2258 redo A;
2259 } elsif ($self->{next_char} == -1) {
2260 !!!cp (146);
2261 !!!parse-error (type => 'unclosed comment');
2262 $self->{state} = DATA_STATE;
2263 ## reconsume
2264
2265 !!!emit ($self->{current_token}); # comment
2266
2267 redo A;
2268 } else {
2269 !!!cp (147);
2270 $self->{current_token}->{data} .= chr ($self->{next_char}); # comment
2271 $self->{read_until}->($self->{current_token}->{data},
2272 q[-],
2273 length $self->{current_token}->{data});
2274
2275 ## Stay in the state
2276 !!!next-input-character;
2277 redo A;
2278 }
2279 } elsif ($self->{state} == COMMENT_END_DASH_STATE) {
2280 if ($self->{next_char} == 0x002D) { # -
2281 !!!cp (148);
2282 $self->{state} = COMMENT_END_STATE;
2283 !!!next-input-character;
2284 redo A;
2285 } elsif ($self->{next_char} == -1) {
2286 !!!cp (149);
2287 !!!parse-error (type => 'unclosed comment');
2288 $self->{state} = DATA_STATE;
2289 ## reconsume
2290
2291 !!!emit ($self->{current_token}); # comment
2292
2293 redo A;
2294 } else {
2295 !!!cp (150);
2296 $self->{current_token}->{data} .= '-' . chr ($self->{next_char}); # comment
2297 $self->{state} = COMMENT_STATE;
2298 !!!next-input-character;
2299 redo A;
2300 }
2301 } elsif ($self->{state} == COMMENT_END_STATE) {
2302 if ($self->{next_char} == 0x003E) { # >
2303 !!!cp (151);
2304 $self->{state} = DATA_STATE;
2305 !!!next-input-character;
2306
2307 !!!emit ($self->{current_token}); # comment
2308
2309 redo A;
2310 } elsif ($self->{next_char} == 0x002D) { # -
2311 !!!cp (152);
2312 !!!parse-error (type => 'dash in comment',
2313 line => $self->{line_prev},
2314 column => $self->{column_prev});
2315 $self->{current_token}->{data} .= '-'; # comment
2316 ## Stay in the state
2317 !!!next-input-character;
2318 redo A;
2319 } elsif ($self->{next_char} == -1) {
2320 !!!cp (153);
2321 !!!parse-error (type => 'unclosed comment');
2322 $self->{state} = DATA_STATE;
2323 ## reconsume
2324
2325 !!!emit ($self->{current_token}); # comment
2326
2327 redo A;
2328 } else {
2329 !!!cp (154);
2330 !!!parse-error (type => 'dash in comment',
2331 line => $self->{line_prev},
2332 column => $self->{column_prev});
2333 $self->{current_token}->{data} .= '--' . chr ($self->{next_char}); # comment
2334 $self->{state} = COMMENT_STATE;
2335 !!!next-input-character;
2336 redo A;
2337 }
2338 } elsif ($self->{state} == DOCTYPE_STATE) {
2339 if ($self->{next_char} == 0x0009 or # HT
2340 $self->{next_char} == 0x000A or # LF
2341 $self->{next_char} == 0x000B or # VT
2342 $self->{next_char} == 0x000C or # FF
2343 $self->{next_char} == 0x0020) { # SP
2344 !!!cp (155);
2345 $self->{state} = BEFORE_DOCTYPE_NAME_STATE;
2346 !!!next-input-character;
2347 redo A;
2348 } else {
2349 !!!cp (156);
2350 !!!parse-error (type => 'no space before DOCTYPE name');
2351 $self->{state} = BEFORE_DOCTYPE_NAME_STATE;
2352 ## reconsume
2353 redo A;
2354 }
2355 } elsif ($self->{state} == BEFORE_DOCTYPE_NAME_STATE) {
2356 if ($self->{next_char} == 0x0009 or # HT
2357 $self->{next_char} == 0x000A or # LF
2358 $self->{next_char} == 0x000B or # VT
2359 $self->{next_char} == 0x000C or # FF
2360 $self->{next_char} == 0x0020) { # SP
2361 !!!cp (157);
2362 ## Stay in the state
2363 !!!next-input-character;
2364 redo A;
2365 } elsif ($self->{next_char} == 0x003E) { # >
2366 !!!cp (158);
2367 !!!parse-error (type => 'no DOCTYPE name');
2368 $self->{state} = DATA_STATE;
2369 !!!next-input-character;
2370
2371 !!!emit ($self->{current_token}); # DOCTYPE (quirks)
2372
2373 redo A;
2374 } elsif ($self->{next_char} == -1) {
2375 !!!cp (159);
2376 !!!parse-error (type => 'no DOCTYPE name');
2377 $self->{state} = DATA_STATE;
2378 ## reconsume
2379
2380 !!!emit ($self->{current_token}); # DOCTYPE (quirks)
2381
2382 redo A;
2383 } else {
2384 !!!cp (160);
2385 $self->{current_token}->{name} = chr $self->{next_char};
2386 delete $self->{current_token}->{quirks};
2387 ## ISSUE: "Set the token's name name to the" in the spec
2388 $self->{state} = DOCTYPE_NAME_STATE;
2389 !!!next-input-character;
2390 redo A;
2391 }
2392 } elsif ($self->{state} == DOCTYPE_NAME_STATE) {
2393 ## ISSUE: Redundant "First," in the spec.
2394 if ($self->{next_char} == 0x0009 or # HT
2395 $self->{next_char} == 0x000A or # LF
2396 $self->{next_char} == 0x000B or # VT
2397 $self->{next_char} == 0x000C or # FF
2398 $self->{next_char} == 0x0020) { # SP
2399 !!!cp (161);
2400 $self->{state} = AFTER_DOCTYPE_NAME_STATE;
2401 !!!next-input-character;
2402 redo A;
2403 } elsif ($self->{next_char} == 0x003E) { # >
2404 !!!cp (162);
2405 $self->{state} = DATA_STATE;
2406 !!!next-input-character;
2407
2408 !!!emit ($self->{current_token}); # DOCTYPE
2409
2410 redo A;
2411 } elsif ($self->{next_char} == -1) {
2412 !!!cp (163);
2413 !!!parse-error (type => 'unclosed DOCTYPE');
2414 $self->{state} = DATA_STATE;
2415 ## reconsume
2416
2417 $self->{current_token}->{quirks} = 1;
2418 !!!emit ($self->{current_token}); # DOCTYPE
2419
2420 redo A;
2421 } else {
2422 !!!cp (164);
2423 $self->{current_token}->{name}
2424 .= chr ($self->{next_char}); # DOCTYPE
2425 ## Stay in the state
2426 !!!next-input-character;
2427 redo A;
2428 }
2429 } elsif ($self->{state} == AFTER_DOCTYPE_NAME_STATE) {
2430 if ($self->{next_char} == 0x0009 or # HT
2431 $self->{next_char} == 0x000A or # LF
2432 $self->{next_char} == 0x000B or # VT
2433 $self->{next_char} == 0x000C or # FF
2434 $self->{next_char} == 0x0020) { # SP
2435 !!!cp (165);
2436 ## Stay in the state
2437 !!!next-input-character;
2438 redo A;
2439 } elsif ($self->{next_char} == 0x003E) { # >
2440 !!!cp (166);
2441 $self->{state} = DATA_STATE;
2442 !!!next-input-character;
2443
2444 !!!emit ($self->{current_token}); # DOCTYPE
2445
2446 redo A;
2447 } elsif ($self->{next_char} == -1) {
2448 !!!cp (167);
2449 !!!parse-error (type => 'unclosed DOCTYPE');
2450 $self->{state} = DATA_STATE;
2451 ## reconsume
2452
2453 $self->{current_token}->{quirks} = 1;
2454 !!!emit ($self->{current_token}); # DOCTYPE
2455
2456 redo A;
2457 } elsif ($self->{next_char} == 0x0050 or # P
2458 $self->{next_char} == 0x0070) { # p
2459 $self->{state} = PUBLIC_STATE;
2460 $self->{state_keyword} = chr $self->{next_char};
2461 !!!next-input-character;
2462 redo A;
2463 } elsif ($self->{next_char} == 0x0053 or # S
2464 $self->{next_char} == 0x0073) { # s
2465 $self->{state} = SYSTEM_STATE;
2466 $self->{state_keyword} = chr $self->{next_char};
2467 !!!next-input-character;
2468 redo A;
2469 } else {
2470 !!!cp (180);
2471 !!!parse-error (type => 'string after DOCTYPE name');
2472 $self->{current_token}->{quirks} = 1;
2473
2474 $self->{state} = BOGUS_DOCTYPE_STATE;
2475 !!!next-input-character;
2476 redo A;
2477 }
2478 } elsif ($self->{state} == PUBLIC_STATE) {
2479 ## ASCII case-insensitive
2480 if ($self->{next_char} == [
2481 undef,
2482 0x0055, # U
2483 0x0042, # B
2484 0x004C, # L
2485 0x0049, # I
2486 ]->[length $self->{state_keyword}] or
2487 $self->{next_char} == [
2488 undef,
2489 0x0075, # u
2490 0x0062, # b
2491 0x006C, # l
2492 0x0069, # i
2493 ]->[length $self->{state_keyword}]) {
2494 !!!cp (175);
2495 ## Stay in the state.
2496 $self->{state_keyword} .= chr $self->{next_char};
2497 !!!next-input-character;
2498 redo A;
2499 } elsif ((length $self->{state_keyword}) == 5 and
2500 ($self->{next_char} == 0x0043 or # C
2501 $self->{next_char} == 0x0063)) { # c
2502 !!!cp (168);
2503 $self->{state} = BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2504 !!!next-input-character;
2505 redo A;
2506 } else {
2507 !!!cp (169);
2508 !!!parse-error (type => 'string after DOCTYPE name',
2509 line => $self->{line_prev},
2510 column => $self->{column_prev} + 1 - length $self->{state_keyword});
2511 $self->{current_token}->{quirks} = 1;
2512
2513 $self->{state} = BOGUS_DOCTYPE_STATE;
2514 ## Reconsume.
2515 redo A;
2516 }
2517 } elsif ($self->{state} == SYSTEM_STATE) {
2518 ## ASCII case-insensitive
2519 if ($self->{next_char} == [
2520 undef,
2521 0x0059, # Y
2522 0x0053, # S
2523 0x0054, # T
2524 0x0045, # E
2525 ]->[length $self->{state_keyword}] or
2526 $self->{next_char} == [
2527 undef,
2528 0x0079, # y
2529 0x0073, # s
2530 0x0074, # t
2531 0x0065, # e
2532 ]->[length $self->{state_keyword}]) {
2533 !!!cp (170);
2534 ## Stay in the state.
2535 $self->{state_keyword} .= chr $self->{next_char};
2536 !!!next-input-character;
2537 redo A;
2538 } elsif ((length $self->{state_keyword}) == 5 and
2539 ($self->{next_char} == 0x004D or # M
2540 $self->{next_char} == 0x006D)) { # m
2541 !!!cp (171);
2542 $self->{state} = BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2543 !!!next-input-character;
2544 redo A;
2545 } else {
2546 !!!cp (172);
2547 !!!parse-error (type => 'string after DOCTYPE name',
2548 line => $self->{line_prev},
2549 column => $self->{column_prev} + 1 - length $self->{state_keyword});
2550 $self->{current_token}->{quirks} = 1;
2551
2552 $self->{state} = BOGUS_DOCTYPE_STATE;
2553 ## Reconsume.
2554 redo A;
2555 }
2556 } elsif ($self->{state} == BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {
2557 if ({
2558 0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,
2559 #0x000D => 1, # HT, LF, VT, FF, SP, CR
2560 }->{$self->{next_char}}) {
2561 !!!cp (181);
2562 ## Stay in the state
2563 !!!next-input-character;
2564 redo A;
2565 } elsif ($self->{next_char} eq 0x0022) { # "
2566 !!!cp (182);
2567 $self->{current_token}->{public_identifier} = ''; # DOCTYPE
2568 $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE;
2569 !!!next-input-character;
2570 redo A;
2571 } elsif ($self->{next_char} eq 0x0027) { # '
2572 !!!cp (183);
2573 $self->{current_token}->{public_identifier} = ''; # DOCTYPE
2574 $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE;
2575 !!!next-input-character;
2576 redo A;
2577 } elsif ($self->{next_char} eq 0x003E) { # >
2578 !!!cp (184);
2579 !!!parse-error (type => 'no PUBLIC literal');
2580
2581 $self->{state} = DATA_STATE;
2582 !!!next-input-character;
2583
2584 $self->{current_token}->{quirks} = 1;
2585 !!!emit ($self->{current_token}); # DOCTYPE
2586
2587 redo A;
2588 } elsif ($self->{next_char} == -1) {
2589 !!!cp (185);
2590 !!!parse-error (type => 'unclosed DOCTYPE');
2591
2592 $self->{state} = DATA_STATE;
2593 ## reconsume
2594
2595 $self->{current_token}->{quirks} = 1;
2596 !!!emit ($self->{current_token}); # DOCTYPE
2597
2598 redo A;
2599 } else {
2600 !!!cp (186);
2601 !!!parse-error (type => 'string after PUBLIC');
2602 $self->{current_token}->{quirks} = 1;
2603
2604 $self->{state} = BOGUS_DOCTYPE_STATE;
2605 !!!next-input-character;
2606 redo A;
2607 }
2608 } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE) {
2609 if ($self->{next_char} == 0x0022) { # "
2610 !!!cp (187);
2611 $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2612 !!!next-input-character;
2613 redo A;
2614 } elsif ($self->{next_char} == 0x003E) { # >
2615 !!!cp (188);
2616 !!!parse-error (type => 'unclosed PUBLIC literal');
2617
2618 $self->{state} = DATA_STATE;
2619 !!!next-input-character;
2620
2621 $self->{current_token}->{quirks} = 1;
2622 !!!emit ($self->{current_token}); # DOCTYPE
2623
2624 redo A;
2625 } elsif ($self->{next_char} == -1) {
2626 !!!cp (189);
2627 !!!parse-error (type => 'unclosed PUBLIC literal');
2628
2629 $self->{state} = DATA_STATE;
2630 ## reconsume
2631
2632 $self->{current_token}->{quirks} = 1;
2633 !!!emit ($self->{current_token}); # DOCTYPE
2634
2635 redo A;
2636 } else {
2637 !!!cp (190);
2638 $self->{current_token}->{public_identifier} # DOCTYPE
2639 .= chr $self->{next_char};
2640 $self->{read_until}->($self->{current_token}->{public_identifier},
2641 q[">],
2642 length $self->{current_token}->{public_identifier});
2643
2644 ## Stay in the state
2645 !!!next-input-character;
2646 redo A;
2647 }
2648 } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE) {
2649 if ($self->{next_char} == 0x0027) { # '
2650 !!!cp (191);
2651 $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2652 !!!next-input-character;
2653 redo A;
2654 } elsif ($self->{next_char} == 0x003E) { # >
2655 !!!cp (192);
2656 !!!parse-error (type => 'unclosed PUBLIC literal');
2657
2658 $self->{state} = DATA_STATE;
2659 !!!next-input-character;
2660
2661 $self->{current_token}->{quirks} = 1;
2662 !!!emit ($self->{current_token}); # DOCTYPE
2663
2664 redo A;
2665 } elsif ($self->{next_char} == -1) {
2666 !!!cp (193);
2667 !!!parse-error (type => 'unclosed PUBLIC literal');
2668
2669 $self->{state} = DATA_STATE;
2670 ## reconsume
2671
2672 $self->{current_token}->{quirks} = 1;
2673 !!!emit ($self->{current_token}); # DOCTYPE
2674
2675 redo A;
2676 } else {
2677 !!!cp (194);
2678 $self->{current_token}->{public_identifier} # DOCTYPE
2679 .= chr $self->{next_char};
2680 $self->{read_until}->($self->{current_token}->{public_identifier},
2681 q['>],
2682 length $self->{current_token}->{public_identifier});
2683
2684 ## Stay in the state
2685 !!!next-input-character;
2686 redo A;
2687 }
2688 } elsif ($self->{state} == AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {
2689 if ({
2690 0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,
2691 #0x000D => 1, # HT, LF, VT, FF, SP, CR
2692 }->{$self->{next_char}}) {
2693 !!!cp (195);
2694 ## Stay in the state
2695 !!!next-input-character;
2696 redo A;
2697 } elsif ($self->{next_char} == 0x0022) { # "
2698 !!!cp (196);
2699 $self->{current_token}->{system_identifier} = ''; # DOCTYPE
2700 $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;
2701 !!!next-input-character;
2702 redo A;
2703 } elsif ($self->{next_char} == 0x0027) { # '
2704 !!!cp (197);
2705 $self->{current_token}->{system_identifier} = ''; # DOCTYPE
2706 $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;
2707 !!!next-input-character;
2708 redo A;
2709 } elsif ($self->{next_char} == 0x003E) { # >
2710 !!!cp (198);
2711 $self->{state} = DATA_STATE;
2712 !!!next-input-character;
2713
2714 !!!emit ($self->{current_token}); # DOCTYPE
2715
2716 redo A;
2717 } elsif ($self->{next_char} == -1) {
2718 !!!cp (199);
2719 !!!parse-error (type => 'unclosed DOCTYPE');
2720
2721 $self->{state} = DATA_STATE;
2722 ## reconsume
2723
2724 $self->{current_token}->{quirks} = 1;
2725 !!!emit ($self->{current_token}); # DOCTYPE
2726
2727 redo A;
2728 } else {
2729 !!!cp (200);
2730 !!!parse-error (type => 'string after PUBLIC literal');
2731 $self->{current_token}->{quirks} = 1;
2732
2733 $self->{state} = BOGUS_DOCTYPE_STATE;
2734 !!!next-input-character;
2735 redo A;
2736 }
2737 } elsif ($self->{state} == BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {
2738 if ({
2739 0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,
2740 #0x000D => 1, # HT, LF, VT, FF, SP, CR
2741 }->{$self->{next_char}}) {
2742 !!!cp (201);
2743 ## Stay in the state
2744 !!!next-input-character;
2745 redo A;
2746 } elsif ($self->{next_char} == 0x0022) { # "
2747 !!!cp (202);
2748 $self->{current_token}->{system_identifier} = ''; # DOCTYPE
2749 $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;
2750 !!!next-input-character;
2751 redo A;
2752 } elsif ($self->{next_char} == 0x0027) { # '
2753 !!!cp (203);
2754 $self->{current_token}->{system_identifier} = ''; # DOCTYPE
2755 $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;
2756 !!!next-input-character;
2757 redo A;
2758 } elsif ($self->{next_char} == 0x003E) { # >
2759 !!!cp (204);
2760 !!!parse-error (type => 'no SYSTEM literal');
2761 $self->{state} = DATA_STATE;
2762 !!!next-input-character;
2763
2764 $self->{current_token}->{quirks} = 1;
2765 !!!emit ($self->{current_token}); # DOCTYPE
2766
2767 redo A;
2768 } elsif ($self->{next_char} == -1) {
2769 !!!cp (205);
2770 !!!parse-error (type => 'unclosed DOCTYPE');
2771
2772 $self->{state} = DATA_STATE;
2773 ## reconsume
2774
2775 $self->{current_token}->{quirks} = 1;
2776 !!!emit ($self->{current_token}); # DOCTYPE
2777
2778 redo A;
2779 } else {
2780 !!!cp (206);
2781 !!!parse-error (type => 'string after SYSTEM');
2782 $self->{current_token}->{quirks} = 1;
2783
2784 $self->{state} = BOGUS_DOCTYPE_STATE;
2785 !!!next-input-character;
2786 redo A;
2787 }
2788 } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE) {
2789 if ($self->{next_char} == 0x0022) { # "
2790 !!!cp (207);
2791 $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2792 !!!next-input-character;
2793 redo A;
2794 } elsif ($self->{next_char} == 0x003E) { # >
2795 !!!cp (208);
2796 !!!parse-error (type => 'unclosed SYSTEM literal');
2797
2798 $self->{state} = DATA_STATE;
2799 !!!next-input-character;
2800
2801 $self->{current_token}->{quirks} = 1;
2802 !!!emit ($self->{current_token}); # DOCTYPE
2803
2804 redo A;
2805 } elsif ($self->{next_char} == -1) {
2806 !!!cp (209);
2807 !!!parse-error (type => 'unclosed SYSTEM literal');
2808
2809 $self->{state} = DATA_STATE;
2810 ## reconsume
2811
2812 $self->{current_token}->{quirks} = 1;
2813 !!!emit ($self->{current_token}); # DOCTYPE
2814
2815 redo A;
2816 } else {
2817 !!!cp (210);
2818 $self->{current_token}->{system_identifier} # DOCTYPE
2819 .= chr $self->{next_char};
2820 $self->{read_until}->($self->{current_token}->{system_identifier},
2821 q[">],
2822 length $self->{current_token}->{system_identifier});
2823
2824 ## Stay in the state
2825 !!!next-input-character;
2826 redo A;
2827 }
2828 } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE) {
2829 if ($self->{next_char} == 0x0027) { # '
2830 !!!cp (211);
2831 $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2832 !!!next-input-character;
2833 redo A;
2834 } elsif ($self->{next_char} == 0x003E) { # >
2835 !!!cp (212);
2836 !!!parse-error (type => 'unclosed SYSTEM literal');
2837
2838 $self->{state} = DATA_STATE;
2839 !!!next-input-character;
2840
2841 $self->{current_token}->{quirks} = 1;
2842 !!!emit ($self->{current_token}); # DOCTYPE
2843
2844 redo A;
2845 } elsif ($self->{next_char} == -1) {
2846 !!!cp (213);
2847 !!!parse-error (type => 'unclosed SYSTEM literal');
2848
2849 $self->{state} = DATA_STATE;
2850 ## reconsume
2851
2852 $self->{current_token}->{quirks} = 1;
2853 !!!emit ($self->{current_token}); # DOCTYPE
2854
2855 redo A;
2856 } else {
2857 !!!cp (214);
2858 $self->{current_token}->{system_identifier} # DOCTYPE
2859 .= chr $self->{next_char};
2860 $self->{read_until}->($self->{current_token}->{system_identifier},
2861 q['>],
2862 length $self->{current_token}->{system_identifier});
2863
2864 ## Stay in the state
2865 !!!next-input-character;
2866 redo A;
2867 }
2868 } elsif ($self->{state} == AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {
2869 if ({
2870 0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,
2871 #0x000D => 1, # HT, LF, VT, FF, SP, CR
2872 }->{$self->{next_char}}) {
2873 !!!cp (215);
2874 ## Stay in the state
2875 !!!next-input-character;
2876 redo A;
2877 } elsif ($self->{next_char} == 0x003E) { # >
2878 !!!cp (216);
2879 $self->{state} = DATA_STATE;
2880 !!!next-input-character;
2881
2882 !!!emit ($self->{current_token}); # DOCTYPE
2883
2884 redo A;
2885 } elsif ($self->{next_char} == -1) {
2886 !!!cp (217);
2887 !!!parse-error (type => 'unclosed DOCTYPE');
2888 $self->{state} = DATA_STATE;
2889 ## reconsume
2890
2891 $self->{current_token}->{quirks} = 1;
2892 !!!emit ($self->{current_token}); # DOCTYPE
2893
2894 redo A;
2895 } else {
2896 !!!cp (218);
2897 !!!parse-error (type => 'string after SYSTEM literal');
2898 #$self->{current_token}->{quirks} = 1;
2899
2900 $self->{state} = BOGUS_DOCTYPE_STATE;
2901 !!!next-input-character;
2902 redo A;
2903 }
2904 } elsif ($self->{state} == BOGUS_DOCTYPE_STATE) {
2905 if ($self->{next_char} == 0x003E) { # >
2906 !!!cp (219);
2907 $self->{state} = DATA_STATE;
2908 !!!next-input-character;
2909
2910 !!!emit ($self->{current_token}); # DOCTYPE
2911
2912 redo A;
2913 } elsif ($self->{next_char} == -1) {
2914 !!!cp (220);
2915 !!!parse-error (type => 'unclosed DOCTYPE');
2916 $self->{state} = DATA_STATE;
2917 ## reconsume
2918
2919 !!!emit ($self->{current_token}); # DOCTYPE
2920
2921 redo A;
2922 } else {
2923 !!!cp (221);
2924 my $s = '';
2925 $self->{read_until}->($s, q[>], 0);
2926
2927 ## Stay in the state
2928 !!!next-input-character;
2929 redo A;
2930 }
2931 } elsif ($self->{state} == CDATA_SECTION_STATE) {
2932 ## NOTE: "CDATA section state" in the state is jointly implemented
2933 ## by three states, |CDATA_SECTION_STATE|, |CDATA_SECTION_MSE1_STATE|,
2934 ## and |CDATA_SECTION_MSE2_STATE|.
2935
2936 if ($self->{next_char} == 0x005D) { # ]
2937 !!!cp (221.1);
2938 $self->{state} = CDATA_SECTION_MSE1_STATE;
2939 !!!next-input-character;
2940 redo A;
2941 } elsif ($self->{next_char} == -1) {
2942 $self->{state} = DATA_STATE;
2943 !!!next-input-character;
2944 if (length $self->{current_token}->{data}) { # character
2945 !!!cp (221.2);
2946 !!!emit ($self->{current_token}); # character
2947 } else {
2948 !!!cp (221.3);
2949 ## No token to emit. $self->{current_token} is discarded.
2950 }
2951 redo A;
2952 } else {
2953 !!!cp (221.4);
2954 $self->{current_token}->{data} .= chr $self->{next_char};
2955 $self->{read_until}->($self->{current_token}->{data},
2956 q<]>,
2957 length $self->{current_token}->{data});
2958
2959 ## Stay in the state.
2960 !!!next-input-character;
2961 redo A;
2962 }
2963
2964 ## ISSUE: "text tokens" in spec.
2965 } elsif ($self->{state} == CDATA_SECTION_MSE1_STATE) {
2966 if ($self->{next_char} == 0x005D) { # ]
2967 !!!cp (221.5);
2968 $self->{state} = CDATA_SECTION_MSE2_STATE;
2969 !!!next-input-character;
2970 redo A;
2971 } else {
2972 !!!cp (221.6);
2973 $self->{current_token}->{data} .= ']';
2974 $self->{state} = CDATA_SECTION_STATE;
2975 ## Reconsume.
2976 redo A;
2977 }
2978 } elsif ($self->{state} == CDATA_SECTION_MSE2_STATE) {
2979 if ($self->{next_char} == 0x003E) { # >
2980 $self->{state} = DATA_STATE;
2981 !!!next-input-character;
2982 if (length $self->{current_token}->{data}) { # character
2983 !!!cp (221.7);
2984 !!!emit ($self->{current_token}); # character
2985 } else {
2986 !!!cp (221.8);
2987 ## No token to emit. $self->{current_token} is discarded.
2988 }
2989 redo A;
2990 } elsif ($self->{next_char} == 0x005D) { # ]
2991 !!!cp (221.9); # character
2992 $self->{current_token}->{data} .= ']'; ## Add first "]" of "]]]".
2993 ## Stay in the state.
2994 !!!next-input-character;
2995 redo A;
2996 } else {
2997 !!!cp (221.11);
2998 $self->{current_token}->{data} .= ']]'; # character
2999 $self->{state} = CDATA_SECTION_STATE;
3000 ## Reconsume.
3001 redo A;
3002 }
3003 } elsif ($self->{state} == ENTITY_STATE) {
3004 if ({
3005 0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, # HT, LF, VT, FF,
3006 0x0020 => 1, 0x003C => 1, 0x0026 => 1, -1 => 1, # SP, <, &
3007 $self->{entity_additional} => 1,
3008 }->{$self->{next_char}}) {
3009 !!!cp (1001);
3010 ## Don't consume
3011 ## No error
3012 ## Return nothing.
3013 #
3014 } elsif ($self->{next_char} == 0x0023) { # #
3015 !!!cp (999);
3016 $self->{state} = ENTITY_HASH_STATE;
3017 $self->{state_keyword} = '#';
3018 !!!next-input-character;
3019 redo A;
3020 } elsif ((0x0041 <= $self->{next_char} and
3021 $self->{next_char} <= 0x005A) or # A..Z
3022 (0x0061 <= $self->{next_char} and
3023 $self->{next_char} <= 0x007A)) { # a..z
3024 !!!cp (998);
3025 require Whatpm::_NamedEntityList;
3026 $self->{state} = ENTITY_NAME_STATE;
3027 $self->{state_keyword} = chr $self->{next_char};
3028 $self->{entity__value} = $self->{state_keyword};
3029 $self->{entity__match} = 0;
3030 !!!next-input-character;
3031 redo A;
3032 } else {
3033 !!!cp (1027);
3034 !!!parse-error (type => 'bare ero');
3035 ## Return nothing.
3036 #
3037 }
3038
3039 ## NOTE: No character is consumed by the "consume a character
3040 ## reference" algorithm. In other word, there is an "&" character
3041 ## that does not introduce a character reference, which would be
3042 ## appended to the parent element or the attribute value in later
3043 ## process of the tokenizer.
3044
3045 if ($self->{prev_state} == DATA_STATE) {
3046 !!!cp (997);
3047 $self->{state} = $self->{prev_state};
3048 ## Reconsume.
3049 !!!emit ({type => CHARACTER_TOKEN, data => '&',
3050 line => $self->{line_prev},
3051 column => $self->{column_prev},
3052 });
3053 redo A;
3054 } else {
3055 !!!cp (996);
3056 $self->{current_attribute}->{value} .= '&';
3057 $self->{state} = $self->{prev_state};
3058 ## Reconsume.
3059 redo A;
3060 }
3061 } elsif ($self->{state} == ENTITY_HASH_STATE) {
3062 if ($self->{next_char} == 0x0078 or # x
3063 $self->{next_char} == 0x0058) { # X
3064 !!!cp (995);
3065 $self->{state} = HEXREF_X_STATE;
3066 $self->{state_keyword} .= chr $self->{next_char};
3067 !!!next-input-character;
3068 redo A;
3069 } elsif (0x0030 <= $self->{next_char} and
3070 $self->{next_char} <= 0x0039) { # 0..9
3071 !!!cp (994);
3072 $self->{state} = NCR_NUM_STATE;
3073 $self->{state_keyword} = $self->{next_char} - 0x0030;
3074 !!!next-input-character;
3075 redo A;
3076 } else {
3077 !!!parse-error (type => 'bare nero',
3078 line => $self->{line_prev},
3079 column => $self->{column_prev} - 1);
3080
3081 ## NOTE: According to the spec algorithm, nothing is returned,
3082 ## and then "&#" is appended to the parent element or the attribute
3083 ## value in the later processing.
3084
3085 if ($self->{prev_state} == DATA_STATE) {
3086 !!!cp (1019);
3087 $self->{state} = $self->{prev_state};
3088 ## Reconsume.
3089 !!!emit ({type => CHARACTER_TOKEN,
3090 data => '&#',
3091 line => $self->{line_prev},
3092 column => $self->{column_prev} - 1,
3093 });
3094 redo A;
3095 } else {
3096 !!!cp (993);
3097 $self->{current_attribute}->{value} .= '&#';
3098 $self->{state} = $self->{prev_state};
3099 ## Reconsume.
3100 redo A;
3101 }
3102 }
3103 } elsif ($self->{state} == NCR_NUM_STATE) {
3104 if (0x0030 <= $self->{next_char} and
3105 $self->{next_char} <= 0x0039) { # 0..9
3106 !!!cp (1012);
3107 $self->{state_keyword} *= 10;
3108 $self->{state_keyword} += $self->{next_char} - 0x0030;
3109
3110 ## Stay in the state.
3111 !!!next-input-character;
3112 redo A;
3113 } elsif ($self->{next_char} == 0x003B) { # ;
3114 !!!cp (1013);
3115 !!!next-input-character;
3116 #
3117 } else {
3118 !!!cp (1014);
3119 !!!parse-error (type => 'no refc');
3120 ## Reconsume.
3121 #
3122 }
3123
3124 my $code = $self->{state_keyword};
3125 my $l = $self->{line_prev};
3126 my $c = $self->{column_prev};
3127 if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) {
3128 !!!cp (1015);
3129 !!!parse-error (type => 'invalid character reference',
3130 text => (sprintf 'U+%04X', $code),
3131 line => $l, column => $c);
3132 $code = 0xFFFD;
3133 } elsif ($code > 0x10FFFF) {
3134 !!!cp (1016);
3135 !!!parse-error (type => 'invalid character reference',
3136 text => (sprintf 'U-%08X', $code),
3137 line => $l, column => $c);
3138 $code = 0xFFFD;
3139 } elsif ($code == 0x000D) {
3140 !!!cp (1017);
3141 !!!parse-error (type => 'CR character reference',
3142 line => $l, column => $c);
3143 $code = 0x000A;
3144 } elsif (0x80 <= $code and $code <= 0x9F) {
3145 !!!cp (1018);
3146 !!!parse-error (type => 'C1 character reference',
3147 text => (sprintf 'U+%04X', $code),
3148 line => $l, column => $c);
3149 $code = $c1_entity_char->{$code};
3150 }
3151
3152 if ($self->{prev_state} == DATA_STATE) {
3153 !!!cp (992);
3154 $self->{state} = $self->{prev_state};
3155 ## Reconsume.
3156 !!!emit ({type => CHARACTER_TOKEN, data => chr $code,
3157 line => $l, column => $c,
3158 });
3159 redo A;
3160 } else {
3161 !!!cp (991);
3162 $self->{current_attribute}->{value} .= chr $code;
3163 $self->{current_attribute}->{has_reference} = 1;
3164 $self->{state} = $self->{prev_state};
3165 ## Reconsume.
3166 redo A;
3167 }
3168 } elsif ($self->{state} == HEXREF_X_STATE) {
3169 if ((0x0030 <= $self->{next_char} and $self->{next_char} <= 0x0039) or
3170 (0x0041 <= $self->{next_char} and $self->{next_char} <= 0x0046) or
3171 (0x0061 <= $self->{next_char} and $self->{next_char} <= 0x0066)) {
3172 # 0..9, A..F, a..f
3173 !!!cp (990);
3174 $self->{state} = HEXREF_HEX_STATE;
3175 $self->{state_keyword} = 0;
3176 ## Reconsume.
3177 redo A;
3178 } else {
3179 !!!parse-error (type => 'bare hcro',
3180 line => $self->{line_prev},
3181 column => $self->{column_prev} - 2);
3182
3183 ## NOTE: According to the spec algorithm, nothing is returned,
3184 ## and then "&#" followed by "X" or "x" is appended to the parent
3185 ## element or the attribute value in the later processing.
3186
3187 if ($self->{prev_state} == DATA_STATE) {
3188 !!!cp (1005);
3189 $self->{state} = $self->{prev_state};
3190 ## Reconsume.
3191 !!!emit ({type => CHARACTER_TOKEN,
3192 data => '&' . $self->{state_keyword},
3193 line => $self->{line_prev},
3194 column => $self->{column_prev} - length $self->{state_keyword},
3195 });
3196 redo A;
3197 } else {
3198 !!!cp (989);
3199 $self->{current_attribute}->{value} .= '&' . $self->{state_keyword};
3200 $self->{state} = $self->{prev_state};
3201 ## Reconsume.
3202 redo A;
3203 }
3204 }
3205 } elsif ($self->{state} == HEXREF_HEX_STATE) {
3206 if (0x0030 <= $self->{next_char} and $self->{next_char} <= 0x0039) {
3207 # 0..9
3208 !!!cp (1002);
3209 $self->{state_keyword} *= 0x10;
3210 $self->{state_keyword} += $self->{next_char} - 0x0030;
3211 ## Stay in the state.
3212 !!!next-input-character;
3213 redo A;
3214 } elsif (0x0061 <= $self->{next_char} and
3215 $self->{next_char} <= 0x0066) { # a..f
3216 !!!cp (1003);
3217 $self->{state_keyword} *= 0x10;
3218 $self->{state_keyword} += $self->{next_char} - 0x0060 + 9;
3219 ## Stay in the state.
3220 !!!next-input-character;
3221 redo A;
3222 } elsif (0x0041 <= $self->{next_char} and
3223 $self->{next_char} <= 0x0046) { # A..F
3224 !!!cp (1004);
3225 $self->{state_keyword} *= 0x10;
3226 $self->{state_keyword} += $self->{next_char} - 0x0040 + 9;
3227 ## Stay in the state.
3228 !!!next-input-character;
3229 redo A;
3230 } elsif ($self->{next_char} == 0x003B) { # ;
3231 !!!cp (1006);
3232 !!!next-input-character;
3233 #
3234 } else {
3235 !!!cp (1007);
3236 !!!parse-error (type => 'no refc',
3237 line => $self->{line},
3238 column => $self->{column});
3239 ## Reconsume.
3240 #
3241 }
3242
3243 my $code = $self->{state_keyword};
3244 my $l = $self->{line_prev};
3245 my $c = $self->{column_prev};
3246 if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) {
3247 !!!cp (1008);
3248 !!!parse-error (type => 'invalid character reference',
3249 text => (sprintf 'U+%04X', $code),
3250 line => $l, column => $c);
3251 $code = 0xFFFD;
3252 } elsif ($code > 0x10FFFF) {
3253 !!!cp (1009);
3254 !!!parse-error (type => 'invalid character reference',
3255 text => (sprintf 'U-%08X', $code),
3256 line => $l, column => $c);
3257 $code = 0xFFFD;
3258 } elsif ($code == 0x000D) {
3259 !!!cp (1010);
3260 !!!parse-error (type => 'CR character reference', line => $l, column => $c);
3261 $code = 0x000A;
3262 } elsif (0x80 <= $code and $code <= 0x9F) {
3263 !!!cp (1011);
3264 !!!parse-error (type => 'C1 character reference', text => (sprintf 'U+%04X', $code), line => $l, column => $c);
3265 $code = $c1_entity_char->{$code};
3266 }
3267
3268 if ($self->{prev_state} == DATA_STATE) {
3269 !!!cp (988);
3270 $self->{state} = $self->{prev_state};
3271 ## Reconsume.
3272 !!!emit ({type => CHARACTER_TOKEN, data => chr $code,
3273 line => $l, column => $c,
3274 });
3275 redo A;
3276 } else {
3277 !!!cp (987);
3278 $self->{current_attribute}->{value} .= chr $code;
3279 $self->{current_attribute}->{has_reference} = 1;
3280 $self->{state} = $self->{prev_state};
3281 ## Reconsume.
3282 redo A;
3283 }
3284 } elsif ($self->{state} == ENTITY_NAME_STATE) {
3285 if (length $self->{state_keyword} < 30 and
3286 ## NOTE: Some number greater than the maximum length of entity name
3287 ((0x0041 <= $self->{next_char} and # a
3288 $self->{next_char} <= 0x005A) or # x
3289 (0x0061 <= $self->{next_char} and # a
3290 $self->{next_char} <= 0x007A) or # z
3291 (0x0030 <= $self->{next_char} and # 0
3292 $self->{next_char} <= 0x0039) or # 9
3293 $self->{next_char} == 0x003B)) { # ;
3294 our $EntityChar;
3295 $self->{state_keyword} .= chr $self->{next_char};
3296 if (defined $EntityChar->{$self->{state_keyword}}) {
3297 if ($self->{next_char} == 0x003B) { # ;
3298 !!!cp (1020);
3299 $self->{entity__value} = $EntityChar->{$self->{state_keyword}};
3300 $self->{entity__match} = 1;
3301 !!!next-input-character;
3302 #
3303 } else {
3304 !!!cp (1021);
3305 $self->{entity__value} = $EntityChar->{$self->{state_keyword}};
3306 $self->{entity__match} = -1;
3307 ## Stay in the state.
3308 !!!next-input-character;
3309 redo A;
3310 }
3311 } else {
3312 !!!cp (1022);
3313 $self->{entity__value} .= chr $self->{next_char};
3314 $self->{entity__match} *= 2;
3315 ## Stay in the state.
3316 !!!next-input-character;
3317 redo A;
3318 }
3319 }
3320
3321 my $data;
3322 my $has_ref;
3323 if ($self->{entity__match} > 0) {
3324 !!!cp (1023);
3325 $data = $self->{entity__value};
3326 $has_ref = 1;
3327 #
3328 } elsif ($self->{entity__match} < 0) {
3329 !!!parse-error (type => 'no refc');
3330 if ($self->{prev_state} != DATA_STATE and # in attribute
3331 $self->{entity__match} < -1) {
3332 !!!cp (1024);
3333 $data = '&' . $self->{state_keyword};
3334 #
3335 } else {
3336 !!!cp (1025);
3337 $data = $self->{entity__value};
3338 $has_ref = 1;
3339 #
3340 }
3341 } else {
3342 !!!cp (1026);
3343 !!!parse-error (type => 'bare ero',
3344 line => $self->{line_prev},
3345 column => $self->{column_prev});
3346 $data = '&' . $self->{state_keyword};
3347 #
3348 }
3349
3350 ## NOTE: In these cases, when a character reference is found,
3351 ## it is consumed and a character token is returned, or, otherwise,
3352 ## nothing is consumed and returned, according to the spec algorithm.
3353 ## In this implementation, anything that has been examined by the
3354 ## tokenizer is appended to the parent element or the attribute value
3355 ## as string, either literal string when no character reference or
3356 ## entity-replaced string otherwise, in this stage, since any characters
3357 ## that would not be consumed are appended in the data state or in an
3358 ## appropriate attribute value state anyway.
3359
3360 if ($self->{prev_state} == DATA_STATE) {
3361 !!!cp (986);
3362 $self->{state} = $self->{prev_state};
3363 ## Reconsume.
3364 !!!emit ({type => CHARACTER_TOKEN,
3365 data => $data,
3366 line => $self->{line_prev},
3367 column => $self->{column_prev} + 1 - length $self->{state_keyword},
3368 });
3369 redo A;
3370 } else {
3371 !!!cp (985);
3372 $self->{current_attribute}->{value} .= $data;
3373 $self->{current_attribute}->{has_reference} = 1 if $has_ref;
3374 $self->{state} = $self->{prev_state};
3375 ## Reconsume.
3376 redo A;
3377 }
3378 } else {
3379 die "$0: $self->{state}: Unknown state";
3380 }
3381 } # A
3382
3383 die "$0: _get_next_token: unexpected case";
3384 } # _get_next_token
3385
3386 sub _initialize_tree_constructor ($) {
3387 my $self = shift;
3388 ## NOTE: $self->{document} MUST be specified before this method is called
3389 $self->{document}->strict_error_checking (0);
3390 ## TODO: Turn mutation events off # MUST
3391 ## TODO: Turn loose Document option (manakai extension) on
3392 $self->{document}->manakai_is_html (1); # MUST
3393 $self->{document}->set_user_data (manakai_source_line => 1);
3394 $self->{document}->set_user_data (manakai_source_column => 1);
3395 } # _initialize_tree_constructor
3396
3397 sub _terminate_tree_constructor ($) {
3398 my $self = shift;
3399 $self->{document}->strict_error_checking (1);
3400 ## TODO: Turn mutation events on
3401 } # _terminate_tree_constructor
3402
3403 ## ISSUE: Should append_child (for example) in script executed in tree construction stage fire mutation events?
3404
3405 { # tree construction stage
3406 my $token;
3407
3408 sub _construct_tree ($) {
3409 my ($self) = @_;
3410
3411 ## When an interactive UA render the $self->{document} available
3412 ## to the user, or when it begin accepting user input, are
3413 ## not defined.
3414
3415 ## Append a character: collect it and all subsequent consecutive
3416 ## characters and insert one Text node whose data is concatenation
3417 ## of all those characters. # MUST
3418
3419 !!!next-token;
3420
3421 undef $self->{form_element};
3422 undef $self->{head_element};
3423 $self->{open_elements} = [];
3424 undef $self->{inner_html_node};
3425
3426 ## NOTE: The "initial" insertion mode.
3427 $self->_tree_construction_initial; # MUST
3428
3429 ## NOTE: The "before html" insertion mode.
3430 $self->_tree_construction_root_element;
3431 $self->{insertion_mode} = BEFORE_HEAD_IM;
3432
3433 ## NOTE: The "before head" insertion mode and so on.
3434 $self->_tree_construction_main;
3435 } # _construct_tree
3436
3437 sub _tree_construction_initial ($) {
3438 my $self = shift;
3439
3440 ## NOTE: "initial" insertion mode
3441
3442 INITIAL: {
3443 if ($token->{type} == DOCTYPE_TOKEN) {
3444 ## NOTE: Conformance checkers MAY, instead of reporting "not HTML5"
3445 ## error, switch to a conformance checking mode for another
3446 ## language.
3447 my $doctype_name = $token->{name};
3448 $doctype_name = '' unless defined $doctype_name;
3449 $doctype_name =~ tr/a-z/A-Z/; # ASCII case-insensitive
3450 if (not defined $token->{name} or # <!DOCTYPE>
3451 defined $token->{system_identifier}) {
3452 !!!cp ('t1');
3453 !!!parse-error (type => 'not HTML5', token => $token);
3454 } elsif ($doctype_name ne 'HTML') {
3455 !!!cp ('t2');
3456 !!!parse-error (type => 'not HTML5', token => $token);
3457 } elsif (defined $token->{public_identifier}) {
3458 if ($token->{public_identifier} eq 'XSLT-compat') {
3459 !!!cp ('t1.2');
3460 !!!parse-error (type => 'XSLT-compat', token => $token,
3461 level => $self->{level}->{should});
3462 } else {
3463 !!!parse-error (type => 'not HTML5', token => $token);
3464 }
3465 } else {
3466 !!!cp ('t3');
3467 #
3468 }
3469
3470 my $doctype = $self->{document}->create_document_type_definition
3471 ($token->{name}); ## ISSUE: If name is missing (e.g. <!DOCTYPE>)?
3472 ## NOTE: Default value for both |public_id| and |system_id| attributes
3473 ## are empty strings, so that we don't set any value in missing cases.
3474 $doctype->public_id ($token->{public_identifier})
3475 if defined $token->{public_identifier};
3476 $doctype->system_id ($token->{system_identifier})
3477 if defined $token->{system_identifier};
3478 ## NOTE: Other DocumentType attributes are null or empty lists.
3479 ## ISSUE: internalSubset = null??
3480 $self->{document}->append_child ($doctype);
3481
3482 if ($token->{quirks} or $doctype_name ne 'HTML') {
3483 !!!cp ('t4');
3484 $self->{document}->manakai_compat_mode ('quirks');
3485 } elsif (defined $token->{public_identifier}) {
3486 my $pubid = $token->{public_identifier};
3487 $pubid =~ tr/a-z/A-z/;
3488 my $prefix = [
3489 "+//SILMARIL//DTD HTML PRO V0R11 19970101//",
3490 "-//ADVASOFT LTD//DTD HTML 3.0 ASWEDIT + EXTENSIONS//",
3491 "-//AS//DTD HTML 3.0 ASWEDIT + EXTENSIONS//",
3492 "-//IETF//DTD HTML 2.0 LEVEL 1//",
3493 "-//IETF//DTD HTML 2.0 LEVEL 2//",
3494 "-//IETF//DTD HTML 2.0 STRICT LEVEL 1//",
3495 "-//IETF//DTD HTML 2.0 STRICT LEVEL 2//",
3496 "-//IETF//DTD HTML 2.0 STRICT//",
3497 "-//IETF//DTD HTML 2.0//",
3498 "-//IETF//DTD HTML 2.1E//",
3499 "-//IETF//DTD HTML 3.0//",
3500 "-//IETF//DTD HTML 3.2 FINAL//",
3501 "-//IETF//DTD HTML 3.2//",
3502 "-//IETF//DTD HTML 3//",
3503 "-//IETF//DTD HTML LEVEL 0//",
3504 "-//IETF//DTD HTML LEVEL 1//",
3505 "-//IETF//DTD HTML LEVEL 2//",
3506 "-//IETF//DTD HTML LEVEL 3//",
3507 "-//IETF//DTD HTML STRICT LEVEL 0//",
3508 "-//IETF//DTD HTML STRICT LEVEL 1//",
3509 "-//IETF//DTD HTML STRICT LEVEL 2//",
3510 "-//IETF//DTD HTML STRICT LEVEL 3//",
3511 "-//IETF//DTD HTML STRICT//",
3512 "-//IETF//DTD HTML//",
3513 "-//METRIUS//DTD METRIUS PRESENTATIONAL//",
3514 "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 HTML STRICT//",
3515 "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 HTML//",
3516 "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 TABLES//",
3517 "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 HTML STRICT//",
3518 "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 HTML//",
3519 "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 TABLES//",
3520 "-//NETSCAPE COMM. CORP.//DTD HTML//",
3521 "-//NETSCAPE COMM. CORP.//DTD STRICT HTML//",
3522 "-//O'REILLY AND ASSOCIATES//DTD HTML 2.0//",
3523 "-//O'REILLY AND ASSOCIATES//DTD HTML EXTENDED 1.0//",
3524 "-//O'REILLY AND ASSOCIATES//DTD HTML EXTENDED RELAXED 1.0//",
3525 "-//SOFTQUAD SOFTWARE//DTD HOTMETAL PRO 6.0::19990601::EXTENSIONS TO HTML 4.0//",
3526 "-//SOFTQUAD//DTD HOTMETAL PRO 4.0::19971010::EXTENSIONS TO HTML 4.0//",
3527 "-//SPYGLASS//DTD HTML 2.0 EXTENDED//",
3528 "-//SQ//DTD HTML 2.0 HOTMETAL + EXTENSIONS//",
3529 "-//SUN MICROSYSTEMS CORP.//DTD HOTJAVA HTML//",
3530 "-//SUN MICROSYSTEMS CORP.//DTD HOTJAVA STRICT HTML//",
3531 "-//W3C//DTD HTML 3 1995-03-24//",
3532 "-//W3C//DTD HTML 3.2 DRAFT//",
3533 "-//W3C//DTD HTML 3.2 FINAL//",
3534 "-//W3C//DTD HTML 3.2//",
3535 "-//W3C//DTD HTML 3.2S DRAFT//",
3536 "-//W3C//DTD HTML 4.0 FRAMESET//",
3537 "-//W3C//DTD HTML 4.0 TRANSITIONAL//",
3538 "-//W3C//DTD HTML EXPERIMETNAL 19960712//",
3539 "-//W3C//DTD HTML EXPERIMENTAL 970421//",
3540 "-//W3C//DTD W3 HTML//",
3541 "-//W3O//DTD W3 HTML 3.0//",
3542 "-//WEBTECHS//DTD MOZILLA HTML 2.0//",
3543 "-//WEBTECHS//DTD MOZILLA HTML//",
3544 ]; # $prefix
3545 my $match;
3546 for (@$prefix) {
3547 if (substr ($prefix, 0, length $_) eq $_) {
3548 $match = 1;
3549 last;
3550 }
3551 }
3552 if ($match or
3553 $pubid eq "-//W3O//DTD W3 HTML STRICT 3.0//EN//" or
3554 $pubid eq "-/W3C/DTD HTML 4.0 TRANSITIONAL/EN" or
3555 $pubid eq "HTML") {
3556 !!!cp ('t5');
3557 $self->{document}->manakai_compat_mode ('quirks');
3558 } elsif ($pubid =~ m[^-//W3C//DTD HTML 4.01 FRAMESET//] or
3559 $pubid =~ m[^-//W3C//DTD HTML 4.01 TRANSITIONAL//]) {
3560 if (defined $token->{system_identifier}) {
3561 !!!cp ('t6');
3562 $self->{document}->manakai_compat_mode ('quirks');
3563 } else {
3564 !!!cp ('t7');
3565 $self->{document}->manakai_compat_mode ('limited quirks');
3566 }
3567 } elsif ($pubid =~ m[^-//W3C//DTD XHTML 1.0 FRAMESET//] or
3568 $pubid =~ m[^-//W3C//DTD XHTML 1.0 TRANSITIONAL//]) {
3569 !!!cp ('t8');
3570 $self->{document}->manakai_compat_mode ('limited quirks');
3571 } else {
3572 !!!cp ('t9');
3573 }
3574 } else {
3575 !!!cp ('t10');
3576 }
3577 if (defined $token->{system_identifier}) {
3578 my $sysid = $token->{system_identifier};
3579 $sysid =~ tr/A-Z/a-z/;
3580 if ($sysid eq "http://www.ibm.com/data/dtd/v11/ibmxhtml1-transitional.dtd") {
3581 ## NOTE: Ensure that |PUBLIC "(limited quirks)" "(quirks)"| is
3582 ## marked as quirks.
3583 $self->{document}->manakai_compat_mode ('quirks');
3584 !!!cp ('t11');
3585 } else {
3586 !!!cp ('t12');
3587 }
3588 } else {
3589 !!!cp ('t13');
3590 }
3591
3592 ## Go to the "before html" insertion mode.
3593 !!!next-token;
3594 return;
3595 } elsif ({
3596 START_TAG_TOKEN, 1,
3597 END_TAG_TOKEN, 1,
3598 END_OF_FILE_TOKEN, 1,
3599 }->{$token->{type}}) {
3600 !!!cp ('t14');
3601 !!!parse-error (type => 'no DOCTYPE', token => $token);
3602 $self->{document}->manakai_compat_mode ('quirks');
3603 ## Go to the "before html" insertion mode.
3604 ## reprocess
3605 !!!ack-later;
3606 return;
3607 } elsif ($token->{type} == CHARACTER_TOKEN) {
3608 if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D
3609 ## Ignore the token
3610
3611 unless (length $token->{data}) {
3612 !!!cp ('t15');
3613 ## Stay in the insertion mode.
3614 !!!next-token;
3615 redo INITIAL;
3616 } else {
3617 !!!cp ('t16');
3618 }
3619 } else {
3620 !!!cp ('t17');
3621 }
3622
3623 !!!parse-error (type => 'no DOCTYPE', token => $token);
3624 $self->{document}->manakai_compat_mode ('quirks');
3625 ## Go to the "before html" insertion mode.
3626 ## reprocess
3627 return;
3628 } elsif ($token->{type} == COMMENT_TOKEN) {
3629 !!!cp ('t18');
3630 my $comment = $self->{document}->create_comment ($token->{data});
3631 $self->{document}->append_child ($comment);
3632
3633 ## Stay in the insertion mode.
3634 !!!next-token;
3635 redo INITIAL;
3636 } else {
3637 die "$0: $token->{type}: Unknown token type";
3638 }
3639 } # INITIAL
3640
3641 die "$0: _tree_construction_initial: This should be never reached";
3642 } # _tree_construction_initial
3643
3644 sub _tree_construction_root_element ($) {
3645 my $self = shift;
3646
3647 ## NOTE: "before html" insertion mode.
3648
3649 B: {
3650 if ($token->{type} == DOCTYPE_TOKEN) {
3651 !!!cp ('t19');
3652 !!!parse-error (type => 'in html:#DOCTYPE', token => $token);
3653 ## Ignore the token
3654 ## Stay in the insertion mode.
3655 !!!next-token;
3656 redo B;
3657 } elsif ($token->{type} == COMMENT_TOKEN) {
3658 !!!cp ('t20');
3659 my $comment = $self->{document}->create_comment ($token->{data});
3660 $self->{document}->append_child ($comment);
3661 ## Stay in the insertion mode.
3662 !!!next-token;
3663 redo B;
3664 } elsif ($token->{type} == CHARACTER_TOKEN) {
3665 if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D
3666 ## Ignore the token.
3667
3668 unless (length $token->{data}) {
3669 !!!cp ('t21');
3670 ## Stay in the insertion mode.
3671 !!!next-token;
3672 redo B;
3673 } else {
3674 !!!cp ('t22');
3675 }
3676 } else {
3677 !!!cp ('t23');
3678 }
3679
3680 $self->{application_cache_selection}->(undef);
3681
3682 #
3683 } elsif ($token->{type} == START_TAG_TOKEN) {
3684 if ($token->{tag_name} eq 'html') {
3685 my $root_element;
3686 !!!create-element ($root_element, $HTML_NS, $token->{tag_name}, $token->{attributes}, $token);
3687 $self->{document}->append_child ($root_element);
3688 push @{$self->{open_elements}},
3689 [$root_element, $el_category->{html}];
3690
3691 if ($token->{attributes}->{manifest}) {
3692 !!!cp ('t24');
3693 $self->{application_cache_selection}
3694 ->($token->{attributes}->{manifest}->{value});
3695 ## ISSUE: Spec is unclear on relative references.
3696 ## According to Hixie (#whatwg 2008-03-19), it should be
3697 ## resolved against the base URI of the document in HTML
3698 ## or xml:base of the element in XHTML.
3699 } else {
3700 !!!cp ('t25');
3701 $self->{application_cache_selection}->(undef);
3702 }
3703
3704 !!!nack ('t25c');
3705
3706 !!!next-token;
3707 return; ## Go to the "before head" insertion mode.
3708 } else {
3709 !!!cp ('t25.1');
3710 #
3711 }
3712 } elsif ({
3713 END_TAG_TOKEN, 1,
3714 END_OF_FILE_TOKEN, 1,
3715 }->{$token->{type}}) {
3716 !!!cp ('t26');
3717 #
3718 } else {
3719 die "$0: $token->{type}: Unknown token type";
3720 }
3721
3722 my $root_element;
3723 !!!create-element ($root_element, $HTML_NS, 'html',, $token);
3724 $self->{document}->append_child ($root_element);
3725 push @{$self->{open_elements}}, [$root_element, $el_category->{html}];
3726
3727 $self->{application_cache_selection}->(undef);
3728
3729 ## NOTE: Reprocess the token.
3730 !!!ack-later;
3731 return; ## Go to the "before head" insertion mode.
3732
3733 ## ISSUE: There is an issue in the spec
3734 } # B
3735
3736 die "$0: _tree_construction_root_element: This should never be reached";
3737 } # _tree_construction_root_element
3738
3739 sub _reset_insertion_mode ($) {
3740 my $self = shift;
3741
3742 ## Step 1
3743 my $last;
3744
3745 ## Step 2
3746 my $i = -1;
3747 my $node = $self->{open_elements}->[$i];
3748
3749 ## Step 3
3750 S3: {
3751 if ($self->{open_elements}->[0]->[0] eq $node->[0]) {
3752 $last = 1;
3753 if (defined $self->{inner_html_node}) {
3754 !!!cp ('t28');
3755 $node = $self->{inner_html_node};
3756 } else {
3757 die "_reset_insertion_mode: t27";
3758 }
3759 }
3760
3761 ## Step 4..14
3762 my $new_mode;
3763 if ($node->[1] & FOREIGN_EL) {
3764 !!!cp ('t28.1');
3765 ## NOTE: Strictly spaking, the line below only applies to MathML and
3766 ## SVG elements. Currently the HTML syntax supports only MathML and
3767 ## SVG elements as foreigners.
3768 $new_mode = IN_BODY_IM | IN_FOREIGN_CONTENT_IM;
3769 } elsif ($node->[1] & TABLE_CELL_EL) {
3770 if ($last) {
3771 !!!cp ('t28.2');
3772 #
3773 } else {
3774 !!!cp ('t28.3');
3775 $new_mode = IN_CELL_IM;
3776 }
3777 } else {
3778 !!!cp ('t28.4');
3779 $new_mode = {
3780 select => IN_SELECT_IM,
3781 ## NOTE: |option| and |optgroup| do not set
3782 ## insertion mode to "in select" by themselves.
3783 tr => IN_ROW_IM,
3784 tbody => IN_TABLE_BODY_IM,
3785 thead => IN_TABLE_BODY_IM,
3786 tfoot => IN_TABLE_BODY_IM,
3787 caption => IN_CAPTION_IM,
3788 colgroup => IN_COLUMN_GROUP_IM,
3789 table => IN_TABLE_IM,
3790 head => IN_BODY_IM, # not in head!
3791 body => IN_BODY_IM,
3792 frameset => IN_FRAMESET_IM,
3793 }->{$node->[0]->manakai_local_name};
3794 }
3795 $self->{insertion_mode} = $new_mode and return if defined $new_mode;
3796
3797 ## Step 15
3798 if ($node->[1] & HTML_EL) {
3799 unless (defined $self->{head_element}) {
3800 !!!cp ('t29');
3801 $self->{insertion_mode} = BEFORE_HEAD_IM;
3802 } else {
3803 ## ISSUE: Can this state be reached?
3804 !!!cp ('t30');
3805 $self->{insertion_mode} = AFTER_HEAD_IM;
3806 }
3807 return;
3808 } else {
3809 !!!cp ('t31');
3810 }
3811
3812 ## Step 16
3813 $self->{insertion_mode} = IN_BODY_IM and return if $last;
3814
3815 ## Step 17
3816 $i--;
3817 $node = $self->{open_elements}->[$i];
3818
3819 ## Step 18
3820 redo S3;
3821 } # S3
3822
3823 die "$0: _reset_insertion_mode: This line should never be reached";
3824 } # _reset_insertion_mode
3825
3826 sub _tree_construction_main ($) {
3827 my $self = shift;
3828
3829 my $active_formatting_elements = [];
3830
3831 my $reconstruct_active_formatting_elements = sub { # MUST
3832 my $insert = shift;
3833
3834 ## Step 1
3835 return unless @$active_formatting_elements;
3836
3837 ## Step 3
3838 my $i = -1;
3839 my $entry = $active_formatting_elements->[$i];
3840
3841 ## Step 2
3842 return if $entry->[0] eq '#marker';
3843 for (@{$self->{open_elements}}) {
3844 if ($entry->[0] eq $_->[0]) {
3845 !!!cp ('t32');
3846 return;
3847 }
3848 }
3849
3850 S4: {
3851 ## Step 4
3852 last S4 if $active_formatting_elements->[0]->[0] eq $entry->[0];
3853
3854 ## Step 5
3855 $i--;
3856 $entry = $active_formatting_elements->[$i];
3857
3858 ## Step 6
3859 if ($entry->[0] eq '#marker') {
3860 !!!cp ('t33_1');
3861 #
3862 } else {
3863 my $in_open_elements;
3864 OE: for (@{$self->{open_elements}}) {
3865 if ($entry->[0] eq $_->[0]) {
3866 !!!cp ('t33');
3867 $in_open_elements = 1;
3868 last OE;
3869 }
3870 }
3871 if ($in_open_elements) {
3872 !!!cp ('t34');
3873 #
3874 } else {
3875 ## NOTE: <!DOCTYPE HTML><p><b><i><u></p> <p>X
3876 !!!cp ('t35');
3877 redo S4;
3878 }
3879 }
3880
3881 ## Step 7
3882 $i++;
3883 $entry = $active_formatting_elements->[$i];
3884 } # S4
3885
3886 S7: {
3887 ## Step 8
3888 my $clone = [$entry->[0]->clone_node (0), $entry->[1]];
3889
3890 ## Step 9
3891 $insert->($clone->[0]);
3892 push @{$self->{open_elements}}, $clone;
3893
3894 ## Step 10
3895 $active_formatting_elements->[$i] = $self->{open_elements}->[-1];
3896
3897 ## Step 11
3898 unless ($clone->[0] eq $active_formatting_elements->[-1]->[0]) {
3899 !!!cp ('t36');
3900 ## Step 7'
3901 $i++;
3902 $entry = $active_formatting_elements->[$i];
3903
3904 redo S7;
3905 }
3906
3907 !!!cp ('t37');
3908 } # S7
3909 }; # $reconstruct_active_formatting_elements
3910
3911 my $clear_up_to_marker = sub {
3912 for (reverse 0..$#$active_formatting_elements) {
3913 if ($active_formatting_elements->[$_]->[0] eq '#marker') {
3914 !!!cp ('t38');
3915 splice @$active_formatting_elements, $_;
3916 return;
3917 }
3918 }
3919
3920 !!!cp ('t39');
3921 }; # $clear_up_to_marker
3922
3923 my $insert;
3924
3925 my $parse_rcdata = sub ($) {
3926 my ($content_model_flag) = @_;
3927
3928 ## Step 1
3929 my $start_tag_name = $token->{tag_name};
3930 my $el;
3931 !!!create-element ($el, $HTML_NS, $start_tag_name, $token->{attributes}, $token);
3932
3933 ## Step 2
3934 $insert->($el);
3935
3936 ## Step 3
3937 $self->{content_model} = $content_model_flag; # CDATA or RCDATA
3938 delete $self->{escape}; # MUST
3939
3940 ## Step 4
3941 my $text = '';
3942 !!!nack ('t40.1');
3943 !!!next-token;
3944 while ($token->{type} == CHARACTER_TOKEN) { # or until stop tokenizing
3945 !!!cp ('t40');
3946 $text .= $token->{data};
3947 !!!next-token;
3948 }
3949
3950 ## Step 5
3951 if (length $text) {
3952 !!!cp ('t41');
3953 my $text = $self->{document}->create_text_node ($text);
3954 $el->append_child ($text);
3955 }
3956
3957 ## Step 6
3958 $self->{content_model} = PCDATA_CONTENT_MODEL;
3959
3960 ## Step 7
3961 if ($token->{type} == END_TAG_TOKEN and
3962 $token->{tag_name} eq $start_tag_name) {
3963 !!!cp ('t42');
3964 ## Ignore the token
3965 } else {
3966 ## NOTE: An end-of-file token.
3967 if ($content_model_flag == CDATA_CONTENT_MODEL) {
3968 !!!cp ('t43');
3969 !!!parse-error (type => 'in CDATA:#eof', token => $token);
3970 } elsif ($content_model_flag == RCDATA_CONTENT_MODEL) {
3971 !!!cp ('t44');
3972 !!!parse-error (type => 'in RCDATA:#eof', token => $token);
3973 } else {
3974 die "$0: $content_model_flag in parse_rcdata";
3975 }
3976 }
3977 !!!next-token;
3978 }; # $parse_rcdata
3979
3980 my $script_start_tag = sub () {
3981 my $script_el;
3982 !!!create-element ($script_el, $HTML_NS, 'script', $token->{attributes}, $token);
3983 ## TODO: mark as "parser-inserted"
3984
3985 $self->{content_model} = CDATA_CONTENT_MODEL;
3986 delete $self->{escape}; # MUST
3987
3988 my $text = '';
3989 !!!nack ('t45.1');
3990 !!!next-token;
3991 while ($token->{type} == CHARACTER_TOKEN) {
3992 !!!cp ('t45');
3993 $text .= $token->{data};
3994 !!!next-token;
3995 } # stop if non-character token or tokenizer stops tokenising
3996 if (length $text) {
3997 !!!cp ('t46');
3998 $script_el->manakai_append_text ($text);
3999 }
4000
4001 $self->{content_model} = PCDATA_CONTENT_MODEL;
4002
4003 if ($token->{type} == END_TAG_TOKEN and
4004 $token->{tag_name} eq 'script') {
4005 !!!cp ('t47');
4006 ## Ignore the token
4007 } else {
4008 !!!cp ('t48');
4009 !!!parse-error (type => 'in CDATA:#eof', token => $token);
4010 ## ISSUE: And ignore?
4011 ## TODO: mark as "already executed"
4012 }
4013
4014 if (defined $self->{inner_html_node}) {
4015 !!!cp ('t49');
4016 ## TODO: mark as "already executed"
4017 } else {
4018 !!!cp ('t50');
4019 ## TODO: $old_insertion_point = current insertion point
4020 ## TODO: insertion point = just before the next input character
4021
4022 $insert->($script_el);
4023
4024 ## TODO: insertion point = $old_insertion_point (might be "undefined")
4025
4026 ## TODO: if there is a script that will execute as soon as the parser resume, then...
4027 }
4028
4029 !!!next-token;
4030 }; # $script_start_tag
4031
4032 ## NOTE: $open_tables->[-1]->[0] is the "current table" element node.
4033 ## NOTE: $open_tables->[-1]->[1] is the "tainted" flag.
4034 my $open_tables = [[$self->{open_elements}->[0]->[0]]];
4035
4036 my $formatting_end_tag = sub {
4037 my $end_tag_token = shift;
4038 my $tag_name = $end_tag_token->{tag_name};
4039
4040 ## NOTE: The adoption agency algorithm (AAA).
4041
4042 FET: {
4043 ## Step 1
4044 my $formatting_element;
4045 my $formatting_element_i_in_active;
4046 AFE: for (reverse 0..$#$active_formatting_elements) {
4047 if ($active_formatting_elements->[$_]->[0] eq '#marker') {
4048 !!!cp ('t52');
4049 last AFE;
4050 } elsif ($active_formatting_elements->[$_]->[0]->manakai_local_name
4051 eq $tag_name) {
4052 !!!cp ('t51');
4053 $formatting_element = $active_formatting_elements->[$_];
4054 $formatting_element_i_in_active = $_;
4055 last AFE;
4056 }
4057 } # AFE
4058 unless (defined $formatting_element) {
4059 !!!cp ('t53');
4060 !!!parse-error (type => 'unmatched end tag', text => $tag_name, token => $end_tag_token);
4061 ## Ignore the token
4062 !!!next-token;
4063 return;
4064 }
4065 ## has an element in scope
4066 my $in_scope = 1;
4067 my $formatting_element_i_in_open;
4068 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
4069 my $node = $self->{open_elements}->[$_];
4070 if ($node->[0] eq $formatting_element->[0]) {
4071 if ($in_scope) {
4072 !!!cp ('t54');
4073 $formatting_element_i_in_open = $_;
4074 last INSCOPE;
4075 } else { # in open elements but not in scope
4076 !!!cp ('t55');
4077 !!!parse-error (type => 'unmatched end tag',
4078 text => $token->{tag_name},
4079 token => $end_tag_token);
4080 ## Ignore the token
4081 !!!next-token;
4082 return;
4083 }
4084 } elsif ($node->[1] & SCOPING_EL) {
4085 !!!cp ('t56');
4086 $in_scope = 0;
4087 }
4088 } # INSCOPE
4089 unless (defined $formatting_element_i_in_open) {
4090 !!!cp ('t57');
4091 !!!parse-error (type => 'unmatched end tag',
4092 text => $token->{tag_name},
4093 token => $end_tag_token);
4094 pop @$active_formatting_elements; # $formatting_element
4095 !!!next-token; ## TODO: ok?
4096 return;
4097 }
4098 if (not $self->{open_elements}->[-1]->[0] eq $formatting_element->[0]) {
4099 !!!cp ('t58');
4100 !!!parse-error (type => 'not closed',
4101 text => $self->{open_elements}->[-1]->[0]
4102 ->manakai_local_name,
4103 token => $end_tag_token);
4104 }
4105
4106 ## Step 2
4107 my $furthest_block;
4108 my $furthest_block_i_in_open;
4109 OE: for (reverse 0..$#{$self->{open_elements}}) {
4110 my $node = $self->{open_elements}->[$_];
4111 if (not ($node->[1] & FORMATTING_EL) and
4112 #not $phrasing_category->{$node->[1]} and
4113 ($node->[1] & SPECIAL_EL or
4114 $node->[1] & SCOPING_EL)) { ## Scoping is redundant, maybe
4115 !!!cp ('t59');
4116 $furthest_block = $node;
4117 $furthest_block_i_in_open = $_;
4118 } elsif ($node->[0] eq $formatting_element->[0]) {
4119 !!!cp ('t60');
4120 last OE;
4121 }
4122 } # OE
4123
4124 ## Step 3
4125 unless (defined $furthest_block) { # MUST
4126 !!!cp ('t61');
4127 splice @{$self->{open_elements}}, $formatting_element_i_in_open;
4128 splice @$active_formatting_elements, $formatting_element_i_in_active, 1;
4129 !!!next-token;
4130 return;
4131 }
4132
4133 ## Step 4
4134 my $common_ancestor_node = $self->{open_elements}->[$formatting_element_i_in_open - 1];
4135
4136 ## Step 5
4137 my $furthest_block_parent = $furthest_block->[0]->parent_node;
4138 if (defined $furthest_block_parent) {
4139 !!!cp ('t62');
4140 $furthest_block_parent->remove_child ($furthest_block->[0]);
4141 }
4142
4143 ## Step 6
4144 my $bookmark_prev_el
4145 = $active_formatting_elements->[$formatting_element_i_in_active - 1]
4146 ->[0];
4147
4148 ## Step 7
4149 my $node = $furthest_block;
4150 my $node_i_in_open = $furthest_block_i_in_open;
4151 my $last_node = $furthest_block;
4152 S7: {
4153 ## Step 1
4154 $node_i_in_open--;
4155 $node = $self->{open_elements}->[$node_i_in_open];
4156
4157 ## Step 2
4158 my $node_i_in_active;
4159 S7S2: {
4160 for (reverse 0..$#$active_formatting_elements) {
4161 if ($active_formatting_elements->[$_]->[0] eq $node->[0]) {
4162 !!!cp ('t63');
4163 $node_i_in_active = $_;
4164 last S7S2;
4165 }
4166 }
4167 splice @{$self->{open_elements}}, $node_i_in_open, 1;
4168 redo S7;
4169 } # S7S2
4170
4171 ## Step 3
4172 last S7 if $node->[0] eq $formatting_element->[0];
4173
4174 ## Step 4
4175 if ($last_node->[0] eq $furthest_block->[0]) {
4176 !!!cp ('t64');
4177 $bookmark_prev_el = $node->[0];
4178 }
4179
4180 ## Step 5
4181 if ($node->[0]->has_child_nodes ()) {
4182 !!!cp ('t65');
4183 my $clone = [$node->[0]->clone_node (0), $node->[1]];
4184 $active_formatting_elements->[$node_i_in_active] = $clone;
4185 $self->{open_elements}->[$node_i_in_open] = $clone;
4186 $node = $clone;
4187 }
4188
4189 ## Step 6
4190 $node->[0]->append_child ($last_node->[0]);
4191
4192 ## Step 7
4193 $last_node = $node;
4194
4195 ## Step 8
4196 redo S7;
4197 } # S7
4198
4199 ## Step 8
4200 if ($common_ancestor_node->[1] & TABLE_ROWS_EL) {
4201 my $foster_parent_element;
4202 my $next_sibling;
4203 OE: for (reverse 0..$#{$self->{open_elements}}) {
4204 if ($self->{open_elements}->[$_]->[1] & TABLE_EL) {
4205 my $parent = $self->{open_elements}->[$_]->[0]->parent_node;
4206 if (defined $parent and $parent->node_type == 1) {
4207 !!!cp ('t65.1');
4208 $foster_parent_element = $parent;
4209 $next_sibling = $self->{open_elements}->[$_]->[0];
4210 } else {
4211 !!!cp ('t65.2');
4212 $foster_parent_element
4213 = $self->{open_elements}->[$_ - 1]->[0];
4214 }
4215 last OE;
4216 }
4217 } # OE
4218 $foster_parent_element = $self->{open_elements}->[0]->[0]
4219 unless defined $foster_parent_element;
4220 $foster_parent_element->insert_before ($last_node->[0], $next_sibling);
4221 $open_tables->[-1]->[1] = 1; # tainted
4222 } else {
4223 !!!cp ('t65.3');
4224 $common_ancestor_node->[0]->append_child ($last_node->[0]);
4225 }
4226
4227 ## Step 9
4228 my $clone = [$formatting_element->[0]->clone_node (0),
4229 $formatting_element->[1]];
4230
4231 ## Step 10
4232 my @cn = @{$furthest_block->[0]->child_nodes};
4233 $clone->[0]->append_child ($_) for @cn;
4234
4235 ## Step 11
4236 $furthest_block->[0]->append_child ($clone->[0]);
4237
4238 ## Step 12
4239 my $i;
4240 AFE: for (reverse 0..$#$active_formatting_elements) {
4241 if ($active_formatting_elements->[$_]->[0] eq $formatting_element->[0]) {
4242 !!!cp ('t66');
4243 splice @$active_formatting_elements, $_, 1;
4244 $i-- and last AFE if defined $i;
4245 } elsif ($active_formatting_elements->[$_]->[0] eq $bookmark_prev_el) {
4246 !!!cp ('t67');
4247 $i = $_;
4248 }
4249 } # AFE
4250 splice @$active_formatting_elements, $i + 1, 0, $clone;
4251
4252 ## Step 13
4253 undef $i;
4254 OE: for (reverse 0..$#{$self->{open_elements}}) {
4255 if ($self->{open_elements}->[$_]->[0] eq $formatting_element->[0]) {
4256 !!!cp ('t68');
4257 splice @{$self->{open_elements}}, $_, 1;
4258 $i-- and last OE if defined $i;
4259 } elsif ($self->{open_elements}->[$_]->[0] eq $furthest_block->[0]) {
4260 !!!cp ('t69');
4261 $i = $_;
4262 }
4263 } # OE
4264 splice @{$self->{open_elements}}, $i + 1, 1, $clone;
4265
4266 ## Step 14
4267 redo FET;
4268 } # FET
4269 }; # $formatting_end_tag
4270
4271 $insert = my $insert_to_current = sub {
4272 $self->{open_elements}->[-1]->[0]->append_child ($_[0]);
4273 }; # $insert_to_current
4274
4275 my $insert_to_foster = sub {
4276 my $child = shift;
4277 if ($self->{open_elements}->[-1]->[1] & TABLE_ROWS_EL) {
4278 # MUST
4279 my $foster_parent_element;
4280 my $next_sibling;
4281 OE: for (reverse 0..$#{$self->{open_elements}}) {
4282 if ($self->{open_elements}->[$_]->[1] & TABLE_EL) {
4283 my $parent = $self->{open_elements}->[$_]->[0]->parent_node;
4284 if (defined $parent and $parent->node_type == 1) {
4285 !!!cp ('t70');
4286 $foster_parent_element = $parent;
4287 $next_sibling = $self->{open_elements}->[$_]->[0];
4288 } else {
4289 !!!cp ('t71');
4290 $foster_parent_element
4291 = $self->{open_elements}->[$_ - 1]->[0];
4292 }
4293 last OE;
4294 }
4295 } # OE
4296 $foster_parent_element = $self->{open_elements}->[0]->[0]
4297 unless defined $foster_parent_element;
4298 $foster_parent_element->insert_before
4299 ($child, $next_sibling);
4300 $open_tables->[-1]->[1] = 1; # tainted
4301 } else {
4302 !!!cp ('t72');
4303 $self->{open_elements}->[-1]->[0]->append_child ($child);
4304 }
4305 }; # $insert_to_foster
4306
4307 B: while (1) {
4308 if ($token->{type} == DOCTYPE_TOKEN) {
4309 !!!cp ('t73');
4310 !!!parse-error (type => 'in html:#DOCTYPE', token => $token);
4311 ## Ignore the token
4312 ## Stay in the phase
4313 !!!next-token;
4314 next B;
4315 } elsif ($token->{type} == START_TAG_TOKEN and
4316 $token->{tag_name} eq 'html') {
4317 if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
4318 !!!cp ('t79');
4319 !!!parse-error (type => 'after html', text => 'html', token => $token);
4320 $self->{insertion_mode} = AFTER_BODY_IM;
4321 } elsif ($self->{insertion_mode} == AFTER_HTML_FRAMESET_IM) {
4322 !!!cp ('t80');
4323 !!!parse-error (type => 'after html', text => 'html', token => $token);
4324 $self->{insertion_mode} = AFTER_FRAMESET_IM;
4325 } else {
4326 !!!cp ('t81');
4327 }
4328
4329 !!!cp ('t82');
4330 !!!parse-error (type => 'not first start tag', token => $token);
4331 my $top_el = $self->{open_elements}->[0]->[0];
4332 for my $attr_name (keys %{$token->{attributes}}) {
4333 unless ($top_el->has_attribute_ns (undef, $attr_name)) {
4334 !!!cp ('t84');
4335 $top_el->set_attribute_ns
4336 (undef, [undef, $attr_name],
4337 $token->{attributes}->{$attr_name}->{value});
4338 }
4339 }
4340 !!!nack ('t84.1');
4341 !!!next-token;
4342 next B;
4343 } elsif ($token->{type} == COMMENT_TOKEN) {
4344 my $comment = $self->{document}->create_comment ($token->{data});
4345 if ($self->{insertion_mode} & AFTER_HTML_IMS) {
4346 !!!cp ('t85');
4347 $self->{document}->append_child ($comment);
4348 } elsif ($self->{insertion_mode} == AFTER_BODY_IM) {
4349 !!!cp ('t86');
4350 $self->{open_elements}->[0]->[0]->append_child ($comment);
4351 } else {
4352 !!!cp ('t87');
4353 $self->{open_elements}->[-1]->[0]->append_child ($comment);
4354 }
4355 !!!next-token;
4356 next B;
4357 } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM) {
4358 if ($token->{type} == CHARACTER_TOKEN) {
4359 !!!cp ('t87.1');
4360 $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});
4361 !!!next-token;
4362 next B;
4363 } elsif ($token->{type} == START_TAG_TOKEN) {
4364 if ((not {mglyph => 1, malignmark => 1}->{$token->{tag_name}} and
4365 $self->{open_elements}->[-1]->[1] & FOREIGN_FLOW_CONTENT_EL) or
4366 not ($self->{open_elements}->[-1]->[1] & FOREIGN_EL) or
4367 ($token->{tag_name} eq 'svg' and
4368 $self->{open_elements}->[-1]->[1] & MML_AXML_EL)) {
4369 ## NOTE: "using the rules for secondary insertion mode"then"continue"
4370 !!!cp ('t87.2');
4371 #
4372 } elsif ({
4373 b => 1, big => 1, blockquote => 1, body => 1, br => 1,
4374 center => 1, code => 1, dd => 1, div => 1, dl => 1, dt => 1,
4375 em => 1, embed => 1, font => 1, h1 => 1, h2 => 1, h3 => 1,
4376 h4 => 1, h5 => 1, h6 => 1, head => 1, hr => 1, i => 1,
4377 img => 1, li => 1, listing => 1, menu => 1, meta => 1,
4378 nobr => 1, ol => 1, p => 1, pre => 1, ruby => 1, s => 1,
4379 small => 1, span => 1, strong => 1, strike => 1, sub => 1,
4380 sup => 1, table => 1, tt => 1, u => 1, ul => 1, var => 1,
4381 }->{$token->{tag_name}}) {
4382 !!!cp ('t87.2');
4383 !!!parse-error (type => 'not closed',
4384 text => $self->{open_elements}->[-1]->[0]
4385 ->manakai_local_name,
4386 token => $token);
4387
4388 pop @{$self->{open_elements}}
4389 while $self->{open_elements}->[-1]->[1] & FOREIGN_EL;
4390
4391 $self->{insertion_mode} &= ~ IN_FOREIGN_CONTENT_IM;
4392 ## Reprocess.
4393 next B;
4394 } else {
4395 my $nsuri = $self->{open_elements}->[-1]->[0]->namespace_uri;
4396 my $tag_name = $token->{tag_name};
4397 if ($nsuri eq $SVG_NS) {
4398 $tag_name = {
4399 altglyph => 'altGlyph',
4400 altglyphdef => 'altGlyphDef',
4401 altglyphitem => 'altGlyphItem',
4402 animatecolor => 'animateColor',
4403 animatemotion => 'animateMotion',
4404 animatetransform => 'animateTransform',
4405 clippath => 'clipPath',
4406 feblend => 'feBlend',
4407 fecolormatrix => 'feColorMatrix',
4408 fecomponenttransfer => 'feComponentTransfer',
4409 fecomposite => 'feComposite',
4410 feconvolvematrix => 'feConvolveMatrix',
4411 fediffuselighting => 'feDiffuseLighting',
4412 fedisplacementmap => 'feDisplacementMap',
4413 fedistantlight => 'feDistantLight',
4414 feflood => 'feFlood',
4415 fefunca => 'feFuncA',
4416 fefuncb => 'feFuncB',
4417 fefuncg => 'feFuncG',
4418 fefuncr => 'feFuncR',
4419 fegaussianblur => 'feGaussianBlur',
4420 feimage => 'feImage',
4421 femerge => 'feMerge',
4422 femergenode => 'feMergeNode',
4423 femorphology => 'feMorphology',
4424 feoffset => 'feOffset',
4425 fepointlight => 'fePointLight',
4426 fespecularlighting => 'feSpecularLighting',
4427 fespotlight => 'feSpotLight',
4428 fetile => 'feTile',
4429 feturbulence => 'feTurbulence',
4430 foreignobject => 'foreignObject',
4431 glyphref => 'glyphRef',
4432 lineargradient => 'linearGradient',
4433 radialgradient => 'radialGradient',
4434 #solidcolor => 'solidColor', ## NOTE: Commented in spec (SVG1.2)
4435 textpath => 'textPath',
4436 }->{$tag_name} || $tag_name;
4437 }
4438
4439 ## "adjust SVG attributes" (SVG only) - done in insert-element-f
4440
4441 ## "adjust foreign attributes" - done in insert-element-f
4442
4443 !!!insert-element-f ($nsuri, $tag_name, $token->{attributes}, $token);
4444
4445 if ($self->{self_closing}) {
4446 pop @{$self->{open_elements}};
4447 !!!ack ('t87.3');
4448 } else {
4449 !!!cp ('t87.4');
4450 }
4451
4452 !!!next-token;
4453 next B;
4454 }
4455 } elsif ($token->{type} == END_TAG_TOKEN) {
4456 ## NOTE: "using the rules for secondary insertion mode" then "continue"
4457 !!!cp ('t87.5');
4458 #
4459 } elsif ($token->{type} == END_OF_FILE_TOKEN) {
4460 !!!cp ('t87.6');
4461 !!!parse-error (type => 'not closed',
4462 text => $self->{open_elements}->[-1]->[0]
4463 ->manakai_local_name,
4464 token => $token);
4465
4466 pop @{$self->{open_elements}}
4467 while $self->{open_elements}->[-1]->[1] & FOREIGN_EL;
4468
4469 $self->{insertion_mode} &= ~ IN_FOREIGN_CONTENT_IM;
4470 ## Reprocess.
4471 next B;
4472 } else {
4473 die "$0: $token->{type}: Unknown token type";
4474 }
4475 }
4476
4477 if ($self->{insertion_mode} & HEAD_IMS) {
4478 if ($token->{type} == CHARACTER_TOKEN) {
4479 if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {
4480 unless ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4481 !!!cp ('t88.2');
4482 $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
4483 } else {
4484 !!!cp ('t88.1');
4485 ## Ignore the token.
4486 !!!next-token;
4487 next B;
4488 }
4489 unless (length $token->{data}) {
4490 !!!cp ('t88');
4491 !!!next-token;
4492 next B;
4493 }
4494 }
4495
4496 if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4497 !!!cp ('t89');
4498 ## As if <head>
4499 !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);
4500 $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
4501 push @{$self->{open_elements}},
4502 [$self->{head_element}, $el_category->{head}];
4503
4504 ## Reprocess in the "in head" insertion mode...
4505 pop @{$self->{open_elements}};
4506
4507 ## Reprocess in the "after head" insertion mode...
4508 } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4509 !!!cp ('t90');
4510 ## As if </noscript>
4511 pop @{$self->{open_elements}};
4512 !!!parse-error (type => 'in noscript:#text', token => $token);
4513
4514 ## Reprocess in the "in head" insertion mode...
4515 ## As if </head>
4516 pop @{$self->{open_elements}};
4517
4518 ## Reprocess in the "after head" insertion mode...
4519 } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
4520 !!!cp ('t91');
4521 pop @{$self->{open_elements}};
4522
4523 ## Reprocess in the "after head" insertion mode...
4524 } else {
4525 !!!cp ('t92');
4526 }
4527
4528 ## "after head" insertion mode
4529 ## As if <body>
4530 !!!insert-element ('body',, $token);
4531 $self->{insertion_mode} = IN_BODY_IM;
4532 ## reprocess
4533 next B;
4534 } elsif ($token->{type} == START_TAG_TOKEN) {
4535 if ($token->{tag_name} eq 'head') {
4536 if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4537 !!!cp ('t93');
4538 !!!create-element ($self->{head_element}, $HTML_NS, $token->{tag_name}, $token->{attributes}, $token);
4539 $self->{open_elements}->[-1]->[0]->append_child
4540 ($self->{head_element});
4541 push @{$self->{open_elements}},
4542 [$self->{head_element}, $el_category->{head}];
4543 $self->{insertion_mode} = IN_HEAD_IM;
4544 !!!nack ('t93.1');
4545 !!!next-token;
4546 next B;
4547 } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4548 !!!cp ('t93.2');
4549 !!!parse-error (type => 'after head', text => 'head',
4550 token => $token);
4551 ## Ignore the token
4552 !!!nack ('t93.3');
4553 !!!next-token;
4554 next B;
4555 } else {
4556 !!!cp ('t95');
4557 !!!parse-error (type => 'in head:head',
4558 token => $token); # or in head noscript
4559 ## Ignore the token
4560 !!!nack ('t95.1');
4561 !!!next-token;
4562 next B;
4563 }
4564 } elsif ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4565 !!!cp ('t96');
4566 ## As if <head>
4567 !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);
4568 $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
4569 push @{$self->{open_elements}},
4570 [$self->{head_element}, $el_category->{head}];
4571
4572 $self->{insertion_mode} = IN_HEAD_IM;
4573 ## Reprocess in the "in head" insertion mode...
4574 } else {
4575 !!!cp ('t97');
4576 }
4577
4578 if ($token->{tag_name} eq 'base') {
4579 if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4580 !!!cp ('t98');
4581 ## As if </noscript>
4582 pop @{$self->{open_elements}};
4583 !!!parse-error (type => 'in noscript', text => 'base',
4584 token => $token);
4585
4586 $self->{insertion_mode} = IN_HEAD_IM;
4587 ## Reprocess in the "in head" insertion mode...
4588 } else {
4589 !!!cp ('t99');
4590 }
4591
4592 ## NOTE: There is a "as if in head" code clone.
4593 if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4594 !!!cp ('t100');
4595 !!!parse-error (type => 'after head',
4596 text => $token->{tag_name}, token => $token);
4597 push @{$self->{open_elements}},
4598 [$self->{head_element}, $el_category->{head}];
4599 } else {
4600 !!!cp ('t101');
4601 }
4602 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4603 pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.
4604 pop @{$self->{open_elements}} # <head>
4605 if $self->{insertion_mode} == AFTER_HEAD_IM;
4606 !!!nack ('t101.1');
4607 !!!next-token;
4608 next B;
4609 } elsif ($token->{tag_name} eq 'link') {
4610 ## NOTE: There is a "as if in head" code clone.
4611 if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4612 !!!cp ('t102');
4613 !!!parse-error (type => 'after head',
4614 text => $token->{tag_name}, token => $token);
4615 push @{$self->{open_elements}},
4616 [$self->{head_element}, $el_category->{head}];
4617 } else {
4618 !!!cp ('t103');
4619 }
4620 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4621 pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.
4622 pop @{$self->{open_elements}} # <head>
4623 if $self->{insertion_mode} == AFTER_HEAD_IM;
4624 !!!ack ('t103.1');
4625 !!!next-token;
4626 next B;
4627 } elsif ($token->{tag_name} eq 'meta') {
4628 ## NOTE: There is a "as if in head" code clone.
4629 if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4630 !!!cp ('t104');
4631 !!!parse-error (type => 'after head',
4632 text => $token->{tag_name}, token => $token);
4633 push @{$self->{open_elements}},
4634 [$self->{head_element}, $el_category->{head}];
4635 } else {
4636 !!!cp ('t105');
4637 }
4638 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4639 my $meta_el = pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.
4640
4641 unless ($self->{confident}) {
4642 if ($token->{attributes}->{charset}) {
4643 !!!cp ('t106');
4644 ## NOTE: Whether the encoding is supported or not is handled
4645 ## in the {change_encoding} callback.
4646 $self->{change_encoding}
4647 ->($self, $token->{attributes}->{charset}->{value},
4648 $token);
4649
4650 $meta_el->[0]->get_attribute_node_ns (undef, 'charset')
4651 ->set_user_data (manakai_has_reference =>
4652 $token->{attributes}->{charset}
4653 ->{has_reference});
4654 } elsif ($token->{attributes}->{content}) {
4655 if ($token->{attributes}->{content}->{value}
4656 =~ /[Cc][Hh][Aa][Rr][Ss][Ee][Tt]
4657 [\x09-\x0D\x20]*=
4658 [\x09-\x0D\x20]*(?>"([^"]*)"|'([^']*)'|
4659 ([^"'\x09-\x0D\x20][^\x09-\x0D\x20\x3B]*))/x) {
4660 !!!cp ('t107');
4661 ## NOTE: Whether the encoding is supported or not is handled
4662 ## in the {change_encoding} callback.
4663 $self->{change_encoding}
4664 ->($self, defined $1 ? $1 : defined $2 ? $2 : $3,
4665 $token);
4666 $meta_el->[0]->get_attribute_node_ns (undef, 'content')
4667 ->set_user_data (manakai_has_reference =>
4668 $token->{attributes}->{content}
4669 ->{has_reference});
4670 } else {
4671 !!!cp ('t108');
4672 }
4673 }
4674 } else {
4675 if ($token->{attributes}->{charset}) {
4676 !!!cp ('t109');
4677 $meta_el->[0]->get_attribute_node_ns (undef, 'charset')
4678 ->set_user_data (manakai_has_reference =>
4679 $token->{attributes}->{charset}
4680 ->{has_reference});
4681 }
4682 if ($token->{attributes}->{content}) {
4683 !!!cp ('t110');
4684 $meta_el->[0]->get_attribute_node_ns (undef, 'content')
4685 ->set_user_data (manakai_has_reference =>
4686 $token->{attributes}->{content}
4687 ->{has_reference});
4688 }
4689 }
4690
4691 pop @{$self->{open_elements}} # <head>
4692 if $self->{insertion_mode} == AFTER_HEAD_IM;
4693 !!!ack ('t110.1');
4694 !!!next-token;
4695 next B;
4696 } elsif ($token->{tag_name} eq 'title') {
4697 if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4698 !!!cp ('t111');
4699 ## As if </noscript>
4700 pop @{$self->{open_elements}};
4701 !!!parse-error (type => 'in noscript', text => 'title',
4702 token => $token);
4703
4704 $self->{insertion_mode} = IN_HEAD_IM;
4705 ## Reprocess in the "in head" insertion mode...
4706 } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4707 !!!cp ('t112');
4708 !!!parse-error (type => 'after head',
4709 text => $token->{tag_name}, token => $token);
4710 push @{$self->{open_elements}},
4711 [$self->{head_element}, $el_category->{head}];
4712 } else {
4713 !!!cp ('t113');
4714 }
4715
4716 ## NOTE: There is a "as if in head" code clone.
4717 my $parent = defined $self->{head_element} ? $self->{head_element}
4718 : $self->{open_elements}->[-1]->[0];
4719 $parse_rcdata->(RCDATA_CONTENT_MODEL);
4720 pop @{$self->{open_elements}} # <head>
4721 if $self->{insertion_mode} == AFTER_HEAD_IM;
4722 next B;
4723 } elsif ($token->{tag_name} eq 'style' or
4724 $token->{tag_name} eq 'noframes') {
4725 ## NOTE: Or (scripting is enabled and tag_name eq 'noscript' and
4726 ## insertion mode IN_HEAD_IM)
4727 ## NOTE: There is a "as if in head" code clone.
4728 if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4729 !!!cp ('t114');
4730 !!!parse-error (type => 'after head',
4731 text => $token->{tag_name}, token => $token);
4732 push @{$self->{open_elements}},
4733 [$self->{head_element}, $el_category->{head}];
4734 } else {
4735 !!!cp ('t115');
4736 }
4737 $parse_rcdata->(CDATA_CONTENT_MODEL);
4738 pop @{$self->{open_elements}} # <head>
4739 if $self->{insertion_mode} == AFTER_HEAD_IM;
4740 next B;
4741 } elsif ($token->{tag_name} eq 'noscript') {
4742 if ($self->{insertion_mode} == IN_HEAD_IM) {
4743 !!!cp ('t116');
4744 ## NOTE: and scripting is disalbed
4745 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4746 $self->{insertion_mode} = IN_HEAD_NOSCRIPT_IM;
4747 !!!nack ('t116.1');
4748 !!!next-token;
4749 next B;
4750 } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4751 !!!cp ('t117');
4752 !!!parse-error (type => 'in noscript', text => 'noscript',
4753 token => $token);
4754 ## Ignore the token
4755 !!!nack ('t117.1');
4756 !!!next-token;
4757 next B;
4758 } else {
4759 !!!cp ('t118');
4760 #
4761 }
4762 } elsif ($token->{tag_name} eq 'script') {
4763 if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4764 !!!cp ('t119');
4765 ## As if </noscript>
4766 pop @{$self->{open_elements}};
4767 !!!parse-error (type => 'in noscript', text => 'script',
4768 token => $token);
4769
4770 $self->{insertion_mode} = IN_HEAD_IM;
4771 ## Reprocess in the "in head" insertion mode...
4772 } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4773 !!!cp ('t120');
4774 !!!parse-error (type => 'after head',
4775 text => $token->{tag_name}, token => $token);
4776 push @{$self->{open_elements}},
4777 [$self->{head_element}, $el_category->{head}];
4778 } else {
4779 !!!cp ('t121');
4780 }
4781
4782 ## NOTE: There is a "as if in head" code clone.
4783 $script_start_tag->();
4784 pop @{$self->{open_elements}} # <head>
4785 if $self->{insertion_mode} == AFTER_HEAD_IM;
4786 next B;
4787 } elsif ($token->{tag_name} eq 'body' or
4788 $token->{tag_name} eq 'frameset') {
4789 if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4790 !!!cp ('t122');
4791 ## As if </noscript>
4792 pop @{$self->{open_elements}};
4793 !!!parse-error (type => 'in noscript',
4794 text => $token->{tag_name}, token => $token);
4795
4796 ## Reprocess in the "in head" insertion mode...
4797 ## As if </head>
4798 pop @{$self->{open_elements}};
4799
4800 ## Reprocess in the "after head" insertion mode...
4801 } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
4802 !!!cp ('t124');
4803 pop @{$self->{open_elements}};
4804
4805 ## Reprocess in the "after head" insertion mode...
4806 } else {
4807 !!!cp ('t125');
4808 }
4809
4810 ## "after head" insertion mode
4811 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4812 if ($token->{tag_name} eq 'body') {
4813 !!!cp ('t126');
4814 $self->{insertion_mode} = IN_BODY_IM;
4815 } elsif ($token->{tag_name} eq 'frameset') {
4816 !!!cp ('t127');
4817 $self->{insertion_mode} = IN_FRAMESET_IM;
4818 } else {
4819 die "$0: tag name: $self->{tag_name}";
4820 }
4821 !!!nack ('t127.1');
4822 !!!next-token;
4823 next B;
4824 } else {
4825 !!!cp ('t128');
4826 #
4827 }
4828
4829 if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4830 !!!cp ('t129');
4831 ## As if </noscript>
4832 pop @{$self->{open_elements}};
4833 !!!parse-error (type => 'in noscript:/',
4834 text => $token->{tag_name}, token => $token);
4835
4836 ## Reprocess in the "in head" insertion mode...
4837 ## As if </head>
4838 pop @{$self->{open_elements}};
4839
4840 ## Reprocess in the "after head" insertion mode...
4841 } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
4842 !!!cp ('t130');
4843 ## As if </head>
4844 pop @{$self->{open_elements}};
4845
4846 ## Reprocess in the "after head" insertion mode...
4847 } else {
4848 !!!cp ('t131');
4849 }
4850
4851 ## "after head" insertion mode
4852 ## As if <body>
4853 !!!insert-element ('body',, $token);
4854 $self->{insertion_mode} = IN_BODY_IM;
4855 ## reprocess
4856 !!!ack-later;
4857 next B;
4858 } elsif ($token->{type} == END_TAG_TOKEN) {
4859 if ($token->{tag_name} eq 'head') {
4860 if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4861 !!!cp ('t132');
4862 ## As if <head>
4863 !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);
4864 $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
4865 push @{$self->{open_elements}},
4866 [$self->{head_element}, $el_category->{head}];
4867
4868 ## Reprocess in the "in head" insertion mode...
4869 pop @{$self->{open_elements}};
4870 $self->{insertion_mode} = AFTER_HEAD_IM;
4871 !!!next-token;
4872 next B;
4873 } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4874 !!!cp ('t133');
4875 ## As if </noscript>
4876 pop @{$self->{open_elements}};
4877 !!!parse-error (type => 'in noscript:/',
4878 text => 'head', token => $token);
4879
4880 ## Reprocess in the "in head" insertion mode...
4881 pop @{$self->{open_elements}};
4882 $self->{insertion_mode} = AFTER_HEAD_IM;
4883 !!!next-token;
4884 next B;
4885 } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
4886 !!!cp ('t134');
4887 pop @{$self->{open_elements}};
4888 $self->{insertion_mode} = AFTER_HEAD_IM;
4889 !!!next-token;
4890 next B;
4891 } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4892 !!!cp ('t134.1');
4893 !!!parse-error (type => 'unmatched end tag', text => 'head',
4894 token => $token);
4895 ## Ignore the token
4896 !!!next-token;
4897 next B;
4898 } else {
4899 die "$0: $self->{insertion_mode}: Unknown insertion mode";
4900 }
4901 } elsif ($token->{tag_name} eq 'noscript') {
4902 if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4903 !!!cp ('t136');
4904 pop @{$self->{open_elements}};
4905 $self->{insertion_mode} = IN_HEAD_IM;
4906 !!!next-token;
4907 next B;
4908 } elsif ($self->{insertion_mode} == BEFORE_HEAD_IM or
4909 $self->{insertion_mode} == AFTER_HEAD_IM) {
4910 !!!cp ('t137');
4911 !!!parse-error (type => 'unmatched end tag',
4912 text => 'noscript', token => $token);
4913 ## Ignore the token ## ISSUE: An issue in the spec.
4914 !!!next-token;
4915 next B;
4916 } else {
4917 !!!cp ('t138');
4918 #
4919 }
4920 } elsif ({
4921 body => 1, html => 1,
4922 }->{$token->{tag_name}}) {
4923 if ($self->{insertion_mode} == BEFORE_HEAD_IM or
4924 $self->{insertion_mode} == IN_HEAD_IM or
4925 $self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4926 !!!cp ('t140');
4927 !!!parse-error (type => 'unmatched end tag',
4928 text => $token->{tag_name}, token => $token);
4929 ## Ignore the token
4930 !!!next-token;
4931 next B;
4932 } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4933 !!!cp ('t140.1');
4934 !!!parse-error (type => 'unmatched end tag',
4935 text => $token->{tag_name}, token => $token);
4936 ## Ignore the token
4937 !!!next-token;
4938 next B;
4939 } else {
4940 die "$0: $self->{insertion_mode}: Unknown insertion mode";
4941 }
4942 } elsif ($token->{tag_name} eq 'p') {
4943 !!!cp ('t142');
4944 !!!parse-error (type => 'unmatched end tag',
4945 text => $token->{tag_name}, token => $token);
4946 ## Ignore the token
4947 !!!next-token;
4948 next B;
4949 } elsif ($token->{tag_name} eq 'br') {
4950 if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4951 !!!cp ('t142.2');
4952 ## (before head) as if <head>, (in head) as if </head>
4953 !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);
4954 $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
4955 $self->{insertion_mode} = AFTER_HEAD_IM;
4956
4957 ## Reprocess in the "after head" insertion mode...
4958 } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
4959 !!!cp ('t143.2');
4960 ## As if </head>
4961 pop @{$self->{open_elements}};
4962 $self->{insertion_mode} = AFTER_HEAD_IM;
4963
4964 ## Reprocess in the "after head" insertion mode...
4965 } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4966 !!!cp ('t143.3');
4967 ## ISSUE: Two parse errors for <head><noscript></br>
4968 !!!parse-error (type => 'unmatched end tag',
4969 text => 'br', token => $token);
4970 ## As if </noscript>
4971 pop @{$self->{open_elements}};
4972 $self->{insertion_mode} = IN_HEAD_IM;
4973
4974 ## Reprocess in the "in head" insertion mode...
4975 ## As if </head>
4976 pop @{$self->{open_elements}};
4977 $self->{insertion_mode} = AFTER_HEAD_IM;
4978
4979 ## Reprocess in the "after head" insertion mode...
4980 } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4981 !!!cp ('t143.4');
4982 #
4983 } else {
4984 die "$0: $self->{insertion_mode}: Unknown insertion mode";
4985 }
4986
4987 ## ISSUE: does not agree with IE7 - it doesn't ignore </br>.
4988 !!!parse-error (type => 'unmatched end tag',
4989 text => 'br', token => $token);
4990 ## Ignore the token
4991 !!!next-token;
4992 next B;
4993 } else {
4994 !!!cp ('t145');
4995 !!!parse-error (type => 'unmatched end tag',
4996 text => $token->{tag_name}, token => $token);
4997 ## Ignore the token
4998 !!!next-token;
4999 next B;
5000 }
5001
5002 if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
5003 !!!cp ('t146');
5004 ## As if </noscript>
5005 pop @{$self->{open_elements}};
5006 !!!parse-error (type => 'in noscript:/',
5007 text => $token->{tag_name}, token => $token);
5008
5009 ## Reprocess in the "in head" insertion mode...
5010 ## As if </head>
5011 pop @{$self->{open_elements}};
5012
5013 ## Reprocess in the "after head" insertion mode...
5014 } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
5015 !!!cp ('t147');
5016 ## As if </head>
5017 pop @{$self->{open_elements}};
5018
5019 ## Reprocess in the "after head" insertion mode...
5020 } elsif ($self->{insertion_mode} == BEFORE_HEAD_IM) {
5021 ## ISSUE: This case cannot be reached?
5022 !!!cp ('t148');
5023 !!!parse-error (type => 'unmatched end tag',
5024 text => $token->{tag_name}, token => $token);
5025 ## Ignore the token ## ISSUE: An issue in the spec.
5026 !!!next-token;
5027 next B;
5028 } else {
5029 !!!cp ('t149');
5030 }
5031
5032 ## "after head" insertion mode
5033 ## As if <body>
5034 !!!insert-element ('body',, $token);
5035 $self->{insertion_mode} = IN_BODY_IM;
5036 ## reprocess
5037 next B;
5038 } elsif ($token->{type} == END_OF_FILE_TOKEN) {
5039 if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
5040 !!!cp ('t149.1');
5041
5042 ## NOTE: As if <head>
5043 !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);
5044 $self->{open_elements}->[-1]->[0]->append_child
5045 ($self->{head_element});
5046 #push @{$self->{open_elements}},
5047 # [$self->{head_element}, $el_category->{head}];
5048 #$self->{insertion_mode} = IN_HEAD_IM;
5049 ## NOTE: Reprocess.
5050
5051 ## NOTE: As if </head>
5052 #pop @{$self->{open_elements}};
5053 #$self->{insertion_mode} = IN_AFTER_HEAD_IM;
5054 ## NOTE: Reprocess.
5055
5056 #
5057 } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
5058 !!!cp ('t149.2');
5059
5060 ## NOTE: As if </head>
5061 pop @{$self->{open_elements}};
5062 #$self->{insertion_mode} = IN_AFTER_HEAD_IM;
5063 ## NOTE: Reprocess.
5064
5065 #
5066 } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
5067 !!!cp ('t149.3');
5068
5069 !!!parse-error (type => 'in noscript:#eof', token => $token);
5070
5071 ## As if </noscript>
5072 pop @{$self->{open_elements}};
5073 #$self->{insertion_mode} = IN_HEAD_IM;
5074 ## NOTE: Reprocess.
5075
5076 ## NOTE: As if </head>
5077 pop @{$self->{open_elements}};
5078 #$self->{insertion_mode} = IN_AFTER_HEAD_IM;
5079 ## NOTE: Reprocess.
5080
5081 #
5082 } else {
5083 !!!cp ('t149.4');
5084 #
5085 }
5086
5087 ## NOTE: As if <body>
5088 !!!insert-element ('body',, $token);
5089 $self->{insertion_mode} = IN_BODY_IM;
5090 ## NOTE: Reprocess.
5091 next B;
5092 } else {
5093 die "$0: $token->{type}: Unknown token type";
5094 }
5095
5096 ## ISSUE: An issue in the spec.
5097 } elsif ($self->{insertion_mode} & BODY_IMS) {
5098 if ($token->{type} == CHARACTER_TOKEN) {
5099 !!!cp ('t150');
5100 ## NOTE: There is a code clone of "character in body".
5101 $reconstruct_active_formatting_elements->($insert_to_current);
5102
5103 $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});
5104
5105 !!!next-token;
5106 next B;
5107 } elsif ($token->{type} == START_TAG_TOKEN) {
5108 if ({
5109 caption => 1, col => 1, colgroup => 1, tbody => 1,
5110 td => 1, tfoot => 1, th => 1, thead => 1, tr => 1,
5111 }->{$token->{tag_name}}) {
5112 if ($self->{insertion_mode} == IN_CELL_IM) {
5113 ## have an element in table scope
5114 for (reverse 0..$#{$self->{open_elements}}) {
5115 my $node = $self->{open_elements}->[$_];
5116 if ($node->[1] & TABLE_CELL_EL) {
5117 !!!cp ('t151');
5118
5119 ## Close the cell
5120 !!!back-token; # <x>
5121 $token = {type => END_TAG_TOKEN,
5122 tag_name => $node->[0]->manakai_local_name,
5123 line => $token->{line},
5124 column => $token->{column}};
5125 next B;
5126 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5127 !!!cp ('t152');
5128 ## ISSUE: This case can never be reached, maybe.
5129 last;
5130 }
5131 }
5132
5133 !!!cp ('t153');
5134 !!!parse-error (type => 'start tag not allowed',
5135 text => $token->{tag_name}, token => $token);
5136 ## Ignore the token
5137 !!!nack ('t153.1');
5138 !!!next-token;
5139 next B;
5140 } elsif ($self->{insertion_mode} == IN_CAPTION_IM) {
5141 !!!parse-error (type => 'not closed', text => 'caption',
5142 token => $token);
5143
5144 ## NOTE: As if </caption>.
5145 ## have a table element in table scope
5146 my $i;
5147 INSCOPE: {
5148 for (reverse 0..$#{$self->{open_elements}}) {
5149 my $node = $self->{open_elements}->[$_];
5150 if ($node->[1] & CAPTION_EL) {
5151 !!!cp ('t155');
5152 $i = $_;
5153 last INSCOPE;
5154 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5155 !!!cp ('t156');
5156 last;
5157 }
5158 }
5159
5160 !!!cp ('t157');
5161 !!!parse-error (type => 'start tag not allowed',
5162 text => $token->{tag_name}, token => $token);
5163 ## Ignore the token
5164 !!!nack ('t157.1');
5165 !!!next-token;
5166 next B;
5167 } # INSCOPE
5168
5169 ## generate implied end tags
5170 while ($self->{open_elements}->[-1]->[1]
5171 & END_TAG_OPTIONAL_EL) {
5172 !!!cp ('t158');
5173 pop @{$self->{open_elements}};
5174 }
5175
5176 unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {
5177 !!!cp ('t159');
5178 !!!parse-error (type => 'not closed',
5179 text => $self->{open_elements}->[-1]->[0]
5180 ->manakai_local_name,
5181 token => $token);
5182 } else {
5183 !!!cp ('t160');
5184 }
5185
5186 splice @{$self->{open_elements}}, $i;
5187
5188 $clear_up_to_marker->();
5189
5190 $self->{insertion_mode} = IN_TABLE_IM;
5191
5192 ## reprocess
5193 !!!ack-later;
5194 next B;
5195 } else {
5196 !!!cp ('t161');
5197 #
5198 }
5199 } else {
5200 !!!cp ('t162');
5201 #
5202 }
5203 } elsif ($token->{type} == END_TAG_TOKEN) {
5204 if ($token->{tag_name} eq 'td' or $token->{tag_name} eq 'th') {
5205 if ($self->{insertion_mode} == IN_CELL_IM) {
5206 ## have an element in table scope
5207 my $i;
5208 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5209 my $node = $self->{open_elements}->[$_];
5210 if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
5211 !!!cp ('t163');
5212 $i = $_;
5213 last INSCOPE;
5214 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5215 !!!cp ('t164');
5216 last INSCOPE;
5217 }
5218 } # INSCOPE
5219 unless (defined $i) {
5220 !!!cp ('t165');
5221 !!!parse-error (type => 'unmatched end tag',
5222 text => $token->{tag_name},
5223 token => $token);
5224 ## Ignore the token
5225 !!!next-token;
5226 next B;
5227 }
5228
5229 ## generate implied end tags
5230 while ($self->{open_elements}->[-1]->[1]
5231 & END_TAG_OPTIONAL_EL) {
5232 !!!cp ('t166');
5233 pop @{$self->{open_elements}};
5234 }
5235
5236 if ($self->{open_elements}->[-1]->[0]->manakai_local_name
5237 ne $token->{tag_name}) {
5238 !!!cp ('t167');
5239 !!!parse-error (type => 'not closed',
5240 text => $self->{open_elements}->[-1]->[0]
5241 ->manakai_local_name,
5242 token => $token);
5243 } else {
5244 !!!cp ('t168');
5245 }
5246
5247 splice @{$self->{open_elements}}, $i;
5248
5249 $clear_up_to_marker->();
5250
5251 $self->{insertion_mode} = IN_ROW_IM;
5252
5253 !!!next-token;
5254 next B;
5255 } elsif ($self->{insertion_mode} == IN_CAPTION_IM) {
5256 !!!cp ('t169');
5257 !!!parse-error (type => 'unmatched end tag',
5258 text => $token->{tag_name}, token => $token);
5259 ## Ignore the token
5260 !!!next-token;
5261 next B;
5262 } else {
5263 !!!cp ('t170');
5264 #
5265 }
5266 } elsif ($token->{tag_name} eq 'caption') {
5267 if ($self->{insertion_mode} == IN_CAPTION_IM) {
5268 ## have a table element in table scope
5269 my $i;
5270 INSCOPE: {
5271 for (reverse 0..$#{$self->{open_elements}}) {
5272 my $node = $self->{open_elements}->[$_];
5273 if ($node->[1] & CAPTION_EL) {
5274 !!!cp ('t171');
5275 $i = $_;
5276 last INSCOPE;
5277 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5278 !!!cp ('t172');
5279 last;
5280 }
5281 }
5282
5283 !!!cp ('t173');
5284 !!!parse-error (type => 'unmatched end tag',
5285 text => $token->{tag_name}, token => $token);
5286 ## Ignore the token
5287 !!!next-token;
5288 next B;
5289 } # INSCOPE
5290
5291 ## generate implied end tags
5292 while ($self->{open_elements}->[-1]->[1]
5293 & END_TAG_OPTIONAL_EL) {
5294 !!!cp ('t174');
5295 pop @{$self->{open_elements}};
5296 }
5297
5298 unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {
5299 !!!cp ('t175');
5300 !!!parse-error (type => 'not closed',
5301 text => $self->{open_elements}->[-1]->[0]
5302 ->manakai_local_name,
5303 token => $token);
5304 } else {
5305 !!!cp ('t176');
5306 }
5307
5308 splice @{$self->{open_elements}}, $i;
5309
5310 $clear_up_to_marker->();
5311
5312 $self->{insertion_mode} = IN_TABLE_IM;
5313
5314 !!!next-token;
5315 next B;
5316 } elsif ($self->{insertion_mode} == IN_CELL_IM) {
5317 !!!cp ('t177');
5318 !!!parse-error (type => 'unmatched end tag',
5319 text => $token->{tag_name}, token => $token);
5320 ## Ignore the token
5321 !!!next-token;
5322 next B;
5323 } else {
5324 !!!cp ('t178');
5325 #
5326 }
5327 } elsif ({
5328 table => 1, tbody => 1, tfoot => 1,
5329 thead => 1, tr => 1,
5330 }->{$token->{tag_name}} and
5331 $self->{insertion_mode} == IN_CELL_IM) {
5332 ## have an element in table scope
5333 my $i;
5334 my $tn;
5335 INSCOPE: {
5336 for (reverse 0..$#{$self->{open_elements}}) {
5337 my $node = $self->{open_elements}->[$_];
5338 if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
5339 !!!cp ('t179');
5340 $i = $_;
5341
5342 ## Close the cell
5343 !!!back-token; # </x>
5344 $token = {type => END_TAG_TOKEN, tag_name => $tn,
5345 line => $token->{line},
5346 column => $token->{column}};
5347 next B;
5348 } elsif ($node->[1] & TABLE_CELL_EL) {
5349 !!!cp ('t180');
5350 $tn = $node->[0]->manakai_local_name;
5351 ## NOTE: There is exactly one |td| or |th| element
5352 ## in scope in the stack of open elements by definition.
5353 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5354 ## ISSUE: Can this be reached?
5355 !!!cp ('t181');
5356 last;
5357 }
5358 }
5359
5360 !!!cp ('t182');
5361 !!!parse-error (type => 'unmatched end tag',
5362 text => $token->{tag_name}, token => $token);
5363 ## Ignore the token
5364 !!!next-token;
5365 next B;
5366 } # INSCOPE
5367 } elsif ($token->{tag_name} eq 'table' and
5368 $self->{insertion_mode} == IN_CAPTION_IM) {
5369 !!!parse-error (type => 'not closed', text => 'caption',
5370 token => $token);
5371
5372 ## As if </caption>
5373 ## have a table element in table scope
5374 my $i;
5375 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5376 my $node = $self->{open_elements}->[$_];
5377 if ($node->[1] & CAPTION_EL) {
5378 !!!cp ('t184');
5379 $i = $_;
5380 last INSCOPE;
5381 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5382 !!!cp ('t185');
5383 last INSCOPE;
5384 }
5385 } # INSCOPE
5386 unless (defined $i) {
5387 !!!cp ('t186');
5388 !!!parse-error (type => 'unmatched end tag',
5389 text => 'caption', token => $token);
5390 ## Ignore the token
5391 !!!next-token;
5392 next B;
5393 }
5394
5395 ## generate implied end tags
5396 while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
5397 !!!cp ('t187');
5398 pop @{$self->{open_elements}};
5399 }
5400
5401 unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {
5402 !!!cp ('t188');
5403 !!!parse-error (type => 'not closed',
5404 text => $self->{open_elements}->[-1]->[0]
5405 ->manakai_local_name,
5406 token => $token);
5407 } else {
5408 !!!cp ('t189');
5409 }
5410
5411 splice @{$self->{open_elements}}, $i;
5412
5413 $clear_up_to_marker->();
5414
5415 $self->{insertion_mode} = IN_TABLE_IM;
5416
5417 ## reprocess
5418 next B;
5419 } elsif ({
5420 body => 1, col => 1, colgroup => 1, html => 1,
5421 }->{$token->{tag_name}}) {
5422 if ($self->{insertion_mode} & BODY_TABLE_IMS) {
5423 !!!cp ('t190');
5424 !!!parse-error (type => 'unmatched end tag',
5425 text => $token->{tag_name}, token => $token);
5426 ## Ignore the token
5427 !!!next-token;
5428 next B;
5429 } else {
5430 !!!cp ('t191');
5431 #
5432 }
5433 } elsif ({
5434 tbody => 1, tfoot => 1,
5435 thead => 1, tr => 1,
5436 }->{$token->{tag_name}} and
5437 $self->{insertion_mode} == IN_CAPTION_IM) {
5438 !!!cp ('t192');
5439 !!!parse-error (type => 'unmatched end tag',
5440 text => $token->{tag_name}, token => $token);
5441 ## Ignore the token
5442 !!!next-token;
5443 next B;
5444 } else {
5445 !!!cp ('t193');
5446 #
5447 }
5448 } elsif ($token->{type} == END_OF_FILE_TOKEN) {
5449 for my $entry (@{$self->{open_elements}}) {
5450 unless ($entry->[1] & ALL_END_TAG_OPTIONAL_EL) {
5451 !!!cp ('t75');
5452 !!!parse-error (type => 'in body:#eof', token => $token);
5453 last;
5454 }
5455 }
5456
5457 ## Stop parsing.
5458 last B;
5459 } else {
5460 die "$0: $token->{type}: Unknown token type";
5461 }
5462
5463 $insert = $insert_to_current;
5464 #
5465 } elsif ($self->{insertion_mode} & TABLE_IMS) {
5466 if ($token->{type} == CHARACTER_TOKEN) {
5467 if (not $open_tables->[-1]->[1] and # tainted
5468 $token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {
5469 $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
5470
5471 unless (length $token->{data}) {
5472 !!!cp ('t194');
5473 !!!next-token;
5474 next B;
5475 } else {
5476 !!!cp ('t195');
5477 }
5478 }
5479
5480 !!!parse-error (type => 'in table:#text', token => $token);
5481
5482 ## As if in body, but insert into foster parent element
5483 ## ISSUE: Spec says that "whenever a node would be inserted
5484 ## into the current node" while characters might not be
5485 ## result in a new Text node.
5486 $reconstruct_active_formatting_elements->($insert_to_foster);
5487
5488 if ($self->{open_elements}->[-1]->[1] & TABLE_ROWS_EL) {
5489 # MUST
5490 my $foster_parent_element;
5491 my $next_sibling;
5492 my $prev_sibling;
5493 OE: for (reverse 0..$#{$self->{open_elements}}) {
5494 if ($self->{open_elements}->[$_]->[1] & TABLE_EL) {
5495 my $parent = $self->{open_elements}->[$_]->[0]->parent_node;
5496 if (defined $parent and $parent->node_type == 1) {
5497 !!!cp ('t196');
5498 $foster_parent_element = $parent;
5499 $next_sibling = $self->{open_elements}->[$_]->[0];
5500 $prev_sibling = $next_sibling->previous_sibling;
5501 } else {
5502 !!!cp ('t197');
5503 $foster_parent_element = $self->{open_elements}->[$_ - 1]->[0];
5504 $prev_sibling = $foster_parent_element->last_child;
5505 }
5506 last OE;
5507 }
5508 } # OE
5509 $foster_parent_element = $self->{open_elements}->[0]->[0] and
5510 $prev_sibling = $foster_parent_element->last_child
5511 unless defined $foster_parent_element;
5512 if (defined $prev_sibling and
5513 $prev_sibling->node_type == 3) {
5514 !!!cp ('t198');
5515 $prev_sibling->manakai_append_text ($token->{data});
5516 } else {
5517 !!!cp ('t199');
5518 $foster_parent_element->insert_before
5519 ($self->{document}->create_text_node ($token->{data}),
5520 $next_sibling);
5521 }
5522 $open_tables->[-1]->[1] = 1; # tainted
5523 } else {
5524 !!!cp ('t200');
5525 $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});
5526 }
5527
5528 !!!next-token;
5529 next B;
5530 } elsif ($token->{type} == START_TAG_TOKEN) {
5531 if ({
5532 tr => ($self->{insertion_mode} != IN_ROW_IM),
5533 th => 1, td => 1,
5534 }->{$token->{tag_name}}) {
5535 if ($self->{insertion_mode} == IN_TABLE_IM) {
5536 ## Clear back to table context
5537 while (not ($self->{open_elements}->[-1]->[1]
5538 & TABLE_SCOPING_EL)) {
5539 !!!cp ('t201');
5540 pop @{$self->{open_elements}};
5541 }
5542
5543 !!!insert-element ('tbody',, $token);
5544 $self->{insertion_mode} = IN_TABLE_BODY_IM;
5545 ## reprocess in the "in table body" insertion mode...
5546 }
5547
5548 if ($self->{insertion_mode} == IN_TABLE_BODY_IM) {
5549 unless ($token->{tag_name} eq 'tr') {
5550 !!!cp ('t202');
5551 !!!parse-error (type => 'missing start tag:tr', token => $token);
5552 }
5553
5554 ## Clear back to table body context
5555 while (not ($self->{open_elements}->[-1]->[1]
5556 & TABLE_ROWS_SCOPING_EL)) {
5557 !!!cp ('t203');
5558 ## ISSUE: Can this case be reached?
5559 pop @{$self->{open_elements}};
5560 }
5561
5562 $self->{insertion_mode} = IN_ROW_IM;
5563 if ($token->{tag_name} eq 'tr') {
5564 !!!cp ('t204');
5565 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5566 !!!nack ('t204');
5567 !!!next-token;
5568 next B;
5569 } else {
5570 !!!cp ('t205');
5571 !!!insert-element ('tr',, $token);
5572 ## reprocess in the "in row" insertion mode
5573 }
5574 } else {
5575 !!!cp ('t206');
5576 }
5577
5578 ## Clear back to table row context
5579 while (not ($self->{open_elements}->[-1]->[1]
5580 & TABLE_ROW_SCOPING_EL)) {
5581 !!!cp ('t207');
5582 pop @{$self->{open_elements}};
5583 }
5584
5585 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5586 $self->{insertion_mode} = IN_CELL_IM;
5587
5588 push @$active_formatting_elements, ['#marker', ''];
5589
5590 !!!nack ('t207.1');
5591 !!!next-token;
5592 next B;
5593 } elsif ({
5594 caption => 1, col => 1, colgroup => 1,
5595 tbody => 1, tfoot => 1, thead => 1,
5596 tr => 1, # $self->{insertion_mode} == IN_ROW_IM
5597 }->{$token->{tag_name}}) {
5598 if ($self->{insertion_mode} == IN_ROW_IM) {
5599 ## As if </tr>
5600 ## have an element in table scope
5601 my $i;
5602 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5603 my $node = $self->{open_elements}->[$_];
5604 if ($node->[1] & TABLE_ROW_EL) {
5605 !!!cp ('t208');
5606 $i = $_;
5607 last INSCOPE;
5608 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5609 !!!cp ('t209');
5610 last INSCOPE;
5611 }
5612 } # INSCOPE
5613 unless (defined $i) {
5614 !!!cp ('t210');
5615 ## TODO: This type is wrong.
5616 !!!parse-error (type => 'unmacthed end tag',
5617 text => $token->{tag_name}, token => $token);
5618 ## Ignore the token
5619 !!!nack ('t210.1');
5620 !!!next-token;
5621 next B;
5622 }
5623
5624 ## Clear back to table row context
5625 while (not ($self->{open_elements}->[-1]->[1]
5626 & TABLE_ROW_SCOPING_EL)) {
5627 !!!cp ('t211');
5628 ## ISSUE: Can this case be reached?
5629 pop @{$self->{open_elements}};
5630 }
5631
5632 pop @{$self->{open_elements}}; # tr
5633 $self->{insertion_mode} = IN_TABLE_BODY_IM;
5634 if ($token->{tag_name} eq 'tr') {
5635 !!!cp ('t212');
5636 ## reprocess
5637 !!!ack-later;
5638 next B;
5639 } else {
5640 !!!cp ('t213');
5641 ## reprocess in the "in table body" insertion mode...
5642 }
5643 }
5644
5645 if ($self->{insertion_mode} == IN_TABLE_BODY_IM) {
5646 ## have an element in table scope
5647 my $i;
5648 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5649 my $node = $self->{open_elements}->[$_];
5650 if ($node->[1] & TABLE_ROW_GROUP_EL) {
5651 !!!cp ('t214');
5652 $i = $_;
5653 last INSCOPE;
5654 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5655 !!!cp ('t215');
5656 last INSCOPE;
5657 }
5658 } # INSCOPE
5659 unless (defined $i) {
5660 !!!cp ('t216');
5661 ## TODO: This erorr type is wrong.
5662 !!!parse-error (type => 'unmatched end tag',
5663 text => $token->{tag_name}, token => $token);
5664 ## Ignore the token
5665 !!!nack ('t216.1');
5666 !!!next-token;
5667 next B;
5668 }
5669
5670 ## Clear back to table body context
5671 while (not ($self->{open_elements}->[-1]->[1]
5672 & TABLE_ROWS_SCOPING_EL)) {
5673 !!!cp ('t217');
5674 ## ISSUE: Can this state be reached?
5675 pop @{$self->{open_elements}};
5676 }
5677
5678 ## As if <{current node}>
5679 ## have an element in table scope
5680 ## true by definition
5681
5682 ## Clear back to table body context
5683 ## nop by definition
5684
5685 pop @{$self->{open_elements}};
5686 $self->{insertion_mode} = IN_TABLE_IM;
5687 ## reprocess in "in table" insertion mode...
5688 } else {
5689 !!!cp ('t218');
5690 }
5691
5692 if ($token->{tag_name} eq 'col') {
5693 ## Clear back to table context
5694 while (not ($self->{open_elements}->[-1]->[1]
5695 & TABLE_SCOPING_EL)) {
5696 !!!cp ('t219');
5697 ## ISSUE: Can this state be reached?
5698 pop @{$self->{open_elements}};
5699 }
5700
5701 !!!insert-element ('colgroup',, $token);
5702 $self->{insertion_mode} = IN_COLUMN_GROUP_IM;
5703 ## reprocess
5704 !!!ack-later;
5705 next B;
5706 } elsif ({
5707 caption => 1,
5708 colgroup => 1,
5709 tbody => 1, tfoot => 1, thead => 1,
5710 }->{$token->{tag_name}}) {
5711 ## Clear back to table context
5712 while (not ($self->{open_elements}->[-1]->[1]
5713 & TABLE_SCOPING_EL)) {
5714 !!!cp ('t220');
5715 ## ISSUE: Can this state be reached?
5716 pop @{$self->{open_elements}};
5717 }
5718
5719 push @$active_formatting_elements, ['#marker', '']
5720 if $token->{tag_name} eq 'caption';
5721
5722 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5723 $self->{insertion_mode} = {
5724 caption => IN_CAPTION_IM,
5725 colgroup => IN_COLUMN_GROUP_IM,
5726 tbody => IN_TABLE_BODY_IM,
5727 tfoot => IN_TABLE_BODY_IM,
5728 thead => IN_TABLE_BODY_IM,
5729 }->{$token->{tag_name}};
5730 !!!next-token;
5731 !!!nack ('t220.1');
5732 next B;
5733 } else {
5734 die "$0: in table: <>: $token->{tag_name}";
5735 }
5736 } elsif ($token->{tag_name} eq 'table') {
5737 !!!parse-error (type => 'not closed',
5738 text => $self->{open_elements}->[-1]->[0]
5739 ->manakai_local_name,
5740 token => $token);
5741
5742 ## As if </table>
5743 ## have a table element in table scope
5744 my $i;
5745 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5746 my $node = $self->{open_elements}->[$_];
5747 if ($node->[1] & TABLE_EL) {
5748 !!!cp ('t221');
5749 $i = $_;
5750 last INSCOPE;
5751 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5752 !!!cp ('t222');
5753 last INSCOPE;
5754 }
5755 } # INSCOPE
5756 unless (defined $i) {
5757 !!!cp ('t223');
5758 ## TODO: The following is wrong, maybe.
5759 !!!parse-error (type => 'unmatched end tag', text => 'table',
5760 token => $token);
5761 ## Ignore tokens </table><table>
5762 !!!nack ('t223.1');
5763 !!!next-token;
5764 next B;
5765 }
5766
5767 ## TODO: Followings are removed from the latest spec.
5768 ## generate implied end tags
5769 while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
5770 !!!cp ('t224');
5771 pop @{$self->{open_elements}};
5772 }
5773
5774 unless ($self->{open_elements}->[-1]->[1] & TABLE_EL) {
5775 !!!cp ('t225');
5776 ## NOTE: |<table><tr><table>|
5777 !!!parse-error (type => 'not closed',
5778 text => $self->{open_elements}->[-1]->[0]
5779 ->manakai_local_name,
5780 token => $token);
5781 } else {
5782 !!!cp ('t226');
5783 }
5784
5785 splice @{$self->{open_elements}}, $i;
5786 pop @{$open_tables};
5787
5788 $self->_reset_insertion_mode;
5789
5790 ## reprocess
5791 !!!ack-later;
5792 next B;
5793 } elsif ($token->{tag_name} eq 'style') {
5794 if (not $open_tables->[-1]->[1]) { # tainted
5795 !!!cp ('t227.8');
5796 ## NOTE: This is a "as if in head" code clone.
5797 $parse_rcdata->(CDATA_CONTENT_MODEL);
5798 next B;
5799 } else {
5800 !!!cp ('t227.7');
5801 #
5802 }
5803 } elsif ($token->{tag_name} eq 'script') {
5804 if (not $open_tables->[-1]->[1]) { # tainted
5805 !!!cp ('t227.6');
5806 ## NOTE: This is a "as if in head" code clone.
5807 $script_start_tag->();
5808 next B;
5809 } else {
5810 !!!cp ('t227.5');
5811 #
5812 }
5813 } elsif ($token->{tag_name} eq 'input') {
5814 if (not $open_tables->[-1]->[1]) { # tainted
5815 if ($token->{attributes}->{type}) { ## TODO: case
5816 my $type = lc $token->{attributes}->{type}->{value};
5817 if ($type eq 'hidden') {
5818 !!!cp ('t227.3');
5819 !!!parse-error (type => 'in table',
5820 text => $token->{tag_name}, token => $token);
5821
5822 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5823
5824 ## TODO: form element pointer
5825
5826 pop @{$self->{open_elements}};
5827
5828 !!!next-token;
5829 !!!ack ('t227.2.1');
5830 next B;
5831 } else {
5832 !!!cp ('t227.2');
5833 #
5834 }
5835 } else {
5836 !!!cp ('t227.1');
5837 #
5838 }
5839 } else {
5840 !!!cp ('t227.4');
5841 #
5842 }
5843 } else {
5844 !!!cp ('t227');
5845 #
5846 }
5847
5848 !!!parse-error (type => 'in table', text => $token->{tag_name},
5849 token => $token);
5850
5851 $insert = $insert_to_foster;
5852 #
5853 } elsif ($token->{type} == END_TAG_TOKEN) {
5854 if ($token->{tag_name} eq 'tr' and
5855 $self->{insertion_mode} == IN_ROW_IM) {
5856 ## have an element in table scope
5857 my $i;
5858 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5859 my $node = $self->{open_elements}->[$_];
5860 if ($node->[1] & TABLE_ROW_EL) {
5861 !!!cp ('t228');
5862 $i = $_;
5863 last INSCOPE;
5864 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5865 !!!cp ('t229');
5866 last INSCOPE;
5867 }
5868 } # INSCOPE
5869 unless (defined $i) {
5870 !!!cp ('t230');
5871 !!!parse-error (type => 'unmatched end tag',
5872 text => $token->{tag_name}, token => $token);
5873 ## Ignore the token
5874 !!!nack ('t230.1');
5875 !!!next-token;
5876 next B;
5877 } else {
5878 !!!cp ('t232');
5879 }
5880
5881 ## Clear back to table row context
5882 while (not ($self->{open_elements}->[-1]->[1]
5883 & TABLE_ROW_SCOPING_EL)) {
5884 !!!cp ('t231');
5885 ## ISSUE: Can this state be reached?
5886 pop @{$self->{open_elements}};
5887 }
5888
5889 pop @{$self->{open_elements}}; # tr
5890 $self->{insertion_mode} = IN_TABLE_BODY_IM;
5891 !!!next-token;
5892 !!!nack ('t231.1');
5893 next B;
5894 } elsif ($token->{tag_name} eq 'table') {
5895 if ($self->{insertion_mode} == IN_ROW_IM) {
5896 ## As if </tr>
5897 ## have an element in table scope
5898 my $i;
5899 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5900 my $node = $self->{open_elements}->[$_];
5901 if ($node->[1] & TABLE_ROW_EL) {
5902 !!!cp ('t233');
5903 $i = $_;
5904 last INSCOPE;
5905 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5906 !!!cp ('t234');
5907 last INSCOPE;
5908 }
5909 } # INSCOPE
5910 unless (defined $i) {
5911 !!!cp ('t235');
5912 ## TODO: The following is wrong.
5913 !!!parse-error (type => 'unmatched end tag',
5914 text => $token->{type}, token => $token);
5915 ## Ignore the token
5916 !!!nack ('t236.1');
5917 !!!next-token;
5918 next B;
5919 }
5920
5921 ## Clear back to table row context
5922 while (not ($self->{open_elements}->[-1]->[1]
5923 & TABLE_ROW_SCOPING_EL)) {
5924 !!!cp ('t236');
5925 ## ISSUE: Can this state be reached?
5926 pop @{$self->{open_elements}};
5927 }
5928
5929 pop @{$self->{open_elements}}; # tr
5930 $self->{insertion_mode} = IN_TABLE_BODY_IM;
5931 ## reprocess in the "in table body" insertion mode...
5932 }
5933
5934 if ($self->{insertion_mode} == IN_TABLE_BODY_IM) {
5935 ## have an element in table scope
5936 my $i;
5937 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5938 my $node = $self->{open_elements}->[$_];
5939 if ($node->[1] & TABLE_ROW_GROUP_EL) {
5940 !!!cp ('t237');
5941 $i = $_;
5942 last INSCOPE;
5943 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5944 !!!cp ('t238');
5945 last INSCOPE;
5946 }
5947 } # INSCOPE
5948 unless (defined $i) {
5949 !!!cp ('t239');
5950 !!!parse-error (type => 'unmatched end tag',
5951 text => $token->{tag_name}, token => $token);
5952 ## Ignore the token
5953 !!!nack ('t239.1');
5954 !!!next-token;
5955 next B;
5956 }
5957
5958 ## Clear back to table body context
5959 while (not ($self->{open_elements}->[-1]->[1]
5960 & TABLE_ROWS_SCOPING_EL)) {
5961 !!!cp ('t240');
5962 pop @{$self->{open_elements}};
5963 }
5964
5965 ## As if <{current node}>
5966 ## have an element in table scope
5967 ## true by definition
5968
5969 ## Clear back to table body context
5970 ## nop by definition
5971
5972 pop @{$self->{open_elements}};
5973 $self->{insertion_mode} = IN_TABLE_IM;
5974 ## reprocess in the "in table" insertion mode...
5975 }
5976
5977 ## NOTE: </table> in the "in table" insertion mode.
5978 ## When you edit the code fragment below, please ensure that
5979 ## the code for <table> in the "in table" insertion mode
5980 ## is synced with it.
5981
5982 ## have a table element in table scope
5983 my $i;
5984 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5985 my $node = $self->{open_elements}->[$_];
5986 if ($node->[1] & TABLE_EL) {
5987 !!!cp ('t241');
5988 $i = $_;
5989 last INSCOPE;
5990 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5991 !!!cp ('t242');
5992 last INSCOPE;
5993 }
5994 } # INSCOPE
5995 unless (defined $i) {
5996 !!!cp ('t243');
5997 !!!parse-error (type => 'unmatched end tag',
5998 text => $token->{tag_name}, token => $token);
5999 ## Ignore the token
6000 !!!nack ('t243.1');
6001 !!!next-token;
6002 next B;
6003 }
6004
6005 splice @{$self->{open_elements}}, $i;
6006 pop @{$open_tables};
6007
6008 $self->_reset_insertion_mode;
6009
6010 !!!next-token;
6011 next B;
6012 } elsif ({
6013 tbody => 1, tfoot => 1, thead => 1,
6014 }->{$token->{tag_name}} and
6015 $self->{insertion_mode} & ROW_IMS) {
6016 if ($self->{insertion_mode} == IN_ROW_IM) {
6017 ## have an element in table scope
6018 my $i;
6019 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6020 my $node = $self->{open_elements}->[$_];
6021 if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
6022 !!!cp ('t247');
6023 $i = $_;
6024 last INSCOPE;
6025 } elsif ($node->[1] & TABLE_SCOPING_EL) {
6026 !!!cp ('t248');
6027 last INSCOPE;
6028 }
6029 } # INSCOPE
6030 unless (defined $i) {
6031 !!!cp ('t249');
6032 !!!parse-error (type => 'unmatched end tag',
6033 text => $token->{tag_name}, token => $token);
6034 ## Ignore the token
6035 !!!nack ('t249.1');
6036 !!!next-token;
6037 next B;
6038 }
6039
6040 ## As if </tr>
6041 ## have an element in table scope
6042 my $i;
6043 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6044 my $node = $self->{open_elements}->[$_];
6045 if ($node->[1] & TABLE_ROW_EL) {
6046 !!!cp ('t250');
6047 $i = $_;
6048 last INSCOPE;
6049 } elsif ($node->[1] & TABLE_SCOPING_EL) {
6050 !!!cp ('t251');
6051 last INSCOPE;
6052 }
6053 } # INSCOPE
6054 unless (defined $i) {
6055 !!!cp ('t252');
6056 !!!parse-error (type => 'unmatched end tag',
6057 text => 'tr', token => $token);
6058 ## Ignore the token
6059 !!!nack ('t252.1');
6060 !!!next-token;
6061 next B;
6062 }
6063
6064 ## Clear back to table row context
6065 while (not ($self->{open_elements}->[-1]->[1]
6066 & TABLE_ROW_SCOPING_EL)) {
6067 !!!cp ('t253');
6068 ## ISSUE: Can this case be reached?
6069 pop @{$self->{open_elements}};
6070 }
6071
6072 pop @{$self->{open_elements}}; # tr
6073 $self->{insertion_mode} = IN_TABLE_BODY_IM;
6074 ## reprocess in the "in table body" insertion mode...
6075 }
6076
6077 ## have an element in table scope
6078 my $i;
6079 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6080 my $node = $self->{open_elements}->[$_];
6081 if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
6082 !!!cp ('t254');
6083 $i = $_;
6084 last INSCOPE;
6085 } elsif ($node->[1] & TABLE_SCOPING_EL) {
6086 !!!cp ('t255');
6087 last INSCOPE;
6088 }
6089 } # INSCOPE
6090 unless (defined $i) {
6091 !!!cp ('t256');
6092 !!!parse-error (type => 'unmatched end tag',
6093 text => $token->{tag_name}, token => $token);
6094 ## Ignore the token
6095 !!!nack ('t256.1');
6096 !!!next-token;
6097 next B;
6098 }
6099
6100 ## Clear back to table body context
6101 while (not ($self->{open_elements}->[-1]->[1]
6102 & TABLE_ROWS_SCOPING_EL)) {
6103 !!!cp ('t257');
6104 ## ISSUE: Can this case be reached?
6105 pop @{$self->{open_elements}};
6106 }
6107
6108 pop @{$self->{open_elements}};
6109 $self->{insertion_mode} = IN_TABLE_IM;
6110 !!!nack ('t257.1');
6111 !!!next-token;
6112 next B;
6113 } elsif ({
6114 body => 1, caption => 1, col => 1, colgroup => 1,
6115 html => 1, td => 1, th => 1,
6116 tr => 1, # $self->{insertion_mode} == IN_ROW_IM
6117 tbody => 1, tfoot => 1, thead => 1, # $self->{insertion_mode} == IN_TABLE_IM
6118 }->{$token->{tag_name}}) {
6119 !!!cp ('t258');
6120 !!!parse-error (type => 'unmatched end tag',
6121 text => $token->{tag_name}, token => $token);
6122 ## Ignore the token
6123 !!!nack ('t258.1');
6124 !!!next-token;
6125 next B;
6126 } else {
6127 !!!cp ('t259');
6128 !!!parse-error (type => 'in table:/',
6129 text => $token->{tag_name}, token => $token);
6130
6131 $insert = $insert_to_foster;
6132 #
6133 }
6134 } elsif ($token->{type} == END_OF_FILE_TOKEN) {
6135 unless ($self->{open_elements}->[-1]->[1] & HTML_EL and
6136 @{$self->{open_elements}} == 1) { # redundant, maybe
6137 !!!parse-error (type => 'in body:#eof', token => $token);
6138 !!!cp ('t259.1');
6139 #
6140 } else {
6141 !!!cp ('t259.2');
6142 #
6143 }
6144
6145 ## Stop parsing
6146 last B;
6147 } else {
6148 die "$0: $token->{type}: Unknown token type";
6149 }
6150 } elsif ($self->{insertion_mode} == IN_COLUMN_GROUP_IM) {
6151 if ($token->{type} == CHARACTER_TOKEN) {
6152 if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {
6153 $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
6154 unless (length $token->{data}) {
6155 !!!cp ('t260');
6156 !!!next-token;
6157 next B;
6158 }
6159 }
6160
6161 !!!cp ('t261');
6162 #
6163 } elsif ($token->{type} == START_TAG_TOKEN) {
6164 if ($token->{tag_name} eq 'col') {
6165 !!!cp ('t262');
6166 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
6167 pop @{$self->{open_elements}};
6168 !!!ack ('t262.1');
6169 !!!next-token;
6170 next B;
6171 } else {
6172 !!!cp ('t263');
6173 #
6174 }
6175 } elsif ($token->{type} == END_TAG_TOKEN) {
6176 if ($token->{tag_name} eq 'colgroup') {
6177 if ($self->{open_elements}->[-1]->[1] & HTML_EL) {
6178 !!!cp ('t264');
6179 !!!parse-error (type => 'unmatched end tag',
6180 text => 'colgroup', token => $token);
6181 ## Ignore the token
6182 !!!next-token;
6183 next B;
6184 } else {
6185 !!!cp ('t265');
6186 pop @{$self->{open_elements}}; # colgroup
6187 $self->{insertion_mode} = IN_TABLE_IM;
6188 !!!next-token;
6189 next B;
6190 }
6191 } elsif ($token->{tag_name} eq 'col') {
6192 !!!cp ('t266');
6193 !!!parse-error (type => 'unmatched end tag',
6194 text => 'col', token => $token);
6195 ## Ignore the token
6196 !!!next-token;
6197 next B;
6198 } else {
6199 !!!cp ('t267');
6200 #
6201 }
6202 } elsif ($token->{type} == END_OF_FILE_TOKEN) {
6203 if ($self->{open_elements}->[-1]->[1] & HTML_EL and
6204 @{$self->{open_elements}} == 1) { # redundant, maybe
6205 !!!cp ('t270.2');
6206 ## Stop parsing.
6207 last B;
6208 } else {
6209 ## NOTE: As if </colgroup>.
6210 !!!cp ('t270.1');
6211 pop @{$self->{open_elements}}; # colgroup
6212 $self->{insertion_mode} = IN_TABLE_IM;
6213 ## Reprocess.
6214 next B;
6215 }
6216 } else {
6217 die "$0: $token->{type}: Unknown token type";
6218 }
6219
6220 ## As if </colgroup>
6221 if ($self->{open_elements}->[-1]->[1] & HTML_EL) {
6222 !!!cp ('t269');
6223 ## TODO: Wrong error type?
6224 !!!parse-error (type => 'unmatched end tag',
6225 text => 'colgroup', token => $token);
6226 ## Ignore the token
6227 !!!nack ('t269.1');
6228 !!!next-token;
6229 next B;
6230 } else {
6231 !!!cp ('t270');
6232 pop @{$self->{open_elements}}; # colgroup
6233 $self->{insertion_mode} = IN_TABLE_IM;
6234 !!!ack-later;
6235 ## reprocess
6236 next B;
6237 }
6238 } elsif ($self->{insertion_mode} & SELECT_IMS) {
6239 if ($token->{type} == CHARACTER_TOKEN) {
6240 !!!cp ('t271');
6241 $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});
6242 !!!next-token;
6243 next B;
6244 } elsif ($token->{type} == START_TAG_TOKEN) {
6245 if ($token->{tag_name} eq 'option') {
6246 if ($self->{open_elements}->[-1]->[1] & OPTION_EL) {
6247 !!!cp ('t272');
6248 ## As if </option>
6249 pop @{$self->{open_elements}};
6250 } else {
6251 !!!cp ('t273');
6252 }
6253
6254 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
6255 !!!nack ('t273.1');
6256 !!!next-token;
6257 next B;
6258 } elsif ($token->{tag_name} eq 'optgroup') {
6259 if ($self->{open_elements}->[-1]->[1] & OPTION_EL) {
6260 !!!cp ('t274');
6261 ## As if </option>
6262 pop @{$self->{open_elements}};
6263 } else {
6264 !!!cp ('t275');
6265 }
6266
6267 if ($self->{open_elements}->[-1]->[1] & OPTGROUP_EL) {
6268 !!!cp ('t276');
6269 ## As if </optgroup>
6270 pop @{$self->{open_elements}};
6271 } else {
6272 !!!cp ('t277');
6273 }
6274
6275 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
6276 !!!nack ('t277.1');
6277 !!!next-token;
6278 next B;
6279 } elsif ({
6280 select => 1, input => 1, textarea => 1,
6281 }->{$token->{tag_name}} or
6282 ($self->{insertion_mode} == IN_SELECT_IN_TABLE_IM and
6283 {
6284 caption => 1, table => 1,
6285 tbody => 1, tfoot => 1, thead => 1,
6286 tr => 1, td => 1, th => 1,
6287 }->{$token->{tag_name}})) {
6288 ## TODO: The type below is not good - <select> is replaced by </select>
6289 !!!parse-error (type => 'not closed', text => 'select',
6290 token => $token);
6291 ## NOTE: As if the token were </select> (<select> case) or
6292 ## as if there were </select> (otherwise).
6293 ## have an element in table scope
6294 my $i;
6295 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6296 my $node = $self->{open_elements}->[$_];
6297 if ($node->[1] & SELECT_EL) {
6298 !!!cp ('t278');
6299 $i = $_;
6300 last INSCOPE;
6301 } elsif ($node->[1] & TABLE_SCOPING_EL) {
6302 !!!cp ('t279');
6303 last INSCOPE;
6304 }
6305 } # INSCOPE
6306 unless (defined $i) {
6307 !!!cp ('t280');
6308 !!!parse-error (type => 'unmatched end tag',
6309 text => 'select', token => $token);
6310 ## Ignore the token
6311 !!!nack ('t280.1');
6312 !!!next-token;
6313 next B;
6314 }
6315
6316 !!!cp ('t281');
6317 splice @{$self->{open_elements}}, $i;
6318
6319 $self->_reset_insertion_mode;
6320
6321 if ($token->{tag_name} eq 'select') {
6322 !!!nack ('t281.2');
6323 !!!next-token;
6324 next B;
6325 } else {
6326 !!!cp ('t281.1');
6327 !!!ack-later;
6328 ## Reprocess the token.
6329 next B;
6330 }
6331 } else {
6332 !!!cp ('t282');
6333 !!!parse-error (type => 'in select',
6334 text => $token->{tag_name}, token => $token);
6335 ## Ignore the token
6336 !!!nack ('t282.1');
6337 !!!next-token;
6338 next B;
6339 }
6340 } elsif ($token->{type} == END_TAG_TOKEN) {
6341 if ($token->{tag_name} eq 'optgroup') {
6342 if ($self->{open_elements}->[-1]->[1] & OPTION_EL and
6343 $self->{open_elements}->[-2]->[1] & OPTGROUP_EL) {
6344 !!!cp ('t283');
6345 ## As if </option>
6346 splice @{$self->{open_elements}}, -2;
6347 } elsif ($self->{open_elements}->[-1]->[1] & OPTGROUP_EL) {
6348 !!!cp ('t284');
6349 pop @{$self->{open_elements}};
6350 } else {
6351 !!!cp ('t285');
6352 !!!parse-error (type => 'unmatched end tag',
6353 text => $token->{tag_name}, token => $token);
6354 ## Ignore the token
6355 }
6356 !!!nack ('t285.1');
6357 !!!next-token;
6358 next B;
6359 } elsif ($token->{tag_name} eq 'option') {
6360 if ($self->{open_elements}->[-1]->[1] & OPTION_EL) {
6361 !!!cp ('t286');
6362 pop @{$self->{open_elements}};
6363 } else {
6364 !!!cp ('t287');
6365 !!!parse-error (type => 'unmatched end tag',
6366 text => $token->{tag_name}, token => $token);
6367 ## Ignore the token
6368 }
6369 !!!nack ('t287.1');
6370 !!!next-token;
6371 next B;
6372 } elsif ($token->{tag_name} eq 'select') {
6373 ## have an element in table scope
6374 my $i;
6375 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6376 my $node = $self->{open_elements}->[$_];
6377 if ($node->[1] & SELECT_EL) {
6378 !!!cp ('t288');
6379 $i = $_;
6380 last INSCOPE;
6381 } elsif ($node->[1] & TABLE_SCOPING_EL) {
6382 !!!cp ('t289');
6383 last INSCOPE;
6384 }
6385 } # INSCOPE
6386 unless (defined $i) {
6387 !!!cp ('t290');
6388 !!!parse-error (type => 'unmatched end tag',
6389 text => $token->{tag_name}, token => $token);
6390 ## Ignore the token
6391 !!!nack ('t290.1');
6392 !!!next-token;
6393 next B;
6394 }
6395
6396 !!!cp ('t291');
6397 splice @{$self->{open_elements}}, $i;
6398
6399 $self->_reset_insertion_mode;
6400
6401 !!!nack ('t291.1');
6402 !!!next-token;
6403 next B;
6404 } elsif ($self->{insertion_mode} == IN_SELECT_IN_TABLE_IM and
6405 {
6406 caption => 1, table => 1, tbody => 1,
6407 tfoot => 1, thead => 1, tr => 1, td => 1, th => 1,
6408 }->{$token->{tag_name}}) {
6409 ## TODO: The following is wrong?
6410 !!!parse-error (type => 'unmatched end tag',
6411 text => $token->{tag_name}, token => $token);
6412
6413 ## have an element in table scope
6414 my $i;
6415 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6416 my $node = $self->{open_elements}->[$_];
6417 if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
6418 !!!cp ('t292');
6419 $i = $_;
6420 last INSCOPE;
6421 } elsif ($node->[1] & TABLE_SCOPING_EL) {
6422 !!!cp ('t293');
6423 last INSCOPE;
6424 }
6425 } # INSCOPE
6426 unless (defined $i) {
6427 !!!cp ('t294');
6428 ## Ignore the token
6429 !!!nack ('t294.1');
6430 !!!next-token;
6431 next B;
6432 }
6433
6434 ## As if </select>
6435 ## have an element in table scope
6436 undef $i;
6437 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6438 my $node = $self->{open_elements}->[$_];
6439 if ($node->[1] & SELECT_EL) {
6440 !!!cp ('t295');
6441 $i = $_;
6442 last INSCOPE;
6443 } elsif ($node->[1] & TABLE_SCOPING_EL) {
6444 ## ISSUE: Can this state be reached?
6445 !!!cp ('t296');
6446 last INSCOPE;
6447 }
6448 } # INSCOPE
6449 unless (defined $i) {
6450 !!!cp ('t297');
6451 ## TODO: The following error type is correct?
6452 !!!parse-error (type => 'unmatched end tag',
6453 text => 'select', token => $token);
6454 ## Ignore the </select> token
6455 !!!nack ('t297.1');
6456 !!!next-token; ## TODO: ok?
6457 next B;
6458 }
6459
6460 !!!cp ('t298');
6461 splice @{$self->{open_elements}}, $i;
6462
6463 $self->_reset_insertion_mode;
6464
6465 !!!ack-later;
6466 ## reprocess
6467 next B;
6468 } else {
6469 !!!cp ('t299');
6470 !!!parse-error (type => 'in select:/',
6471 text => $token->{tag_name}, token => $token);
6472 ## Ignore the token
6473 !!!nack ('t299.3');
6474 !!!next-token;
6475 next B;
6476 }
6477 } elsif ($token->{type} == END_OF_FILE_TOKEN) {
6478 unless ($self->{open_elements}->[-1]->[1] & HTML_EL and
6479 @{$self->{open_elements}} == 1) { # redundant, maybe
6480 !!!cp ('t299.1');
6481 !!!parse-error (type => 'in body:#eof', token => $token);
6482 } else {
6483 !!!cp ('t299.2');
6484 }
6485
6486 ## Stop parsing.
6487 last B;
6488 } else {
6489 die "$0: $token->{type}: Unknown token type";
6490 }
6491 } elsif ($self->{insertion_mode} & BODY_AFTER_IMS) {
6492 if ($token->{type} == CHARACTER_TOKEN) {
6493 if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {
6494 my $data = $1;
6495 ## As if in body
6496 $reconstruct_active_formatting_elements->($insert_to_current);
6497
6498 $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
6499
6500 unless (length $token->{data}) {
6501 !!!cp ('t300');
6502 !!!next-token;
6503 next B;
6504 }
6505 }
6506
6507 if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
6508 !!!cp ('t301');
6509 !!!parse-error (type => 'after html:#text', token => $token);
6510
6511 ## Reprocess in the "after body" insertion mode.
6512 } else {
6513 !!!cp ('t302');
6514 }
6515
6516 ## "after body" insertion mode
6517 !!!parse-error (type => 'after body:#text', token => $token);
6518
6519 $self->{insertion_mode} = IN_BODY_IM;
6520 ## reprocess
6521 next B;
6522 } elsif ($token->{type} == START_TAG_TOKEN) {
6523 if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
6524 !!!cp ('t303');
6525 !!!parse-error (type => 'after html',
6526 text => $token->{tag_name}, token => $token);
6527
6528 ## Reprocess in the "after body" insertion mode.
6529 } else {
6530 !!!cp ('t304');
6531 }
6532
6533 ## "after body" insertion mode
6534 !!!parse-error (type => 'after body',
6535 text => $token->{tag_name}, token => $token);
6536
6537 $self->{insertion_mode} = IN_BODY_IM;
6538 !!!ack-later;
6539 ## reprocess
6540 next B;
6541 } elsif ($token->{type} == END_TAG_TOKEN) {
6542 if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
6543 !!!cp ('t305');
6544 !!!parse-error (type => 'after html:/',
6545 text => $token->{tag_name}, token => $token);
6546
6547 $self->{insertion_mode} = AFTER_BODY_IM;
6548 ## Reprocess in the "after body" insertion mode.
6549 } else {
6550 !!!cp ('t306');
6551 }
6552
6553 ## "after body" insertion mode
6554 if ($token->{tag_name} eq 'html') {
6555 if (defined $self->{inner_html_node}) {
6556 !!!cp ('t307');
6557 !!!parse-error (type => 'unmatched end tag',
6558 text => 'html', token => $token);
6559 ## Ignore the token
6560 !!!next-token;
6561 next B;
6562 } else {
6563 !!!cp ('t308');
6564 $self->{insertion_mode} = AFTER_HTML_BODY_IM;
6565 !!!next-token;
6566 next B;
6567 }
6568 } else {
6569 !!!cp ('t309');
6570 !!!parse-error (type => 'after body:/',
6571 text => $token->{tag_name}, token => $token);
6572
6573 $self->{insertion_mode} = IN_BODY_IM;
6574 ## reprocess
6575 next B;
6576 }
6577 } elsif ($token->{type} == END_OF_FILE_TOKEN) {
6578 !!!cp ('t309.2');
6579 ## Stop parsing
6580 last B;
6581 } else {
6582 die "$0: $token->{type}: Unknown token type";
6583 }
6584 } elsif ($self->{insertion_mode} & FRAME_IMS) {
6585 if ($token->{type} == CHARACTER_TOKEN) {
6586 if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {
6587 $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
6588
6589 unless (length $token->{data}) {
6590 !!!cp ('t310');
6591 !!!next-token;
6592 next B;
6593 }
6594 }
6595
6596 if ($token->{data} =~ s/^[^\x09\x0A\x0B\x0C\x20]+//) {
6597 if ($self->{insertion_mode} == IN_FRAMESET_IM) {
6598 !!!cp ('t311');
6599 !!!parse-error (type => 'in frameset:#text', token => $token);
6600 } elsif ($self->{insertion_mode} == AFTER_FRAMESET_IM) {
6601 !!!cp ('t312');
6602 !!!parse-error (type => 'after frameset:#text', token => $token);
6603 } else { # "after after frameset"
6604 !!!cp ('t313');
6605 !!!parse-error (type => 'after html:#text', token => $token);
6606 }
6607
6608 ## Ignore the token.
6609 if (length $token->{data}) {
6610 !!!cp ('t314');
6611 ## reprocess the rest of characters
6612 } else {
6613 !!!cp ('t315');
6614 !!!next-token;
6615 }
6616 next B;
6617 }
6618
6619 die qq[$0: Character "$token->{data}"];
6620 } elsif ($token->{type} == START_TAG_TOKEN) {
6621 if ($token->{tag_name} eq 'frameset' and
6622 $self->{insertion_mode} == IN_FRAMESET_IM) {
6623 !!!cp ('t318');
6624 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
6625 !!!nack ('t318.1');
6626 !!!next-token;
6627 next B;
6628 } elsif ($token->{tag_name} eq 'frame' and
6629 $self->{insertion_mode} == IN_FRAMESET_IM) {
6630 !!!cp ('t319');
6631 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
6632 pop @{$self->{open_elements}};
6633 !!!ack ('t319.1');
6634 !!!next-token;
6635 next B;
6636 } elsif ($token->{tag_name} eq 'noframes') {
6637 !!!cp ('t320');
6638 ## NOTE: As if in head.
6639 $parse_rcdata->(CDATA_CONTENT_MODEL);
6640 next B;
6641
6642 ## NOTE: |<!DOCTYPE HTML><frameset></frameset></html><noframes></noframes>|
6643 ## has no parse error.
6644 } else {
6645 if ($self->{insertion_mode} == IN_FRAMESET_IM) {
6646 !!!cp ('t321');
6647 !!!parse-error (type => 'in frameset',
6648 text => $token->{tag_name}, token => $token);
6649 } elsif ($self->{insertion_mode} == AFTER_FRAMESET_IM) {
6650 !!!cp ('t322');
6651 !!!parse-error (type => 'after frameset',
6652 text => $token->{tag_name}, token => $token);
6653 } else { # "after after frameset"
6654 !!!cp ('t322.2');
6655 !!!parse-error (type => 'after after frameset',
6656 text => $token->{tag_name}, token => $token);
6657 }
6658 ## Ignore the token
6659 !!!nack ('t322.1');
6660 !!!next-token;
6661 next B;
6662 }
6663 } elsif ($token->{type} == END_TAG_TOKEN) {
6664 if ($token->{tag_name} eq 'frameset' and
6665 $self->{insertion_mode} == IN_FRAMESET_IM) {
6666 if ($self->{open_elements}->[-1]->[1] & HTML_EL and
6667 @{$self->{open_elements}} == 1) {
6668 !!!cp ('t325');
6669 !!!parse-error (type => 'unmatched end tag',
6670 text => $token->{tag_name}, token => $token);
6671 ## Ignore the token
6672 !!!next-token;
6673 } else {
6674 !!!cp ('t326');
6675 pop @{$self->{open_elements}};
6676 !!!next-token;
6677 }
6678
6679 if (not defined $self->{inner_html_node} and
6680 not ($self->{open_elements}->[-1]->[1] & FRAMESET_EL)) {
6681 !!!cp ('t327');
6682 $self->{insertion_mode} = AFTER_FRAMESET_IM;
6683 } else {
6684 !!!cp ('t328');
6685 }
6686 next B;
6687 } elsif ($token->{tag_name} eq 'html' and
6688 $self->{insertion_mode} == AFTER_FRAMESET_IM) {
6689 !!!cp ('t329');
6690 $self->{insertion_mode} = AFTER_HTML_FRAMESET_IM;
6691 !!!next-token;
6692 next B;
6693 } else {
6694 if ($self->{insertion_mode} == IN_FRAMESET_IM) {
6695 !!!cp ('t330');
6696 !!!parse-error (type => 'in frameset:/',
6697 text => $token->{tag_name}, token => $token);
6698 } elsif ($self->{insertion_mode} == AFTER_FRAMESET_IM) {
6699 !!!cp ('t330.1');
6700 !!!parse-error (type => 'after frameset:/',
6701 text => $token->{tag_name}, token => $token);
6702 } else { # "after after html"
6703 !!!cp ('t331');
6704 !!!parse-error (type => 'after after frameset:/',
6705 text => $token->{tag_name}, token => $token);
6706 }
6707 ## Ignore the token
6708 !!!next-token;
6709 next B;
6710 }
6711 } elsif ($token->{type} == END_OF_FILE_TOKEN) {
6712 unless ($self->{open_elements}->[-1]->[1] & HTML_EL and
6713 @{$self->{open_elements}} == 1) { # redundant, maybe
6714 !!!cp ('t331.1');
6715 !!!parse-error (type => 'in body:#eof', token => $token);
6716 } else {
6717 !!!cp ('t331.2');
6718 }
6719
6720 ## Stop parsing
6721 last B;
6722 } else {
6723 die "$0: $token->{type}: Unknown token type";
6724 }
6725
6726 ## ISSUE: An issue in spec here
6727 } else {
6728 die "$0: $self->{insertion_mode}: Unknown insertion mode";
6729 }
6730
6731 ## "in body" insertion mode
6732 if ($token->{type} == START_TAG_TOKEN) {
6733 if ($token->{tag_name} eq 'script') {
6734 !!!cp ('t332');
6735 ## NOTE: This is an "as if in head" code clone
6736 $script_start_tag->();
6737 next B;
6738 } elsif ($token->{tag_name} eq 'style') {
6739 !!!cp ('t333');
6740 ## NOTE: This is an "as if in head" code clone
6741 $parse_rcdata->(CDATA_CONTENT_MODEL);
6742 next B;
6743 } elsif ({
6744 base => 1, link => 1,
6745 }->{$token->{tag_name}}) {
6746 !!!cp ('t334');
6747 ## NOTE: This is an "as if in head" code clone, only "-t" differs
6748 !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
6749 pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.
6750 !!!ack ('t334.1');
6751 !!!next-token;
6752 next B;
6753 } elsif ($token->{tag_name} eq 'meta') {
6754 ## NOTE: This is an "as if in head" code clone, only "-t" differs
6755 !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
6756 my $meta_el = pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.
6757
6758 unless ($self->{confident}) {
6759 if ($token->{attributes}->{charset}) {
6760 !!!cp ('t335');
6761 ## NOTE: Whether the encoding is supported or not is handled
6762 ## in the {change_encoding} callback.
6763 $self->{change_encoding}
6764 ->($self, $token->{attributes}->{charset}->{value}, $token);
6765
6766 $meta_el->[0]->get_attribute_node_ns (undef, 'charset')
6767 ->set_user_data (manakai_has_reference =>
6768 $token->{attributes}->{charset}
6769 ->{has_reference});
6770 } elsif ($token->{attributes}->{content}) {
6771 if ($token->{attributes}->{content}->{value}
6772 =~ /[Cc][Hh][Aa][Rr][Ss][Ee][Tt]
6773 [\x09-\x0D\x20]*=
6774 [\x09-\x0D\x20]*(?>"([^"]*)"|'([^']*)'|
6775 ([^"'\x09-\x0D\x20][^\x09-\x0D\x20\x3B]*))/x) {
6776 !!!cp ('t336');
6777 ## NOTE: Whether the encoding is supported or not is handled
6778 ## in the {change_encoding} callback.
6779 $self->{change_encoding}
6780 ->($self, defined $1 ? $1 : defined $2 ? $2 : $3, $token);
6781 $meta_el->[0]->get_attribute_node_ns (undef, 'content')
6782 ->set_user_data (manakai_has_reference =>
6783 $token->{attributes}->{content}
6784 ->{has_reference});
6785 }
6786 }
6787 } else {
6788 if ($token->{attributes}->{charset}) {
6789 !!!cp ('t337');
6790 $meta_el->[0]->get_attribute_node_ns (undef, 'charset')
6791 ->set_user_data (manakai_has_reference =>
6792 $token->{attributes}->{charset}
6793 ->{has_reference});
6794 }
6795 if ($token->{attributes}->{content}) {
6796 !!!cp ('t338');
6797 $meta_el->[0]->get_attribute_node_ns (undef, 'content')
6798 ->set_user_data (manakai_has_reference =>
6799 $token->{attributes}->{content}
6800 ->{has_reference});
6801 }
6802 }
6803
6804 !!!ack ('t338.1');
6805 !!!next-token;
6806 next B;
6807 } elsif ($token->{tag_name} eq 'title') {
6808 !!!cp ('t341');
6809 ## NOTE: This is an "as if in head" code clone
6810 $parse_rcdata->(RCDATA_CONTENT_MODEL);
6811 next B;
6812 } elsif ($token->{tag_name} eq 'body') {
6813 !!!parse-error (type => 'in body', text => 'body', token => $token);
6814
6815 if (@{$self->{open_elements}} == 1 or
6816 not ($self->{open_elements}->[1]->[1] & BODY_EL)) {
6817 !!!cp ('t342');
6818 ## Ignore the token
6819 } else {
6820 my $body_el = $self->{open_elements}->[1]->[0];
6821 for my $attr_name (keys %{$token->{attributes}}) {
6822 unless ($body_el->has_attribute_ns (undef, $attr_name)) {
6823 !!!cp ('t343');
6824 $body_el->set_attribute_ns
6825 (undef, [undef, $attr_name],
6826 $token->{attributes}->{$attr_name}->{value});
6827 }
6828 }
6829 }
6830 !!!nack ('t343.1');
6831 !!!next-token;
6832 next B;
6833 } elsif ({
6834 address => 1, blockquote => 1, center => 1, dir => 1,
6835 div => 1, dl => 1, fieldset => 1,
6836 h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,
6837 menu => 1, ol => 1, p => 1, ul => 1,
6838 pre => 1, listing => 1,
6839 form => 1,
6840 table => 1,
6841 hr => 1,
6842 }->{$token->{tag_name}}) {
6843 if ($token->{tag_name} eq 'form' and defined $self->{form_element}) {
6844 !!!cp ('t350');
6845 !!!parse-error (type => 'in form:form', token => $token);
6846 ## Ignore the token
6847 !!!nack ('t350.1');
6848 !!!next-token;
6849 next B;
6850 }
6851
6852 ## has a p element in scope
6853 INSCOPE: for (reverse @{$self->{open_elements}}) {
6854 if ($_->[1] & P_EL) {
6855 !!!cp ('t344');
6856 !!!back-token; # <form>
6857 $token = {type => END_TAG_TOKEN, tag_name => 'p',
6858 line => $token->{line}, column => $token->{column}};
6859 next B;
6860 } elsif ($_->[1] & SCOPING_EL) {
6861 !!!cp ('t345');
6862 last INSCOPE;
6863 }
6864 } # INSCOPE
6865
6866 !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
6867 if ($token->{tag_name} eq 'pre' or $token->{tag_name} eq 'listing') {
6868 !!!nack ('t346.1');
6869 !!!next-token;
6870 if ($token->{type} == CHARACTER_TOKEN) {
6871 $token->{data} =~ s/^\x0A//;
6872 unless (length $token->{data}) {
6873 !!!cp ('t346');
6874 !!!next-token;
6875 } else {
6876 !!!cp ('t349');
6877 }
6878 } else {
6879 !!!cp ('t348');
6880 }
6881 } elsif ($token->{tag_name} eq 'form') {
6882 !!!cp ('t347.1');
6883 $self->{form_element} = $self->{open_elements}->[-1]->[0];
6884
6885 !!!nack ('t347.2');
6886 !!!next-token;
6887 } elsif ($token->{tag_name} eq 'table') {
6888 !!!cp ('t382');
6889 push @{$open_tables}, [$self->{open_elements}->[-1]->[0]];
6890
6891 $self->{insertion_mode} = IN_TABLE_IM;
6892
6893 !!!nack ('t382.1');
6894 !!!next-token;
6895 } elsif ($token->{tag_name} eq 'hr') {
6896 !!!cp ('t386');
6897 pop @{$self->{open_elements}};
6898
6899 !!!nack ('t386.1');
6900 !!!next-token;
6901 } else {
6902 !!!nack ('t347.1');
6903 !!!next-token;
6904 }
6905 next B;
6906 } elsif ({li => 1, dt => 1, dd => 1}->{$token->{tag_name}}) {
6907 ## has a p element in scope
6908 INSCOPE: for (reverse @{$self->{open_elements}}) {
6909 if ($_->[1] & P_EL) {
6910 !!!cp ('t353');
6911 !!!back-token; # <x>
6912 $token = {type => END_TAG_TOKEN, tag_name => 'p',
6913 line => $token->{line}, column => $token->{column}};
6914 next B;
6915 } elsif ($_->[1] & SCOPING_EL) {
6916 !!!cp ('t354');
6917 last INSCOPE;
6918 }
6919 } # INSCOPE
6920
6921 ## Step 1
6922 my $i = -1;
6923 my $node = $self->{open_elements}->[$i];
6924 my $li_or_dtdd = {li => {li => 1},
6925 dt => {dt => 1, dd => 1},
6926 dd => {dt => 1, dd => 1}}->{$token->{tag_name}};
6927 LI: {
6928 ## Step 2
6929 if ($li_or_dtdd->{$node->[0]->manakai_local_name}) {
6930 if ($i != -1) {
6931 !!!cp ('t355');
6932 !!!parse-error (type => 'not closed',
6933 text => $self->{open_elements}->[-1]->[0]
6934 ->manakai_local_name,
6935 token => $token);
6936 } else {
6937 !!!cp ('t356');
6938 }
6939 splice @{$self->{open_elements}}, $i;
6940 last LI;
6941 } else {
6942 !!!cp ('t357');
6943 }
6944
6945 ## Step 3
6946 if (not ($node->[1] & FORMATTING_EL) and
6947 #not $phrasing_category->{$node->[1]} and
6948 ($node->[1] & SPECIAL_EL or
6949 $node->[1] & SCOPING_EL) and
6950 not ($node->[1] & ADDRESS_EL) and
6951 not ($node->[1] & DIV_EL)) {
6952 !!!cp ('t358');
6953 last LI;
6954 }
6955
6956 !!!cp ('t359');
6957 ## Step 4
6958 $i--;
6959 $node = $self->{open_elements}->[$i];
6960 redo LI;
6961 } # LI
6962
6963 !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
6964 !!!nack ('t359.1');
6965 !!!next-token;
6966 next B;
6967 } elsif ($token->{tag_name} eq 'plaintext') {
6968 ## has a p element in scope
6969 INSCOPE: for (reverse @{$self->{open_elements}}) {
6970 if ($_->[1] & P_EL) {
6971 !!!cp ('t367');
6972 !!!back-token; # <plaintext>
6973 $token = {type => END_TAG_TOKEN, tag_name => 'p',
6974 line => $token->{line}, column => $token->{column}};
6975 next B;
6976 } elsif ($_->[1] & SCOPING_EL) {
6977 !!!cp ('t368');
6978 last INSCOPE;
6979 }
6980 } # INSCOPE
6981
6982 !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
6983
6984 $self->{content_model} = PLAINTEXT_CONTENT_MODEL;
6985
6986 !!!nack ('t368.1');
6987 !!!next-token;
6988 next B;
6989 } elsif ($token->{tag_name} eq 'a') {
6990 AFE: for my $i (reverse 0..$#$active_formatting_elements) {
6991 my $node = $active_formatting_elements->[$i];
6992 if ($node->[1] & A_EL) {
6993 !!!cp ('t371');
6994 !!!parse-error (type => 'in a:a', token => $token);
6995
6996 !!!back-token; # <a>
6997 $token = {type => END_TAG_TOKEN, tag_name => 'a',
6998 line => $token->{line}, column => $token->{column}};
6999 $formatting_end_tag->($token);
7000
7001 AFE2: for (reverse 0..$#$active_formatting_elements) {
7002 if ($active_formatting_elements->[$_]->[0] eq $node->[0]) {
7003 !!!cp ('t372');
7004 splice @$active_formatting_elements, $_, 1;
7005 last AFE2;
7006 }
7007 } # AFE2
7008 OE: for (reverse 0..$#{$self->{open_elements}}) {
7009 if ($self->{open_elements}->[$_]->[0] eq $node->[0]) {
7010 !!!cp ('t373');
7011 splice @{$self->{open_elements}}, $_, 1;
7012 last OE;
7013 }
7014 } # OE
7015 last AFE;
7016 } elsif ($node->[0] eq '#marker') {
7017 !!!cp ('t374');
7018 last AFE;
7019 }
7020 } # AFE
7021
7022 $reconstruct_active_formatting_elements->($insert_to_current);
7023
7024 !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7025 push @$active_formatting_elements, $self->{open_elements}->[-1];
7026
7027 !!!nack ('t374.1');
7028 !!!next-token;
7029 next B;
7030 } elsif ($token->{tag_name} eq 'nobr') {
7031 $reconstruct_active_formatting_elements->($insert_to_current);
7032
7033 ## has a |nobr| element in scope
7034 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7035 my $node = $self->{open_elements}->[$_];
7036 if ($node->[1] & NOBR_EL) {
7037 !!!cp ('t376');
7038 !!!parse-error (type => 'in nobr:nobr', token => $token);
7039 !!!back-token; # <nobr>
7040 $token = {type => END_TAG_TOKEN, tag_name => 'nobr',
7041 line => $token->{line}, column => $token->{column}};
7042 next B;
7043 } elsif ($node->[1] & SCOPING_EL) {
7044 !!!cp ('t377');
7045 last INSCOPE;
7046 }
7047 } # INSCOPE
7048
7049 !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7050 push @$active_formatting_elements, $self->{open_elements}->[-1];
7051
7052 !!!nack ('t377.1');
7053 !!!next-token;
7054 next B;
7055 } elsif ($token->{tag_name} eq 'button') {
7056 ## has a button element in scope
7057 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7058 my $node = $self->{open_elements}->[$_];
7059 if ($node->[1] & BUTTON_EL) {
7060 !!!cp ('t378');
7061 !!!parse-error (type => 'in button:button', token => $token);
7062 !!!back-token; # <button>
7063 $token = {type => END_TAG_TOKEN, tag_name => 'button',
7064 line => $token->{line}, column => $token->{column}};
7065 next B;
7066 } elsif ($node->[1] & SCOPING_EL) {
7067 !!!cp ('t379');
7068 last INSCOPE;
7069 }
7070 } # INSCOPE
7071
7072 $reconstruct_active_formatting_elements->($insert_to_current);
7073
7074 !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7075
7076 ## TODO: associate with $self->{form_element} if defined
7077
7078 push @$active_formatting_elements, ['#marker', ''];
7079
7080 !!!nack ('t379.1');
7081 !!!next-token;
7082 next B;
7083 } elsif ({
7084 xmp => 1,
7085 iframe => 1,
7086 noembed => 1,
7087 noframes => 1, ## NOTE: This is an "as if in head" code clone.
7088 noscript => 0, ## TODO: 1 if scripting is enabled
7089 }->{$token->{tag_name}}) {
7090 if ($token->{tag_name} eq 'xmp') {
7091 !!!cp ('t381');
7092 $reconstruct_active_formatting_elements->($insert_to_current);
7093 } else {
7094 !!!cp ('t399');
7095 }
7096 ## NOTE: There is an "as if in body" code clone.
7097 $parse_rcdata->(CDATA_CONTENT_MODEL);
7098 next B;
7099 } elsif ($token->{tag_name} eq 'isindex') {
7100 !!!parse-error (type => 'isindex', token => $token);
7101
7102 if (defined $self->{form_element}) {
7103 !!!cp ('t389');
7104 ## Ignore the token
7105 !!!nack ('t389'); ## NOTE: Not acknowledged.
7106 !!!next-token;
7107 next B;
7108 } else {
7109 !!!ack ('t391.1');
7110
7111 my $at = $token->{attributes};
7112 my $form_attrs;
7113 $form_attrs->{action} = $at->{action} if $at->{action};
7114 my $prompt_attr = $at->{prompt};
7115 $at->{name} = {name => 'name', value => 'isindex'};
7116 delete $at->{action};
7117 delete $at->{prompt};
7118 my @tokens = (
7119 {type => START_TAG_TOKEN, tag_name => 'form',
7120 attributes => $form_attrs,
7121 line => $token->{line}, column => $token->{column}},
7122 {type => START_TAG_TOKEN, tag_name => 'hr',
7123 line => $token->{line}, column => $token->{column}},
7124 {type => START_TAG_TOKEN, tag_name => 'p',
7125 line => $token->{line}, column => $token->{column}},
7126 {type => START_TAG_TOKEN, tag_name => 'label',
7127 line => $token->{line}, column => $token->{column}},
7128 );
7129 if ($prompt_attr) {
7130 !!!cp ('t390');
7131 push @tokens, {type => CHARACTER_TOKEN, data => $prompt_attr->{value},
7132 #line => $token->{line}, column => $token->{column},
7133 };
7134 } else {
7135 !!!cp ('t391');
7136 push @tokens, {type => CHARACTER_TOKEN,
7137 data => 'This is a searchable index. Insert your search keywords here: ',
7138 #line => $token->{line}, column => $token->{column},
7139 }; # SHOULD
7140 ## TODO: make this configurable
7141 }
7142 push @tokens,
7143 {type => START_TAG_TOKEN, tag_name => 'input', attributes => $at,
7144 line => $token->{line}, column => $token->{column}},
7145 #{type => CHARACTER_TOKEN, data => ''}, # SHOULD
7146 {type => END_TAG_TOKEN, tag_name => 'label',
7147 line => $token->{line}, column => $token->{column}},
7148 {type => END_TAG_TOKEN, tag_name => 'p',
7149 line => $token->{line}, column => $token->{column}},
7150 {type => START_TAG_TOKEN, tag_name => 'hr',
7151 line => $token->{line}, column => $token->{column}},
7152 {type => END_TAG_TOKEN, tag_name => 'form',
7153 line => $token->{line}, column => $token->{column}};
7154 !!!back-token (@tokens);
7155 !!!next-token;
7156 next B;
7157 }
7158 } elsif ($token->{tag_name} eq 'textarea') {
7159 my $tag_name = $token->{tag_name};
7160 my $el;
7161 !!!create-element ($el, $HTML_NS, $token->{tag_name}, $token->{attributes}, $token);
7162
7163 ## TODO: $self->{form_element} if defined
7164 $self->{content_model} = RCDATA_CONTENT_MODEL;
7165 delete $self->{escape}; # MUST
7166
7167 $insert->($el);
7168
7169 my $text = '';
7170 !!!nack ('t392.1');
7171 !!!next-token;
7172 if ($token->{type} == CHARACTER_TOKEN) {
7173 $token->{data} =~ s/^\x0A//;
7174 unless (length $token->{data}) {
7175 !!!cp ('t392');
7176 !!!next-token;
7177 } else {
7178 !!!cp ('t393');
7179 }
7180 } else {
7181 !!!cp ('t394');
7182 }
7183 while ($token->{type} == CHARACTER_TOKEN) {
7184 !!!cp ('t395');
7185 $text .= $token->{data};
7186 !!!next-token;
7187 }
7188 if (length $text) {
7189 !!!cp ('t396');
7190 $el->manakai_append_text ($text);
7191 }
7192
7193 $self->{content_model} = PCDATA_CONTENT_MODEL;
7194
7195 if ($token->{type} == END_TAG_TOKEN and
7196 $token->{tag_name} eq $tag_name) {
7197 !!!cp ('t397');
7198 ## Ignore the token
7199 } else {
7200 !!!cp ('t398');
7201 !!!parse-error (type => 'in RCDATA:#eof', token => $token);
7202 }
7203 !!!next-token;
7204 next B;
7205 } elsif ($token->{tag_name} eq 'rt' or
7206 $token->{tag_name} eq 'rp') {
7207 ## has a |ruby| element in scope
7208 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7209 my $node = $self->{open_elements}->[$_];
7210 if ($node->[1] & RUBY_EL) {
7211 !!!cp ('t398.1');
7212 ## generate implied end tags
7213 while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
7214 !!!cp ('t398.2');
7215 pop @{$self->{open_elements}};
7216 }
7217 unless ($self->{open_elements}->[-1]->[1] & RUBY_EL) {
7218 !!!cp ('t398.3');
7219 !!!parse-error (type => 'not closed',
7220 text => $self->{open_elements}->[-1]->[0]
7221 ->manakai_local_name,
7222 token => $token);
7223 pop @{$self->{open_elements}}
7224 while not $self->{open_elements}->[-1]->[1] & RUBY_EL;
7225 }
7226 last INSCOPE;
7227 } elsif ($node->[1] & SCOPING_EL) {
7228 !!!cp ('t398.4');
7229 last INSCOPE;
7230 }
7231 } # INSCOPE
7232
7233 !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7234
7235 !!!nack ('t398.5');
7236 !!!next-token;
7237 redo B;
7238 } elsif ($token->{tag_name} eq 'math' or
7239 $token->{tag_name} eq 'svg') {
7240 $reconstruct_active_formatting_elements->($insert_to_current);
7241
7242 ## "Adjust MathML attributes" ('math' only) - done in insert-element-f
7243
7244 ## "adjust SVG attributes" ('svg' only) - done in insert-element-f
7245
7246 ## "adjust foreign attributes" - done in insert-element-f
7247
7248 !!!insert-element-f ($token->{tag_name} eq 'math' ? $MML_NS : $SVG_NS, $token->{tag_name}, $token->{attributes}, $token);
7249
7250 if ($self->{self_closing}) {
7251 pop @{$self->{open_elements}};
7252 !!!ack ('t398.1');
7253 } else {
7254 !!!cp ('t398.2');
7255 $self->{insertion_mode} |= IN_FOREIGN_CONTENT_IM;
7256 ## NOTE: |<body><math><mi><svg>| -> "in foreign content" insertion
7257 ## mode, "in body" (not "in foreign content") secondary insertion
7258 ## mode, maybe.
7259 }
7260
7261 !!!next-token;
7262 next B;
7263 } elsif ({
7264 caption => 1, col => 1, colgroup => 1, frame => 1,
7265 frameset => 1, head => 1, option => 1, optgroup => 1,
7266 tbody => 1, td => 1, tfoot => 1, th => 1,
7267 thead => 1, tr => 1,
7268 }->{$token->{tag_name}}) {
7269 !!!cp ('t401');
7270 !!!parse-error (type => 'in body',
7271 text => $token->{tag_name}, token => $token);
7272 ## Ignore the token
7273 !!!nack ('t401.1'); ## NOTE: |<col/>| or |<frame/>| here is an error.
7274 !!!next-token;
7275 next B;
7276
7277 ## ISSUE: An issue on HTML5 new elements in the spec.
7278 } else {
7279 if ($token->{tag_name} eq 'image') {
7280 !!!cp ('t384');
7281 !!!parse-error (type => 'image', token => $token);
7282 $token->{tag_name} = 'img';
7283 } else {
7284 !!!cp ('t385');
7285 }
7286
7287 ## NOTE: There is an "as if <br>" code clone.
7288 $reconstruct_active_formatting_elements->($insert_to_current);
7289
7290 !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7291
7292 if ({
7293 applet => 1, marquee => 1, object => 1,
7294 }->{$token->{tag_name}}) {
7295 !!!cp ('t380');
7296 push @$active_formatting_elements, ['#marker', ''];
7297 !!!nack ('t380.1');
7298 } elsif ({
7299 b => 1, big => 1, em => 1, font => 1, i => 1,
7300 s => 1, small => 1, strile => 1,
7301 strong => 1, tt => 1, u => 1,
7302 }->{$token->{tag_name}}) {
7303 !!!cp ('t375');
7304 push @$active_formatting_elements, $self->{open_elements}->[-1];
7305 !!!nack ('t375.1');
7306 } elsif ($token->{tag_name} eq 'input') {
7307 !!!cp ('t388');
7308 ## TODO: associate with $self->{form_element} if defined
7309 pop @{$self->{open_elements}};
7310 !!!ack ('t388.2');
7311 } elsif ({
7312 area => 1, basefont => 1, bgsound => 1, br => 1,
7313 embed => 1, img => 1, param => 1, spacer => 1, wbr => 1,
7314 #image => 1,
7315 }->{$token->{tag_name}}) {
7316 !!!cp ('t388.1');
7317 pop @{$self->{open_elements}};
7318 !!!ack ('t388.3');
7319 } elsif ($token->{tag_name} eq 'select') {
7320 ## TODO: associate with $self->{form_element} if defined
7321
7322 if ($self->{insertion_mode} & TABLE_IMS or
7323 $self->{insertion_mode} & BODY_TABLE_IMS or
7324 $self->{insertion_mode} == IN_COLUMN_GROUP_IM) {
7325 !!!cp ('t400.1');
7326 $self->{insertion_mode} = IN_SELECT_IN_TABLE_IM;
7327 } else {
7328 !!!cp ('t400.2');
7329 $self->{insertion_mode} = IN_SELECT_IM;
7330 }
7331 !!!nack ('t400.3');
7332 } else {
7333 !!!nack ('t402');
7334 }
7335
7336 !!!next-token;
7337 next B;
7338 }
7339 } elsif ($token->{type} == END_TAG_TOKEN) {
7340 if ($token->{tag_name} eq 'body') {
7341 ## has a |body| element in scope
7342 my $i;
7343 INSCOPE: {
7344 for (reverse @{$self->{open_elements}}) {
7345 if ($_->[1] & BODY_EL) {
7346 !!!cp ('t405');
7347 $i = $_;
7348 last INSCOPE;
7349 } elsif ($_->[1] & SCOPING_EL) {
7350 !!!cp ('t405.1');
7351 last;
7352 }
7353 }
7354
7355 !!!parse-error (type => 'start tag not allowed',
7356 text => $token->{tag_name}, token => $token);
7357 ## NOTE: Ignore the token.
7358 !!!next-token;
7359 next B;
7360 } # INSCOPE
7361
7362 for (@{$self->{open_elements}}) {
7363 unless ($_->[1] & ALL_END_TAG_OPTIONAL_EL) {
7364 !!!cp ('t403');
7365 !!!parse-error (type => 'not closed',
7366 text => $_->[0]->manakai_local_name,
7367 token => $token);
7368 last;
7369 } else {
7370 !!!cp ('t404');
7371 }
7372 }
7373
7374 $self->{insertion_mode} = AFTER_BODY_IM;
7375 !!!next-token;
7376 next B;
7377 } elsif ($token->{tag_name} eq 'html') {
7378 ## TODO: Update this code. It seems that the code below is not
7379 ## up-to-date, though it has same effect as speced.
7380 if (@{$self->{open_elements}} > 1 and
7381 $self->{open_elements}->[1]->[1] & BODY_EL) {
7382 ## ISSUE: There is an issue in the spec.
7383 unless ($self->{open_elements}->[-1]->[1] & BODY_EL) {
7384 !!!cp ('t406');
7385 !!!parse-error (type => 'not closed',
7386 text => $self->{open_elements}->[1]->[0]
7387 ->manakai_local_name,
7388 token => $token);
7389 } else {
7390 !!!cp ('t407');
7391 }
7392 $self->{insertion_mode} = AFTER_BODY_IM;
7393 ## reprocess
7394 next B;
7395 } else {
7396 !!!cp ('t408');
7397 !!!parse-error (type => 'unmatched end tag',
7398 text => $token->{tag_name}, token => $token);
7399 ## Ignore the token
7400 !!!next-token;
7401 next B;
7402 }
7403 } elsif ({
7404 address => 1, blockquote => 1, center => 1, dir => 1,
7405 div => 1, dl => 1, fieldset => 1, listing => 1,
7406 menu => 1, ol => 1, pre => 1, ul => 1,
7407 dd => 1, dt => 1, li => 1,
7408 applet => 1, button => 1, marquee => 1, object => 1,
7409 }->{$token->{tag_name}}) {
7410 ## has an element in scope
7411 my $i;
7412 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7413 my $node = $self->{open_elements}->[$_];
7414 if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
7415 !!!cp ('t410');
7416 $i = $_;
7417 last INSCOPE;
7418 } elsif ($node->[1] & SCOPING_EL) {
7419 !!!cp ('t411');
7420 last INSCOPE;
7421 }
7422 } # INSCOPE
7423
7424 unless (defined $i) { # has an element in scope
7425 !!!cp ('t413');
7426 !!!parse-error (type => 'unmatched end tag',
7427 text => $token->{tag_name}, token => $token);
7428 ## NOTE: Ignore the token.
7429 } else {
7430 ## Step 1. generate implied end tags
7431 while ({
7432 ## END_TAG_OPTIONAL_EL
7433 dd => ($token->{tag_name} ne 'dd'),
7434 dt => ($token->{tag_name} ne 'dt'),
7435 li => ($token->{tag_name} ne 'li'),
7436 p => 1,
7437 rt => 1,
7438 rp => 1,
7439 }->{$self->{open_elements}->[-1]->[0]->manakai_local_name}) {
7440 !!!cp ('t409');
7441 pop @{$self->{open_elements}};
7442 }
7443
7444 ## Step 2.
7445 if ($self->{open_elements}->[-1]->[0]->manakai_local_name
7446 ne $token->{tag_name}) {
7447 !!!cp ('t412');
7448 !!!parse-error (type => 'not closed',
7449 text => $self->{open_elements}->[-1]->[0]
7450 ->manakai_local_name,
7451 token => $token);
7452 } else {
7453 !!!cp ('t414');
7454 }
7455
7456 ## Step 3.
7457 splice @{$self->{open_elements}}, $i;
7458
7459 ## Step 4.
7460 $clear_up_to_marker->()
7461 if {
7462 applet => 1, button => 1, marquee => 1, object => 1,
7463 }->{$token->{tag_name}};
7464 }
7465 !!!next-token;
7466 next B;
7467 } elsif ($token->{tag_name} eq 'form') {
7468 undef $self->{form_element};
7469
7470 ## has an element in scope
7471 my $i;
7472 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7473 my $node = $self->{open_elements}->[$_];
7474 if ($node->[1] & FORM_EL) {
7475 !!!cp ('t418');
7476 $i = $_;
7477 last INSCOPE;
7478 } elsif ($node->[1] & SCOPING_EL) {
7479 !!!cp ('t419');
7480 last INSCOPE;
7481 }
7482 } # INSCOPE
7483
7484 unless (defined $i) { # has an element in scope
7485 !!!cp ('t421');
7486 !!!parse-error (type => 'unmatched end tag',
7487 text => $token->{tag_name}, token => $token);
7488 ## NOTE: Ignore the token.
7489 } else {
7490 ## Step 1. generate implied end tags
7491 while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
7492 !!!cp ('t417');
7493 pop @{$self->{open_elements}};
7494 }
7495
7496 ## Step 2.
7497 if ($self->{open_elements}->[-1]->[0]->manakai_local_name
7498 ne $token->{tag_name}) {
7499 !!!cp ('t417.1');
7500 !!!parse-error (type => 'not closed',
7501 text => $self->{open_elements}->[-1]->[0]
7502 ->manakai_local_name,
7503 token => $token);
7504 } else {
7505 !!!cp ('t420');
7506 }
7507
7508 ## Step 3.
7509 splice @{$self->{open_elements}}, $i;
7510 }
7511
7512 !!!next-token;
7513 next B;
7514 } elsif ({
7515 h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,
7516 }->{$token->{tag_name}}) {
7517 ## has an element in scope
7518 my $i;
7519 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7520 my $node = $self->{open_elements}->[$_];
7521 if ($node->[1] & HEADING_EL) {
7522 !!!cp ('t423');
7523 $i = $_;
7524 last INSCOPE;
7525 } elsif ($node->[1] & SCOPING_EL) {
7526 !!!cp ('t424');
7527 last INSCOPE;
7528 }
7529 } # INSCOPE
7530
7531 unless (defined $i) { # has an element in scope
7532 !!!cp ('t425.1');
7533 !!!parse-error (type => 'unmatched end tag',
7534 text => $token->{tag_name}, token => $token);
7535 ## NOTE: Ignore the token.
7536 } else {
7537 ## Step 1. generate implied end tags
7538 while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
7539 !!!cp ('t422');
7540 pop @{$self->{open_elements}};
7541 }
7542
7543 ## Step 2.
7544 if ($self->{open_elements}->[-1]->[0]->manakai_local_name
7545 ne $token->{tag_name}) {
7546 !!!cp ('t425');
7547 !!!parse-error (type => 'unmatched end tag',
7548 text => $token->{tag_name}, token => $token);
7549 } else {
7550 !!!cp ('t426');
7551 }
7552
7553 ## Step 3.
7554 splice @{$self->{open_elements}}, $i;
7555 }
7556
7557 !!!next-token;
7558 next B;
7559 } elsif ($token->{tag_name} eq 'p') {
7560 ## has an element in scope
7561 my $i;
7562 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7563 my $node = $self->{open_elements}->[$_];
7564 if ($node->[1] & P_EL) {
7565 !!!cp ('t410.1');
7566 $i = $_;
7567 last INSCOPE;
7568 } elsif ($node->[1] & SCOPING_EL) {
7569 !!!cp ('t411.1');
7570 last INSCOPE;
7571 }
7572 } # INSCOPE
7573
7574 if (defined $i) {
7575 if ($self->{open_elements}->[-1]->[0]->manakai_local_name
7576 ne $token->{tag_name}) {
7577 !!!cp ('t412.1');
7578 !!!parse-error (type => 'not closed',
7579 text => $self->{open_elements}->[-1]->[0]
7580 ->manakai_local_name,
7581 token => $token);
7582 } else {
7583 !!!cp ('t414.1');
7584 }
7585
7586 splice @{$self->{open_elements}}, $i;
7587 } else {
7588 !!!cp ('t413.1');
7589 !!!parse-error (type => 'unmatched end tag',
7590 text => $token->{tag_name}, token => $token);
7591
7592 !!!cp ('t415.1');
7593 ## As if <p>, then reprocess the current token
7594 my $el;
7595 !!!create-element ($el, $HTML_NS, 'p',, $token);
7596 $insert->($el);
7597 ## NOTE: Not inserted into |$self->{open_elements}|.
7598 }
7599
7600 !!!next-token;
7601 next B;
7602 } elsif ({
7603 a => 1,
7604 b => 1, big => 1, em => 1, font => 1, i => 1,
7605 nobr => 1, s => 1, small => 1, strile => 1,
7606 strong => 1, tt => 1, u => 1,
7607 }->{$token->{tag_name}}) {
7608 !!!cp ('t427');
7609 $formatting_end_tag->($token);
7610 next B;
7611 } elsif ($token->{tag_name} eq 'br') {
7612 !!!cp ('t428');
7613 !!!parse-error (type => 'unmatched end tag',
7614 text => 'br', token => $token);
7615
7616 ## As if <br>
7617 $reconstruct_active_formatting_elements->($insert_to_current);
7618
7619 my $el;
7620 !!!create-element ($el, $HTML_NS, 'br',, $token);
7621 $insert->($el);
7622
7623 ## Ignore the token.
7624 !!!next-token;
7625 next B;
7626 } elsif ({
7627 caption => 1, col => 1, colgroup => 1, frame => 1,
7628 frameset => 1, head => 1, option => 1, optgroup => 1,
7629 tbody => 1, td => 1, tfoot => 1, th => 1,
7630 thead => 1, tr => 1,
7631 area => 1, basefont => 1, bgsound => 1,
7632 embed => 1, hr => 1, iframe => 1, image => 1,
7633 img => 1, input => 1, isindex => 1, noembed => 1,
7634 noframes => 1, param => 1, select => 1, spacer => 1,
7635 table => 1, textarea => 1, wbr => 1,
7636 noscript => 0, ## TODO: if scripting is enabled
7637 }->{$token->{tag_name}}) {
7638 !!!cp ('t429');
7639 !!!parse-error (type => 'unmatched end tag',
7640 text => $token->{tag_name}, token => $token);
7641 ## Ignore the token
7642 !!!next-token;
7643 next B;
7644
7645 ## ISSUE: Issue on HTML5 new elements in spec
7646
7647 } else {
7648 ## Step 1
7649 my $node_i = -1;
7650 my $node = $self->{open_elements}->[$node_i];
7651
7652 ## Step 2
7653 S2: {
7654 if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
7655 ## Step 1
7656 ## generate implied end tags
7657 while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
7658 !!!cp ('t430');
7659 ## NOTE: |<ruby><rt></ruby>|.
7660 ## ISSUE: <ruby><rt></rt> will also take this code path,
7661 ## which seems wrong.
7662 pop @{$self->{open_elements}};
7663 $node_i++;
7664 }
7665
7666 ## Step 2
7667 if ($self->{open_elements}->[-1]->[0]->manakai_local_name
7668 ne $token->{tag_name}) {
7669 !!!cp ('t431');
7670 ## NOTE: <x><y></x>
7671 !!!parse-error (type => 'not closed',
7672 text => $self->{open_elements}->[-1]->[0]
7673 ->manakai_local_name,
7674 token => $token);
7675 } else {
7676 !!!cp ('t432');
7677 }
7678
7679 ## Step 3
7680 splice @{$self->{open_elements}}, $node_i if $node_i < 0;
7681
7682 !!!next-token;
7683 last S2;
7684 } else {
7685 ## Step 3
7686 if (not ($node->[1] & FORMATTING_EL) and
7687 #not $phrasing_category->{$node->[1]} and
7688 ($node->[1] & SPECIAL_EL or
7689 $node->[1] & SCOPING_EL)) {
7690 !!!cp ('t433');
7691 !!!parse-error (type => 'unmatched end tag',
7692 text => $token->{tag_name}, token => $token);
7693 ## Ignore the token
7694 !!!next-token;
7695 last S2;
7696 }
7697
7698 !!!cp ('t434');
7699 }
7700
7701 ## Step 4
7702 $node_i--;
7703 $node = $self->{open_elements}->[$node_i];
7704
7705 ## Step 5;
7706 redo S2;
7707 } # S2
7708 next B;
7709 }
7710 }
7711 next B;
7712 } continue { # B
7713 if ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM) {
7714 ## NOTE: The code below is executed in cases where it does not have
7715 ## to be, but it it is harmless even in those cases.
7716 ## has an element in scope
7717 INSCOPE: {
7718 for (reverse 0..$#{$self->{open_elements}}) {
7719 my $node = $self->{open_elements}->[$_];
7720 if ($node->[1] & FOREIGN_EL) {
7721 last INSCOPE;
7722 } elsif ($node->[1] & SCOPING_EL) {
7723 last;
7724 }
7725 }
7726
7727 ## NOTE: No foreign element in scope.
7728 $self->{insertion_mode} &= ~ IN_FOREIGN_CONTENT_IM;
7729 } # INSCOPE
7730 }
7731 } # B
7732
7733 ## Stop parsing # MUST
7734
7735 ## TODO: script stuffs
7736 } # _tree_construct_main
7737
7738 sub set_inner_html ($$$;$) {
7739 my $class = shift;
7740 my $node = shift;
7741 my $s = \$_[0];
7742 my $onerror = $_[1];
7743 my $get_wrapper = $_[2] || sub ($) { return $_[0] };
7744
7745 ## ISSUE: Should {confident} be true?
7746
7747 my $nt = $node->node_type;
7748 if ($nt == 9) {
7749 # MUST
7750
7751 ## Step 1 # MUST
7752 ## TODO: If the document has an active parser, ...
7753 ## ISSUE: There is an issue in the spec.
7754
7755 ## Step 2 # MUST
7756 my @cn = @{$node->child_nodes};
7757 for (@cn) {
7758 $node->remove_child ($_);
7759 }
7760
7761 ## Step 3, 4, 5 # MUST
7762 $class->parse_char_string ($$s => $node, $onerror, $get_wrapper);
7763 } elsif ($nt == 1) {
7764 ## TODO: If non-html element
7765
7766 ## NOTE: Most of this code is copied from |parse_string|
7767
7768 ## TODO: Support for $get_wrapper
7769
7770 ## Step 1 # MUST
7771 my $this_doc = $node->owner_document;
7772 my $doc = $this_doc->implementation->create_document;
7773 $doc->manakai_is_html (1);
7774 my $p = $class->new;
7775 $p->{document} = $doc;
7776
7777 ## Step 8 # MUST
7778 my $i = 0;
7779 $p->{line_prev} = $p->{line} = 1;
7780 $p->{column_prev} = $p->{column} = 0;
7781 $p->{set_next_char} = sub {
7782 my $self = shift;
7783
7784 pop @{$self->{prev_char}};
7785 unshift @{$self->{prev_char}}, $self->{next_char};
7786
7787 $self->{next_char} = -1 and return if $i >= length $$s;
7788 $self->{next_char} = ord substr $$s, $i++, 1;
7789
7790 ($p->{line_prev}, $p->{column_prev}) = ($p->{line}, $p->{column});
7791 $p->{column}++;
7792
7793 if ($self->{next_char} == 0x000A) { # LF
7794 $p->{line}++;
7795 $p->{column} = 0;
7796 !!!cp ('i1');
7797 } elsif ($self->{next_char} == 0x000D) { # CR
7798 $i++ if substr ($$s, $i, 1) eq "\x0A";
7799 $self->{next_char} = 0x000A; # LF # MUST
7800 $p->{line}++;
7801 $p->{column} = 0;
7802 !!!cp ('i2');
7803 } elsif ($self->{next_char} > 0x10FFFF) {
7804 $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
7805 !!!cp ('i3');
7806 } elsif ($self->{next_char} == 0x0000) { # NULL
7807 !!!cp ('i4');
7808 !!!parse-error (type => 'NULL');
7809 $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
7810 } elsif ($self->{next_char} <= 0x0008 or
7811 (0x000E <= $self->{next_char} and
7812 $self->{next_char} <= 0x001F) or
7813 (0x007F <= $self->{next_char} and
7814 $self->{next_char} <= 0x009F) or
7815 (0xD800 <= $self->{next_char} and
7816 $self->{next_char} <= 0xDFFF) or
7817 (0xFDD0 <= $self->{next_char} and
7818 $self->{next_char} <= 0xFDDF) or
7819 {
7820 0xFFFE => 1, 0xFFFF => 1, 0x1FFFE => 1, 0x1FFFF => 1,
7821 0x2FFFE => 1, 0x2FFFF => 1, 0x3FFFE => 1, 0x3FFFF => 1,
7822 0x4FFFE => 1, 0x4FFFF => 1, 0x5FFFE => 1, 0x5FFFF => 1,
7823 0x6FFFE => 1, 0x6FFFF => 1, 0x7FFFE => 1, 0x7FFFF => 1,
7824 0x8FFFE => 1, 0x8FFFF => 1, 0x9FFFE => 1, 0x9FFFF => 1,
7825 0xAFFFE => 1, 0xAFFFF => 1, 0xBFFFE => 1, 0xBFFFF => 1,
7826 0xCFFFE => 1, 0xCFFFF => 1, 0xDFFFE => 1, 0xDFFFF => 1,
7827 0xEFFFE => 1, 0xEFFFF => 1, 0xFFFFE => 1, 0xFFFFF => 1,
7828 0x10FFFE => 1, 0x10FFFF => 1,
7829 }->{$self->{next_char}}) {
7830 !!!cp ('i4.1');
7831 if ($self->{next_char} < 0x10000) {
7832 !!!parse-error (type => 'control char',
7833 text => (sprintf 'U+%04X', $self->{next_char}));
7834 } else {
7835 !!!parse-error (type => 'control char',
7836 text => (sprintf 'U-%08X', $self->{next_char}));
7837 }
7838 }
7839 };
7840 $p->{prev_char} = [-1, -1, -1];
7841 $p->{next_char} = -1;
7842
7843 $p->{read_until} = sub {
7844 ## TODO: ...
7845 return 0;
7846 }; # $p->{read_until};
7847
7848 my $ponerror = $onerror || sub {
7849 my (%opt) = @_;
7850 my $line = $opt{line};
7851 my $column = $opt{column};
7852 if (defined $opt{token} and defined $opt{token}->{line}) {
7853 $line = $opt{token}->{line};
7854 $column = $opt{token}->{column};
7855 }
7856 warn "Parse error ($opt{type}) at line $line column $column\n";
7857 };
7858 $p->{parse_error} = sub {
7859 $ponerror->(line => $p->{line}, column => $p->{column}, @_);
7860 };
7861
7862 $p->_initialize_tokenizer;
7863 $p->_initialize_tree_constructor;
7864
7865 ## Step 2
7866 my $node_ln = $node->manakai_local_name;
7867 $p->{content_model} = {
7868 title => RCDATA_CONTENT_MODEL,
7869 textarea => RCDATA_CONTENT_MODEL,
7870 style => CDATA_CONTENT_MODEL,
7871 script => CDATA_CONTENT_MODEL,
7872 xmp => CDATA_CONTENT_MODEL,
7873 iframe => CDATA_CONTENT_MODEL,
7874 noembed => CDATA_CONTENT_MODEL,
7875 noframes => CDATA_CONTENT_MODEL,
7876 noscript => CDATA_CONTENT_MODEL,
7877 plaintext => PLAINTEXT_CONTENT_MODEL,
7878 }->{$node_ln};
7879 $p->{content_model} = PCDATA_CONTENT_MODEL
7880 unless defined $p->{content_model};
7881 ## ISSUE: What is "the name of the element"? local name?
7882
7883 $p->{inner_html_node} = [$node, $el_category->{$node_ln}];
7884 ## TODO: Foreign element OK?
7885
7886 ## Step 3
7887 my $root = $doc->create_element_ns
7888 ('http://www.w3.org/1999/xhtml', [undef, 'html']);
7889
7890 ## Step 4 # MUST
7891 $doc->append_child ($root);
7892
7893 ## Step 5 # MUST
7894 push @{$p->{open_elements}}, [$root, $el_category->{html}];
7895
7896 undef $p->{head_element};
7897
7898 ## Step 6 # MUST
7899 $p->_reset_insertion_mode;
7900
7901 ## Step 7 # MUST
7902 my $anode = $node;
7903 AN: while (defined $anode) {
7904 if ($anode->node_type == 1) {
7905 my $nsuri = $anode->namespace_uri;
7906 if (defined $nsuri and $nsuri eq 'http://www.w3.org/1999/xhtml') {
7907 if ($anode->manakai_local_name eq 'form') {
7908 !!!cp ('i5');
7909 $p->{form_element} = $anode;
7910 last AN;
7911 }
7912 }
7913 }
7914 $anode = $anode->parent_node;
7915 } # AN
7916
7917 ## Step 9 # MUST
7918 {
7919 my $self = $p;
7920 !!!next-token;
7921 }
7922 $p->_tree_construction_main;
7923
7924 ## Step 10 # MUST
7925 my @cn = @{$node->child_nodes};
7926 for (@cn) {
7927 $node->remove_child ($_);
7928 }
7929 ## ISSUE: mutation events? read-only?
7930
7931 ## Step 11 # MUST
7932 @cn = @{$root->child_nodes};
7933 for (@cn) {
7934 $this_doc->adopt_node ($_);
7935 $node->append_child ($_);
7936 }
7937 ## ISSUE: mutation events?
7938
7939 $p->_terminate_tree_constructor;
7940
7941 delete $p->{parse_error}; # delete loop
7942 } else {
7943 die "$0: |set_inner_html| is not defined for node of type $nt";
7944 }
7945 } # set_inner_html
7946
7947 } # tree construction stage
7948
7949 package Whatpm::HTML::RestartParser;
7950 push our @ISA, 'Error';
7951
7952 1;
7953 # $Date: 2008/09/14 03:59:08 $

[email protected]
ViewVC Help
Powered by ViewVC 1.1.24