/[suikacvs]/markup/html/whatpm/Whatpm/HTML.pm.src
Suika

Contents of /markup/html/whatpm/Whatpm/HTML.pm.src

Parent Directory Parent Directory | Revision Log Revision Log


Revision 1.151 - (show annotations) (download) (as text)
Sun Jun 8 05:08:42 2008 UTC (18 years, 4 months ago) by wakaba
Branch: MAIN
Changes since 1.150: +67 -8 lines
File MIME type: application/x-wais-source
++ whatpm/t/ChangeLog	8 Jun 2008 05:04:08 -0000
2008-06-08  Wakaba  <wakaba@suika.fam.cx>

	* tree-test-1.dat: Test data added for ruby parsing (HTML5 revision
	1704).

++ whatpm/Whatpm/ChangeLog	8 Jun 2008 05:03:48 -0000
2008-06-08  Wakaba  <wakaba@suika.fam.cx>

	* HTML.pm.src: Support for ruby parsing (HTML5 revision 1704).

1 package Whatpm::HTML;
2 use strict;
3 our $VERSION=do{my @r=(q$Revision: 1.150 $=~/\d+/g);sprintf "%d."."%02d" x $#r,@r};
4 use Error qw(:try);
5
6 ## ISSUE:
7 ## var doc = implementation.createDocument (null, null, null);
8 ## doc.write ('');
9 ## alert (doc.compatMode);
10
11 require IO::Handle;
12
13 my $HTML_NS = q<http://www.w3.org/1999/xhtml>;
14 my $MML_NS = q<http://www.w3.org/1998/Math/MathML>;
15 my $SVG_NS = q<http://www.w3.org/2000/svg>;
16 my $XLINK_NS = q<http://www.w3.org/1999/xlink>;
17 my $XML_NS = q<http://www.w3.org/XML/1998/namespace>;
18 my $XMLNS_NS = q<http://www.w3.org/2000/xmlns/>;
19
20 sub A_EL () { 0b1 }
21 sub ADDRESS_EL () { 0b10 }
22 sub BODY_EL () { 0b100 }
23 sub BUTTON_EL () { 0b1000 }
24 sub CAPTION_EL () { 0b10000 }
25 sub DD_EL () { 0b100000 }
26 sub DIV_EL () { 0b1000000 }
27 sub DT_EL () { 0b10000000 }
28 sub FORM_EL () { 0b100000000 }
29 sub FORMATTING_EL () { 0b1000000000 }
30 sub FRAMESET_EL () { 0b10000000000 }
31 sub HEADING_EL () { 0b100000000000 }
32 sub HTML_EL () { 0b1000000000000 }
33 sub LI_EL () { 0b10000000000000 }
34 sub NOBR_EL () { 0b100000000000000 }
35 sub OPTION_EL () { 0b1000000000000000 }
36 sub OPTGROUP_EL () { 0b10000000000000000 }
37 sub P_EL () { 0b100000000000000000 }
38 sub SELECT_EL () { 0b1000000000000000000 }
39 sub TABLE_EL () { 0b10000000000000000000 }
40 sub TABLE_CELL_EL () { 0b100000000000000000000 }
41 sub TABLE_ROW_EL () { 0b1000000000000000000000 }
42 sub TABLE_ROW_GROUP_EL () { 0b10000000000000000000000 }
43 sub MISC_SCOPING_EL () { 0b100000000000000000000000 }
44 sub MISC_SPECIAL_EL () { 0b1000000000000000000000000 }
45 sub FOREIGN_EL () { 0b10000000000000000000000000 }
46 sub FOREIGN_FLOW_CONTENT_EL () { 0b100000000000000000000000000 }
47 sub MML_AXML_EL () { 0b1000000000000000000000000000 }
48 sub RUBY_EL () { 0b10000000000000000000000000000 }
49 sub RUBY_COMPONENT_EL () { 0b100000000000000000000000000000 }
50
51 sub TABLE_ROWS_EL () {
52 TABLE_EL |
53 TABLE_ROW_EL |
54 TABLE_ROW_GROUP_EL
55 }
56
57 ## NOTE: Used in "generate implied end tags" algorithm.
58 ## NOTE: There is a code where a modified version of END_TAG_OPTIONAL_EL
59 ## is used in "generate implied end tags" implementation (search for the
60 ## function mae).
61 sub END_TAG_OPTIONAL_EL () {
62 DD_EL |
63 DT_EL |
64 LI_EL |
65 P_EL |
66 RUBY_COMPONENT_EL
67 }
68
69 ## NOTE: Used in </body> and EOF algorithms.
70 sub ALL_END_TAG_OPTIONAL_EL () {
71 DD_EL |
72 DT_EL |
73 LI_EL |
74 P_EL |
75
76 BODY_EL |
77 HTML_EL |
78 TABLE_CELL_EL |
79 TABLE_ROW_EL |
80 TABLE_ROW_GROUP_EL
81 }
82
83 sub SCOPING_EL () {
84 BUTTON_EL |
85 CAPTION_EL |
86 HTML_EL |
87 TABLE_EL |
88 TABLE_CELL_EL |
89 MISC_SCOPING_EL
90 }
91
92 sub TABLE_SCOPING_EL () {
93 HTML_EL |
94 TABLE_EL
95 }
96
97 sub TABLE_ROWS_SCOPING_EL () {
98 HTML_EL |
99 TABLE_ROW_GROUP_EL
100 }
101
102 sub TABLE_ROW_SCOPING_EL () {
103 HTML_EL |
104 TABLE_ROW_EL
105 }
106
107 sub SPECIAL_EL () {
108 ADDRESS_EL |
109 BODY_EL |
110 DIV_EL |
111
112 DD_EL |
113 DT_EL |
114 LI_EL |
115 P_EL |
116
117 FORM_EL |
118 FRAMESET_EL |
119 HEADING_EL |
120 OPTION_EL |
121 OPTGROUP_EL |
122 SELECT_EL |
123 TABLE_ROW_EL |
124 TABLE_ROW_GROUP_EL |
125 MISC_SPECIAL_EL
126 }
127
128 my $el_category = {
129 a => A_EL | FORMATTING_EL,
130 address => ADDRESS_EL,
131 applet => MISC_SCOPING_EL,
132 area => MISC_SPECIAL_EL,
133 b => FORMATTING_EL,
134 base => MISC_SPECIAL_EL,
135 basefont => MISC_SPECIAL_EL,
136 bgsound => MISC_SPECIAL_EL,
137 big => FORMATTING_EL,
138 blockquote => MISC_SPECIAL_EL,
139 body => BODY_EL,
140 br => MISC_SPECIAL_EL,
141 button => BUTTON_EL,
142 caption => CAPTION_EL,
143 center => MISC_SPECIAL_EL,
144 col => MISC_SPECIAL_EL,
145 colgroup => MISC_SPECIAL_EL,
146 dd => DD_EL,
147 dir => MISC_SPECIAL_EL,
148 div => DIV_EL,
149 dl => MISC_SPECIAL_EL,
150 dt => DT_EL,
151 em => FORMATTING_EL,
152 embed => MISC_SPECIAL_EL,
153 fieldset => MISC_SPECIAL_EL,
154 font => FORMATTING_EL,
155 form => FORM_EL,
156 frame => MISC_SPECIAL_EL,
157 frameset => FRAMESET_EL,
158 h1 => HEADING_EL,
159 h2 => HEADING_EL,
160 h3 => HEADING_EL,
161 h4 => HEADING_EL,
162 h5 => HEADING_EL,
163 h6 => HEADING_EL,
164 head => MISC_SPECIAL_EL,
165 hr => MISC_SPECIAL_EL,
166 html => HTML_EL,
167 i => FORMATTING_EL,
168 iframe => MISC_SPECIAL_EL,
169 img => MISC_SPECIAL_EL,
170 input => MISC_SPECIAL_EL,
171 isindex => MISC_SPECIAL_EL,
172 li => LI_EL,
173 link => MISC_SPECIAL_EL,
174 listing => MISC_SPECIAL_EL,
175 marquee => MISC_SCOPING_EL,
176 menu => MISC_SPECIAL_EL,
177 meta => MISC_SPECIAL_EL,
178 nobr => NOBR_EL | FORMATTING_EL,
179 noembed => MISC_SPECIAL_EL,
180 noframes => MISC_SPECIAL_EL,
181 noscript => MISC_SPECIAL_EL,
182 object => MISC_SCOPING_EL,
183 ol => MISC_SPECIAL_EL,
184 optgroup => OPTGROUP_EL,
185 option => OPTION_EL,
186 p => P_EL,
187 param => MISC_SPECIAL_EL,
188 plaintext => MISC_SPECIAL_EL,
189 pre => MISC_SPECIAL_EL,
190 rp => RUBY_COMPONENT_EL,
191 rt => RUBY_COMPONENT_EL,
192 ruby => RUBY_EL,
193 s => FORMATTING_EL,
194 script => MISC_SPECIAL_EL,
195 select => SELECT_EL,
196 small => FORMATTING_EL,
197 spacer => MISC_SPECIAL_EL,
198 strike => FORMATTING_EL,
199 strong => FORMATTING_EL,
200 style => MISC_SPECIAL_EL,
201 table => TABLE_EL,
202 tbody => TABLE_ROW_GROUP_EL,
203 td => TABLE_CELL_EL,
204 textarea => MISC_SPECIAL_EL,
205 tfoot => TABLE_ROW_GROUP_EL,
206 th => TABLE_CELL_EL,
207 thead => TABLE_ROW_GROUP_EL,
208 title => MISC_SPECIAL_EL,
209 tr => TABLE_ROW_EL,
210 tt => FORMATTING_EL,
211 u => FORMATTING_EL,
212 ul => MISC_SPECIAL_EL,
213 wbr => MISC_SPECIAL_EL,
214 };
215
216 my $el_category_f = {
217 $MML_NS => {
218 'annotation-xml' => MML_AXML_EL,
219 mi => FOREIGN_FLOW_CONTENT_EL,
220 mo => FOREIGN_FLOW_CONTENT_EL,
221 mn => FOREIGN_FLOW_CONTENT_EL,
222 ms => FOREIGN_FLOW_CONTENT_EL,
223 mtext => FOREIGN_FLOW_CONTENT_EL,
224 },
225 $SVG_NS => {
226 foreignObject => FOREIGN_FLOW_CONTENT_EL,
227 desc => FOREIGN_FLOW_CONTENT_EL,
228 title => FOREIGN_FLOW_CONTENT_EL,
229 },
230 ## NOTE: In addition, FOREIGN_EL is set to non-HTML elements.
231 };
232
233 my $svg_attr_name = {
234 attributename => 'attributeName',
235 attributetype => 'attributeType',
236 basefrequency => 'baseFrequency',
237 baseprofile => 'baseProfile',
238 calcmode => 'calcMode',
239 clippathunits => 'clipPathUnits',
240 contentscripttype => 'contentScriptType',
241 contentstyletype => 'contentStyleType',
242 diffuseconstant => 'diffuseConstant',
243 edgemode => 'edgeMode',
244 externalresourcesrequired => 'externalResourcesRequired',
245 filterres => 'filterRes',
246 filterunits => 'filterUnits',
247 glyphref => 'glyphRef',
248 gradienttransform => 'gradientTransform',
249 gradientunits => 'gradientUnits',
250 kernelmatrix => 'kernelMatrix',
251 kernelunitlength => 'kernelUnitLength',
252 keypoints => 'keyPoints',
253 keysplines => 'keySplines',
254 keytimes => 'keyTimes',
255 lengthadjust => 'lengthAdjust',
256 limitingconeangle => 'limitingConeAngle',
257 markerheight => 'markerHeight',
258 markerunits => 'markerUnits',
259 markerwidth => 'markerWidth',
260 maskcontentunits => 'maskContentUnits',
261 maskunits => 'maskUnits',
262 numoctaves => 'numOctaves',
263 pathlength => 'pathLength',
264 patterncontentunits => 'patternContentUnits',
265 patterntransform => 'patternTransform',
266 patternunits => 'patternUnits',
267 pointsatx => 'pointsAtX',
268 pointsaty => 'pointsAtY',
269 pointsatz => 'pointsAtZ',
270 preservealpha => 'preserveAlpha',
271 preserveaspectratio => 'preserveAspectRatio',
272 primitiveunits => 'primitiveUnits',
273 refx => 'refX',
274 refy => 'refY',
275 repeatcount => 'repeatCount',
276 repeatdur => 'repeatDur',
277 requiredextensions => 'requiredExtensions',
278 requiredfeatures => 'requiredFeatures',
279 specularconstant => 'specularConstant',
280 specularexponent => 'specularExponent',
281 spreadmethod => 'spreadMethod',
282 startoffset => 'startOffset',
283 stddeviation => 'stdDeviation',
284 stitchtiles => 'stitchTiles',
285 surfacescale => 'surfaceScale',
286 systemlanguage => 'systemLanguage',
287 tablevalues => 'tableValues',
288 targetx => 'targetX',
289 targety => 'targetY',
290 textlength => 'textLength',
291 viewbox => 'viewBox',
292 viewtarget => 'viewTarget',
293 xchannelselector => 'xChannelSelector',
294 ychannelselector => 'yChannelSelector',
295 zoomandpan => 'zoomAndPan',
296 };
297
298 my $foreign_attr_xname = {
299 'xlink:actuate' => [$XLINK_NS, ['xlink', 'actuate']],
300 'xlink:arcrole' => [$XLINK_NS, ['xlink', 'arcrole']],
301 'xlink:href' => [$XLINK_NS, ['xlink', 'href']],
302 'xlink:role' => [$XLINK_NS, ['xlink', 'role']],
303 'xlink:show' => [$XLINK_NS, ['xlink', 'show']],
304 'xlink:title' => [$XLINK_NS, ['xlink', 'title']],
305 'xlink:type' => [$XLINK_NS, ['xlink', 'type']],
306 'xml:base' => [$XML_NS, ['xml', 'base']],
307 'xml:lang' => [$XML_NS, ['xml', 'lang']],
308 'xml:space' => [$XML_NS, ['xml', 'space']],
309 'xmlns' => [$XMLNS_NS, [undef, 'xmlns']],
310 'xmlns:xlink' => [$XMLNS_NS, ['xmlns', 'xlink']],
311 };
312
313 ## ISSUE: xmlns:xlink="non-xlink-ns" is not an error.
314
315 my $c1_entity_char = {
316 0x80 => 0x20AC,
317 0x81 => 0xFFFD,
318 0x82 => 0x201A,
319 0x83 => 0x0192,
320 0x84 => 0x201E,
321 0x85 => 0x2026,
322 0x86 => 0x2020,
323 0x87 => 0x2021,
324 0x88 => 0x02C6,
325 0x89 => 0x2030,
326 0x8A => 0x0160,
327 0x8B => 0x2039,
328 0x8C => 0x0152,
329 0x8D => 0xFFFD,
330 0x8E => 0x017D,
331 0x8F => 0xFFFD,
332 0x90 => 0xFFFD,
333 0x91 => 0x2018,
334 0x92 => 0x2019,
335 0x93 => 0x201C,
336 0x94 => 0x201D,
337 0x95 => 0x2022,
338 0x96 => 0x2013,
339 0x97 => 0x2014,
340 0x98 => 0x02DC,
341 0x99 => 0x2122,
342 0x9A => 0x0161,
343 0x9B => 0x203A,
344 0x9C => 0x0153,
345 0x9D => 0xFFFD,
346 0x9E => 0x017E,
347 0x9F => 0x0178,
348 }; # $c1_entity_char
349
350 sub parse_byte_string ($$$$;$) {
351 my $self = shift;
352 my $charset_name = shift;
353 open my $input, '<', ref $_[0] ? $_[0] : \($_[0]);
354 return $self->parse_byte_stream ($charset_name, $input, @_[1..$#_]);
355 } # parse_byte_string
356
357 sub parse_byte_stream ($$$$;$) {
358 my $self = ref $_[0] ? shift : shift->new;
359 my $charset_name = shift;
360 my $byte_stream = $_[0];
361
362 my $onerror = $_[2] || sub {
363 my (%opt) = @_;
364 warn "Parse error ($opt{type})\n";
365 };
366 $self->{parse_error} = $onerror; # updated later by parse_char_string
367
368 ## HTML5 encoding sniffing algorithm
369 require Message::Charset::Info;
370 my $charset;
371 my $buffer;
372 my ($char_stream, $e_status);
373
374 SNIFFING: {
375
376 ## Step 1
377 if (defined $charset_name) {
378 $charset = Message::Charset::Info->get_by_iana_name ($charset_name);
379
380 ## ISSUE: Unsupported encoding is not ignored according to the spec.
381 ($char_stream, $e_status) = $charset->get_decode_handle
382 ($byte_stream, allow_error_reporting => 1,
383 allow_fallback => 1);
384 if ($char_stream) {
385 $self->{confident} = 1;
386 last SNIFFING;
387 } else {
388 ## TODO: unsupported error
389 }
390 }
391
392 ## Step 2
393 my $byte_buffer = '';
394 for (1..1024) {
395 my $char = $byte_stream->getc;
396 last unless defined $char;
397 $byte_buffer .= $char;
398 } ## TODO: timeout
399
400 ## Step 3
401 if ($byte_buffer =~ /^\xFE\xFF/) {
402 $charset = Message::Charset::Info->get_by_iana_name ('utf-16be');
403 ($char_stream, $e_status) = $charset->get_decode_handle
404 ($byte_stream, allow_error_reporting => 1,
405 allow_fallback => 1, byte_buffer => \$byte_buffer);
406 $self->{confident} = 1;
407 last SNIFFING;
408 } elsif ($byte_buffer =~ /^\xFF\xFE/) {
409 $charset = Message::Charset::Info->get_by_iana_name ('utf-16le');
410 ($char_stream, $e_status) = $charset->get_decode_handle
411 ($byte_stream, allow_error_reporting => 1,
412 allow_fallback => 1, byte_buffer => \$byte_buffer);
413 $self->{confident} = 1;
414 last SNIFFING;
415 } elsif ($byte_buffer =~ /^\xEF\xBB\xBF/) {
416 $charset = Message::Charset::Info->get_by_iana_name ('utf-8');
417 ($char_stream, $e_status) = $charset->get_decode_handle
418 ($byte_stream, allow_error_reporting => 1,
419 allow_fallback => 1, byte_buffer => \$byte_buffer);
420 $self->{confident} = 1;
421 last SNIFFING;
422 }
423
424 ## Step 4
425 ## TODO: <meta charset>
426
427 ## Step 5
428 ## TODO: from history
429
430 ## Step 6
431 require Whatpm::Charset::UniversalCharDet;
432 $charset_name = Whatpm::Charset::UniversalCharDet->detect_byte_string
433 ($byte_buffer);
434 if (defined $charset_name) {
435 $charset = Message::Charset::Info->get_by_iana_name ($charset_name);
436
437 ## ISSUE: Unsupported encoding is not ignored according to the spec.
438 require Whatpm::Charset::DecodeHandle;
439 $buffer = Whatpm::Charset::DecodeHandle::ByteBuffer->new
440 ($byte_stream);
441 ($char_stream, $e_status) = $charset->get_decode_handle
442 ($buffer, allow_error_reporting => 1,
443 allow_fallback => 1, byte_buffer => \$byte_buffer);
444 if ($char_stream) {
445 $buffer->{buffer} = $byte_buffer;
446 !!!parse-error (type => 'sniffing:chardet', ## TODO: type name
447 value => $charset_name,
448 level => $self->{info_level},
449 line => 1, column => 1);
450 $self->{confident} = 0;
451 last SNIFFING;
452 }
453 }
454
455 ## Step 7: default
456 ## TODO: Make this configurable.
457 $charset = Message::Charset::Info->get_by_iana_name ('windows-1252');
458 ## NOTE: We choose |windows-1252| here, since |utf-8| should be
459 ## detectable in the step 6.
460 require Whatpm::Charset::DecodeHandle;
461 $buffer = Whatpm::Charset::DecodeHandle::ByteBuffer->new
462 ($byte_stream);
463 ($char_stream, $e_status)
464 = $charset->get_decode_handle ($buffer,
465 allow_error_reporting => 1,
466 allow_fallback => 1,
467 byte_buffer => \$byte_buffer);
468 $buffer->{buffer} = $byte_buffer;
469 !!!parse-error (type => 'sniffing:default', ## TODO: type name
470 value => 'windows-1252',
471 level => $self->{info_level},
472 line => 1, column => 1);
473 $self->{confident} = 0;
474 } # SNIFFING
475
476 $self->{input_encoding} = $charset->get_iana_name;
477 if ($e_status & Message::Charset::Info::FALLBACK_ENCODING_IMPL ()) {
478 !!!parse-error (type => 'chardecode:fallback', ## TODO: type name
479 value => $self->{input_encoding},
480 level => $self->{unsupported_level},
481 line => 1, column => 1);
482 } elsif (not ($e_status &
483 Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL())) {
484 !!!parse-error (type => 'chardecode:no error', ## TODO: type name
485 value => $self->{input_encoding},
486 level => $self->{unsupported_level},
487 line => 1, column => 1);
488 }
489
490 $self->{change_encoding} = sub {
491 my $self = shift;
492 $charset_name = shift;
493 my $token = shift;
494
495 $charset = Message::Charset::Info->get_by_iana_name ($charset_name);
496 ($char_stream, $e_status) = $charset->get_decode_handle
497 ($byte_stream, allow_error_reporting => 1, allow_fallback => 1,
498 byte_buffer => \ $buffer->{buffer});
499
500 if ($char_stream) { # if supported
501 ## "Change the encoding" algorithm:
502
503 ## Step 1
504 if ($charset->{category} &
505 Message::Charset::Info::CHARSET_CATEGORY_UTF16 ()) {
506 $charset = Message::Charset::Info->get_by_iana_name ('utf-8');
507 ($char_stream, $e_status) = $charset->get_decode_handle
508 ($byte_stream,
509 byte_buffer => \ $buffer->{buffer});
510 }
511 $charset_name = $charset->get_iana_name;
512
513 ## Step 2
514 if (defined $self->{input_encoding} and
515 $self->{input_encoding} eq $charset_name) {
516 !!!parse-error (type => 'charset label:matching', ## TODO: type
517 value => $charset_name,
518 level => $self->{info_level});
519 $self->{confident} = 1;
520 return;
521 }
522
523 !!!parse-error (type => 'charset label detected:'.$self->{input_encoding}.
524 ':'.$charset_name, level => 'w', token => $token);
525
526 ## Step 3
527 # if (can) {
528 ## change the encoding on the fly.
529 #$self->{confident} = 1;
530 #return;
531 # }
532
533 ## Step 4
534 throw Whatpm::HTML::RestartParser ();
535 }
536 }; # $self->{change_encoding}
537
538 my $char_onerror = sub {
539 my (undef, $type, %opt) = @_;
540 !!!parse-error (%opt, type => $type,
541 line => $self->{line}, column => $self->{column} + 1);
542 if ($opt{octets}) {
543 ${$opt{octets}} = "\x{FFFD}"; # relacement character
544 }
545 };
546 $char_stream->onerror ($char_onerror);
547
548 my @args = @_; shift @args; # $s
549 my $return;
550 try {
551 $return = $self->parse_char_stream ($char_stream, @args);
552 } catch Whatpm::HTML::RestartParser with {
553 ## NOTE: Invoked after {change_encoding}.
554
555 $self->{input_encoding} = $charset->get_iana_name;
556 if ($e_status & Message::Charset::Info::FALLBACK_ENCODING_IMPL ()) {
557 !!!parse-error (type => 'chardecode:fallback', ## TODO: type name
558 value => $self->{input_encoding},
559 level => $self->{unsupported_level},
560 line => 1, column => 1);
561 } elsif (not ($e_status &
562 Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL())) {
563 !!!parse-error (type => 'chardecode:no error', ## TODO: type name
564 value => $self->{input_encoding},
565 level => $self->{unsupported_level},
566 line => 1, column => 1);
567 }
568 $self->{confident} = 1;
569 $char_stream->onerror ($char_onerror);
570 $return = $self->parse_char_stream ($char_stream, @args);
571 };
572 return $return;
573 } # parse_byte_stream
574
575 ## NOTE: HTML5 spec says that the encoding layer MUST NOT strip BOM
576 ## and the HTML layer MUST ignore it. However, we does strip BOM in
577 ## the encoding layer and the HTML layer does not ignore any U+FEFF,
578 ## because the core part of our HTML parser expects a string of character,
579 ## not a string of bytes or code units or anything which might contain a BOM.
580 ## Therefore, any parser interface that accepts a string of bytes,
581 ## such as |parse_byte_string| in this module, must ensure that it does
582 ## strip the BOM and never strip any ZWNBSP.
583
584 sub parse_char_string ($$$;$) {
585 my $self = shift;
586 require utf8;
587 my $s = ref $_[0] ? $_[0] : \($_[0]);
588 open my $input, '<' . (utf8::is_utf8 ($$s) ? ':utf8' : ''), $s;
589 return $self->parse_char_stream ($input, @_[1..$#_]);
590 } # parse_char_string
591 *parse_string = \&parse_char_string;
592
593 sub parse_char_stream ($$$;$) {
594 my $self = ref $_[0] ? shift : shift->new;
595 my $input = $_[0];
596 $self->{document} = $_[1];
597 @{$self->{document}->child_nodes} = ();
598
599 ## NOTE: |set_inner_html| copies most of this method's code
600
601 $self->{confident} = 1 unless exists $self->{confident};
602 $self->{document}->input_encoding ($self->{input_encoding})
603 if defined $self->{input_encoding};
604
605 my $i = 0;
606 $self->{line_prev} = $self->{line} = 1;
607 $self->{column_prev} = $self->{column} = 0;
608 $self->{set_next_char} = sub {
609 my $self = shift;
610
611 pop @{$self->{prev_char}};
612 unshift @{$self->{prev_char}}, $self->{next_char};
613
614 my $char;
615 if (defined $self->{next_next_char}) {
616 $char = $self->{next_next_char};
617 delete $self->{next_next_char};
618 } else {
619 $char = $input->getc;
620 }
621 $self->{next_char} = -1 and return unless defined $char;
622 $self->{next_char} = ord $char;
623
624 ($self->{line_prev}, $self->{column_prev})
625 = ($self->{line}, $self->{column});
626 $self->{column}++;
627
628 if ($self->{next_char} == 0x000A) { # LF
629 !!!cp ('j1');
630 $self->{line}++;
631 $self->{column} = 0;
632 } elsif ($self->{next_char} == 0x000D) { # CR
633 !!!cp ('j2');
634 my $next = $input->getc;
635 if (defined $next and $next ne "\x0A") {
636 $self->{next_next_char} = $next;
637 }
638 $self->{next_char} = 0x000A; # LF # MUST
639 $self->{line}++;
640 $self->{column} = 0;
641 } elsif ($self->{next_char} > 0x10FFFF) {
642 !!!cp ('j3');
643 $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
644 } elsif ($self->{next_char} == 0x0000) { # NULL
645 !!!cp ('j4');
646 !!!parse-error (type => 'NULL');
647 $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
648 } elsif ($self->{next_char} <= 0x0008 or
649 (0x000E <= $self->{next_char} and $self->{next_char} <= 0x001F) or
650 (0x007F <= $self->{next_char} and $self->{next_char} <= 0x009F) or
651 (0xD800 <= $self->{next_char} and $self->{next_char} <= 0xDFFF) or
652 (0xFDD0 <= $self->{next_char} and $self->{next_char} <= 0xFDDF) or
653 {
654 0xFFFE => 1, 0xFFFF => 1, 0x1FFFE => 1, 0x1FFFF => 1,
655 0x2FFFE => 1, 0x2FFFF => 1, 0x3FFFE => 1, 0x3FFFF => 1,
656 0x4FFFE => 1, 0x4FFFF => 1, 0x5FFFE => 1, 0x5FFFF => 1,
657 0x6FFFE => 1, 0x6FFFF => 1, 0x7FFFE => 1, 0x7FFFF => 1,
658 0x8FFFE => 1, 0x8FFFF => 1, 0x9FFFE => 1, 0x9FFFF => 1,
659 0xAFFFE => 1, 0xAFFFF => 1, 0xBFFFE => 1, 0xBFFFF => 1,
660 0xCFFFE => 1, 0xCFFFF => 1, 0xDFFFE => 1, 0xDFFFF => 1,
661 0xEFFFE => 1, 0xEFFFF => 1, 0xFFFFE => 1, 0xFFFFF => 1,
662 0x10FFFE => 1, 0x10FFFF => 1,
663 }->{$self->{next_char}}) {
664 !!!cp ('j5');
665 !!!parse-error (type => 'control char', level => $self->{must_level});
666 ## TODO: error type documentation
667 }
668 };
669 $self->{prev_char} = [-1, -1, -1];
670 $self->{next_char} = -1;
671
672 my $onerror = $_[2] || sub {
673 my (%opt) = @_;
674 my $line = $opt{token} ? $opt{token}->{line} : $opt{line};
675 my $column = $opt{token} ? $opt{token}->{column} : $opt{column};
676 warn "Parse error ($opt{type}) at line $line column $column\n";
677 };
678 $self->{parse_error} = sub {
679 $onerror->(line => $self->{line}, column => $self->{column}, @_);
680 };
681
682 $self->_initialize_tokenizer;
683 $self->_initialize_tree_constructor;
684 $self->_construct_tree;
685 $self->_terminate_tree_constructor;
686
687 delete $self->{parse_error}; # remove loop
688
689 return $self->{document};
690 } # parse_char_stream
691
692 sub new ($) {
693 my $class = shift;
694 my $self = bless {
695 must_level => 'm',
696 should_level => 's',
697 good_level => 'w',
698 warn_level => 'w',
699 info_level => 'i',
700 unsupported_level => 'u',
701 }, $class;
702 $self->{set_next_char} = sub {
703 $self->{next_char} = -1;
704 };
705 $self->{parse_error} = sub {
706 #
707 };
708 $self->{change_encoding} = sub {
709 # if ($_[0] is a supported encoding) {
710 # run "change the encoding" algorithm;
711 # throw Whatpm::HTML::RestartParser (charset => $new_encoding);
712 # }
713 };
714 $self->{application_cache_selection} = sub {
715 #
716 };
717 return $self;
718 } # new
719
720 sub CM_ENTITY () { 0b001 } # & markup in data
721 sub CM_LIMITED_MARKUP () { 0b010 } # < markup in data (limited)
722 sub CM_FULL_MARKUP () { 0b100 } # < markup in data (any)
723
724 sub PLAINTEXT_CONTENT_MODEL () { 0 }
725 sub CDATA_CONTENT_MODEL () { CM_LIMITED_MARKUP }
726 sub RCDATA_CONTENT_MODEL () { CM_ENTITY | CM_LIMITED_MARKUP }
727 sub PCDATA_CONTENT_MODEL () { CM_ENTITY | CM_FULL_MARKUP }
728
729 sub DATA_STATE () { 0 }
730 sub ENTITY_DATA_STATE () { 1 }
731 sub TAG_OPEN_STATE () { 2 }
732 sub CLOSE_TAG_OPEN_STATE () { 3 }
733 sub TAG_NAME_STATE () { 4 }
734 sub BEFORE_ATTRIBUTE_NAME_STATE () { 5 }
735 sub ATTRIBUTE_NAME_STATE () { 6 }
736 sub AFTER_ATTRIBUTE_NAME_STATE () { 7 }
737 sub BEFORE_ATTRIBUTE_VALUE_STATE () { 8 }
738 sub ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE () { 9 }
739 sub ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE () { 10 }
740 sub ATTRIBUTE_VALUE_UNQUOTED_STATE () { 11 }
741 sub ENTITY_IN_ATTRIBUTE_VALUE_STATE () { 12 }
742 sub MARKUP_DECLARATION_OPEN_STATE () { 13 }
743 sub COMMENT_START_STATE () { 14 }
744 sub COMMENT_START_DASH_STATE () { 15 }
745 sub COMMENT_STATE () { 16 }
746 sub COMMENT_END_STATE () { 17 }
747 sub COMMENT_END_DASH_STATE () { 18 }
748 sub BOGUS_COMMENT_STATE () { 19 }
749 sub DOCTYPE_STATE () { 20 }
750 sub BEFORE_DOCTYPE_NAME_STATE () { 21 }
751 sub DOCTYPE_NAME_STATE () { 22 }
752 sub AFTER_DOCTYPE_NAME_STATE () { 23 }
753 sub BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE () { 24 }
754 sub DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE () { 25 }
755 sub DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE () { 26 }
756 sub AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE () { 27 }
757 sub BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE () { 28 }
758 sub DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE () { 29 }
759 sub DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE () { 30 }
760 sub AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE () { 31 }
761 sub BOGUS_DOCTYPE_STATE () { 32 }
762 sub AFTER_ATTRIBUTE_VALUE_QUOTED_STATE () { 33 }
763 sub SELF_CLOSING_START_TAG_STATE () { 34 }
764 sub CDATA_BLOCK_STATE () { 35 }
765
766 sub DOCTYPE_TOKEN () { 1 }
767 sub COMMENT_TOKEN () { 2 }
768 sub START_TAG_TOKEN () { 3 }
769 sub END_TAG_TOKEN () { 4 }
770 sub END_OF_FILE_TOKEN () { 5 }
771 sub CHARACTER_TOKEN () { 6 }
772
773 sub AFTER_HTML_IMS () { 0b100 }
774 sub HEAD_IMS () { 0b1000 }
775 sub BODY_IMS () { 0b10000 }
776 sub BODY_TABLE_IMS () { 0b100000 }
777 sub TABLE_IMS () { 0b1000000 }
778 sub ROW_IMS () { 0b10000000 }
779 sub BODY_AFTER_IMS () { 0b100000000 }
780 sub FRAME_IMS () { 0b1000000000 }
781 sub SELECT_IMS () { 0b10000000000 }
782 sub IN_FOREIGN_CONTENT_IM () { 0b100000000000 }
783 ## NOTE: "in foreign content" insertion mode is special; it is combined
784 ## with the secondary insertion mode. In this parser, they are stored
785 ## together in the bit-or'ed form.
786
787 ## NOTE: "initial" and "before html" insertion modes have no constants.
788
789 ## NOTE: "after after body" insertion mode.
790 sub AFTER_HTML_BODY_IM () { AFTER_HTML_IMS | BODY_AFTER_IMS }
791
792 ## NOTE: "after after frameset" insertion mode.
793 sub AFTER_HTML_FRAMESET_IM () { AFTER_HTML_IMS | FRAME_IMS }
794
795 sub IN_HEAD_IM () { HEAD_IMS | 0b00 }
796 sub IN_HEAD_NOSCRIPT_IM () { HEAD_IMS | 0b01 }
797 sub AFTER_HEAD_IM () { HEAD_IMS | 0b10 }
798 sub BEFORE_HEAD_IM () { HEAD_IMS | 0b11 }
799 sub IN_BODY_IM () { BODY_IMS }
800 sub IN_CELL_IM () { BODY_IMS | BODY_TABLE_IMS | 0b01 }
801 sub IN_CAPTION_IM () { BODY_IMS | BODY_TABLE_IMS | 0b10 }
802 sub IN_ROW_IM () { TABLE_IMS | ROW_IMS | 0b01 }
803 sub IN_TABLE_BODY_IM () { TABLE_IMS | ROW_IMS | 0b10 }
804 sub IN_TABLE_IM () { TABLE_IMS }
805 sub AFTER_BODY_IM () { BODY_AFTER_IMS }
806 sub IN_FRAMESET_IM () { FRAME_IMS | 0b01 }
807 sub AFTER_FRAMESET_IM () { FRAME_IMS | 0b10 }
808 sub IN_SELECT_IM () { SELECT_IMS | 0b01 }
809 sub IN_SELECT_IN_TABLE_IM () { SELECT_IMS | 0b10 }
810 sub IN_COLUMN_GROUP_IM () { 0b10 }
811
812 ## Implementations MUST act as if state machine in the spec
813
814 sub _initialize_tokenizer ($) {
815 my $self = shift;
816 $self->{state} = DATA_STATE; # MUST
817 $self->{content_model} = PCDATA_CONTENT_MODEL; # be
818 undef $self->{current_token}; # start tag, end tag, comment, or DOCTYPE
819 undef $self->{current_attribute};
820 undef $self->{last_emitted_start_tag_name};
821 undef $self->{last_attribute_value_state};
822 delete $self->{self_closing};
823 $self->{char} = [];
824 # $self->{next_char}
825 !!!next-input-character;
826 $self->{token} = [];
827 # $self->{escape}
828 } # _initialize_tokenizer
829
830 ## A token has:
831 ## ->{type} == DOCTYPE_TOKEN, START_TAG_TOKEN, END_TAG_TOKEN, COMMENT_TOKEN,
832 ## CHARACTER_TOKEN, or END_OF_FILE_TOKEN
833 ## ->{name} (DOCTYPE_TOKEN)
834 ## ->{tag_name} (START_TAG_TOKEN, END_TAG_TOKEN)
835 ## ->{public_identifier} (DOCTYPE_TOKEN)
836 ## ->{system_identifier} (DOCTYPE_TOKEN)
837 ## ->{quirks} == 1 or 0 (DOCTYPE_TOKEN): "force-quirks" flag
838 ## ->{attributes} isa HASH (START_TAG_TOKEN, END_TAG_TOKEN)
839 ## ->{name}
840 ## ->{value}
841 ## ->{has_reference} == 1 or 0
842 ## ->{data} (COMMENT_TOKEN, CHARACTER_TOKEN)
843 ## NOTE: The "self-closing flag" is hold as |$self->{self_closing}|.
844 ## |->{self_closing}| is used to save the value of |$self->{self_closing}|
845 ## while the token is pushed back to the stack.
846
847 ## Emitted token MUST immediately be handled by the tree construction state.
848
849 ## Before each step, UA MAY check to see if either one of the scripts in
850 ## "list of scripts that will execute as soon as possible" or the first
851 ## script in the "list of scripts that will execute asynchronously",
852 ## has completed loading. If one has, then it MUST be executed
853 ## and removed from the list.
854
855 ## NOTE: HTML5 "Writing HTML documents" section, applied to
856 ## documents and not to user agents and conformance checkers,
857 ## contains some requirements that are not detected by the
858 ## parsing algorithm:
859 ## - Some requirements on character encoding declarations. ## TODO
860 ## - "Elements MUST NOT contain content that their content model disallows."
861 ## ... Some are parse error, some are not (will be reported by c.c.).
862 ## - Polytheistic slash SHOULD NOT be used. (Applied only to atheists.) ## TODO
863 ## - Text (in elements, attributes, and comments) SHOULD NOT contain
864 ## control characters other than space characters. ## TODO: (what is control character? C0, C1 and DEL? Unicode control character?)
865
866 ## TODO: HTML5 poses authors two SHOULD-level requirements that cannot
867 ## be detected by the HTML5 parsing algorithm:
868 ## - Text,
869
870 sub _get_next_token ($) {
871 my $self = shift;
872
873 if ($self->{self_closing}) {
874 !!!parse-error (type => 'nestc', token => $self->{current_token});
875 ## NOTE: The |self_closing| flag is only set by start tag token.
876 ## In addition, when a start tag token is emitted, it is always set to
877 ## |current_token|.
878 delete $self->{self_closing};
879 }
880
881 if (@{$self->{token}}) {
882 $self->{self_closing} = $self->{token}->[0]->{self_closing};
883 return shift @{$self->{token}};
884 }
885
886 A: {
887 if ($self->{state} == DATA_STATE) {
888 if ($self->{next_char} == 0x0026) { # &
889 if ($self->{content_model} & CM_ENTITY and # PCDATA | RCDATA
890 not $self->{escape}) {
891 !!!cp (1);
892 $self->{state} = ENTITY_DATA_STATE;
893 !!!next-input-character;
894 redo A;
895 } else {
896 !!!cp (2);
897 #
898 }
899 } elsif ($self->{next_char} == 0x002D) { # -
900 if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
901 unless ($self->{escape}) {
902 if ($self->{prev_char}->[0] == 0x002D and # -
903 $self->{prev_char}->[1] == 0x0021 and # !
904 $self->{prev_char}->[2] == 0x003C) { # <
905 !!!cp (3);
906 $self->{escape} = 1;
907 } else {
908 !!!cp (4);
909 }
910 } else {
911 !!!cp (5);
912 }
913 }
914
915 #
916 } elsif ($self->{next_char} == 0x003C) { # <
917 if ($self->{content_model} & CM_FULL_MARKUP or # PCDATA
918 (($self->{content_model} & CM_LIMITED_MARKUP) and # CDATA | RCDATA
919 not $self->{escape})) {
920 !!!cp (6);
921 $self->{state} = TAG_OPEN_STATE;
922 !!!next-input-character;
923 redo A;
924 } else {
925 !!!cp (7);
926 #
927 }
928 } elsif ($self->{next_char} == 0x003E) { # >
929 if ($self->{escape} and
930 ($self->{content_model} & CM_LIMITED_MARKUP)) { # RCDATA | CDATA
931 if ($self->{prev_char}->[0] == 0x002D and # -
932 $self->{prev_char}->[1] == 0x002D) { # -
933 !!!cp (8);
934 delete $self->{escape};
935 } else {
936 !!!cp (9);
937 }
938 } else {
939 !!!cp (10);
940 }
941
942 #
943 } elsif ($self->{next_char} == -1) {
944 !!!cp (11);
945 !!!emit ({type => END_OF_FILE_TOKEN,
946 line => $self->{line}, column => $self->{column}});
947 last A; ## TODO: ok?
948 } else {
949 !!!cp (12);
950 }
951 # Anything else
952 my $token = {type => CHARACTER_TOKEN,
953 data => chr $self->{next_char},
954 line => $self->{line}, column => $self->{column},
955 };
956 ## Stay in the data state
957 !!!next-input-character;
958
959 !!!emit ($token);
960
961 redo A;
962 } elsif ($self->{state} == ENTITY_DATA_STATE) {
963 ## (cannot happen in CDATA state)
964
965 my ($l, $c) = ($self->{line_prev}, $self->{column_prev});
966
967 my $token = $self->_tokenize_attempt_to_consume_an_entity (0, -1);
968
969 $self->{state} = DATA_STATE;
970 # next-input-character is already done
971
972 unless (defined $token) {
973 !!!cp (13);
974 !!!emit ({type => CHARACTER_TOKEN, data => '&',
975 line => $l, column => $c,
976 });
977 } else {
978 !!!cp (14);
979 !!!emit ($token);
980 }
981
982 redo A;
983 } elsif ($self->{state} == TAG_OPEN_STATE) {
984 if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
985 if ($self->{next_char} == 0x002F) { # /
986 !!!cp (15);
987 !!!next-input-character;
988 $self->{state} = CLOSE_TAG_OPEN_STATE;
989 redo A;
990 } else {
991 !!!cp (16);
992 ## reconsume
993 $self->{state} = DATA_STATE;
994
995 !!!emit ({type => CHARACTER_TOKEN, data => '<',
996 line => $self->{line_prev},
997 column => $self->{column_prev},
998 });
999
1000 redo A;
1001 }
1002 } elsif ($self->{content_model} & CM_FULL_MARKUP) { # PCDATA
1003 if ($self->{next_char} == 0x0021) { # !
1004 !!!cp (17);
1005 $self->{state} = MARKUP_DECLARATION_OPEN_STATE;
1006 !!!next-input-character;
1007 redo A;
1008 } elsif ($self->{next_char} == 0x002F) { # /
1009 !!!cp (18);
1010 $self->{state} = CLOSE_TAG_OPEN_STATE;
1011 !!!next-input-character;
1012 redo A;
1013 } elsif (0x0041 <= $self->{next_char} and
1014 $self->{next_char} <= 0x005A) { # A..Z
1015 !!!cp (19);
1016 $self->{current_token}
1017 = {type => START_TAG_TOKEN,
1018 tag_name => chr ($self->{next_char} + 0x0020),
1019 line => $self->{line_prev},
1020 column => $self->{column_prev}};
1021 $self->{state} = TAG_NAME_STATE;
1022 !!!next-input-character;
1023 redo A;
1024 } elsif (0x0061 <= $self->{next_char} and
1025 $self->{next_char} <= 0x007A) { # a..z
1026 !!!cp (20);
1027 $self->{current_token} = {type => START_TAG_TOKEN,
1028 tag_name => chr ($self->{next_char}),
1029 line => $self->{line_prev},
1030 column => $self->{column_prev}};
1031 $self->{state} = TAG_NAME_STATE;
1032 !!!next-input-character;
1033 redo A;
1034 } elsif ($self->{next_char} == 0x003E) { # >
1035 !!!cp (21);
1036 !!!parse-error (type => 'empty start tag',
1037 line => $self->{line_prev},
1038 column => $self->{column_prev});
1039 $self->{state} = DATA_STATE;
1040 !!!next-input-character;
1041
1042 !!!emit ({type => CHARACTER_TOKEN, data => '<>',
1043 line => $self->{line_prev},
1044 column => $self->{column_prev},
1045 });
1046
1047 redo A;
1048 } elsif ($self->{next_char} == 0x003F) { # ?
1049 !!!cp (22);
1050 !!!parse-error (type => 'pio',
1051 line => $self->{line_prev},
1052 column => $self->{column_prev});
1053 $self->{state} = BOGUS_COMMENT_STATE;
1054 $self->{current_token} = {type => COMMENT_TOKEN, data => '',
1055 line => $self->{line_prev},
1056 column => $self->{column_prev},
1057 };
1058 ## $self->{next_char} is intentionally left as is
1059 redo A;
1060 } else {
1061 !!!cp (23);
1062 !!!parse-error (type => 'bare stago',
1063 line => $self->{line_prev},
1064 column => $self->{column_prev});
1065 $self->{state} = DATA_STATE;
1066 ## reconsume
1067
1068 !!!emit ({type => CHARACTER_TOKEN, data => '<',
1069 line => $self->{line_prev},
1070 column => $self->{column_prev},
1071 });
1072
1073 redo A;
1074 }
1075 } else {
1076 die "$0: $self->{content_model} in tag open";
1077 }
1078 } elsif ($self->{state} == CLOSE_TAG_OPEN_STATE) {
1079 my ($l, $c) = ($self->{line_prev}, $self->{column_prev} - 1); # "<"of"</"
1080 if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
1081 if (defined $self->{last_emitted_start_tag_name}) {
1082
1083 ## NOTE: <http://krijnhoetmer.nl/irc-logs/whatwg/20070626#l-564>
1084 my @next_char;
1085 TAGNAME: for (my $i = 0; $i < length $self->{last_emitted_start_tag_name}; $i++) {
1086 push @next_char, $self->{next_char};
1087 my $c = ord substr ($self->{last_emitted_start_tag_name}, $i, 1);
1088 my $C = 0x0061 <= $c && $c <= 0x007A ? $c - 0x0020 : $c;
1089 if ($self->{next_char} == $c or $self->{next_char} == $C) {
1090 !!!cp (24);
1091 !!!next-input-character;
1092 next TAGNAME;
1093 } else {
1094 !!!cp (25);
1095 $self->{next_char} = shift @next_char; # reconsume
1096 !!!back-next-input-character (@next_char);
1097 $self->{state} = DATA_STATE;
1098
1099 !!!emit ({type => CHARACTER_TOKEN, data => '</',
1100 line => $l, column => $c,
1101 });
1102
1103 redo A;
1104 }
1105 }
1106 push @next_char, $self->{next_char};
1107
1108 unless ($self->{next_char} == 0x0009 or # HT
1109 $self->{next_char} == 0x000A or # LF
1110 $self->{next_char} == 0x000B or # VT
1111 $self->{next_char} == 0x000C or # FF
1112 $self->{next_char} == 0x0020 or # SP
1113 $self->{next_char} == 0x003E or # >
1114 $self->{next_char} == 0x002F or # /
1115 $self->{next_char} == -1) {
1116 !!!cp (26);
1117 $self->{next_char} = shift @next_char; # reconsume
1118 !!!back-next-input-character (@next_char);
1119 $self->{state} = DATA_STATE;
1120 !!!emit ({type => CHARACTER_TOKEN, data => '</',
1121 line => $l, column => $c,
1122 });
1123 redo A;
1124 } else {
1125 !!!cp (27);
1126 $self->{next_char} = shift @next_char;
1127 !!!back-next-input-character (@next_char);
1128 # and consume...
1129 }
1130 } else {
1131 ## No start tag token has ever been emitted
1132 !!!cp (28);
1133 # next-input-character is already done
1134 $self->{state} = DATA_STATE;
1135 !!!emit ({type => CHARACTER_TOKEN, data => '</',
1136 line => $l, column => $c,
1137 });
1138 redo A;
1139 }
1140 }
1141
1142 if (0x0041 <= $self->{next_char} and
1143 $self->{next_char} <= 0x005A) { # A..Z
1144 !!!cp (29);
1145 $self->{current_token}
1146 = {type => END_TAG_TOKEN,
1147 tag_name => chr ($self->{next_char} + 0x0020),
1148 line => $l, column => $c};
1149 $self->{state} = TAG_NAME_STATE;
1150 !!!next-input-character;
1151 redo A;
1152 } elsif (0x0061 <= $self->{next_char} and
1153 $self->{next_char} <= 0x007A) { # a..z
1154 !!!cp (30);
1155 $self->{current_token} = {type => END_TAG_TOKEN,
1156 tag_name => chr ($self->{next_char}),
1157 line => $l, column => $c};
1158 $self->{state} = TAG_NAME_STATE;
1159 !!!next-input-character;
1160 redo A;
1161 } elsif ($self->{next_char} == 0x003E) { # >
1162 !!!cp (31);
1163 !!!parse-error (type => 'empty end tag',
1164 line => $self->{line_prev}, ## "<" in "</>"
1165 column => $self->{column_prev} - 1);
1166 $self->{state} = DATA_STATE;
1167 !!!next-input-character;
1168 redo A;
1169 } elsif ($self->{next_char} == -1) {
1170 !!!cp (32);
1171 !!!parse-error (type => 'bare etago');
1172 $self->{state} = DATA_STATE;
1173 # reconsume
1174
1175 !!!emit ({type => CHARACTER_TOKEN, data => '</',
1176 line => $l, column => $c,
1177 });
1178
1179 redo A;
1180 } else {
1181 !!!cp (33);
1182 !!!parse-error (type => 'bogus end tag');
1183 $self->{state} = BOGUS_COMMENT_STATE;
1184 $self->{current_token} = {type => COMMENT_TOKEN, data => '',
1185 line => $self->{line_prev}, # "<" of "</"
1186 column => $self->{column_prev} - 1,
1187 };
1188 ## $self->{next_char} is intentionally left as is
1189 redo A;
1190 }
1191 } elsif ($self->{state} == TAG_NAME_STATE) {
1192 if ($self->{next_char} == 0x0009 or # HT
1193 $self->{next_char} == 0x000A or # LF
1194 $self->{next_char} == 0x000B or # VT
1195 $self->{next_char} == 0x000C or # FF
1196 $self->{next_char} == 0x0020) { # SP
1197 !!!cp (34);
1198 $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1199 !!!next-input-character;
1200 redo A;
1201 } elsif ($self->{next_char} == 0x003E) { # >
1202 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1203 !!!cp (35);
1204 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1205 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1206 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1207 #if ($self->{current_token}->{attributes}) {
1208 # ## NOTE: This should never be reached.
1209 # !!! cp (36);
1210 # !!! parse-error (type => 'end tag attribute');
1211 #} else {
1212 !!!cp (37);
1213 #}
1214 } else {
1215 die "$0: $self->{current_token}->{type}: Unknown token type";
1216 }
1217 $self->{state} = DATA_STATE;
1218 !!!next-input-character;
1219
1220 !!!emit ($self->{current_token}); # start tag or end tag
1221
1222 redo A;
1223 } elsif (0x0041 <= $self->{next_char} and
1224 $self->{next_char} <= 0x005A) { # A..Z
1225 !!!cp (38);
1226 $self->{current_token}->{tag_name} .= chr ($self->{next_char} + 0x0020);
1227 # start tag or end tag
1228 ## Stay in this state
1229 !!!next-input-character;
1230 redo A;
1231 } elsif ($self->{next_char} == -1) {
1232 !!!parse-error (type => 'unclosed tag');
1233 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1234 !!!cp (39);
1235 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1236 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1237 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1238 #if ($self->{current_token}->{attributes}) {
1239 # ## NOTE: This state should never be reached.
1240 # !!! cp (40);
1241 # !!! parse-error (type => 'end tag attribute');
1242 #} else {
1243 !!!cp (41);
1244 #}
1245 } else {
1246 die "$0: $self->{current_token}->{type}: Unknown token type";
1247 }
1248 $self->{state} = DATA_STATE;
1249 # reconsume
1250
1251 !!!emit ($self->{current_token}); # start tag or end tag
1252
1253 redo A;
1254 } elsif ($self->{next_char} == 0x002F) { # /
1255 !!!cp (42);
1256 $self->{state} = SELF_CLOSING_START_TAG_STATE;
1257 !!!next-input-character;
1258 redo A;
1259 } else {
1260 !!!cp (44);
1261 $self->{current_token}->{tag_name} .= chr $self->{next_char};
1262 # start tag or end tag
1263 ## Stay in the state
1264 !!!next-input-character;
1265 redo A;
1266 }
1267 } elsif ($self->{state} == BEFORE_ATTRIBUTE_NAME_STATE) {
1268 if ($self->{next_char} == 0x0009 or # HT
1269 $self->{next_char} == 0x000A or # LF
1270 $self->{next_char} == 0x000B or # VT
1271 $self->{next_char} == 0x000C or # FF
1272 $self->{next_char} == 0x0020) { # SP
1273 !!!cp (45);
1274 ## Stay in the state
1275 !!!next-input-character;
1276 redo A;
1277 } elsif ($self->{next_char} == 0x003E) { # >
1278 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1279 !!!cp (46);
1280 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1281 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1282 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1283 if ($self->{current_token}->{attributes}) {
1284 !!!cp (47);
1285 !!!parse-error (type => 'end tag attribute');
1286 } else {
1287 !!!cp (48);
1288 }
1289 } else {
1290 die "$0: $self->{current_token}->{type}: Unknown token type";
1291 }
1292 $self->{state} = DATA_STATE;
1293 !!!next-input-character;
1294
1295 !!!emit ($self->{current_token}); # start tag or end tag
1296
1297 redo A;
1298 } elsif (0x0041 <= $self->{next_char} and
1299 $self->{next_char} <= 0x005A) { # A..Z
1300 !!!cp (49);
1301 $self->{current_attribute}
1302 = {name => chr ($self->{next_char} + 0x0020),
1303 value => '',
1304 line => $self->{line}, column => $self->{column}};
1305 $self->{state} = ATTRIBUTE_NAME_STATE;
1306 !!!next-input-character;
1307 redo A;
1308 } elsif ($self->{next_char} == 0x002F) { # /
1309 !!!cp (50);
1310 $self->{state} = SELF_CLOSING_START_TAG_STATE;
1311 !!!next-input-character;
1312 redo A;
1313 } elsif ($self->{next_char} == -1) {
1314 !!!parse-error (type => 'unclosed tag');
1315 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1316 !!!cp (52);
1317 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1318 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1319 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1320 if ($self->{current_token}->{attributes}) {
1321 !!!cp (53);
1322 !!!parse-error (type => 'end tag attribute');
1323 } else {
1324 !!!cp (54);
1325 }
1326 } else {
1327 die "$0: $self->{current_token}->{type}: Unknown token type";
1328 }
1329 $self->{state} = DATA_STATE;
1330 # reconsume
1331
1332 !!!emit ($self->{current_token}); # start tag or end tag
1333
1334 redo A;
1335 } else {
1336 if ({
1337 0x0022 => 1, # "
1338 0x0027 => 1, # '
1339 0x003D => 1, # =
1340 }->{$self->{next_char}}) {
1341 !!!cp (55);
1342 !!!parse-error (type => 'bad attribute name');
1343 } else {
1344 !!!cp (56);
1345 }
1346 $self->{current_attribute}
1347 = {name => chr ($self->{next_char}),
1348 value => '',
1349 line => $self->{line}, column => $self->{column}};
1350 $self->{state} = ATTRIBUTE_NAME_STATE;
1351 !!!next-input-character;
1352 redo A;
1353 }
1354 } elsif ($self->{state} == ATTRIBUTE_NAME_STATE) {
1355 my $before_leave = sub {
1356 if (exists $self->{current_token}->{attributes} # start tag or end tag
1357 ->{$self->{current_attribute}->{name}}) { # MUST
1358 !!!cp (57);
1359 !!!parse-error (type => 'duplicate attribute:'.$self->{current_attribute}->{name}, line => $self->{current_attribute}->{line}, column => $self->{current_attribute}->{column});
1360 ## Discard $self->{current_attribute} # MUST
1361 } else {
1362 !!!cp (58);
1363 $self->{current_token}->{attributes}->{$self->{current_attribute}->{name}}
1364 = $self->{current_attribute};
1365 }
1366 }; # $before_leave
1367
1368 if ($self->{next_char} == 0x0009 or # HT
1369 $self->{next_char} == 0x000A or # LF
1370 $self->{next_char} == 0x000B or # VT
1371 $self->{next_char} == 0x000C or # FF
1372 $self->{next_char} == 0x0020) { # SP
1373 !!!cp (59);
1374 $before_leave->();
1375 $self->{state} = AFTER_ATTRIBUTE_NAME_STATE;
1376 !!!next-input-character;
1377 redo A;
1378 } elsif ($self->{next_char} == 0x003D) { # =
1379 !!!cp (60);
1380 $before_leave->();
1381 $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;
1382 !!!next-input-character;
1383 redo A;
1384 } elsif ($self->{next_char} == 0x003E) { # >
1385 $before_leave->();
1386 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1387 !!!cp (61);
1388 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1389 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1390 !!!cp (62);
1391 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1392 if ($self->{current_token}->{attributes}) {
1393 !!!parse-error (type => 'end tag attribute');
1394 }
1395 } else {
1396 die "$0: $self->{current_token}->{type}: Unknown token type";
1397 }
1398 $self->{state} = DATA_STATE;
1399 !!!next-input-character;
1400
1401 !!!emit ($self->{current_token}); # start tag or end tag
1402
1403 redo A;
1404 } elsif (0x0041 <= $self->{next_char} and
1405 $self->{next_char} <= 0x005A) { # A..Z
1406 !!!cp (63);
1407 $self->{current_attribute}->{name} .= chr ($self->{next_char} + 0x0020);
1408 ## Stay in the state
1409 !!!next-input-character;
1410 redo A;
1411 } elsif ($self->{next_char} == 0x002F) { # /
1412 !!!cp (64);
1413 $before_leave->();
1414 $self->{state} = SELF_CLOSING_START_TAG_STATE;
1415 !!!next-input-character;
1416 redo A;
1417 } elsif ($self->{next_char} == -1) {
1418 !!!parse-error (type => 'unclosed tag');
1419 $before_leave->();
1420 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1421 !!!cp (66);
1422 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1423 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1424 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1425 if ($self->{current_token}->{attributes}) {
1426 !!!cp (67);
1427 !!!parse-error (type => 'end tag attribute');
1428 } else {
1429 ## NOTE: This state should never be reached.
1430 !!!cp (68);
1431 }
1432 } else {
1433 die "$0: $self->{current_token}->{type}: Unknown token type";
1434 }
1435 $self->{state} = DATA_STATE;
1436 # reconsume
1437
1438 !!!emit ($self->{current_token}); # start tag or end tag
1439
1440 redo A;
1441 } else {
1442 if ($self->{next_char} == 0x0022 or # "
1443 $self->{next_char} == 0x0027) { # '
1444 !!!cp (69);
1445 !!!parse-error (type => 'bad attribute name');
1446 } else {
1447 !!!cp (70);
1448 }
1449 $self->{current_attribute}->{name} .= chr ($self->{next_char});
1450 ## Stay in the state
1451 !!!next-input-character;
1452 redo A;
1453 }
1454 } elsif ($self->{state} == AFTER_ATTRIBUTE_NAME_STATE) {
1455 if ($self->{next_char} == 0x0009 or # HT
1456 $self->{next_char} == 0x000A or # LF
1457 $self->{next_char} == 0x000B or # VT
1458 $self->{next_char} == 0x000C or # FF
1459 $self->{next_char} == 0x0020) { # SP
1460 !!!cp (71);
1461 ## Stay in the state
1462 !!!next-input-character;
1463 redo A;
1464 } elsif ($self->{next_char} == 0x003D) { # =
1465 !!!cp (72);
1466 $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;
1467 !!!next-input-character;
1468 redo A;
1469 } elsif ($self->{next_char} == 0x003E) { # >
1470 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1471 !!!cp (73);
1472 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1473 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1474 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1475 if ($self->{current_token}->{attributes}) {
1476 !!!cp (74);
1477 !!!parse-error (type => 'end tag attribute');
1478 } else {
1479 ## NOTE: This state should never be reached.
1480 !!!cp (75);
1481 }
1482 } else {
1483 die "$0: $self->{current_token}->{type}: Unknown token type";
1484 }
1485 $self->{state} = DATA_STATE;
1486 !!!next-input-character;
1487
1488 !!!emit ($self->{current_token}); # start tag or end tag
1489
1490 redo A;
1491 } elsif (0x0041 <= $self->{next_char} and
1492 $self->{next_char} <= 0x005A) { # A..Z
1493 !!!cp (76);
1494 $self->{current_attribute}
1495 = {name => chr ($self->{next_char} + 0x0020),
1496 value => '',
1497 line => $self->{line}, column => $self->{column}};
1498 $self->{state} = ATTRIBUTE_NAME_STATE;
1499 !!!next-input-character;
1500 redo A;
1501 } elsif ($self->{next_char} == 0x002F) { # /
1502 !!!cp (77);
1503 $self->{state} = SELF_CLOSING_START_TAG_STATE;
1504 !!!next-input-character;
1505 redo A;
1506 } elsif ($self->{next_char} == -1) {
1507 !!!parse-error (type => 'unclosed tag');
1508 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1509 !!!cp (79);
1510 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1511 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1512 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1513 if ($self->{current_token}->{attributes}) {
1514 !!!cp (80);
1515 !!!parse-error (type => 'end tag attribute');
1516 } else {
1517 ## NOTE: This state should never be reached.
1518 !!!cp (81);
1519 }
1520 } else {
1521 die "$0: $self->{current_token}->{type}: Unknown token type";
1522 }
1523 $self->{state} = DATA_STATE;
1524 # reconsume
1525
1526 !!!emit ($self->{current_token}); # start tag or end tag
1527
1528 redo A;
1529 } else {
1530 !!!cp (82);
1531 $self->{current_attribute}
1532 = {name => chr ($self->{next_char}),
1533 value => '',
1534 line => $self->{line}, column => $self->{column}};
1535 $self->{state} = ATTRIBUTE_NAME_STATE;
1536 !!!next-input-character;
1537 redo A;
1538 }
1539 } elsif ($self->{state} == BEFORE_ATTRIBUTE_VALUE_STATE) {
1540 if ($self->{next_char} == 0x0009 or # HT
1541 $self->{next_char} == 0x000A or # LF
1542 $self->{next_char} == 0x000B or # VT
1543 $self->{next_char} == 0x000C or # FF
1544 $self->{next_char} == 0x0020) { # SP
1545 !!!cp (83);
1546 ## Stay in the state
1547 !!!next-input-character;
1548 redo A;
1549 } elsif ($self->{next_char} == 0x0022) { # "
1550 !!!cp (84);
1551 $self->{state} = ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE;
1552 !!!next-input-character;
1553 redo A;
1554 } elsif ($self->{next_char} == 0x0026) { # &
1555 !!!cp (85);
1556 $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;
1557 ## reconsume
1558 redo A;
1559 } elsif ($self->{next_char} == 0x0027) { # '
1560 !!!cp (86);
1561 $self->{state} = ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE;
1562 !!!next-input-character;
1563 redo A;
1564 } elsif ($self->{next_char} == 0x003E) { # >
1565 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1566 !!!cp (87);
1567 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1568 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1569 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1570 if ($self->{current_token}->{attributes}) {
1571 !!!cp (88);
1572 !!!parse-error (type => 'end tag attribute');
1573 } else {
1574 ## NOTE: This state should never be reached.
1575 !!!cp (89);
1576 }
1577 } else {
1578 die "$0: $self->{current_token}->{type}: Unknown token type";
1579 }
1580 $self->{state} = DATA_STATE;
1581 !!!next-input-character;
1582
1583 !!!emit ($self->{current_token}); # start tag or end tag
1584
1585 redo A;
1586 } elsif ($self->{next_char} == -1) {
1587 !!!parse-error (type => 'unclosed tag');
1588 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1589 !!!cp (90);
1590 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1591 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1592 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1593 if ($self->{current_token}->{attributes}) {
1594 !!!cp (91);
1595 !!!parse-error (type => 'end tag attribute');
1596 } else {
1597 ## NOTE: This state should never be reached.
1598 !!!cp (92);
1599 }
1600 } else {
1601 die "$0: $self->{current_token}->{type}: Unknown token type";
1602 }
1603 $self->{state} = DATA_STATE;
1604 ## reconsume
1605
1606 !!!emit ($self->{current_token}); # start tag or end tag
1607
1608 redo A;
1609 } else {
1610 if ($self->{next_char} == 0x003D) { # =
1611 !!!cp (93);
1612 !!!parse-error (type => 'bad attribute value');
1613 } else {
1614 !!!cp (94);
1615 }
1616 $self->{current_attribute}->{value} .= chr ($self->{next_char});
1617 $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;
1618 !!!next-input-character;
1619 redo A;
1620 }
1621 } elsif ($self->{state} == ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE) {
1622 if ($self->{next_char} == 0x0022) { # "
1623 !!!cp (95);
1624 $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;
1625 !!!next-input-character;
1626 redo A;
1627 } elsif ($self->{next_char} == 0x0026) { # &
1628 !!!cp (96);
1629 $self->{last_attribute_value_state} = $self->{state};
1630 $self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE;
1631 !!!next-input-character;
1632 redo A;
1633 } elsif ($self->{next_char} == -1) {
1634 !!!parse-error (type => 'unclosed attribute value');
1635 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1636 !!!cp (97);
1637 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1638 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1639 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1640 if ($self->{current_token}->{attributes}) {
1641 !!!cp (98);
1642 !!!parse-error (type => 'end tag attribute');
1643 } else {
1644 ## NOTE: This state should never be reached.
1645 !!!cp (99);
1646 }
1647 } else {
1648 die "$0: $self->{current_token}->{type}: Unknown token type";
1649 }
1650 $self->{state} = DATA_STATE;
1651 ## reconsume
1652
1653 !!!emit ($self->{current_token}); # start tag or end tag
1654
1655 redo A;
1656 } else {
1657 !!!cp (100);
1658 $self->{current_attribute}->{value} .= chr ($self->{next_char});
1659 ## Stay in the state
1660 !!!next-input-character;
1661 redo A;
1662 }
1663 } elsif ($self->{state} == ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE) {
1664 if ($self->{next_char} == 0x0027) { # '
1665 !!!cp (101);
1666 $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;
1667 !!!next-input-character;
1668 redo A;
1669 } elsif ($self->{next_char} == 0x0026) { # &
1670 !!!cp (102);
1671 $self->{last_attribute_value_state} = $self->{state};
1672 $self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE;
1673 !!!next-input-character;
1674 redo A;
1675 } elsif ($self->{next_char} == -1) {
1676 !!!parse-error (type => 'unclosed attribute value');
1677 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1678 !!!cp (103);
1679 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1680 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1681 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1682 if ($self->{current_token}->{attributes}) {
1683 !!!cp (104);
1684 !!!parse-error (type => 'end tag attribute');
1685 } else {
1686 ## NOTE: This state should never be reached.
1687 !!!cp (105);
1688 }
1689 } else {
1690 die "$0: $self->{current_token}->{type}: Unknown token type";
1691 }
1692 $self->{state} = DATA_STATE;
1693 ## reconsume
1694
1695 !!!emit ($self->{current_token}); # start tag or end tag
1696
1697 redo A;
1698 } else {
1699 !!!cp (106);
1700 $self->{current_attribute}->{value} .= chr ($self->{next_char});
1701 ## Stay in the state
1702 !!!next-input-character;
1703 redo A;
1704 }
1705 } elsif ($self->{state} == ATTRIBUTE_VALUE_UNQUOTED_STATE) {
1706 if ($self->{next_char} == 0x0009 or # HT
1707 $self->{next_char} == 0x000A or # LF
1708 $self->{next_char} == 0x000B or # HT
1709 $self->{next_char} == 0x000C or # FF
1710 $self->{next_char} == 0x0020) { # SP
1711 !!!cp (107);
1712 $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1713 !!!next-input-character;
1714 redo A;
1715 } elsif ($self->{next_char} == 0x0026) { # &
1716 !!!cp (108);
1717 $self->{last_attribute_value_state} = $self->{state};
1718 $self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE;
1719 !!!next-input-character;
1720 redo A;
1721 } elsif ($self->{next_char} == 0x003E) { # >
1722 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1723 !!!cp (109);
1724 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1725 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1726 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1727 if ($self->{current_token}->{attributes}) {
1728 !!!cp (110);
1729 !!!parse-error (type => 'end tag attribute');
1730 } else {
1731 ## NOTE: This state should never be reached.
1732 !!!cp (111);
1733 }
1734 } else {
1735 die "$0: $self->{current_token}->{type}: Unknown token type";
1736 }
1737 $self->{state} = DATA_STATE;
1738 !!!next-input-character;
1739
1740 !!!emit ($self->{current_token}); # start tag or end tag
1741
1742 redo A;
1743 } elsif ($self->{next_char} == -1) {
1744 !!!parse-error (type => 'unclosed tag');
1745 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1746 !!!cp (112);
1747 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1748 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1749 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1750 if ($self->{current_token}->{attributes}) {
1751 !!!cp (113);
1752 !!!parse-error (type => 'end tag attribute');
1753 } else {
1754 ## NOTE: This state should never be reached.
1755 !!!cp (114);
1756 }
1757 } else {
1758 die "$0: $self->{current_token}->{type}: Unknown token type";
1759 }
1760 $self->{state} = DATA_STATE;
1761 ## reconsume
1762
1763 !!!emit ($self->{current_token}); # start tag or end tag
1764
1765 redo A;
1766 } else {
1767 if ({
1768 0x0022 => 1, # "
1769 0x0027 => 1, # '
1770 0x003D => 1, # =
1771 }->{$self->{next_char}}) {
1772 !!!cp (115);
1773 !!!parse-error (type => 'bad attribute value');
1774 } else {
1775 !!!cp (116);
1776 }
1777 $self->{current_attribute}->{value} .= chr ($self->{next_char});
1778 ## Stay in the state
1779 !!!next-input-character;
1780 redo A;
1781 }
1782 } elsif ($self->{state} == ENTITY_IN_ATTRIBUTE_VALUE_STATE) {
1783 my $token = $self->_tokenize_attempt_to_consume_an_entity
1784 (1,
1785 $self->{last_attribute_value_state}
1786 == ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE ? 0x0022 : # "
1787 $self->{last_attribute_value_state}
1788 == ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE ? 0x0027 : # '
1789 -1);
1790
1791 unless (defined $token) {
1792 !!!cp (117);
1793 $self->{current_attribute}->{value} .= '&';
1794 } else {
1795 !!!cp (118);
1796 $self->{current_attribute}->{value} .= $token->{data};
1797 $self->{current_attribute}->{has_reference} = $token->{has_reference};
1798 ## ISSUE: spec says "append the returned character token to the current attribute's value"
1799 }
1800
1801 $self->{state} = $self->{last_attribute_value_state};
1802 # next-input-character is already done
1803 redo A;
1804 } elsif ($self->{state} == AFTER_ATTRIBUTE_VALUE_QUOTED_STATE) {
1805 if ($self->{next_char} == 0x0009 or # HT
1806 $self->{next_char} == 0x000A or # LF
1807 $self->{next_char} == 0x000B or # VT
1808 $self->{next_char} == 0x000C or # FF
1809 $self->{next_char} == 0x0020) { # SP
1810 !!!cp (118);
1811 $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1812 !!!next-input-character;
1813 redo A;
1814 } elsif ($self->{next_char} == 0x003E) { # >
1815 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1816 !!!cp (119);
1817 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1818 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1819 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1820 if ($self->{current_token}->{attributes}) {
1821 !!!cp (120);
1822 !!!parse-error (type => 'end tag attribute');
1823 } else {
1824 ## NOTE: This state should never be reached.
1825 !!!cp (121);
1826 }
1827 } else {
1828 die "$0: $self->{current_token}->{type}: Unknown token type";
1829 }
1830 $self->{state} = DATA_STATE;
1831 !!!next-input-character;
1832
1833 !!!emit ($self->{current_token}); # start tag or end tag
1834
1835 redo A;
1836 } elsif ($self->{next_char} == 0x002F) { # /
1837 !!!cp (122);
1838 $self->{state} = SELF_CLOSING_START_TAG_STATE;
1839 !!!next-input-character;
1840 redo A;
1841 } elsif ($self->{next_char} == -1) {
1842 !!!parse-error (type => 'unclosed tag');
1843 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1844 !!!cp (122.3);
1845 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1846 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1847 if ($self->{current_token}->{attributes}) {
1848 !!!cp (122.1);
1849 !!!parse-error (type => 'end tag attribute');
1850 } else {
1851 ## NOTE: This state should never be reached.
1852 !!!cp (122.2);
1853 }
1854 } else {
1855 die "$0: $self->{current_token}->{type}: Unknown token type";
1856 }
1857 $self->{state} = DATA_STATE;
1858 ## Reconsume.
1859 !!!emit ($self->{current_token}); # start tag or end tag
1860 redo A;
1861 } else {
1862 !!!cp ('124.1');
1863 !!!parse-error (type => 'no space between attributes');
1864 $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1865 ## reconsume
1866 redo A;
1867 }
1868 } elsif ($self->{state} == SELF_CLOSING_START_TAG_STATE) {
1869 if ($self->{next_char} == 0x003E) { # >
1870 if ($self->{current_token}->{type} == END_TAG_TOKEN) {
1871 !!!cp ('124.2');
1872 !!!parse-error (type => 'nestc', token => $self->{current_token});
1873 ## TODO: Different type than slash in start tag
1874 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1875 if ($self->{current_token}->{attributes}) {
1876 !!!cp ('124.4');
1877 !!!parse-error (type => 'end tag attribute');
1878 } else {
1879 !!!cp ('124.5');
1880 }
1881 ## TODO: Test |<title></title/>|
1882 } else {
1883 !!!cp ('124.3');
1884 $self->{self_closing} = 1;
1885 }
1886
1887 $self->{state} = DATA_STATE;
1888 !!!next-input-character;
1889
1890 !!!emit ($self->{current_token}); # start tag or end tag
1891
1892 redo A;
1893 } elsif ($self->{next_char} == -1) {
1894 !!!parse-error (type => 'unclosed tag');
1895 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1896 !!!cp (124.7);
1897 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1898 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1899 if ($self->{current_token}->{attributes}) {
1900 !!!cp (124.5);
1901 !!!parse-error (type => 'end tag attribute');
1902 } else {
1903 ## NOTE: This state should never be reached.
1904 !!!cp (124.6);
1905 }
1906 } else {
1907 die "$0: $self->{current_token}->{type}: Unknown token type";
1908 }
1909 $self->{state} = DATA_STATE;
1910 ## Reconsume.
1911 !!!emit ($self->{current_token}); # start tag or end tag
1912 redo A;
1913 } else {
1914 !!!cp ('124.4');
1915 !!!parse-error (type => 'nestc');
1916 ## TODO: This error type is wrong.
1917 $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1918 ## Reconsume.
1919 redo A;
1920 }
1921 } elsif ($self->{state} == BOGUS_COMMENT_STATE) {
1922 ## (only happen if PCDATA state)
1923
1924 ## NOTE: Set by the previous state
1925 #my $token = {type => COMMENT_TOKEN, data => ''};
1926
1927 BC: {
1928 if ($self->{next_char} == 0x003E) { # >
1929 !!!cp (124);
1930 $self->{state} = DATA_STATE;
1931 !!!next-input-character;
1932
1933 !!!emit ($self->{current_token}); # comment
1934
1935 redo A;
1936 } elsif ($self->{next_char} == -1) {
1937 !!!cp (125);
1938 $self->{state} = DATA_STATE;
1939 ## reconsume
1940
1941 !!!emit ($self->{current_token}); # comment
1942
1943 redo A;
1944 } else {
1945 !!!cp (126);
1946 $self->{current_token}->{data} .= chr ($self->{next_char}); # comment
1947 !!!next-input-character;
1948 redo BC;
1949 }
1950 } # BC
1951
1952 die "$0: _get_next_token: unexpected case [BC]";
1953 } elsif ($self->{state} == MARKUP_DECLARATION_OPEN_STATE) {
1954 ## (only happen if PCDATA state)
1955
1956 my ($l, $c) = ($self->{line_prev}, $self->{column_prev} - 1);
1957
1958 my @next_char;
1959 push @next_char, $self->{next_char};
1960
1961 if ($self->{next_char} == 0x002D) { # -
1962 !!!next-input-character;
1963 push @next_char, $self->{next_char};
1964 if ($self->{next_char} == 0x002D) { # -
1965 !!!cp (127);
1966 $self->{current_token} = {type => COMMENT_TOKEN, data => '',
1967 line => $l, column => $c,
1968 };
1969 $self->{state} = COMMENT_START_STATE;
1970 !!!next-input-character;
1971 redo A;
1972 } else {
1973 !!!cp (128);
1974 }
1975 } elsif ($self->{next_char} == 0x0044 or # D
1976 $self->{next_char} == 0x0064) { # d
1977 !!!next-input-character;
1978 push @next_char, $self->{next_char};
1979 if ($self->{next_char} == 0x004F or # O
1980 $self->{next_char} == 0x006F) { # o
1981 !!!next-input-character;
1982 push @next_char, $self->{next_char};
1983 if ($self->{next_char} == 0x0043 or # C
1984 $self->{next_char} == 0x0063) { # c
1985 !!!next-input-character;
1986 push @next_char, $self->{next_char};
1987 if ($self->{next_char} == 0x0054 or # T
1988 $self->{next_char} == 0x0074) { # t
1989 !!!next-input-character;
1990 push @next_char, $self->{next_char};
1991 if ($self->{next_char} == 0x0059 or # Y
1992 $self->{next_char} == 0x0079) { # y
1993 !!!next-input-character;
1994 push @next_char, $self->{next_char};
1995 if ($self->{next_char} == 0x0050 or # P
1996 $self->{next_char} == 0x0070) { # p
1997 !!!next-input-character;
1998 push @next_char, $self->{next_char};
1999 if ($self->{next_char} == 0x0045 or # E
2000 $self->{next_char} == 0x0065) { # e
2001 !!!cp (129);
2002 ## TODO: What a stupid code this is!
2003 $self->{state} = DOCTYPE_STATE;
2004 $self->{current_token} = {type => DOCTYPE_TOKEN,
2005 quirks => 1,
2006 line => $l, column => $c,
2007 };
2008 !!!next-input-character;
2009 redo A;
2010 } else {
2011 !!!cp (130);
2012 }
2013 } else {
2014 !!!cp (131);
2015 }
2016 } else {
2017 !!!cp (132);
2018 }
2019 } else {
2020 !!!cp (133);
2021 }
2022 } else {
2023 !!!cp (134);
2024 }
2025 } else {
2026 !!!cp (135);
2027 }
2028 } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM and
2029 $self->{open_elements}->[-1]->[1] & FOREIGN_EL and
2030 $self->{next_char} == 0x005B) { # [
2031 !!!next-input-character;
2032 push @next_char, $self->{next_char};
2033 if ($self->{next_char} == 0x0043) { # C
2034 !!!next-input-character;
2035 push @next_char, $self->{next_char};
2036 if ($self->{next_char} == 0x0044) { # D
2037 !!!next-input-character;
2038 push @next_char, $self->{next_char};
2039 if ($self->{next_char} == 0x0041) { # A
2040 !!!next-input-character;
2041 push @next_char, $self->{next_char};
2042 if ($self->{next_char} == 0x0054) { # T
2043 !!!next-input-character;
2044 push @next_char, $self->{next_char};
2045 if ($self->{next_char} == 0x0041) { # A
2046 !!!next-input-character;
2047 push @next_char, $self->{next_char};
2048 if ($self->{next_char} == 0x005B) { # [
2049 !!!cp (135.1);
2050 $self->{state} = CDATA_BLOCK_STATE;
2051 !!!next-input-character;
2052 redo A;
2053 } else {
2054 !!!cp (135.2);
2055 }
2056 } else {
2057 !!!cp (135.3);
2058 }
2059 } else {
2060 !!!cp (135.4);
2061 }
2062 } else {
2063 !!!cp (135.5);
2064 }
2065 } else {
2066 !!!cp (135.6);
2067 }
2068 } else {
2069 !!!cp (135.7);
2070 }
2071 } else {
2072 !!!cp (136);
2073 }
2074
2075 !!!parse-error (type => 'bogus comment');
2076 $self->{next_char} = shift @next_char;
2077 !!!back-next-input-character (@next_char);
2078 $self->{state} = BOGUS_COMMENT_STATE;
2079 $self->{current_token} = {type => COMMENT_TOKEN, data => '',
2080 line => $l, column => $c,
2081 };
2082 redo A;
2083
2084 ## ISSUE: typos in spec: chacacters, is is a parse error
2085 ## ISSUE: spec is somewhat unclear on "is the first character that will be in the comment"; what is "that will be in the comment" is what the algorithm defines, isn't it?
2086 } elsif ($self->{state} == COMMENT_START_STATE) {
2087 if ($self->{next_char} == 0x002D) { # -
2088 !!!cp (137);
2089 $self->{state} = COMMENT_START_DASH_STATE;
2090 !!!next-input-character;
2091 redo A;
2092 } elsif ($self->{next_char} == 0x003E) { # >
2093 !!!cp (138);
2094 !!!parse-error (type => 'bogus comment');
2095 $self->{state} = DATA_STATE;
2096 !!!next-input-character;
2097
2098 !!!emit ($self->{current_token}); # comment
2099
2100 redo A;
2101 } elsif ($self->{next_char} == -1) {
2102 !!!cp (139);
2103 !!!parse-error (type => 'unclosed comment');
2104 $self->{state} = DATA_STATE;
2105 ## reconsume
2106
2107 !!!emit ($self->{current_token}); # comment
2108
2109 redo A;
2110 } else {
2111 !!!cp (140);
2112 $self->{current_token}->{data} # comment
2113 .= chr ($self->{next_char});
2114 $self->{state} = COMMENT_STATE;
2115 !!!next-input-character;
2116 redo A;
2117 }
2118 } elsif ($self->{state} == COMMENT_START_DASH_STATE) {
2119 if ($self->{next_char} == 0x002D) { # -
2120 !!!cp (141);
2121 $self->{state} = COMMENT_END_STATE;
2122 !!!next-input-character;
2123 redo A;
2124 } elsif ($self->{next_char} == 0x003E) { # >
2125 !!!cp (142);
2126 !!!parse-error (type => 'bogus comment');
2127 $self->{state} = DATA_STATE;
2128 !!!next-input-character;
2129
2130 !!!emit ($self->{current_token}); # comment
2131
2132 redo A;
2133 } elsif ($self->{next_char} == -1) {
2134 !!!cp (143);
2135 !!!parse-error (type => 'unclosed comment');
2136 $self->{state} = DATA_STATE;
2137 ## reconsume
2138
2139 !!!emit ($self->{current_token}); # comment
2140
2141 redo A;
2142 } else {
2143 !!!cp (144);
2144 $self->{current_token}->{data} # comment
2145 .= '-' . chr ($self->{next_char});
2146 $self->{state} = COMMENT_STATE;
2147 !!!next-input-character;
2148 redo A;
2149 }
2150 } elsif ($self->{state} == COMMENT_STATE) {
2151 if ($self->{next_char} == 0x002D) { # -
2152 !!!cp (145);
2153 $self->{state} = COMMENT_END_DASH_STATE;
2154 !!!next-input-character;
2155 redo A;
2156 } elsif ($self->{next_char} == -1) {
2157 !!!cp (146);
2158 !!!parse-error (type => 'unclosed comment');
2159 $self->{state} = DATA_STATE;
2160 ## reconsume
2161
2162 !!!emit ($self->{current_token}); # comment
2163
2164 redo A;
2165 } else {
2166 !!!cp (147);
2167 $self->{current_token}->{data} .= chr ($self->{next_char}); # comment
2168 ## Stay in the state
2169 !!!next-input-character;
2170 redo A;
2171 }
2172 } elsif ($self->{state} == COMMENT_END_DASH_STATE) {
2173 if ($self->{next_char} == 0x002D) { # -
2174 !!!cp (148);
2175 $self->{state} = COMMENT_END_STATE;
2176 !!!next-input-character;
2177 redo A;
2178 } elsif ($self->{next_char} == -1) {
2179 !!!cp (149);
2180 !!!parse-error (type => 'unclosed comment');
2181 $self->{state} = DATA_STATE;
2182 ## reconsume
2183
2184 !!!emit ($self->{current_token}); # comment
2185
2186 redo A;
2187 } else {
2188 !!!cp (150);
2189 $self->{current_token}->{data} .= '-' . chr ($self->{next_char}); # comment
2190 $self->{state} = COMMENT_STATE;
2191 !!!next-input-character;
2192 redo A;
2193 }
2194 } elsif ($self->{state} == COMMENT_END_STATE) {
2195 if ($self->{next_char} == 0x003E) { # >
2196 !!!cp (151);
2197 $self->{state} = DATA_STATE;
2198 !!!next-input-character;
2199
2200 !!!emit ($self->{current_token}); # comment
2201
2202 redo A;
2203 } elsif ($self->{next_char} == 0x002D) { # -
2204 !!!cp (152);
2205 !!!parse-error (type => 'dash in comment',
2206 line => $self->{line_prev},
2207 column => $self->{column_prev});
2208 $self->{current_token}->{data} .= '-'; # comment
2209 ## Stay in the state
2210 !!!next-input-character;
2211 redo A;
2212 } elsif ($self->{next_char} == -1) {
2213 !!!cp (153);
2214 !!!parse-error (type => 'unclosed comment');
2215 $self->{state} = DATA_STATE;
2216 ## reconsume
2217
2218 !!!emit ($self->{current_token}); # comment
2219
2220 redo A;
2221 } else {
2222 !!!cp (154);
2223 !!!parse-error (type => 'dash in comment',
2224 line => $self->{line_prev},
2225 column => $self->{column_prev});
2226 $self->{current_token}->{data} .= '--' . chr ($self->{next_char}); # comment
2227 $self->{state} = COMMENT_STATE;
2228 !!!next-input-character;
2229 redo A;
2230 }
2231 } elsif ($self->{state} == DOCTYPE_STATE) {
2232 if ($self->{next_char} == 0x0009 or # HT
2233 $self->{next_char} == 0x000A or # LF
2234 $self->{next_char} == 0x000B or # VT
2235 $self->{next_char} == 0x000C or # FF
2236 $self->{next_char} == 0x0020) { # SP
2237 !!!cp (155);
2238 $self->{state} = BEFORE_DOCTYPE_NAME_STATE;
2239 !!!next-input-character;
2240 redo A;
2241 } else {
2242 !!!cp (156);
2243 !!!parse-error (type => 'no space before DOCTYPE name');
2244 $self->{state} = BEFORE_DOCTYPE_NAME_STATE;
2245 ## reconsume
2246 redo A;
2247 }
2248 } elsif ($self->{state} == BEFORE_DOCTYPE_NAME_STATE) {
2249 if ($self->{next_char} == 0x0009 or # HT
2250 $self->{next_char} == 0x000A or # LF
2251 $self->{next_char} == 0x000B or # VT
2252 $self->{next_char} == 0x000C or # FF
2253 $self->{next_char} == 0x0020) { # SP
2254 !!!cp (157);
2255 ## Stay in the state
2256 !!!next-input-character;
2257 redo A;
2258 } elsif ($self->{next_char} == 0x003E) { # >
2259 !!!cp (158);
2260 !!!parse-error (type => 'no DOCTYPE name');
2261 $self->{state} = DATA_STATE;
2262 !!!next-input-character;
2263
2264 !!!emit ($self->{current_token}); # DOCTYPE (quirks)
2265
2266 redo A;
2267 } elsif ($self->{next_char} == -1) {
2268 !!!cp (159);
2269 !!!parse-error (type => 'no DOCTYPE name');
2270 $self->{state} = DATA_STATE;
2271 ## reconsume
2272
2273 !!!emit ($self->{current_token}); # DOCTYPE (quirks)
2274
2275 redo A;
2276 } else {
2277 !!!cp (160);
2278 $self->{current_token}->{name} = chr $self->{next_char};
2279 delete $self->{current_token}->{quirks};
2280 ## ISSUE: "Set the token's name name to the" in the spec
2281 $self->{state} = DOCTYPE_NAME_STATE;
2282 !!!next-input-character;
2283 redo A;
2284 }
2285 } elsif ($self->{state} == DOCTYPE_NAME_STATE) {
2286 ## ISSUE: Redundant "First," in the spec.
2287 if ($self->{next_char} == 0x0009 or # HT
2288 $self->{next_char} == 0x000A or # LF
2289 $self->{next_char} == 0x000B or # VT
2290 $self->{next_char} == 0x000C or # FF
2291 $self->{next_char} == 0x0020) { # SP
2292 !!!cp (161);
2293 $self->{state} = AFTER_DOCTYPE_NAME_STATE;
2294 !!!next-input-character;
2295 redo A;
2296 } elsif ($self->{next_char} == 0x003E) { # >
2297 !!!cp (162);
2298 $self->{state} = DATA_STATE;
2299 !!!next-input-character;
2300
2301 !!!emit ($self->{current_token}); # DOCTYPE
2302
2303 redo A;
2304 } elsif ($self->{next_char} == -1) {
2305 !!!cp (163);
2306 !!!parse-error (type => 'unclosed DOCTYPE');
2307 $self->{state} = DATA_STATE;
2308 ## reconsume
2309
2310 $self->{current_token}->{quirks} = 1;
2311 !!!emit ($self->{current_token}); # DOCTYPE
2312
2313 redo A;
2314 } else {
2315 !!!cp (164);
2316 $self->{current_token}->{name}
2317 .= chr ($self->{next_char}); # DOCTYPE
2318 ## Stay in the state
2319 !!!next-input-character;
2320 redo A;
2321 }
2322 } elsif ($self->{state} == AFTER_DOCTYPE_NAME_STATE) {
2323 if ($self->{next_char} == 0x0009 or # HT
2324 $self->{next_char} == 0x000A or # LF
2325 $self->{next_char} == 0x000B or # VT
2326 $self->{next_char} == 0x000C or # FF
2327 $self->{next_char} == 0x0020) { # SP
2328 !!!cp (165);
2329 ## Stay in the state
2330 !!!next-input-character;
2331 redo A;
2332 } elsif ($self->{next_char} == 0x003E) { # >
2333 !!!cp (166);
2334 $self->{state} = DATA_STATE;
2335 !!!next-input-character;
2336
2337 !!!emit ($self->{current_token}); # DOCTYPE
2338
2339 redo A;
2340 } elsif ($self->{next_char} == -1) {
2341 !!!cp (167);
2342 !!!parse-error (type => 'unclosed DOCTYPE');
2343 $self->{state} = DATA_STATE;
2344 ## reconsume
2345
2346 $self->{current_token}->{quirks} = 1;
2347 !!!emit ($self->{current_token}); # DOCTYPE
2348
2349 redo A;
2350 } elsif ($self->{next_char} == 0x0050 or # P
2351 $self->{next_char} == 0x0070) { # p
2352 !!!next-input-character;
2353 if ($self->{next_char} == 0x0055 or # U
2354 $self->{next_char} == 0x0075) { # u
2355 !!!next-input-character;
2356 if ($self->{next_char} == 0x0042 or # B
2357 $self->{next_char} == 0x0062) { # b
2358 !!!next-input-character;
2359 if ($self->{next_char} == 0x004C or # L
2360 $self->{next_char} == 0x006C) { # l
2361 !!!next-input-character;
2362 if ($self->{next_char} == 0x0049 or # I
2363 $self->{next_char} == 0x0069) { # i
2364 !!!next-input-character;
2365 if ($self->{next_char} == 0x0043 or # C
2366 $self->{next_char} == 0x0063) { # c
2367 !!!cp (168);
2368 $self->{state} = BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2369 !!!next-input-character;
2370 redo A;
2371 } else {
2372 !!!cp (169);
2373 }
2374 } else {
2375 !!!cp (170);
2376 }
2377 } else {
2378 !!!cp (171);
2379 }
2380 } else {
2381 !!!cp (172);
2382 }
2383 } else {
2384 !!!cp (173);
2385 }
2386
2387 #
2388 } elsif ($self->{next_char} == 0x0053 or # S
2389 $self->{next_char} == 0x0073) { # s
2390 !!!next-input-character;
2391 if ($self->{next_char} == 0x0059 or # Y
2392 $self->{next_char} == 0x0079) { # y
2393 !!!next-input-character;
2394 if ($self->{next_char} == 0x0053 or # S
2395 $self->{next_char} == 0x0073) { # s
2396 !!!next-input-character;
2397 if ($self->{next_char} == 0x0054 or # T
2398 $self->{next_char} == 0x0074) { # t
2399 !!!next-input-character;
2400 if ($self->{next_char} == 0x0045 or # E
2401 $self->{next_char} == 0x0065) { # e
2402 !!!next-input-character;
2403 if ($self->{next_char} == 0x004D or # M
2404 $self->{next_char} == 0x006D) { # m
2405 !!!cp (174);
2406 $self->{state} = BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2407 !!!next-input-character;
2408 redo A;
2409 } else {
2410 !!!cp (175);
2411 }
2412 } else {
2413 !!!cp (176);
2414 }
2415 } else {
2416 !!!cp (177);
2417 }
2418 } else {
2419 !!!cp (178);
2420 }
2421 } else {
2422 !!!cp (179);
2423 }
2424
2425 #
2426 } else {
2427 !!!cp (180);
2428 !!!next-input-character;
2429 #
2430 }
2431
2432 !!!parse-error (type => 'string after DOCTYPE name');
2433 $self->{current_token}->{quirks} = 1;
2434
2435 $self->{state} = BOGUS_DOCTYPE_STATE;
2436 # next-input-character is already done
2437 redo A;
2438 } elsif ($self->{state} == BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {
2439 if ({
2440 0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,
2441 #0x000D => 1, # HT, LF, VT, FF, SP, CR
2442 }->{$self->{next_char}}) {
2443 !!!cp (181);
2444 ## Stay in the state
2445 !!!next-input-character;
2446 redo A;
2447 } elsif ($self->{next_char} eq 0x0022) { # "
2448 !!!cp (182);
2449 $self->{current_token}->{public_identifier} = ''; # DOCTYPE
2450 $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE;
2451 !!!next-input-character;
2452 redo A;
2453 } elsif ($self->{next_char} eq 0x0027) { # '
2454 !!!cp (183);
2455 $self->{current_token}->{public_identifier} = ''; # DOCTYPE
2456 $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE;
2457 !!!next-input-character;
2458 redo A;
2459 } elsif ($self->{next_char} eq 0x003E) { # >
2460 !!!cp (184);
2461 !!!parse-error (type => 'no PUBLIC literal');
2462
2463 $self->{state} = DATA_STATE;
2464 !!!next-input-character;
2465
2466 $self->{current_token}->{quirks} = 1;
2467 !!!emit ($self->{current_token}); # DOCTYPE
2468
2469 redo A;
2470 } elsif ($self->{next_char} == -1) {
2471 !!!cp (185);
2472 !!!parse-error (type => 'unclosed DOCTYPE');
2473
2474 $self->{state} = DATA_STATE;
2475 ## reconsume
2476
2477 $self->{current_token}->{quirks} = 1;
2478 !!!emit ($self->{current_token}); # DOCTYPE
2479
2480 redo A;
2481 } else {
2482 !!!cp (186);
2483 !!!parse-error (type => 'string after PUBLIC');
2484 $self->{current_token}->{quirks} = 1;
2485
2486 $self->{state} = BOGUS_DOCTYPE_STATE;
2487 !!!next-input-character;
2488 redo A;
2489 }
2490 } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE) {
2491 if ($self->{next_char} == 0x0022) { # "
2492 !!!cp (187);
2493 $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2494 !!!next-input-character;
2495 redo A;
2496 } elsif ($self->{next_char} == 0x003E) { # >
2497 !!!cp (188);
2498 !!!parse-error (type => 'unclosed PUBLIC literal');
2499
2500 $self->{state} = DATA_STATE;
2501 !!!next-input-character;
2502
2503 $self->{current_token}->{quirks} = 1;
2504 !!!emit ($self->{current_token}); # DOCTYPE
2505
2506 redo A;
2507 } elsif ($self->{next_char} == -1) {
2508 !!!cp (189);
2509 !!!parse-error (type => 'unclosed PUBLIC literal');
2510
2511 $self->{state} = DATA_STATE;
2512 ## reconsume
2513
2514 $self->{current_token}->{quirks} = 1;
2515 !!!emit ($self->{current_token}); # DOCTYPE
2516
2517 redo A;
2518 } else {
2519 !!!cp (190);
2520 $self->{current_token}->{public_identifier} # DOCTYPE
2521 .= chr $self->{next_char};
2522 ## Stay in the state
2523 !!!next-input-character;
2524 redo A;
2525 }
2526 } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE) {
2527 if ($self->{next_char} == 0x0027) { # '
2528 !!!cp (191);
2529 $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2530 !!!next-input-character;
2531 redo A;
2532 } elsif ($self->{next_char} == 0x003E) { # >
2533 !!!cp (192);
2534 !!!parse-error (type => 'unclosed PUBLIC literal');
2535
2536 $self->{state} = DATA_STATE;
2537 !!!next-input-character;
2538
2539 $self->{current_token}->{quirks} = 1;
2540 !!!emit ($self->{current_token}); # DOCTYPE
2541
2542 redo A;
2543 } elsif ($self->{next_char} == -1) {
2544 !!!cp (193);
2545 !!!parse-error (type => 'unclosed PUBLIC literal');
2546
2547 $self->{state} = DATA_STATE;
2548 ## reconsume
2549
2550 $self->{current_token}->{quirks} = 1;
2551 !!!emit ($self->{current_token}); # DOCTYPE
2552
2553 redo A;
2554 } else {
2555 !!!cp (194);
2556 $self->{current_token}->{public_identifier} # DOCTYPE
2557 .= chr $self->{next_char};
2558 ## Stay in the state
2559 !!!next-input-character;
2560 redo A;
2561 }
2562 } elsif ($self->{state} == AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {
2563 if ({
2564 0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,
2565 #0x000D => 1, # HT, LF, VT, FF, SP, CR
2566 }->{$self->{next_char}}) {
2567 !!!cp (195);
2568 ## Stay in the state
2569 !!!next-input-character;
2570 redo A;
2571 } elsif ($self->{next_char} == 0x0022) { # "
2572 !!!cp (196);
2573 $self->{current_token}->{system_identifier} = ''; # DOCTYPE
2574 $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;
2575 !!!next-input-character;
2576 redo A;
2577 } elsif ($self->{next_char} == 0x0027) { # '
2578 !!!cp (197);
2579 $self->{current_token}->{system_identifier} = ''; # DOCTYPE
2580 $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;
2581 !!!next-input-character;
2582 redo A;
2583 } elsif ($self->{next_char} == 0x003E) { # >
2584 !!!cp (198);
2585 $self->{state} = DATA_STATE;
2586 !!!next-input-character;
2587
2588 !!!emit ($self->{current_token}); # DOCTYPE
2589
2590 redo A;
2591 } elsif ($self->{next_char} == -1) {
2592 !!!cp (199);
2593 !!!parse-error (type => 'unclosed DOCTYPE');
2594
2595 $self->{state} = DATA_STATE;
2596 ## reconsume
2597
2598 $self->{current_token}->{quirks} = 1;
2599 !!!emit ($self->{current_token}); # DOCTYPE
2600
2601 redo A;
2602 } else {
2603 !!!cp (200);
2604 !!!parse-error (type => 'string after PUBLIC literal');
2605 $self->{current_token}->{quirks} = 1;
2606
2607 $self->{state} = BOGUS_DOCTYPE_STATE;
2608 !!!next-input-character;
2609 redo A;
2610 }
2611 } elsif ($self->{state} == BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {
2612 if ({
2613 0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,
2614 #0x000D => 1, # HT, LF, VT, FF, SP, CR
2615 }->{$self->{next_char}}) {
2616 !!!cp (201);
2617 ## Stay in the state
2618 !!!next-input-character;
2619 redo A;
2620 } elsif ($self->{next_char} == 0x0022) { # "
2621 !!!cp (202);
2622 $self->{current_token}->{system_identifier} = ''; # DOCTYPE
2623 $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;
2624 !!!next-input-character;
2625 redo A;
2626 } elsif ($self->{next_char} == 0x0027) { # '
2627 !!!cp (203);
2628 $self->{current_token}->{system_identifier} = ''; # DOCTYPE
2629 $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;
2630 !!!next-input-character;
2631 redo A;
2632 } elsif ($self->{next_char} == 0x003E) { # >
2633 !!!cp (204);
2634 !!!parse-error (type => 'no SYSTEM literal');
2635 $self->{state} = DATA_STATE;
2636 !!!next-input-character;
2637
2638 $self->{current_token}->{quirks} = 1;
2639 !!!emit ($self->{current_token}); # DOCTYPE
2640
2641 redo A;
2642 } elsif ($self->{next_char} == -1) {
2643 !!!cp (205);
2644 !!!parse-error (type => 'unclosed DOCTYPE');
2645
2646 $self->{state} = DATA_STATE;
2647 ## reconsume
2648
2649 $self->{current_token}->{quirks} = 1;
2650 !!!emit ($self->{current_token}); # DOCTYPE
2651
2652 redo A;
2653 } else {
2654 !!!cp (206);
2655 !!!parse-error (type => 'string after SYSTEM');
2656 $self->{current_token}->{quirks} = 1;
2657
2658 $self->{state} = BOGUS_DOCTYPE_STATE;
2659 !!!next-input-character;
2660 redo A;
2661 }
2662 } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE) {
2663 if ($self->{next_char} == 0x0022) { # "
2664 !!!cp (207);
2665 $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2666 !!!next-input-character;
2667 redo A;
2668 } elsif ($self->{next_char} == 0x003E) { # >
2669 !!!cp (208);
2670 !!!parse-error (type => 'unclosed PUBLIC literal');
2671
2672 $self->{state} = DATA_STATE;
2673 !!!next-input-character;
2674
2675 $self->{current_token}->{quirks} = 1;
2676 !!!emit ($self->{current_token}); # DOCTYPE
2677
2678 redo A;
2679 } elsif ($self->{next_char} == -1) {
2680 !!!cp (209);
2681 !!!parse-error (type => 'unclosed SYSTEM literal');
2682
2683 $self->{state} = DATA_STATE;
2684 ## reconsume
2685
2686 $self->{current_token}->{quirks} = 1;
2687 !!!emit ($self->{current_token}); # DOCTYPE
2688
2689 redo A;
2690 } else {
2691 !!!cp (210);
2692 $self->{current_token}->{system_identifier} # DOCTYPE
2693 .= chr $self->{next_char};
2694 ## Stay in the state
2695 !!!next-input-character;
2696 redo A;
2697 }
2698 } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE) {
2699 if ($self->{next_char} == 0x0027) { # '
2700 !!!cp (211);
2701 $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2702 !!!next-input-character;
2703 redo A;
2704 } elsif ($self->{next_char} == 0x003E) { # >
2705 !!!cp (212);
2706 !!!parse-error (type => 'unclosed PUBLIC literal');
2707
2708 $self->{state} = DATA_STATE;
2709 !!!next-input-character;
2710
2711 $self->{current_token}->{quirks} = 1;
2712 !!!emit ($self->{current_token}); # DOCTYPE
2713
2714 redo A;
2715 } elsif ($self->{next_char} == -1) {
2716 !!!cp (213);
2717 !!!parse-error (type => 'unclosed SYSTEM literal');
2718
2719 $self->{state} = DATA_STATE;
2720 ## reconsume
2721
2722 $self->{current_token}->{quirks} = 1;
2723 !!!emit ($self->{current_token}); # DOCTYPE
2724
2725 redo A;
2726 } else {
2727 !!!cp (214);
2728 $self->{current_token}->{system_identifier} # DOCTYPE
2729 .= chr $self->{next_char};
2730 ## Stay in the state
2731 !!!next-input-character;
2732 redo A;
2733 }
2734 } elsif ($self->{state} == AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {
2735 if ({
2736 0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,
2737 #0x000D => 1, # HT, LF, VT, FF, SP, CR
2738 }->{$self->{next_char}}) {
2739 !!!cp (215);
2740 ## Stay in the state
2741 !!!next-input-character;
2742 redo A;
2743 } elsif ($self->{next_char} == 0x003E) { # >
2744 !!!cp (216);
2745 $self->{state} = DATA_STATE;
2746 !!!next-input-character;
2747
2748 !!!emit ($self->{current_token}); # DOCTYPE
2749
2750 redo A;
2751 } elsif ($self->{next_char} == -1) {
2752 !!!cp (217);
2753 !!!parse-error (type => 'unclosed DOCTYPE');
2754 $self->{state} = DATA_STATE;
2755 ## reconsume
2756
2757 $self->{current_token}->{quirks} = 1;
2758 !!!emit ($self->{current_token}); # DOCTYPE
2759
2760 redo A;
2761 } else {
2762 !!!cp (218);
2763 !!!parse-error (type => 'string after SYSTEM literal');
2764 #$self->{current_token}->{quirks} = 1;
2765
2766 $self->{state} = BOGUS_DOCTYPE_STATE;
2767 !!!next-input-character;
2768 redo A;
2769 }
2770 } elsif ($self->{state} == BOGUS_DOCTYPE_STATE) {
2771 if ($self->{next_char} == 0x003E) { # >
2772 !!!cp (219);
2773 $self->{state} = DATA_STATE;
2774 !!!next-input-character;
2775
2776 !!!emit ($self->{current_token}); # DOCTYPE
2777
2778 redo A;
2779 } elsif ($self->{next_char} == -1) {
2780 !!!cp (220);
2781 !!!parse-error (type => 'unclosed DOCTYPE');
2782 $self->{state} = DATA_STATE;
2783 ## reconsume
2784
2785 !!!emit ($self->{current_token}); # DOCTYPE
2786
2787 redo A;
2788 } else {
2789 !!!cp (221);
2790 ## Stay in the state
2791 !!!next-input-character;
2792 redo A;
2793 }
2794 } elsif ($self->{state} == CDATA_BLOCK_STATE) {
2795 my $s = '';
2796
2797 my ($l, $c) = ($self->{line}, $self->{column});
2798
2799 CS: while ($self->{next_char} != -1) {
2800 if ($self->{next_char} == 0x005D) { # ]
2801 !!!next-input-character;
2802 if ($self->{next_char} == 0x005D) { # ]
2803 !!!next-input-character;
2804 MDC: {
2805 if ($self->{next_char} == 0x003E) { # >
2806 !!!cp (221.1);
2807 !!!next-input-character;
2808 last CS;
2809 } elsif ($self->{next_char} == 0x005D) { # ]
2810 !!!cp (221.2);
2811 $s .= ']';
2812 !!!next-input-character;
2813 redo MDC;
2814 } else {
2815 !!!cp (221.3);
2816 $s .= ']]';
2817 #
2818 }
2819 } # MDC
2820 } else {
2821 !!!cp (221.4);
2822 $s .= ']';
2823 #
2824 }
2825 } else {
2826 !!!cp (221.5);
2827 #
2828 }
2829 $s .= chr $self->{next_char};
2830 !!!next-input-character;
2831 } # CS
2832
2833 $self->{state} = DATA_STATE;
2834 ## next-input-character done or EOF, which is reconsumed.
2835
2836 if (length $s) {
2837 !!!cp (221.6);
2838 !!!emit ({type => CHARACTER_TOKEN, data => $s,
2839 line => $l, column => $c});
2840 } else {
2841 !!!cp (221.7);
2842 }
2843
2844 redo A;
2845
2846 ## ISSUE: "text tokens" in spec.
2847 ## TODO: Streaming support
2848 } else {
2849 die "$0: $self->{state}: Unknown state";
2850 }
2851 } # A
2852
2853 die "$0: _get_next_token: unexpected case";
2854 } # _get_next_token
2855
2856 sub _tokenize_attempt_to_consume_an_entity ($$$) {
2857 my ($self, $in_attr, $additional) = @_;
2858
2859 my ($l, $c) = ($self->{line_prev}, $self->{column_prev});
2860
2861 if ({
2862 0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, # HT, LF, VT, FF,
2863 0x0020 => 1, 0x003C => 1, 0x0026 => 1, -1 => 1, # SP, <, & # 0x000D # CR
2864 $additional => 1,
2865 }->{$self->{next_char}}) {
2866 !!!cp (1001);
2867 ## Don't consume
2868 ## No error
2869 return undef;
2870 } elsif ($self->{next_char} == 0x0023) { # #
2871 !!!next-input-character;
2872 if ($self->{next_char} == 0x0078 or # x
2873 $self->{next_char} == 0x0058) { # X
2874 my $code;
2875 X: {
2876 my $x_char = $self->{next_char};
2877 !!!next-input-character;
2878 if (0x0030 <= $self->{next_char} and
2879 $self->{next_char} <= 0x0039) { # 0..9
2880 !!!cp (1002);
2881 $code ||= 0;
2882 $code *= 0x10;
2883 $code += $self->{next_char} - 0x0030;
2884 redo X;
2885 } elsif (0x0061 <= $self->{next_char} and
2886 $self->{next_char} <= 0x0066) { # a..f
2887 !!!cp (1003);
2888 $code ||= 0;
2889 $code *= 0x10;
2890 $code += $self->{next_char} - 0x0060 + 9;
2891 redo X;
2892 } elsif (0x0041 <= $self->{next_char} and
2893 $self->{next_char} <= 0x0046) { # A..F
2894 !!!cp (1004);
2895 $code ||= 0;
2896 $code *= 0x10;
2897 $code += $self->{next_char} - 0x0040 + 9;
2898 redo X;
2899 } elsif (not defined $code) { # no hexadecimal digit
2900 !!!cp (1005);
2901 !!!parse-error (type => 'bare hcro', line => $l, column => $c);
2902 !!!back-next-input-character ($x_char, $self->{next_char});
2903 $self->{next_char} = 0x0023; # #
2904 return undef;
2905 } elsif ($self->{next_char} == 0x003B) { # ;
2906 !!!cp (1006);
2907 !!!next-input-character;
2908 } else {
2909 !!!cp (1007);
2910 !!!parse-error (type => 'no refc', line => $l, column => $c);
2911 }
2912
2913 if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) {
2914 !!!cp (1008);
2915 !!!parse-error (type => (sprintf 'invalid character reference:U+%04X', $code), line => $l, column => $c);
2916 $code = 0xFFFD;
2917 } elsif ($code > 0x10FFFF) {
2918 !!!cp (1009);
2919 !!!parse-error (type => (sprintf 'invalid character reference:U-%08X', $code), line => $l, column => $c);
2920 $code = 0xFFFD;
2921 } elsif ($code == 0x000D) {
2922 !!!cp (1010);
2923 !!!parse-error (type => 'CR character reference', line => $l, column => $c);
2924 $code = 0x000A;
2925 } elsif (0x80 <= $code and $code <= 0x9F) {
2926 !!!cp (1011);
2927 !!!parse-error (type => (sprintf 'C1 character reference:U+%04X', $code), line => $l, column => $c);
2928 $code = $c1_entity_char->{$code};
2929 }
2930
2931 return {type => CHARACTER_TOKEN, data => chr $code,
2932 has_reference => 1,
2933 line => $l, column => $c,
2934 };
2935 } # X
2936 } elsif (0x0030 <= $self->{next_char} and
2937 $self->{next_char} <= 0x0039) { # 0..9
2938 my $code = $self->{next_char} - 0x0030;
2939 !!!next-input-character;
2940
2941 while (0x0030 <= $self->{next_char} and
2942 $self->{next_char} <= 0x0039) { # 0..9
2943 !!!cp (1012);
2944 $code *= 10;
2945 $code += $self->{next_char} - 0x0030;
2946
2947 !!!next-input-character;
2948 }
2949
2950 if ($self->{next_char} == 0x003B) { # ;
2951 !!!cp (1013);
2952 !!!next-input-character;
2953 } else {
2954 !!!cp (1014);
2955 !!!parse-error (type => 'no refc', line => $l, column => $c);
2956 }
2957
2958 if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) {
2959 !!!cp (1015);
2960 !!!parse-error (type => (sprintf 'invalid character reference:U+%04X', $code), line => $l, column => $c);
2961 $code = 0xFFFD;
2962 } elsif ($code > 0x10FFFF) {
2963 !!!cp (1016);
2964 !!!parse-error (type => (sprintf 'invalid character reference:U-%08X', $code), line => $l, column => $c);
2965 $code = 0xFFFD;
2966 } elsif ($code == 0x000D) {
2967 !!!cp (1017);
2968 !!!parse-error (type => 'CR character reference', line => $l, column => $c);
2969 $code = 0x000A;
2970 } elsif (0x80 <= $code and $code <= 0x9F) {
2971 !!!cp (1018);
2972 !!!parse-error (type => (sprintf 'C1 character reference:U+%04X', $code), line => $l, column => $c);
2973 $code = $c1_entity_char->{$code};
2974 }
2975
2976 return {type => CHARACTER_TOKEN, data => chr $code, has_reference => 1,
2977 line => $l, column => $c,
2978 };
2979 } else {
2980 !!!cp (1019);
2981 !!!parse-error (type => 'bare nero', line => $l, column => $c);
2982 !!!back-next-input-character ($self->{next_char});
2983 $self->{next_char} = 0x0023; # #
2984 return undef;
2985 }
2986 } elsif ((0x0041 <= $self->{next_char} and
2987 $self->{next_char} <= 0x005A) or
2988 (0x0061 <= $self->{next_char} and
2989 $self->{next_char} <= 0x007A)) {
2990 my $entity_name = chr $self->{next_char};
2991 !!!next-input-character;
2992
2993 my $value = $entity_name;
2994 my $match = 0;
2995 require Whatpm::_NamedEntityList;
2996 our $EntityChar;
2997
2998 while (length $entity_name < 30 and
2999 ## NOTE: Some number greater than the maximum length of entity name
3000 ((0x0041 <= $self->{next_char} and # a
3001 $self->{next_char} <= 0x005A) or # x
3002 (0x0061 <= $self->{next_char} and # a
3003 $self->{next_char} <= 0x007A) or # z
3004 (0x0030 <= $self->{next_char} and # 0
3005 $self->{next_char} <= 0x0039) or # 9
3006 $self->{next_char} == 0x003B)) { # ;
3007 $entity_name .= chr $self->{next_char};
3008 if (defined $EntityChar->{$entity_name}) {
3009 if ($self->{next_char} == 0x003B) { # ;
3010 !!!cp (1020);
3011 $value = $EntityChar->{$entity_name};
3012 $match = 1;
3013 !!!next-input-character;
3014 last;
3015 } else {
3016 !!!cp (1021);
3017 $value = $EntityChar->{$entity_name};
3018 $match = -1;
3019 !!!next-input-character;
3020 }
3021 } else {
3022 !!!cp (1022);
3023 $value .= chr $self->{next_char};
3024 $match *= 2;
3025 !!!next-input-character;
3026 }
3027 }
3028
3029 if ($match > 0) {
3030 !!!cp (1023);
3031 return {type => CHARACTER_TOKEN, data => $value, has_reference => 1,
3032 line => $l, column => $c,
3033 };
3034 } elsif ($match < 0) {
3035 !!!parse-error (type => 'no refc', line => $l, column => $c);
3036 if ($in_attr and $match < -1) {
3037 !!!cp (1024);
3038 return {type => CHARACTER_TOKEN, data => '&'.$entity_name,
3039 line => $l, column => $c,
3040 };
3041 } else {
3042 !!!cp (1025);
3043 return {type => CHARACTER_TOKEN, data => $value, has_reference => 1,
3044 line => $l, column => $c,
3045 };
3046 }
3047 } else {
3048 !!!cp (1026);
3049 !!!parse-error (type => 'bare ero', line => $l, column => $c);
3050 ## NOTE: "No characters are consumed" in the spec.
3051 return {type => CHARACTER_TOKEN, data => '&'.$value,
3052 line => $l, column => $c,
3053 };
3054 }
3055 } else {
3056 !!!cp (1027);
3057 ## no characters are consumed
3058 !!!parse-error (type => 'bare ero', line => $l, column => $c);
3059 return undef;
3060 }
3061 } # _tokenize_attempt_to_consume_an_entity
3062
3063 sub _initialize_tree_constructor ($) {
3064 my $self = shift;
3065 ## NOTE: $self->{document} MUST be specified before this method is called
3066 $self->{document}->strict_error_checking (0);
3067 ## TODO: Turn mutation events off # MUST
3068 ## TODO: Turn loose Document option (manakai extension) on
3069 $self->{document}->manakai_is_html (1); # MUST
3070 } # _initialize_tree_constructor
3071
3072 sub _terminate_tree_constructor ($) {
3073 my $self = shift;
3074 $self->{document}->strict_error_checking (1);
3075 ## TODO: Turn mutation events on
3076 } # _terminate_tree_constructor
3077
3078 ## ISSUE: Should append_child (for example) in script executed in tree construction stage fire mutation events?
3079
3080 { # tree construction stage
3081 my $token;
3082
3083 sub _construct_tree ($) {
3084 my ($self) = @_;
3085
3086 ## When an interactive UA render the $self->{document} available
3087 ## to the user, or when it begin accepting user input, are
3088 ## not defined.
3089
3090 ## Append a character: collect it and all subsequent consecutive
3091 ## characters and insert one Text node whose data is concatenation
3092 ## of all those characters. # MUST
3093
3094 !!!next-token;
3095
3096 undef $self->{form_element};
3097 undef $self->{head_element};
3098 $self->{open_elements} = [];
3099 undef $self->{inner_html_node};
3100
3101 ## NOTE: The "initial" insertion mode.
3102 $self->_tree_construction_initial; # MUST
3103
3104 ## NOTE: The "before html" insertion mode.
3105 $self->_tree_construction_root_element;
3106 $self->{insertion_mode} = BEFORE_HEAD_IM;
3107
3108 ## NOTE: The "before head" insertion mode and so on.
3109 $self->_tree_construction_main;
3110 } # _construct_tree
3111
3112 sub _tree_construction_initial ($) {
3113 my $self = shift;
3114
3115 ## NOTE: "initial" insertion mode
3116
3117 INITIAL: {
3118 if ($token->{type} == DOCTYPE_TOKEN) {
3119 ## NOTE: Conformance checkers MAY, instead of reporting "not HTML5"
3120 ## error, switch to a conformance checking mode for another
3121 ## language.
3122 my $doctype_name = $token->{name};
3123 $doctype_name = '' unless defined $doctype_name;
3124 $doctype_name =~ tr/a-z/A-Z/;
3125 if (not defined $token->{name} or # <!DOCTYPE>
3126 defined $token->{public_identifier} or
3127 defined $token->{system_identifier}) {
3128 !!!cp ('t1');
3129 !!!parse-error (type => 'not HTML5', token => $token);
3130 } elsif ($doctype_name ne 'HTML') {
3131 !!!cp ('t2');
3132 ## ISSUE: ASCII case-insensitive? (in fact it does not matter)
3133 !!!parse-error (type => 'not HTML5', token => $token);
3134 } else {
3135 !!!cp ('t3');
3136 }
3137
3138 my $doctype = $self->{document}->create_document_type_definition
3139 ($token->{name}); ## ISSUE: If name is missing (e.g. <!DOCTYPE>)?
3140 ## NOTE: Default value for both |public_id| and |system_id| attributes
3141 ## are empty strings, so that we don't set any value in missing cases.
3142 $doctype->public_id ($token->{public_identifier})
3143 if defined $token->{public_identifier};
3144 $doctype->system_id ($token->{system_identifier})
3145 if defined $token->{system_identifier};
3146 ## NOTE: Other DocumentType attributes are null or empty lists.
3147 ## ISSUE: internalSubset = null??
3148 $self->{document}->append_child ($doctype);
3149
3150 if ($token->{quirks} or $doctype_name ne 'HTML') {
3151 !!!cp ('t4');
3152 $self->{document}->manakai_compat_mode ('quirks');
3153 } elsif (defined $token->{public_identifier}) {
3154 my $pubid = $token->{public_identifier};
3155 $pubid =~ tr/a-z/A-z/;
3156 my $prefix = [
3157 "+//SILMARIL//DTD HTML PRO V0R11 19970101//",
3158 "-//ADVASOFT LTD//DTD HTML 3.0 ASWEDIT + EXTENSIONS//",
3159 "-//AS//DTD HTML 3.0 ASWEDIT + EXTENSIONS//",
3160 "-//IETF//DTD HTML 2.0 LEVEL 1//",
3161 "-//IETF//DTD HTML 2.0 LEVEL 2//",
3162 "-//IETF//DTD HTML 2.0 STRICT LEVEL 1//",
3163 "-//IETF//DTD HTML 2.0 STRICT LEVEL 2//",
3164 "-//IETF//DTD HTML 2.0 STRICT//",
3165 "-//IETF//DTD HTML 2.0//",
3166 "-//IETF//DTD HTML 2.1E//",
3167 "-//IETF//DTD HTML 3.0//",
3168 "-//IETF//DTD HTML 3.2 FINAL//",
3169 "-//IETF//DTD HTML 3.2//",
3170 "-//IETF//DTD HTML 3//",
3171 "-//IETF//DTD HTML LEVEL 0//",
3172 "-//IETF//DTD HTML LEVEL 1//",
3173 "-//IETF//DTD HTML LEVEL 2//",
3174 "-//IETF//DTD HTML LEVEL 3//",
3175 "-//IETF//DTD HTML STRICT LEVEL 0//",
3176 "-//IETF//DTD HTML STRICT LEVEL 1//",
3177 "-//IETF//DTD HTML STRICT LEVEL 2//",
3178 "-//IETF//DTD HTML STRICT LEVEL 3//",
3179 "-//IETF//DTD HTML STRICT//",
3180 "-//IETF//DTD HTML//",
3181 "-//METRIUS//DTD METRIUS PRESENTATIONAL//",
3182 "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 HTML STRICT//",
3183 "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 HTML//",
3184 "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 TABLES//",
3185 "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 HTML STRICT//",
3186 "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 HTML//",
3187 "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 TABLES//",
3188 "-//NETSCAPE COMM. CORP.//DTD HTML//",
3189 "-//NETSCAPE COMM. CORP.//DTD STRICT HTML//",
3190 "-//O'REILLY AND ASSOCIATES//DTD HTML 2.0//",
3191 "-//O'REILLY AND ASSOCIATES//DTD HTML EXTENDED 1.0//",
3192 "-//O'REILLY AND ASSOCIATES//DTD HTML EXTENDED RELAXED 1.0//",
3193 "-//SOFTQUAD SOFTWARE//DTD HOTMETAL PRO 6.0::19990601::EXTENSIONS TO HTML 4.0//",
3194 "-//SOFTQUAD//DTD HOTMETAL PRO 4.0::19971010::EXTENSIONS TO HTML 4.0//",
3195 "-//SPYGLASS//DTD HTML 2.0 EXTENDED//",
3196 "-//SQ//DTD HTML 2.0 HOTMETAL + EXTENSIONS//",
3197 "-//SUN MICROSYSTEMS CORP.//DTD HOTJAVA HTML//",
3198 "-//SUN MICROSYSTEMS CORP.//DTD HOTJAVA STRICT HTML//",
3199 "-//W3C//DTD HTML 3 1995-03-24//",
3200 "-//W3C//DTD HTML 3.2 DRAFT//",
3201 "-//W3C//DTD HTML 3.2 FINAL//",
3202 "-//W3C//DTD HTML 3.2//",
3203 "-//W3C//DTD HTML 3.2S DRAFT//",
3204 "-//W3C//DTD HTML 4.0 FRAMESET//",
3205 "-//W3C//DTD HTML 4.0 TRANSITIONAL//",
3206 "-//W3C//DTD HTML EXPERIMETNAL 19960712//",
3207 "-//W3C//DTD HTML EXPERIMENTAL 970421//",
3208 "-//W3C//DTD W3 HTML//",
3209 "-//W3O//DTD W3 HTML 3.0//",
3210 "-//WEBTECHS//DTD MOZILLA HTML 2.0//",
3211 "-//WEBTECHS//DTD MOZILLA HTML//",
3212 ]; # $prefix
3213 my $match;
3214 for (@$prefix) {
3215 if (substr ($prefix, 0, length $_) eq $_) {
3216 $match = 1;
3217 last;
3218 }
3219 }
3220 if ($match or
3221 $pubid eq "-//W3O//DTD W3 HTML STRICT 3.0//EN//" or
3222 $pubid eq "-/W3C/DTD HTML 4.0 TRANSITIONAL/EN" or
3223 $pubid eq "HTML") {
3224 !!!cp ('t5');
3225 $self->{document}->manakai_compat_mode ('quirks');
3226 } elsif ($pubid =~ m[^-//W3C//DTD HTML 4.01 FRAMESET//] or
3227 $pubid =~ m[^-//W3C//DTD HTML 4.01 TRANSITIONAL//]) {
3228 if (defined $token->{system_identifier}) {
3229 !!!cp ('t6');
3230 $self->{document}->manakai_compat_mode ('quirks');
3231 } else {
3232 !!!cp ('t7');
3233 $self->{document}->manakai_compat_mode ('limited quirks');
3234 }
3235 } elsif ($pubid =~ m[^-//W3C//DTD XHTML 1.0 FRAMESET//] or
3236 $pubid =~ m[^-//W3C//DTD XHTML 1.0 TRANSITIONAL//]) {
3237 !!!cp ('t8');
3238 $self->{document}->manakai_compat_mode ('limited quirks');
3239 } else {
3240 !!!cp ('t9');
3241 }
3242 } else {
3243 !!!cp ('t10');
3244 }
3245 if (defined $token->{system_identifier}) {
3246 my $sysid = $token->{system_identifier};
3247 $sysid =~ tr/A-Z/a-z/;
3248 if ($sysid eq "http://www.ibm.com/data/dtd/v11/ibmxhtml1-transitional.dtd") {
3249 ## NOTE: Ensure that |PUBLIC "(limited quirks)" "(quirks)"| is
3250 ## marked as quirks.
3251 $self->{document}->manakai_compat_mode ('quirks');
3252 !!!cp ('t11');
3253 } else {
3254 !!!cp ('t12');
3255 }
3256 } else {
3257 !!!cp ('t13');
3258 }
3259
3260 ## Go to the "before html" insertion mode.
3261 !!!next-token;
3262 return;
3263 } elsif ({
3264 START_TAG_TOKEN, 1,
3265 END_TAG_TOKEN, 1,
3266 END_OF_FILE_TOKEN, 1,
3267 }->{$token->{type}}) {
3268 !!!cp ('t14');
3269 !!!parse-error (type => 'no DOCTYPE', token => $token);
3270 $self->{document}->manakai_compat_mode ('quirks');
3271 ## Go to the "before html" insertion mode.
3272 ## reprocess
3273 !!!ack-later;
3274 return;
3275 } elsif ($token->{type} == CHARACTER_TOKEN) {
3276 if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D
3277 ## Ignore the token
3278
3279 unless (length $token->{data}) {
3280 !!!cp ('t15');
3281 ## Stay in the insertion mode.
3282 !!!next-token;
3283 redo INITIAL;
3284 } else {
3285 !!!cp ('t16');
3286 }
3287 } else {
3288 !!!cp ('t17');
3289 }
3290
3291 !!!parse-error (type => 'no DOCTYPE', token => $token);
3292 $self->{document}->manakai_compat_mode ('quirks');
3293 ## Go to the "before html" insertion mode.
3294 ## reprocess
3295 return;
3296 } elsif ($token->{type} == COMMENT_TOKEN) {
3297 !!!cp ('t18');
3298 my $comment = $self->{document}->create_comment ($token->{data});
3299 $self->{document}->append_child ($comment);
3300
3301 ## Stay in the insertion mode.
3302 !!!next-token;
3303 redo INITIAL;
3304 } else {
3305 die "$0: $token->{type}: Unknown token type";
3306 }
3307 } # INITIAL
3308
3309 die "$0: _tree_construction_initial: This should be never reached";
3310 } # _tree_construction_initial
3311
3312 sub _tree_construction_root_element ($) {
3313 my $self = shift;
3314
3315 ## NOTE: "before html" insertion mode.
3316
3317 B: {
3318 if ($token->{type} == DOCTYPE_TOKEN) {
3319 !!!cp ('t19');
3320 !!!parse-error (type => 'in html:#DOCTYPE', token => $token);
3321 ## Ignore the token
3322 ## Stay in the insertion mode.
3323 !!!next-token;
3324 redo B;
3325 } elsif ($token->{type} == COMMENT_TOKEN) {
3326 !!!cp ('t20');
3327 my $comment = $self->{document}->create_comment ($token->{data});
3328 $self->{document}->append_child ($comment);
3329 ## Stay in the insertion mode.
3330 !!!next-token;
3331 redo B;
3332 } elsif ($token->{type} == CHARACTER_TOKEN) {
3333 if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D
3334 ## Ignore the token.
3335
3336 unless (length $token->{data}) {
3337 !!!cp ('t21');
3338 ## Stay in the insertion mode.
3339 !!!next-token;
3340 redo B;
3341 } else {
3342 !!!cp ('t22');
3343 }
3344 } else {
3345 !!!cp ('t23');
3346 }
3347
3348 $self->{application_cache_selection}->(undef);
3349
3350 #
3351 } elsif ($token->{type} == START_TAG_TOKEN) {
3352 if ($token->{tag_name} eq 'html') {
3353 my $root_element;
3354 !!!create-element ($root_element, $HTML_NS, $token->{tag_name}, $token->{attributes}, $token);
3355 $self->{document}->append_child ($root_element);
3356 push @{$self->{open_elements}},
3357 [$root_element, $el_category->{html}];
3358
3359 if ($token->{attributes}->{manifest}) {
3360 !!!cp ('t24');
3361 $self->{application_cache_selection}
3362 ->($token->{attributes}->{manifest}->{value});
3363 ## ISSUE: Spec is unclear on relative references.
3364 ## According to Hixie (#whatwg 2008-03-19), it should be
3365 ## resolved against the base URI of the document in HTML
3366 ## or xml:base of the element in XHTML.
3367 } else {
3368 !!!cp ('t25');
3369 $self->{application_cache_selection}->(undef);
3370 }
3371
3372 !!!nack ('t25c');
3373
3374 !!!next-token;
3375 return; ## Go to the "before head" insertion mode.
3376 } else {
3377 !!!cp ('t25.1');
3378 #
3379 }
3380 } elsif ({
3381 END_TAG_TOKEN, 1,
3382 END_OF_FILE_TOKEN, 1,
3383 }->{$token->{type}}) {
3384 !!!cp ('t26');
3385 #
3386 } else {
3387 die "$0: $token->{type}: Unknown token type";
3388 }
3389
3390 my $root_element;
3391 !!!create-element ($root_element, $HTML_NS, 'html',, $token);
3392 $self->{document}->append_child ($root_element);
3393 push @{$self->{open_elements}}, [$root_element, $el_category->{html}];
3394
3395 $self->{application_cache_selection}->(undef);
3396
3397 ## NOTE: Reprocess the token.
3398 !!!ack-later;
3399 return; ## Go to the "before head" insertion mode.
3400
3401 ## ISSUE: There is an issue in the spec
3402 } # B
3403
3404 die "$0: _tree_construction_root_element: This should never be reached";
3405 } # _tree_construction_root_element
3406
3407 sub _reset_insertion_mode ($) {
3408 my $self = shift;
3409
3410 ## Step 1
3411 my $last;
3412
3413 ## Step 2
3414 my $i = -1;
3415 my $node = $self->{open_elements}->[$i];
3416
3417 ## Step 3
3418 S3: {
3419 if ($self->{open_elements}->[0]->[0] eq $node->[0]) {
3420 $last = 1;
3421 if (defined $self->{inner_html_node}) {
3422 !!!cp ('t28');
3423 $node = $self->{inner_html_node};
3424 } else {
3425 die "_reset_insertion_mode: t27";
3426 }
3427 }
3428
3429 ## Step 4..14
3430 my $new_mode;
3431 if ($node->[1] & FOREIGN_EL) {
3432 !!!cp ('t28.1');
3433 ## NOTE: Strictly spaking, the line below only applies to MathML and
3434 ## SVG elements. Currently the HTML syntax supports only MathML and
3435 ## SVG elements as foreigners.
3436 $new_mode = IN_BODY_IM | IN_FOREIGN_CONTENT_IM;
3437 } elsif ($node->[1] & TABLE_CELL_EL) {
3438 if ($last) {
3439 !!!cp ('t28.2');
3440 #
3441 } else {
3442 !!!cp ('t28.3');
3443 $new_mode = IN_CELL_IM;
3444 }
3445 } else {
3446 !!!cp ('t28.4');
3447 $new_mode = {
3448 select => IN_SELECT_IM,
3449 ## NOTE: |option| and |optgroup| do not set
3450 ## insertion mode to "in select" by themselves.
3451 tr => IN_ROW_IM,
3452 tbody => IN_TABLE_BODY_IM,
3453 thead => IN_TABLE_BODY_IM,
3454 tfoot => IN_TABLE_BODY_IM,
3455 caption => IN_CAPTION_IM,
3456 colgroup => IN_COLUMN_GROUP_IM,
3457 table => IN_TABLE_IM,
3458 head => IN_BODY_IM, # not in head!
3459 body => IN_BODY_IM,
3460 frameset => IN_FRAMESET_IM,
3461 }->{$node->[0]->manakai_local_name};
3462 }
3463 $self->{insertion_mode} = $new_mode and return if defined $new_mode;
3464
3465 ## Step 15
3466 if ($node->[1] & HTML_EL) {
3467 unless (defined $self->{head_element}) {
3468 !!!cp ('t29');
3469 $self->{insertion_mode} = BEFORE_HEAD_IM;
3470 } else {
3471 ## ISSUE: Can this state be reached?
3472 !!!cp ('t30');
3473 $self->{insertion_mode} = AFTER_HEAD_IM;
3474 }
3475 return;
3476 } else {
3477 !!!cp ('t31');
3478 }
3479
3480 ## Step 16
3481 $self->{insertion_mode} = IN_BODY_IM and return if $last;
3482
3483 ## Step 17
3484 $i--;
3485 $node = $self->{open_elements}->[$i];
3486
3487 ## Step 18
3488 redo S3;
3489 } # S3
3490
3491 die "$0: _reset_insertion_mode: This line should never be reached";
3492 } # _reset_insertion_mode
3493
3494 sub _tree_construction_main ($) {
3495 my $self = shift;
3496
3497 my $active_formatting_elements = [];
3498
3499 my $reconstruct_active_formatting_elements = sub { # MUST
3500 my $insert = shift;
3501
3502 ## Step 1
3503 return unless @$active_formatting_elements;
3504
3505 ## Step 3
3506 my $i = -1;
3507 my $entry = $active_formatting_elements->[$i];
3508
3509 ## Step 2
3510 return if $entry->[0] eq '#marker';
3511 for (@{$self->{open_elements}}) {
3512 if ($entry->[0] eq $_->[0]) {
3513 !!!cp ('t32');
3514 return;
3515 }
3516 }
3517
3518 S4: {
3519 ## Step 4
3520 last S4 if $active_formatting_elements->[0]->[0] eq $entry->[0];
3521
3522 ## Step 5
3523 $i--;
3524 $entry = $active_formatting_elements->[$i];
3525
3526 ## Step 6
3527 if ($entry->[0] eq '#marker') {
3528 !!!cp ('t33_1');
3529 #
3530 } else {
3531 my $in_open_elements;
3532 OE: for (@{$self->{open_elements}}) {
3533 if ($entry->[0] eq $_->[0]) {
3534 !!!cp ('t33');
3535 $in_open_elements = 1;
3536 last OE;
3537 }
3538 }
3539 if ($in_open_elements) {
3540 !!!cp ('t34');
3541 #
3542 } else {
3543 ## NOTE: <!DOCTYPE HTML><p><b><i><u></p> <p>X
3544 !!!cp ('t35');
3545 redo S4;
3546 }
3547 }
3548
3549 ## Step 7
3550 $i++;
3551 $entry = $active_formatting_elements->[$i];
3552 } # S4
3553
3554 S7: {
3555 ## Step 8
3556 my $clone = [$entry->[0]->clone_node (0), $entry->[1]];
3557
3558 ## Step 9
3559 $insert->($clone->[0]);
3560 push @{$self->{open_elements}}, $clone;
3561
3562 ## Step 10
3563 $active_formatting_elements->[$i] = $self->{open_elements}->[-1];
3564
3565 ## Step 11
3566 unless ($clone->[0] eq $active_formatting_elements->[-1]->[0]) {
3567 !!!cp ('t36');
3568 ## Step 7'
3569 $i++;
3570 $entry = $active_formatting_elements->[$i];
3571
3572 redo S7;
3573 }
3574
3575 !!!cp ('t37');
3576 } # S7
3577 }; # $reconstruct_active_formatting_elements
3578
3579 my $clear_up_to_marker = sub {
3580 for (reverse 0..$#$active_formatting_elements) {
3581 if ($active_formatting_elements->[$_]->[0] eq '#marker') {
3582 !!!cp ('t38');
3583 splice @$active_formatting_elements, $_;
3584 return;
3585 }
3586 }
3587
3588 !!!cp ('t39');
3589 }; # $clear_up_to_marker
3590
3591 my $insert;
3592
3593 my $parse_rcdata = sub ($) {
3594 my ($content_model_flag) = @_;
3595
3596 ## Step 1
3597 my $start_tag_name = $token->{tag_name};
3598 my $el;
3599 !!!create-element ($el, $HTML_NS, $start_tag_name, $token->{attributes}, $token);
3600
3601 ## Step 2
3602 $insert->($el);
3603
3604 ## Step 3
3605 $self->{content_model} = $content_model_flag; # CDATA or RCDATA
3606 delete $self->{escape}; # MUST
3607
3608 ## Step 4
3609 my $text = '';
3610 !!!nack ('t40.1');
3611 !!!next-token;
3612 while ($token->{type} == CHARACTER_TOKEN) { # or until stop tokenizing
3613 !!!cp ('t40');
3614 $text .= $token->{data};
3615 !!!next-token;
3616 }
3617
3618 ## Step 5
3619 if (length $text) {
3620 !!!cp ('t41');
3621 my $text = $self->{document}->create_text_node ($text);
3622 $el->append_child ($text);
3623 }
3624
3625 ## Step 6
3626 $self->{content_model} = PCDATA_CONTENT_MODEL;
3627
3628 ## Step 7
3629 if ($token->{type} == END_TAG_TOKEN and
3630 $token->{tag_name} eq $start_tag_name) {
3631 !!!cp ('t42');
3632 ## Ignore the token
3633 } else {
3634 ## NOTE: An end-of-file token.
3635 if ($content_model_flag == CDATA_CONTENT_MODEL) {
3636 !!!cp ('t43');
3637 !!!parse-error (type => 'in CDATA:#'.$token->{type}, token => $token);
3638 } elsif ($content_model_flag == RCDATA_CONTENT_MODEL) {
3639 !!!cp ('t44');
3640 !!!parse-error (type => 'in RCDATA:#'.$token->{type}, token => $token);
3641 } else {
3642 die "$0: $content_model_flag in parse_rcdata";
3643 }
3644 }
3645 !!!next-token;
3646 }; # $parse_rcdata
3647
3648 my $script_start_tag = sub () {
3649 my $script_el;
3650 !!!create-element ($script_el, $HTML_NS, 'script', $token->{attributes}, $token);
3651 ## TODO: mark as "parser-inserted"
3652
3653 $self->{content_model} = CDATA_CONTENT_MODEL;
3654 delete $self->{escape}; # MUST
3655
3656 my $text = '';
3657 !!!nack ('t45.1');
3658 !!!next-token;
3659 while ($token->{type} == CHARACTER_TOKEN) {
3660 !!!cp ('t45');
3661 $text .= $token->{data};
3662 !!!next-token;
3663 } # stop if non-character token or tokenizer stops tokenising
3664 if (length $text) {
3665 !!!cp ('t46');
3666 $script_el->manakai_append_text ($text);
3667 }
3668
3669 $self->{content_model} = PCDATA_CONTENT_MODEL;
3670
3671 if ($token->{type} == END_TAG_TOKEN and
3672 $token->{tag_name} eq 'script') {
3673 !!!cp ('t47');
3674 ## Ignore the token
3675 } else {
3676 !!!cp ('t48');
3677 !!!parse-error (type => 'in CDATA:#'.$token->{type}, token => $token);
3678 ## ISSUE: And ignore?
3679 ## TODO: mark as "already executed"
3680 }
3681
3682 if (defined $self->{inner_html_node}) {
3683 !!!cp ('t49');
3684 ## TODO: mark as "already executed"
3685 } else {
3686 !!!cp ('t50');
3687 ## TODO: $old_insertion_point = current insertion point
3688 ## TODO: insertion point = just before the next input character
3689
3690 $insert->($script_el);
3691
3692 ## TODO: insertion point = $old_insertion_point (might be "undefined")
3693
3694 ## TODO: if there is a script that will execute as soon as the parser resume, then...
3695 }
3696
3697 !!!next-token;
3698 }; # $script_start_tag
3699
3700 ## NOTE: $open_tables->[-1]->[0] is the "current table" element node.
3701 ## NOTE: $open_tables->[-1]->[1] is the "tainted" flag.
3702 my $open_tables = [[$self->{open_elements}->[0]->[0]]];
3703
3704 my $formatting_end_tag = sub {
3705 my $end_tag_token = shift;
3706 my $tag_name = $end_tag_token->{tag_name};
3707
3708 ## NOTE: The adoption agency algorithm (AAA).
3709
3710 FET: {
3711 ## Step 1
3712 my $formatting_element;
3713 my $formatting_element_i_in_active;
3714 AFE: for (reverse 0..$#$active_formatting_elements) {
3715 if ($active_formatting_elements->[$_]->[0] eq '#marker') {
3716 !!!cp ('t52');
3717 last AFE;
3718 } elsif ($active_formatting_elements->[$_]->[0]->manakai_local_name
3719 eq $tag_name) {
3720 !!!cp ('t51');
3721 $formatting_element = $active_formatting_elements->[$_];
3722 $formatting_element_i_in_active = $_;
3723 last AFE;
3724 }
3725 } # AFE
3726 unless (defined $formatting_element) {
3727 !!!cp ('t53');
3728 !!!parse-error (type => 'unmatched end tag:'.$tag_name, token => $end_tag_token);
3729 ## Ignore the token
3730 !!!next-token;
3731 return;
3732 }
3733 ## has an element in scope
3734 my $in_scope = 1;
3735 my $formatting_element_i_in_open;
3736 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3737 my $node = $self->{open_elements}->[$_];
3738 if ($node->[0] eq $formatting_element->[0]) {
3739 if ($in_scope) {
3740 !!!cp ('t54');
3741 $formatting_element_i_in_open = $_;
3742 last INSCOPE;
3743 } else { # in open elements but not in scope
3744 !!!cp ('t55');
3745 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name},
3746 token => $end_tag_token);
3747 ## Ignore the token
3748 !!!next-token;
3749 return;
3750 }
3751 } elsif ($node->[1] & SCOPING_EL) {
3752 !!!cp ('t56');
3753 $in_scope = 0;
3754 }
3755 } # INSCOPE
3756 unless (defined $formatting_element_i_in_open) {
3757 !!!cp ('t57');
3758 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name},
3759 token => $end_tag_token);
3760 pop @$active_formatting_elements; # $formatting_element
3761 !!!next-token; ## TODO: ok?
3762 return;
3763 }
3764 if (not $self->{open_elements}->[-1]->[0] eq $formatting_element->[0]) {
3765 !!!cp ('t58');
3766 !!!parse-error (type => 'not closed',
3767 value => $self->{open_elements}->[-1]->[0]
3768 ->manakai_local_name,
3769 token => $end_tag_token);
3770 }
3771
3772 ## Step 2
3773 my $furthest_block;
3774 my $furthest_block_i_in_open;
3775 OE: for (reverse 0..$#{$self->{open_elements}}) {
3776 my $node = $self->{open_elements}->[$_];
3777 if (not ($node->[1] & FORMATTING_EL) and
3778 #not $phrasing_category->{$node->[1]} and
3779 ($node->[1] & SPECIAL_EL or
3780 $node->[1] & SCOPING_EL)) { ## Scoping is redundant, maybe
3781 !!!cp ('t59');
3782 $furthest_block = $node;
3783 $furthest_block_i_in_open = $_;
3784 } elsif ($node->[0] eq $formatting_element->[0]) {
3785 !!!cp ('t60');
3786 last OE;
3787 }
3788 } # OE
3789
3790 ## Step 3
3791 unless (defined $furthest_block) { # MUST
3792 !!!cp ('t61');
3793 splice @{$self->{open_elements}}, $formatting_element_i_in_open;
3794 splice @$active_formatting_elements, $formatting_element_i_in_active, 1;
3795 !!!next-token;
3796 return;
3797 }
3798
3799 ## Step 4
3800 my $common_ancestor_node = $self->{open_elements}->[$formatting_element_i_in_open - 1];
3801
3802 ## Step 5
3803 my $furthest_block_parent = $furthest_block->[0]->parent_node;
3804 if (defined $furthest_block_parent) {
3805 !!!cp ('t62');
3806 $furthest_block_parent->remove_child ($furthest_block->[0]);
3807 }
3808
3809 ## Step 6
3810 my $bookmark_prev_el
3811 = $active_formatting_elements->[$formatting_element_i_in_active - 1]
3812 ->[0];
3813
3814 ## Step 7
3815 my $node = $furthest_block;
3816 my $node_i_in_open = $furthest_block_i_in_open;
3817 my $last_node = $furthest_block;
3818 S7: {
3819 ## Step 1
3820 $node_i_in_open--;
3821 $node = $self->{open_elements}->[$node_i_in_open];
3822
3823 ## Step 2
3824 my $node_i_in_active;
3825 S7S2: {
3826 for (reverse 0..$#$active_formatting_elements) {
3827 if ($active_formatting_elements->[$_]->[0] eq $node->[0]) {
3828 !!!cp ('t63');
3829 $node_i_in_active = $_;
3830 last S7S2;
3831 }
3832 }
3833 splice @{$self->{open_elements}}, $node_i_in_open, 1;
3834 redo S7;
3835 } # S7S2
3836
3837 ## Step 3
3838 last S7 if $node->[0] eq $formatting_element->[0];
3839
3840 ## Step 4
3841 if ($last_node->[0] eq $furthest_block->[0]) {
3842 !!!cp ('t64');
3843 $bookmark_prev_el = $node->[0];
3844 }
3845
3846 ## Step 5
3847 if ($node->[0]->has_child_nodes ()) {
3848 !!!cp ('t65');
3849 my $clone = [$node->[0]->clone_node (0), $node->[1]];
3850 $active_formatting_elements->[$node_i_in_active] = $clone;
3851 $self->{open_elements}->[$node_i_in_open] = $clone;
3852 $node = $clone;
3853 }
3854
3855 ## Step 6
3856 $node->[0]->append_child ($last_node->[0]);
3857
3858 ## Step 7
3859 $last_node = $node;
3860
3861 ## Step 8
3862 redo S7;
3863 } # S7
3864
3865 ## Step 8
3866 if ($common_ancestor_node->[1] & TABLE_ROWS_EL) {
3867 my $foster_parent_element;
3868 my $next_sibling;
3869 OE: for (reverse 0..$#{$self->{open_elements}}) {
3870 if ($self->{open_elements}->[$_]->[1] & TABLE_EL) {
3871 my $parent = $self->{open_elements}->[$_]->[0]->parent_node;
3872 if (defined $parent and $parent->node_type == 1) {
3873 !!!cp ('t65.1');
3874 $foster_parent_element = $parent;
3875 $next_sibling = $self->{open_elements}->[$_]->[0];
3876 } else {
3877 !!!cp ('t65.2');
3878 $foster_parent_element
3879 = $self->{open_elements}->[$_ - 1]->[0];
3880 }
3881 last OE;
3882 }
3883 } # OE
3884 $foster_parent_element = $self->{open_elements}->[0]->[0]
3885 unless defined $foster_parent_element;
3886 $foster_parent_element->insert_before ($last_node->[0], $next_sibling);
3887 $open_tables->[-1]->[1] = 1; # tainted
3888 } else {
3889 !!!cp ('t65.3');
3890 $common_ancestor_node->[0]->append_child ($last_node->[0]);
3891 }
3892
3893 ## Step 9
3894 my $clone = [$formatting_element->[0]->clone_node (0),
3895 $formatting_element->[1]];
3896
3897 ## Step 10
3898 my @cn = @{$furthest_block->[0]->child_nodes};
3899 $clone->[0]->append_child ($_) for @cn;
3900
3901 ## Step 11
3902 $furthest_block->[0]->append_child ($clone->[0]);
3903
3904 ## Step 12
3905 my $i;
3906 AFE: for (reverse 0..$#$active_formatting_elements) {
3907 if ($active_formatting_elements->[$_]->[0] eq $formatting_element->[0]) {
3908 !!!cp ('t66');
3909 splice @$active_formatting_elements, $_, 1;
3910 $i-- and last AFE if defined $i;
3911 } elsif ($active_formatting_elements->[$_]->[0] eq $bookmark_prev_el) {
3912 !!!cp ('t67');
3913 $i = $_;
3914 }
3915 } # AFE
3916 splice @$active_formatting_elements, $i + 1, 0, $clone;
3917
3918 ## Step 13
3919 undef $i;
3920 OE: for (reverse 0..$#{$self->{open_elements}}) {
3921 if ($self->{open_elements}->[$_]->[0] eq $formatting_element->[0]) {
3922 !!!cp ('t68');
3923 splice @{$self->{open_elements}}, $_, 1;
3924 $i-- and last OE if defined $i;
3925 } elsif ($self->{open_elements}->[$_]->[0] eq $furthest_block->[0]) {
3926 !!!cp ('t69');
3927 $i = $_;
3928 }
3929 } # OE
3930 splice @{$self->{open_elements}}, $i + 1, 1, $clone;
3931
3932 ## Step 14
3933 redo FET;
3934 } # FET
3935 }; # $formatting_end_tag
3936
3937 $insert = my $insert_to_current = sub {
3938 $self->{open_elements}->[-1]->[0]->append_child ($_[0]);
3939 }; # $insert_to_current
3940
3941 my $insert_to_foster = sub {
3942 my $child = shift;
3943 if ($self->{open_elements}->[-1]->[1] & TABLE_ROWS_EL) {
3944 # MUST
3945 my $foster_parent_element;
3946 my $next_sibling;
3947 OE: for (reverse 0..$#{$self->{open_elements}}) {
3948 if ($self->{open_elements}->[$_]->[1] & TABLE_EL) {
3949 my $parent = $self->{open_elements}->[$_]->[0]->parent_node;
3950 if (defined $parent and $parent->node_type == 1) {
3951 !!!cp ('t70');
3952 $foster_parent_element = $parent;
3953 $next_sibling = $self->{open_elements}->[$_]->[0];
3954 } else {
3955 !!!cp ('t71');
3956 $foster_parent_element
3957 = $self->{open_elements}->[$_ - 1]->[0];
3958 }
3959 last OE;
3960 }
3961 } # OE
3962 $foster_parent_element = $self->{open_elements}->[0]->[0]
3963 unless defined $foster_parent_element;
3964 $foster_parent_element->insert_before
3965 ($child, $next_sibling);
3966 $open_tables->[-1]->[1] = 1; # tainted
3967 } else {
3968 !!!cp ('t72');
3969 $self->{open_elements}->[-1]->[0]->append_child ($child);
3970 }
3971 }; # $insert_to_foster
3972
3973 B: while (1) {
3974 if ($token->{type} == DOCTYPE_TOKEN) {
3975 !!!cp ('t73');
3976 !!!parse-error (type => 'DOCTYPE in the middle', token => $token);
3977 ## Ignore the token
3978 ## Stay in the phase
3979 !!!next-token;
3980 next B;
3981 } elsif ($token->{type} == START_TAG_TOKEN and
3982 $token->{tag_name} eq 'html') {
3983 if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
3984 !!!cp ('t79');
3985 !!!parse-error (type => 'after html:html', token => $token);
3986 $self->{insertion_mode} = AFTER_BODY_IM;
3987 } elsif ($self->{insertion_mode} == AFTER_HTML_FRAMESET_IM) {
3988 !!!cp ('t80');
3989 !!!parse-error (type => 'after html:html', token => $token);
3990 $self->{insertion_mode} = AFTER_FRAMESET_IM;
3991 } else {
3992 !!!cp ('t81');
3993 }
3994
3995 !!!cp ('t82');
3996 !!!parse-error (type => 'not first start tag', token => $token);
3997 my $top_el = $self->{open_elements}->[0]->[0];
3998 for my $attr_name (keys %{$token->{attributes}}) {
3999 unless ($top_el->has_attribute_ns (undef, $attr_name)) {
4000 !!!cp ('t84');
4001 $top_el->set_attribute_ns
4002 (undef, [undef, $attr_name],
4003 $token->{attributes}->{$attr_name}->{value});
4004 }
4005 }
4006 !!!nack ('t84.1');
4007 !!!next-token;
4008 next B;
4009 } elsif ($token->{type} == COMMENT_TOKEN) {
4010 my $comment = $self->{document}->create_comment ($token->{data});
4011 if ($self->{insertion_mode} & AFTER_HTML_IMS) {
4012 !!!cp ('t85');
4013 $self->{document}->append_child ($comment);
4014 } elsif ($self->{insertion_mode} == AFTER_BODY_IM) {
4015 !!!cp ('t86');
4016 $self->{open_elements}->[0]->[0]->append_child ($comment);
4017 } else {
4018 !!!cp ('t87');
4019 $self->{open_elements}->[-1]->[0]->append_child ($comment);
4020 }
4021 !!!next-token;
4022 next B;
4023 } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM) {
4024 if ($token->{type} == CHARACTER_TOKEN) {
4025 !!!cp ('t87.1');
4026 $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});
4027 !!!next-token;
4028 next B;
4029 } elsif ($token->{type} == START_TAG_TOKEN) {
4030 if ((not {mglyph => 1, malignmark => 1}->{$token->{tag_name}} and
4031 $self->{open_elements}->[-1]->[1] & FOREIGN_FLOW_CONTENT_EL) or
4032 not ($self->{open_elements}->[-1]->[1] & FOREIGN_EL) or
4033 ($token->{tag_name} eq 'svg' and
4034 $self->{open_elements}->[-1]->[1] & MML_AXML_EL)) {
4035 ## NOTE: "using the rules for secondary insertion mode"then"continue"
4036 !!!cp ('t87.2');
4037 #
4038 } elsif ({
4039 b => 1, big => 1, blockquote => 1, body => 1, br => 1,
4040 center => 1, code => 1, dd => 1, div => 1, dl => 1, dt => 1,
4041 em => 1, embed => 1, font => 1, h1 => 1, h2 => 1, h3 => 1,
4042 h4 => 1, h5 => 1, h6 => 1, head => 1, hr => 1, i => 1,
4043 img => 1, li => 1, listing => 1, menu => 1, meta => 1,
4044 nobr => 1, ol => 1, p => 1, pre => 1, ruby => 1, s => 1,
4045 small => 1, span => 1, strong => 1, strike => 1, sub => 1,
4046 sup => 1, table => 1, tt => 1, u => 1, ul => 1, var => 1,
4047 }->{$token->{tag_name}}) {
4048 !!!cp ('t87.2');
4049 !!!parse-error (type => 'not closed',
4050 value => $self->{open_elements}->[-1]->[0]
4051 ->manakai_local_name,
4052 token => $token);
4053
4054 pop @{$self->{open_elements}}
4055 while $self->{open_elements}->[-1]->[1] & FOREIGN_EL;
4056
4057 $self->{insertion_mode} &= ~ IN_FOREIGN_CONTENT_IM;
4058 ## Reprocess.
4059 next B;
4060 } else {
4061 my $nsuri = $self->{open_elements}->[-1]->[0]->namespace_uri;
4062 my $tag_name = $token->{tag_name};
4063 if ($nsuri eq $SVG_NS) {
4064 $tag_name = {
4065 altglyph => 'altGlyph',
4066 altglyphdef => 'altGlyphDef',
4067 altglyphitem => 'altGlyphItem',
4068 animatecolor => 'animateColor',
4069 animatemotion => 'animateMotion',
4070 animatetransform => 'animateTransform',
4071 clippath => 'clipPath',
4072 feblend => 'feBlend',
4073 fecolormatrix => 'feColorMatrix',
4074 fecomponenttransfer => 'feComponentTransfer',
4075 fecomposite => 'feComposite',
4076 feconvolvematrix => 'feConvolveMatrix',
4077 fediffuselighting => 'feDiffuseLighting',
4078 fedisplacementmap => 'feDisplacementMap',
4079 fedistantlight => 'feDistantLight',
4080 feflood => 'feFlood',
4081 fefunca => 'feFuncA',
4082 fefuncb => 'feFuncB',
4083 fefuncg => 'feFuncG',
4084 fefuncr => 'feFuncR',
4085 fegaussianblur => 'feGaussianBlur',
4086 feimage => 'feImage',
4087 femerge => 'feMerge',
4088 femergenode => 'feMergeNode',
4089 femorphology => 'feMorphology',
4090 feoffset => 'feOffset',
4091 fepointlight => 'fePointLight',
4092 fespecularlighting => 'feSpecularLighting',
4093 fespotlight => 'feSpotLight',
4094 fetile => 'feTile',
4095 feturbulence => 'feTurbulence',
4096 foreignobject => 'foreignObject',
4097 glyphref => 'glyphRef',
4098 lineargradient => 'linearGradient',
4099 radialgradient => 'radialGradient',
4100 #solidcolor => 'solidColor', ## NOTE: Commented in spec (SVG1.2)
4101 textpath => 'textPath',
4102 }->{$tag_name} || $tag_name;
4103 }
4104
4105 ## "adjust SVG attributes" (SVG only) - done in insert-element-f
4106
4107 ## "adjust foreign attributes" - done in insert-element-f
4108
4109 !!!insert-element-f ($nsuri, $tag_name, $token->{attributes}, $token);
4110
4111 if ($self->{self_closing}) {
4112 pop @{$self->{open_elements}};
4113 !!!ack ('t87.3');
4114 } else {
4115 !!!cp ('t87.4');
4116 }
4117
4118 !!!next-token;
4119 next B;
4120 }
4121 } elsif ($token->{type} == END_TAG_TOKEN) {
4122 ## NOTE: "using the rules for secondary insertion mode" then "continue"
4123 !!!cp ('t87.5');
4124 #
4125 } elsif ($token->{type} == END_OF_FILE_TOKEN) {
4126 !!!cp ('t87.6');
4127 !!!parse-error (type => 'not closed',
4128 value => $self->{open_elements}->[-1]->[0]
4129 ->manakai_local_name,
4130 token => $token);
4131
4132 pop @{$self->{open_elements}}
4133 while $self->{open_elements}->[-1]->[1] & FOREIGN_EL;
4134
4135 $self->{insertion_mode} &= ~ IN_FOREIGN_CONTENT_IM;
4136 ## Reprocess.
4137 next B;
4138 } else {
4139 die "$0: $token->{type}: Unknown token type";
4140 }
4141 }
4142
4143 if ($self->{insertion_mode} & HEAD_IMS) {
4144 if ($token->{type} == CHARACTER_TOKEN) {
4145 if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {
4146 unless ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4147 !!!cp ('t88.2');
4148 $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
4149 } else {
4150 !!!cp ('t88.1');
4151 ## Ignore the token.
4152 !!!next-token;
4153 next B;
4154 }
4155 unless (length $token->{data}) {
4156 !!!cp ('t88');
4157 !!!next-token;
4158 next B;
4159 }
4160 }
4161
4162 if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4163 !!!cp ('t89');
4164 ## As if <head>
4165 !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);
4166 $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
4167 push @{$self->{open_elements}},
4168 [$self->{head_element}, $el_category->{head}];
4169
4170 ## Reprocess in the "in head" insertion mode...
4171 pop @{$self->{open_elements}};
4172
4173 ## Reprocess in the "after head" insertion mode...
4174 } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4175 !!!cp ('t90');
4176 ## As if </noscript>
4177 pop @{$self->{open_elements}};
4178 !!!parse-error (type => 'in noscript:#character', token => $token);
4179
4180 ## Reprocess in the "in head" insertion mode...
4181 ## As if </head>
4182 pop @{$self->{open_elements}};
4183
4184 ## Reprocess in the "after head" insertion mode...
4185 } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
4186 !!!cp ('t91');
4187 pop @{$self->{open_elements}};
4188
4189 ## Reprocess in the "after head" insertion mode...
4190 } else {
4191 !!!cp ('t92');
4192 }
4193
4194 ## "after head" insertion mode
4195 ## As if <body>
4196 !!!insert-element ('body',, $token);
4197 $self->{insertion_mode} = IN_BODY_IM;
4198 ## reprocess
4199 next B;
4200 } elsif ($token->{type} == START_TAG_TOKEN) {
4201 if ($token->{tag_name} eq 'head') {
4202 if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4203 !!!cp ('t93');
4204 !!!create-element ($self->{head_element}, $HTML_NS, $token->{tag_name}, $token->{attributes}, $token);
4205 $self->{open_elements}->[-1]->[0]->append_child
4206 ($self->{head_element});
4207 push @{$self->{open_elements}},
4208 [$self->{head_element}, $el_category->{head}];
4209 $self->{insertion_mode} = IN_HEAD_IM;
4210 !!!nack ('t93.1');
4211 !!!next-token;
4212 next B;
4213 } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4214 !!!cp ('t93.2');
4215 !!!parse-error (type => 'after head:head', token => $token); ## TODO: error type
4216 ## Ignore the token
4217 !!!nack ('t93.3');
4218 !!!next-token;
4219 next B;
4220 } else {
4221 !!!cp ('t95');
4222 !!!parse-error (type => 'in head:head', token => $token); # or in head noscript
4223 ## Ignore the token
4224 !!!nack ('t95.1');
4225 !!!next-token;
4226 next B;
4227 }
4228 } elsif ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4229 !!!cp ('t96');
4230 ## As if <head>
4231 !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);
4232 $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
4233 push @{$self->{open_elements}},
4234 [$self->{head_element}, $el_category->{head}];
4235
4236 $self->{insertion_mode} = IN_HEAD_IM;
4237 ## Reprocess in the "in head" insertion mode...
4238 } else {
4239 !!!cp ('t97');
4240 }
4241
4242 if ($token->{tag_name} eq 'base') {
4243 if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4244 !!!cp ('t98');
4245 ## As if </noscript>
4246 pop @{$self->{open_elements}};
4247 !!!parse-error (type => 'in noscript:base', token => $token);
4248
4249 $self->{insertion_mode} = IN_HEAD_IM;
4250 ## Reprocess in the "in head" insertion mode...
4251 } else {
4252 !!!cp ('t99');
4253 }
4254
4255 ## NOTE: There is a "as if in head" code clone.
4256 if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4257 !!!cp ('t100');
4258 !!!parse-error (type => 'after head:'.$token->{tag_name}, token => $token);
4259 push @{$self->{open_elements}},
4260 [$self->{head_element}, $el_category->{head}];
4261 } else {
4262 !!!cp ('t101');
4263 }
4264 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4265 pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.
4266 pop @{$self->{open_elements}} # <head>
4267 if $self->{insertion_mode} == AFTER_HEAD_IM;
4268 !!!nack ('t101.1');
4269 !!!next-token;
4270 next B;
4271 } elsif ($token->{tag_name} eq 'link') {
4272 ## NOTE: There is a "as if in head" code clone.
4273 if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4274 !!!cp ('t102');
4275 !!!parse-error (type => 'after head:'.$token->{tag_name}, token => $token);
4276 push @{$self->{open_elements}},
4277 [$self->{head_element}, $el_category->{head}];
4278 } else {
4279 !!!cp ('t103');
4280 }
4281 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4282 pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.
4283 pop @{$self->{open_elements}} # <head>
4284 if $self->{insertion_mode} == AFTER_HEAD_IM;
4285 !!!ack ('t103.1');
4286 !!!next-token;
4287 next B;
4288 } elsif ($token->{tag_name} eq 'meta') {
4289 ## NOTE: There is a "as if in head" code clone.
4290 if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4291 !!!cp ('t104');
4292 !!!parse-error (type => 'after head:'.$token->{tag_name}, token => $token);
4293 push @{$self->{open_elements}},
4294 [$self->{head_element}, $el_category->{head}];
4295 } else {
4296 !!!cp ('t105');
4297 }
4298 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4299 my $meta_el = pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.
4300
4301 unless ($self->{confident}) {
4302 if ($token->{attributes}->{charset}) {
4303 !!!cp ('t106');
4304 ## NOTE: Whether the encoding is supported or not is handled
4305 ## in the {change_encoding} callback.
4306 $self->{change_encoding}
4307 ->($self, $token->{attributes}->{charset}->{value},
4308 $token);
4309
4310 $meta_el->[0]->get_attribute_node_ns (undef, 'charset')
4311 ->set_user_data (manakai_has_reference =>
4312 $token->{attributes}->{charset}
4313 ->{has_reference});
4314 } elsif ($token->{attributes}->{content}) {
4315 if ($token->{attributes}->{content}->{value}
4316 =~ /[Cc][Hh][Aa][Rr][Ss][Ee][Tt]
4317 [\x09-\x0D\x20]*=
4318 [\x09-\x0D\x20]*(?>"([^"]*)"|'([^']*)'|
4319 ([^"'\x09-\x0D\x20][^\x09-\x0D\x20\x3B]*))/x) {
4320 !!!cp ('t107');
4321 ## NOTE: Whether the encoding is supported or not is handled
4322 ## in the {change_encoding} callback.
4323 $self->{change_encoding}
4324 ->($self, defined $1 ? $1 : defined $2 ? $2 : $3,
4325 $token);
4326 $meta_el->[0]->get_attribute_node_ns (undef, 'content')
4327 ->set_user_data (manakai_has_reference =>
4328 $token->{attributes}->{content}
4329 ->{has_reference});
4330 } else {
4331 !!!cp ('t108');
4332 }
4333 }
4334 } else {
4335 if ($token->{attributes}->{charset}) {
4336 !!!cp ('t109');
4337 $meta_el->[0]->get_attribute_node_ns (undef, 'charset')
4338 ->set_user_data (manakai_has_reference =>
4339 $token->{attributes}->{charset}
4340 ->{has_reference});
4341 }
4342 if ($token->{attributes}->{content}) {
4343 !!!cp ('t110');
4344 $meta_el->[0]->get_attribute_node_ns (undef, 'content')
4345 ->set_user_data (manakai_has_reference =>
4346 $token->{attributes}->{content}
4347 ->{has_reference});
4348 }
4349 }
4350
4351 pop @{$self->{open_elements}} # <head>
4352 if $self->{insertion_mode} == AFTER_HEAD_IM;
4353 !!!ack ('t110.1');
4354 !!!next-token;
4355 next B;
4356 } elsif ($token->{tag_name} eq 'title') {
4357 if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4358 !!!cp ('t111');
4359 ## As if </noscript>
4360 pop @{$self->{open_elements}};
4361 !!!parse-error (type => 'in noscript:title', token => $token);
4362
4363 $self->{insertion_mode} = IN_HEAD_IM;
4364 ## Reprocess in the "in head" insertion mode...
4365 } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4366 !!!cp ('t112');
4367 !!!parse-error (type => 'after head:'.$token->{tag_name}, token => $token);
4368 push @{$self->{open_elements}},
4369 [$self->{head_element}, $el_category->{head}];
4370 } else {
4371 !!!cp ('t113');
4372 }
4373
4374 ## NOTE: There is a "as if in head" code clone.
4375 my $parent = defined $self->{head_element} ? $self->{head_element}
4376 : $self->{open_elements}->[-1]->[0];
4377 $parse_rcdata->(RCDATA_CONTENT_MODEL);
4378 pop @{$self->{open_elements}} # <head>
4379 if $self->{insertion_mode} == AFTER_HEAD_IM;
4380 next B;
4381 } elsif ($token->{tag_name} eq 'style' or
4382 $token->{tag_name} eq 'noframes') {
4383 ## NOTE: Or (scripting is enabled and tag_name eq 'noscript' and
4384 ## insertion mode IN_HEAD_IM)
4385 ## NOTE: There is a "as if in head" code clone.
4386 if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4387 !!!cp ('t114');
4388 !!!parse-error (type => 'after head:'.$token->{tag_name}, token => $token);
4389 push @{$self->{open_elements}},
4390 [$self->{head_element}, $el_category->{head}];
4391 } else {
4392 !!!cp ('t115');
4393 }
4394 $parse_rcdata->(CDATA_CONTENT_MODEL);
4395 pop @{$self->{open_elements}} # <head>
4396 if $self->{insertion_mode} == AFTER_HEAD_IM;
4397 next B;
4398 } elsif ($token->{tag_name} eq 'noscript') {
4399 if ($self->{insertion_mode} == IN_HEAD_IM) {
4400 !!!cp ('t116');
4401 ## NOTE: and scripting is disalbed
4402 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4403 $self->{insertion_mode} = IN_HEAD_NOSCRIPT_IM;
4404 !!!nack ('t116.1');
4405 !!!next-token;
4406 next B;
4407 } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4408 !!!cp ('t117');
4409 !!!parse-error (type => 'in noscript:noscript', token => $token);
4410 ## Ignore the token
4411 !!!nack ('t117.1');
4412 !!!next-token;
4413 next B;
4414 } else {
4415 !!!cp ('t118');
4416 #
4417 }
4418 } elsif ($token->{tag_name} eq 'script') {
4419 if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4420 !!!cp ('t119');
4421 ## As if </noscript>
4422 pop @{$self->{open_elements}};
4423 !!!parse-error (type => 'in noscript:script', token => $token);
4424
4425 $self->{insertion_mode} = IN_HEAD_IM;
4426 ## Reprocess in the "in head" insertion mode...
4427 } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4428 !!!cp ('t120');
4429 !!!parse-error (type => 'after head:'.$token->{tag_name}, token => $token);
4430 push @{$self->{open_elements}},
4431 [$self->{head_element}, $el_category->{head}];
4432 } else {
4433 !!!cp ('t121');
4434 }
4435
4436 ## NOTE: There is a "as if in head" code clone.
4437 $script_start_tag->();
4438 pop @{$self->{open_elements}} # <head>
4439 if $self->{insertion_mode} == AFTER_HEAD_IM;
4440 next B;
4441 } elsif ($token->{tag_name} eq 'body' or
4442 $token->{tag_name} eq 'frameset') {
4443 if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4444 !!!cp ('t122');
4445 ## As if </noscript>
4446 pop @{$self->{open_elements}};
4447 !!!parse-error (type => 'in noscript:'.$token->{tag_name}, token => $token);
4448
4449 ## Reprocess in the "in head" insertion mode...
4450 ## As if </head>
4451 pop @{$self->{open_elements}};
4452
4453 ## Reprocess in the "after head" insertion mode...
4454 } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
4455 !!!cp ('t124');
4456 pop @{$self->{open_elements}};
4457
4458 ## Reprocess in the "after head" insertion mode...
4459 } else {
4460 !!!cp ('t125');
4461 }
4462
4463 ## "after head" insertion mode
4464 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4465 if ($token->{tag_name} eq 'body') {
4466 !!!cp ('t126');
4467 $self->{insertion_mode} = IN_BODY_IM;
4468 } elsif ($token->{tag_name} eq 'frameset') {
4469 !!!cp ('t127');
4470 $self->{insertion_mode} = IN_FRAMESET_IM;
4471 } else {
4472 die "$0: tag name: $self->{tag_name}";
4473 }
4474 !!!nack ('t127.1');
4475 !!!next-token;
4476 next B;
4477 } else {
4478 !!!cp ('t128');
4479 #
4480 }
4481
4482 if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4483 !!!cp ('t129');
4484 ## As if </noscript>
4485 pop @{$self->{open_elements}};
4486 !!!parse-error (type => 'in noscript:/'.$token->{tag_name}, token => $token);
4487
4488 ## Reprocess in the "in head" insertion mode...
4489 ## As if </head>
4490 pop @{$self->{open_elements}};
4491
4492 ## Reprocess in the "after head" insertion mode...
4493 } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
4494 !!!cp ('t130');
4495 ## As if </head>
4496 pop @{$self->{open_elements}};
4497
4498 ## Reprocess in the "after head" insertion mode...
4499 } else {
4500 !!!cp ('t131');
4501 }
4502
4503 ## "after head" insertion mode
4504 ## As if <body>
4505 !!!insert-element ('body',, $token);
4506 $self->{insertion_mode} = IN_BODY_IM;
4507 ## reprocess
4508 !!!ack-later;
4509 next B;
4510 } elsif ($token->{type} == END_TAG_TOKEN) {
4511 if ($token->{tag_name} eq 'head') {
4512 if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4513 !!!cp ('t132');
4514 ## As if <head>
4515 !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);
4516 $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
4517 push @{$self->{open_elements}},
4518 [$self->{head_element}, $el_category->{head}];
4519
4520 ## Reprocess in the "in head" insertion mode...
4521 pop @{$self->{open_elements}};
4522 $self->{insertion_mode} = AFTER_HEAD_IM;
4523 !!!next-token;
4524 next B;
4525 } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4526 !!!cp ('t133');
4527 ## As if </noscript>
4528 pop @{$self->{open_elements}};
4529 !!!parse-error (type => 'in noscript:/head', token => $token);
4530
4531 ## Reprocess in the "in head" insertion mode...
4532 pop @{$self->{open_elements}};
4533 $self->{insertion_mode} = AFTER_HEAD_IM;
4534 !!!next-token;
4535 next B;
4536 } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
4537 !!!cp ('t134');
4538 pop @{$self->{open_elements}};
4539 $self->{insertion_mode} = AFTER_HEAD_IM;
4540 !!!next-token;
4541 next B;
4542 } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4543 !!!cp ('t134.1');
4544 !!!parse-error (type => 'unmatched end tag:head', token => $token);
4545 ## Ignore the token
4546 !!!next-token;
4547 next B;
4548 } else {
4549 die "$0: $self->{insertion_mode}: Unknown insertion mode";
4550 }
4551 } elsif ($token->{tag_name} eq 'noscript') {
4552 if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4553 !!!cp ('t136');
4554 pop @{$self->{open_elements}};
4555 $self->{insertion_mode} = IN_HEAD_IM;
4556 !!!next-token;
4557 next B;
4558 } elsif ($self->{insertion_mode} == BEFORE_HEAD_IM or
4559 $self->{insertion_mode} == AFTER_HEAD_IM) {
4560 !!!cp ('t137');
4561 !!!parse-error (type => 'unmatched end tag:noscript', token => $token);
4562 ## Ignore the token ## ISSUE: An issue in the spec.
4563 !!!next-token;
4564 next B;
4565 } else {
4566 !!!cp ('t138');
4567 #
4568 }
4569 } elsif ({
4570 body => 1, html => 1,
4571 }->{$token->{tag_name}}) {
4572 if ($self->{insertion_mode} == BEFORE_HEAD_IM or
4573 $self->{insertion_mode} == IN_HEAD_IM or
4574 $self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4575 !!!cp ('t140');
4576 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
4577 ## Ignore the token
4578 !!!next-token;
4579 next B;
4580 } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4581 !!!cp ('t140.1');
4582 !!!parse-error (type => 'unmatched end tag:' . $token->{tag_name}, token => $token);
4583 ## Ignore the token
4584 !!!next-token;
4585 next B;
4586 } else {
4587 die "$0: $self->{insertion_mode}: Unknown insertion mode";
4588 }
4589 } elsif ($token->{tag_name} eq 'p') {
4590 !!!cp ('t142');
4591 !!!parse-error (type => 'unmatched end tag:p', token => $token);
4592 ## Ignore the token
4593 !!!next-token;
4594 next B;
4595 } elsif ($token->{tag_name} eq 'br') {
4596 if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4597 !!!cp ('t142.2');
4598 ## (before head) as if <head>, (in head) as if </head>
4599 !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);
4600 $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
4601 $self->{insertion_mode} = AFTER_HEAD_IM;
4602
4603 ## Reprocess in the "after head" insertion mode...
4604 } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
4605 !!!cp ('t143.2');
4606 ## As if </head>
4607 pop @{$self->{open_elements}};
4608 $self->{insertion_mode} = AFTER_HEAD_IM;
4609
4610 ## Reprocess in the "after head" insertion mode...
4611 } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4612 !!!cp ('t143.3');
4613 ## ISSUE: Two parse errors for <head><noscript></br>
4614 !!!parse-error (type => 'unmatched end tag:br', token => $token);
4615 ## As if </noscript>
4616 pop @{$self->{open_elements}};
4617 $self->{insertion_mode} = IN_HEAD_IM;
4618
4619 ## Reprocess in the "in head" insertion mode...
4620 ## As if </head>
4621 pop @{$self->{open_elements}};
4622 $self->{insertion_mode} = AFTER_HEAD_IM;
4623
4624 ## Reprocess in the "after head" insertion mode...
4625 } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4626 !!!cp ('t143.4');
4627 #
4628 } else {
4629 die "$0: $self->{insertion_mode}: Unknown insertion mode";
4630 }
4631
4632 ## ISSUE: does not agree with IE7 - it doesn't ignore </br>.
4633 !!!parse-error (type => 'unmatched end tag:br', token => $token);
4634 ## Ignore the token
4635 !!!next-token;
4636 next B;
4637 } else {
4638 !!!cp ('t145');
4639 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
4640 ## Ignore the token
4641 !!!next-token;
4642 next B;
4643 }
4644
4645 if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4646 !!!cp ('t146');
4647 ## As if </noscript>
4648 pop @{$self->{open_elements}};
4649 !!!parse-error (type => 'in noscript:/'.$token->{tag_name}, token => $token);
4650
4651 ## Reprocess in the "in head" insertion mode...
4652 ## As if </head>
4653 pop @{$self->{open_elements}};
4654
4655 ## Reprocess in the "after head" insertion mode...
4656 } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
4657 !!!cp ('t147');
4658 ## As if </head>
4659 pop @{$self->{open_elements}};
4660
4661 ## Reprocess in the "after head" insertion mode...
4662 } elsif ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4663 ## ISSUE: This case cannot be reached?
4664 !!!cp ('t148');
4665 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
4666 ## Ignore the token ## ISSUE: An issue in the spec.
4667 !!!next-token;
4668 next B;
4669 } else {
4670 !!!cp ('t149');
4671 }
4672
4673 ## "after head" insertion mode
4674 ## As if <body>
4675 !!!insert-element ('body',, $token);
4676 $self->{insertion_mode} = IN_BODY_IM;
4677 ## reprocess
4678 next B;
4679 } elsif ($token->{type} == END_OF_FILE_TOKEN) {
4680 if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4681 !!!cp ('t149.1');
4682
4683 ## NOTE: As if <head>
4684 !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);
4685 $self->{open_elements}->[-1]->[0]->append_child
4686 ($self->{head_element});
4687 #push @{$self->{open_elements}},
4688 # [$self->{head_element}, $el_category->{head}];
4689 #$self->{insertion_mode} = IN_HEAD_IM;
4690 ## NOTE: Reprocess.
4691
4692 ## NOTE: As if </head>
4693 #pop @{$self->{open_elements}};
4694 #$self->{insertion_mode} = IN_AFTER_HEAD_IM;
4695 ## NOTE: Reprocess.
4696
4697 #
4698 } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
4699 !!!cp ('t149.2');
4700
4701 ## NOTE: As if </head>
4702 pop @{$self->{open_elements}};
4703 #$self->{insertion_mode} = IN_AFTER_HEAD_IM;
4704 ## NOTE: Reprocess.
4705
4706 #
4707 } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4708 !!!cp ('t149.3');
4709
4710 !!!parse-error (type => 'in noscript:#eof', token => $token);
4711
4712 ## As if </noscript>
4713 pop @{$self->{open_elements}};
4714 #$self->{insertion_mode} = IN_HEAD_IM;
4715 ## NOTE: Reprocess.
4716
4717 ## NOTE: As if </head>
4718 pop @{$self->{open_elements}};
4719 #$self->{insertion_mode} = IN_AFTER_HEAD_IM;
4720 ## NOTE: Reprocess.
4721
4722 #
4723 } else {
4724 !!!cp ('t149.4');
4725 #
4726 }
4727
4728 ## NOTE: As if <body>
4729 !!!insert-element ('body',, $token);
4730 $self->{insertion_mode} = IN_BODY_IM;
4731 ## NOTE: Reprocess.
4732 next B;
4733 } else {
4734 die "$0: $token->{type}: Unknown token type";
4735 }
4736
4737 ## ISSUE: An issue in the spec.
4738 } elsif ($self->{insertion_mode} & BODY_IMS) {
4739 if ($token->{type} == CHARACTER_TOKEN) {
4740 !!!cp ('t150');
4741 ## NOTE: There is a code clone of "character in body".
4742 $reconstruct_active_formatting_elements->($insert_to_current);
4743
4744 $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});
4745
4746 !!!next-token;
4747 next B;
4748 } elsif ($token->{type} == START_TAG_TOKEN) {
4749 if ({
4750 caption => 1, col => 1, colgroup => 1, tbody => 1,
4751 td => 1, tfoot => 1, th => 1, thead => 1, tr => 1,
4752 }->{$token->{tag_name}}) {
4753 if ($self->{insertion_mode} == IN_CELL_IM) {
4754 ## have an element in table scope
4755 for (reverse 0..$#{$self->{open_elements}}) {
4756 my $node = $self->{open_elements}->[$_];
4757 if ($node->[1] & TABLE_CELL_EL) {
4758 !!!cp ('t151');
4759
4760 ## Close the cell
4761 !!!back-token; # <x>
4762 $token = {type => END_TAG_TOKEN,
4763 tag_name => $node->[0]->manakai_local_name,
4764 line => $token->{line},
4765 column => $token->{column}};
4766 next B;
4767 } elsif ($node->[1] & TABLE_SCOPING_EL) {
4768 !!!cp ('t152');
4769 ## ISSUE: This case can never be reached, maybe.
4770 last;
4771 }
4772 }
4773
4774 !!!cp ('t153');
4775 !!!parse-error (type => 'start tag not allowed',
4776 value => $token->{tag_name}, token => $token);
4777 ## Ignore the token
4778 !!!nack ('t153.1');
4779 !!!next-token;
4780 next B;
4781 } elsif ($self->{insertion_mode} == IN_CAPTION_IM) {
4782 !!!parse-error (type => 'not closed:caption', token => $token);
4783
4784 ## NOTE: As if </caption>.
4785 ## have a table element in table scope
4786 my $i;
4787 INSCOPE: {
4788 for (reverse 0..$#{$self->{open_elements}}) {
4789 my $node = $self->{open_elements}->[$_];
4790 if ($node->[1] & CAPTION_EL) {
4791 !!!cp ('t155');
4792 $i = $_;
4793 last INSCOPE;
4794 } elsif ($node->[1] & TABLE_SCOPING_EL) {
4795 !!!cp ('t156');
4796 last;
4797 }
4798 }
4799
4800 !!!cp ('t157');
4801 !!!parse-error (type => 'start tag not allowed',
4802 value => $token->{tag_name}, token => $token);
4803 ## Ignore the token
4804 !!!nack ('t157.1');
4805 !!!next-token;
4806 next B;
4807 } # INSCOPE
4808
4809 ## generate implied end tags
4810 while ($self->{open_elements}->[-1]->[1]
4811 & END_TAG_OPTIONAL_EL) {
4812 !!!cp ('t158');
4813 pop @{$self->{open_elements}};
4814 }
4815
4816 unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {
4817 !!!cp ('t159');
4818 !!!parse-error (type => 'not closed',
4819 value => $self->{open_elements}->[-1]->[0]
4820 ->manakai_local_name,
4821 token => $token);
4822 } else {
4823 !!!cp ('t160');
4824 }
4825
4826 splice @{$self->{open_elements}}, $i;
4827
4828 $clear_up_to_marker->();
4829
4830 $self->{insertion_mode} = IN_TABLE_IM;
4831
4832 ## reprocess
4833 !!!ack-later;
4834 next B;
4835 } else {
4836 !!!cp ('t161');
4837 #
4838 }
4839 } else {
4840 !!!cp ('t162');
4841 #
4842 }
4843 } elsif ($token->{type} == END_TAG_TOKEN) {
4844 if ($token->{tag_name} eq 'td' or $token->{tag_name} eq 'th') {
4845 if ($self->{insertion_mode} == IN_CELL_IM) {
4846 ## have an element in table scope
4847 my $i;
4848 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
4849 my $node = $self->{open_elements}->[$_];
4850 if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
4851 !!!cp ('t163');
4852 $i = $_;
4853 last INSCOPE;
4854 } elsif ($node->[1] & TABLE_SCOPING_EL) {
4855 !!!cp ('t164');
4856 last INSCOPE;
4857 }
4858 } # INSCOPE
4859 unless (defined $i) {
4860 !!!cp ('t165');
4861 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
4862 ## Ignore the token
4863 !!!next-token;
4864 next B;
4865 }
4866
4867 ## generate implied end tags
4868 while ($self->{open_elements}->[-1]->[1]
4869 & END_TAG_OPTIONAL_EL) {
4870 !!!cp ('t166');
4871 pop @{$self->{open_elements}};
4872 }
4873
4874 if ($self->{open_elements}->[-1]->[0]->manakai_local_name
4875 ne $token->{tag_name}) {
4876 !!!cp ('t167');
4877 !!!parse-error (type => 'not closed',
4878 value => $self->{open_elements}->[-1]->[0]
4879 ->manakai_local_name,
4880 token => $token);
4881 } else {
4882 !!!cp ('t168');
4883 }
4884
4885 splice @{$self->{open_elements}}, $i;
4886
4887 $clear_up_to_marker->();
4888
4889 $self->{insertion_mode} = IN_ROW_IM;
4890
4891 !!!next-token;
4892 next B;
4893 } elsif ($self->{insertion_mode} == IN_CAPTION_IM) {
4894 !!!cp ('t169');
4895 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
4896 ## Ignore the token
4897 !!!next-token;
4898 next B;
4899 } else {
4900 !!!cp ('t170');
4901 #
4902 }
4903 } elsif ($token->{tag_name} eq 'caption') {
4904 if ($self->{insertion_mode} == IN_CAPTION_IM) {
4905 ## have a table element in table scope
4906 my $i;
4907 INSCOPE: {
4908 for (reverse 0..$#{$self->{open_elements}}) {
4909 my $node = $self->{open_elements}->[$_];
4910 if ($node->[1] & CAPTION_EL) {
4911 !!!cp ('t171');
4912 $i = $_;
4913 last INSCOPE;
4914 } elsif ($node->[1] & TABLE_SCOPING_EL) {
4915 !!!cp ('t172');
4916 last;
4917 }
4918 }
4919
4920 !!!cp ('t173');
4921 !!!parse-error (type => 'unmatched end tag',
4922 value => $token->{tag_name}, token => $token);
4923 ## Ignore the token
4924 !!!next-token;
4925 next B;
4926 } # INSCOPE
4927
4928 ## generate implied end tags
4929 while ($self->{open_elements}->[-1]->[1]
4930 & END_TAG_OPTIONAL_EL) {
4931 !!!cp ('t174');
4932 pop @{$self->{open_elements}};
4933 }
4934
4935 unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {
4936 !!!cp ('t175');
4937 !!!parse-error (type => 'not closed',
4938 value => $self->{open_elements}->[-1]->[0]
4939 ->manakai_local_name,
4940 token => $token);
4941 } else {
4942 !!!cp ('t176');
4943 }
4944
4945 splice @{$self->{open_elements}}, $i;
4946
4947 $clear_up_to_marker->();
4948
4949 $self->{insertion_mode} = IN_TABLE_IM;
4950
4951 !!!next-token;
4952 next B;
4953 } elsif ($self->{insertion_mode} == IN_CELL_IM) {
4954 !!!cp ('t177');
4955 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
4956 ## Ignore the token
4957 !!!next-token;
4958 next B;
4959 } else {
4960 !!!cp ('t178');
4961 #
4962 }
4963 } elsif ({
4964 table => 1, tbody => 1, tfoot => 1,
4965 thead => 1, tr => 1,
4966 }->{$token->{tag_name}} and
4967 $self->{insertion_mode} == IN_CELL_IM) {
4968 ## have an element in table scope
4969 my $i;
4970 my $tn;
4971 INSCOPE: {
4972 for (reverse 0..$#{$self->{open_elements}}) {
4973 my $node = $self->{open_elements}->[$_];
4974 if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
4975 !!!cp ('t179');
4976 $i = $_;
4977
4978 ## Close the cell
4979 !!!back-token; # </x>
4980 $token = {type => END_TAG_TOKEN, tag_name => $tn,
4981 line => $token->{line},
4982 column => $token->{column}};
4983 next B;
4984 } elsif ($node->[1] & TABLE_CELL_EL) {
4985 !!!cp ('t180');
4986 $tn = $node->[0]->manakai_local_name;
4987 ## NOTE: There is exactly one |td| or |th| element
4988 ## in scope in the stack of open elements by definition.
4989 } elsif ($node->[1] & TABLE_SCOPING_EL) {
4990 ## ISSUE: Can this be reached?
4991 !!!cp ('t181');
4992 last;
4993 }
4994 }
4995
4996 !!!cp ('t182');
4997 !!!parse-error (type => 'unmatched end tag',
4998 value => $token->{tag_name}, token => $token);
4999 ## Ignore the token
5000 !!!next-token;
5001 next B;
5002 } # INSCOPE
5003 } elsif ($token->{tag_name} eq 'table' and
5004 $self->{insertion_mode} == IN_CAPTION_IM) {
5005 !!!parse-error (type => 'not closed:caption', token => $token);
5006
5007 ## As if </caption>
5008 ## have a table element in table scope
5009 my $i;
5010 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5011 my $node = $self->{open_elements}->[$_];
5012 if ($node->[1] & CAPTION_EL) {
5013 !!!cp ('t184');
5014 $i = $_;
5015 last INSCOPE;
5016 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5017 !!!cp ('t185');
5018 last INSCOPE;
5019 }
5020 } # INSCOPE
5021 unless (defined $i) {
5022 !!!cp ('t186');
5023 !!!parse-error (type => 'unmatched end tag:caption', token => $token);
5024 ## Ignore the token
5025 !!!next-token;
5026 next B;
5027 }
5028
5029 ## generate implied end tags
5030 while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
5031 !!!cp ('t187');
5032 pop @{$self->{open_elements}};
5033 }
5034
5035 unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {
5036 !!!cp ('t188');
5037 !!!parse-error (type => 'not closed',
5038 value => $self->{open_elements}->[-1]->[0]
5039 ->manakai_local_name,
5040 token => $token);
5041 } else {
5042 !!!cp ('t189');
5043 }
5044
5045 splice @{$self->{open_elements}}, $i;
5046
5047 $clear_up_to_marker->();
5048
5049 $self->{insertion_mode} = IN_TABLE_IM;
5050
5051 ## reprocess
5052 next B;
5053 } elsif ({
5054 body => 1, col => 1, colgroup => 1, html => 1,
5055 }->{$token->{tag_name}}) {
5056 if ($self->{insertion_mode} & BODY_TABLE_IMS) {
5057 !!!cp ('t190');
5058 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
5059 ## Ignore the token
5060 !!!next-token;
5061 next B;
5062 } else {
5063 !!!cp ('t191');
5064 #
5065 }
5066 } elsif ({
5067 tbody => 1, tfoot => 1,
5068 thead => 1, tr => 1,
5069 }->{$token->{tag_name}} and
5070 $self->{insertion_mode} == IN_CAPTION_IM) {
5071 !!!cp ('t192');
5072 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
5073 ## Ignore the token
5074 !!!next-token;
5075 next B;
5076 } else {
5077 !!!cp ('t193');
5078 #
5079 }
5080 } elsif ($token->{type} == END_OF_FILE_TOKEN) {
5081 for my $entry (@{$self->{open_elements}}) {
5082 unless ($entry->[1] & ALL_END_TAG_OPTIONAL_EL) {
5083 !!!cp ('t75');
5084 !!!parse-error (type => 'in body:#eof', token => $token);
5085 last;
5086 }
5087 }
5088
5089 ## Stop parsing.
5090 last B;
5091 } else {
5092 die "$0: $token->{type}: Unknown token type";
5093 }
5094
5095 $insert = $insert_to_current;
5096 #
5097 } elsif ($self->{insertion_mode} & TABLE_IMS) {
5098 if ($token->{type} == CHARACTER_TOKEN) {
5099 if (not $open_tables->[-1]->[1] and # tainted
5100 $token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {
5101 $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
5102
5103 unless (length $token->{data}) {
5104 !!!cp ('t194');
5105 !!!next-token;
5106 next B;
5107 } else {
5108 !!!cp ('t195');
5109 }
5110 }
5111
5112 !!!parse-error (type => 'in table:#character', token => $token);
5113
5114 ## As if in body, but insert into foster parent element
5115 ## ISSUE: Spec says that "whenever a node would be inserted
5116 ## into the current node" while characters might not be
5117 ## result in a new Text node.
5118 $reconstruct_active_formatting_elements->($insert_to_foster);
5119
5120 if ($self->{open_elements}->[-1]->[1] & TABLE_ROWS_EL) {
5121 # MUST
5122 my $foster_parent_element;
5123 my $next_sibling;
5124 my $prev_sibling;
5125 OE: for (reverse 0..$#{$self->{open_elements}}) {
5126 if ($self->{open_elements}->[$_]->[1] & TABLE_EL) {
5127 my $parent = $self->{open_elements}->[$_]->[0]->parent_node;
5128 if (defined $parent and $parent->node_type == 1) {
5129 !!!cp ('t196');
5130 $foster_parent_element = $parent;
5131 $next_sibling = $self->{open_elements}->[$_]->[0];
5132 $prev_sibling = $next_sibling->previous_sibling;
5133 } else {
5134 !!!cp ('t197');
5135 $foster_parent_element = $self->{open_elements}->[$_ - 1]->[0];
5136 $prev_sibling = $foster_parent_element->last_child;
5137 }
5138 last OE;
5139 }
5140 } # OE
5141 $foster_parent_element = $self->{open_elements}->[0]->[0] and
5142 $prev_sibling = $foster_parent_element->last_child
5143 unless defined $foster_parent_element;
5144 if (defined $prev_sibling and
5145 $prev_sibling->node_type == 3) {
5146 !!!cp ('t198');
5147 $prev_sibling->manakai_append_text ($token->{data});
5148 } else {
5149 !!!cp ('t199');
5150 $foster_parent_element->insert_before
5151 ($self->{document}->create_text_node ($token->{data}),
5152 $next_sibling);
5153 }
5154 $open_tables->[-1]->[1] = 1; # tainted
5155 } else {
5156 !!!cp ('t200');
5157 $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});
5158 }
5159
5160 !!!next-token;
5161 next B;
5162 } elsif ($token->{type} == START_TAG_TOKEN) {
5163 if ({
5164 tr => ($self->{insertion_mode} != IN_ROW_IM),
5165 th => 1, td => 1,
5166 }->{$token->{tag_name}}) {
5167 if ($self->{insertion_mode} == IN_TABLE_IM) {
5168 ## Clear back to table context
5169 while (not ($self->{open_elements}->[-1]->[1]
5170 & TABLE_SCOPING_EL)) {
5171 !!!cp ('t201');
5172 pop @{$self->{open_elements}};
5173 }
5174
5175 !!!insert-element ('tbody',, $token);
5176 $self->{insertion_mode} = IN_TABLE_BODY_IM;
5177 ## reprocess in the "in table body" insertion mode...
5178 }
5179
5180 if ($self->{insertion_mode} == IN_TABLE_BODY_IM) {
5181 unless ($token->{tag_name} eq 'tr') {
5182 !!!cp ('t202');
5183 !!!parse-error (type => 'missing start tag:tr', token => $token);
5184 }
5185
5186 ## Clear back to table body context
5187 while (not ($self->{open_elements}->[-1]->[1]
5188 & TABLE_ROWS_SCOPING_EL)) {
5189 !!!cp ('t203');
5190 ## ISSUE: Can this case be reached?
5191 pop @{$self->{open_elements}};
5192 }
5193
5194 $self->{insertion_mode} = IN_ROW_IM;
5195 if ($token->{tag_name} eq 'tr') {
5196 !!!cp ('t204');
5197 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5198 !!!nack ('t204');
5199 !!!next-token;
5200 next B;
5201 } else {
5202 !!!cp ('t205');
5203 !!!insert-element ('tr',, $token);
5204 ## reprocess in the "in row" insertion mode
5205 }
5206 } else {
5207 !!!cp ('t206');
5208 }
5209
5210 ## Clear back to table row context
5211 while (not ($self->{open_elements}->[-1]->[1]
5212 & TABLE_ROW_SCOPING_EL)) {
5213 !!!cp ('t207');
5214 pop @{$self->{open_elements}};
5215 }
5216
5217 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5218 $self->{insertion_mode} = IN_CELL_IM;
5219
5220 push @$active_formatting_elements, ['#marker', ''];
5221
5222 !!!nack ('t207.1');
5223 !!!next-token;
5224 next B;
5225 } elsif ({
5226 caption => 1, col => 1, colgroup => 1,
5227 tbody => 1, tfoot => 1, thead => 1,
5228 tr => 1, # $self->{insertion_mode} == IN_ROW_IM
5229 }->{$token->{tag_name}}) {
5230 if ($self->{insertion_mode} == IN_ROW_IM) {
5231 ## As if </tr>
5232 ## have an element in table scope
5233 my $i;
5234 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5235 my $node = $self->{open_elements}->[$_];
5236 if ($node->[1] & TABLE_ROW_EL) {
5237 !!!cp ('t208');
5238 $i = $_;
5239 last INSCOPE;
5240 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5241 !!!cp ('t209');
5242 last INSCOPE;
5243 }
5244 } # INSCOPE
5245 unless (defined $i) {
5246 !!!cp ('t210');
5247 ## TODO: This type is wrong.
5248 !!!parse-error (type => 'unmacthed end tag:'.$token->{tag_name}, token => $token);
5249 ## Ignore the token
5250 !!!nack ('t210.1');
5251 !!!next-token;
5252 next B;
5253 }
5254
5255 ## Clear back to table row context
5256 while (not ($self->{open_elements}->[-1]->[1]
5257 & TABLE_ROW_SCOPING_EL)) {
5258 !!!cp ('t211');
5259 ## ISSUE: Can this case be reached?
5260 pop @{$self->{open_elements}};
5261 }
5262
5263 pop @{$self->{open_elements}}; # tr
5264 $self->{insertion_mode} = IN_TABLE_BODY_IM;
5265 if ($token->{tag_name} eq 'tr') {
5266 !!!cp ('t212');
5267 ## reprocess
5268 !!!ack-later;
5269 next B;
5270 } else {
5271 !!!cp ('t213');
5272 ## reprocess in the "in table body" insertion mode...
5273 }
5274 }
5275
5276 if ($self->{insertion_mode} == IN_TABLE_BODY_IM) {
5277 ## have an element in table scope
5278 my $i;
5279 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5280 my $node = $self->{open_elements}->[$_];
5281 if ($node->[1] & TABLE_ROW_GROUP_EL) {
5282 !!!cp ('t214');
5283 $i = $_;
5284 last INSCOPE;
5285 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5286 !!!cp ('t215');
5287 last INSCOPE;
5288 }
5289 } # INSCOPE
5290 unless (defined $i) {
5291 !!!cp ('t216');
5292 ## TODO: This erorr type ios wrong.
5293 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
5294 ## Ignore the token
5295 !!!nack ('t216.1');
5296 !!!next-token;
5297 next B;
5298 }
5299
5300 ## Clear back to table body context
5301 while (not ($self->{open_elements}->[-1]->[1]
5302 & TABLE_ROWS_SCOPING_EL)) {
5303 !!!cp ('t217');
5304 ## ISSUE: Can this state be reached?
5305 pop @{$self->{open_elements}};
5306 }
5307
5308 ## As if <{current node}>
5309 ## have an element in table scope
5310 ## true by definition
5311
5312 ## Clear back to table body context
5313 ## nop by definition
5314
5315 pop @{$self->{open_elements}};
5316 $self->{insertion_mode} = IN_TABLE_IM;
5317 ## reprocess in "in table" insertion mode...
5318 } else {
5319 !!!cp ('t218');
5320 }
5321
5322 if ($token->{tag_name} eq 'col') {
5323 ## Clear back to table context
5324 while (not ($self->{open_elements}->[-1]->[1]
5325 & TABLE_SCOPING_EL)) {
5326 !!!cp ('t219');
5327 ## ISSUE: Can this state be reached?
5328 pop @{$self->{open_elements}};
5329 }
5330
5331 !!!insert-element ('colgroup',, $token);
5332 $self->{insertion_mode} = IN_COLUMN_GROUP_IM;
5333 ## reprocess
5334 !!!ack-later;
5335 next B;
5336 } elsif ({
5337 caption => 1,
5338 colgroup => 1,
5339 tbody => 1, tfoot => 1, thead => 1,
5340 }->{$token->{tag_name}}) {
5341 ## Clear back to table context
5342 while (not ($self->{open_elements}->[-1]->[1]
5343 & TABLE_SCOPING_EL)) {
5344 !!!cp ('t220');
5345 ## ISSUE: Can this state be reached?
5346 pop @{$self->{open_elements}};
5347 }
5348
5349 push @$active_formatting_elements, ['#marker', '']
5350 if $token->{tag_name} eq 'caption';
5351
5352 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5353 $self->{insertion_mode} = {
5354 caption => IN_CAPTION_IM,
5355 colgroup => IN_COLUMN_GROUP_IM,
5356 tbody => IN_TABLE_BODY_IM,
5357 tfoot => IN_TABLE_BODY_IM,
5358 thead => IN_TABLE_BODY_IM,
5359 }->{$token->{tag_name}};
5360 !!!next-token;
5361 !!!nack ('t220.1');
5362 next B;
5363 } else {
5364 die "$0: in table: <>: $token->{tag_name}";
5365 }
5366 } elsif ($token->{tag_name} eq 'table') {
5367 !!!parse-error (type => 'not closed',
5368 value => $self->{open_elements}->[-1]->[0]
5369 ->manakai_local_name,
5370 token => $token);
5371
5372 ## As if </table>
5373 ## have a table element in table scope
5374 my $i;
5375 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5376 my $node = $self->{open_elements}->[$_];
5377 if ($node->[1] & TABLE_EL) {
5378 !!!cp ('t221');
5379 $i = $_;
5380 last INSCOPE;
5381 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5382 !!!cp ('t222');
5383 last INSCOPE;
5384 }
5385 } # INSCOPE
5386 unless (defined $i) {
5387 !!!cp ('t223');
5388 ## TODO: The following is wrong, maybe.
5389 !!!parse-error (type => 'unmatched end tag:table', token => $token);
5390 ## Ignore tokens </table><table>
5391 !!!nack ('t223.1');
5392 !!!next-token;
5393 next B;
5394 }
5395
5396 ## TODO: Followings are removed from the latest spec.
5397 ## generate implied end tags
5398 while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
5399 !!!cp ('t224');
5400 pop @{$self->{open_elements}};
5401 }
5402
5403 unless ($self->{open_elements}->[-1]->[1] & TABLE_EL) {
5404 !!!cp ('t225');
5405 ## NOTE: |<table><tr><table>|
5406 !!!parse-error (type => 'not closed',
5407 value => $self->{open_elements}->[-1]->[0]
5408 ->manakai_local_name,
5409 token => $token);
5410 } else {
5411 !!!cp ('t226');
5412 }
5413
5414 splice @{$self->{open_elements}}, $i;
5415 pop @{$open_tables};
5416
5417 $self->_reset_insertion_mode;
5418
5419 ## reprocess
5420 !!!ack-later;
5421 next B;
5422 } elsif ($token->{tag_name} eq 'style') {
5423 if (not $open_tables->[-1]->[1]) { # tainted
5424 !!!cp ('t227.8');
5425 ## NOTE: This is a "as if in head" code clone.
5426 $parse_rcdata->(CDATA_CONTENT_MODEL);
5427 next B;
5428 } else {
5429 !!!cp ('t227.7');
5430 #
5431 }
5432 } elsif ($token->{tag_name} eq 'script') {
5433 if (not $open_tables->[-1]->[1]) { # tainted
5434 !!!cp ('t227.6');
5435 ## NOTE: This is a "as if in head" code clone.
5436 $script_start_tag->();
5437 next B;
5438 } else {
5439 !!!cp ('t227.5');
5440 #
5441 }
5442 } elsif ($token->{tag_name} eq 'input') {
5443 if (not $open_tables->[-1]->[1]) { # tainted
5444 if ($token->{attributes}->{type}) { ## TODO: case
5445 my $type = lc $token->{attributes}->{type}->{value};
5446 if ($type eq 'hidden') {
5447 !!!cp ('t227.3');
5448 !!!parse-error (type => 'in table:'.$token->{tag_name}, token => $token);
5449
5450 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5451
5452 ## TODO: form element pointer
5453
5454 pop @{$self->{open_elements}};
5455
5456 !!!next-token;
5457 !!!ack ('t227.2.1');
5458 next B;
5459 } else {
5460 !!!cp ('t227.2');
5461 #
5462 }
5463 } else {
5464 !!!cp ('t227.1');
5465 #
5466 }
5467 } else {
5468 !!!cp ('t227.4');
5469 #
5470 }
5471 } else {
5472 !!!cp ('t227');
5473 #
5474 }
5475
5476 !!!parse-error (type => 'in table:'.$token->{tag_name}, token => $token);
5477
5478 $insert = $insert_to_foster;
5479 #
5480 } elsif ($token->{type} == END_TAG_TOKEN) {
5481 if ($token->{tag_name} eq 'tr' and
5482 $self->{insertion_mode} == IN_ROW_IM) {
5483 ## have an element in table scope
5484 my $i;
5485 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5486 my $node = $self->{open_elements}->[$_];
5487 if ($node->[1] & TABLE_ROW_EL) {
5488 !!!cp ('t228');
5489 $i = $_;
5490 last INSCOPE;
5491 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5492 !!!cp ('t229');
5493 last INSCOPE;
5494 }
5495 } # INSCOPE
5496 unless (defined $i) {
5497 !!!cp ('t230');
5498 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
5499 ## Ignore the token
5500 !!!nack ('t230.1');
5501 !!!next-token;
5502 next B;
5503 } else {
5504 !!!cp ('t232');
5505 }
5506
5507 ## Clear back to table row context
5508 while (not ($self->{open_elements}->[-1]->[1]
5509 & TABLE_ROW_SCOPING_EL)) {
5510 !!!cp ('t231');
5511 ## ISSUE: Can this state be reached?
5512 pop @{$self->{open_elements}};
5513 }
5514
5515 pop @{$self->{open_elements}}; # tr
5516 $self->{insertion_mode} = IN_TABLE_BODY_IM;
5517 !!!next-token;
5518 !!!nack ('t231.1');
5519 next B;
5520 } elsif ($token->{tag_name} eq 'table') {
5521 if ($self->{insertion_mode} == IN_ROW_IM) {
5522 ## As if </tr>
5523 ## have an element in table scope
5524 my $i;
5525 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5526 my $node = $self->{open_elements}->[$_];
5527 if ($node->[1] & TABLE_ROW_EL) {
5528 !!!cp ('t233');
5529 $i = $_;
5530 last INSCOPE;
5531 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5532 !!!cp ('t234');
5533 last INSCOPE;
5534 }
5535 } # INSCOPE
5536 unless (defined $i) {
5537 !!!cp ('t235');
5538 ## TODO: The following is wrong.
5539 !!!parse-error (type => 'unmatched end tag:'.$token->{type}, token => $token);
5540 ## Ignore the token
5541 !!!nack ('t236.1');
5542 !!!next-token;
5543 next B;
5544 }
5545
5546 ## Clear back to table row context
5547 while (not ($self->{open_elements}->[-1]->[1]
5548 & TABLE_ROW_SCOPING_EL)) {
5549 !!!cp ('t236');
5550 ## ISSUE: Can this state be reached?
5551 pop @{$self->{open_elements}};
5552 }
5553
5554 pop @{$self->{open_elements}}; # tr
5555 $self->{insertion_mode} = IN_TABLE_BODY_IM;
5556 ## reprocess in the "in table body" insertion mode...
5557 }
5558
5559 if ($self->{insertion_mode} == IN_TABLE_BODY_IM) {
5560 ## have an element in table scope
5561 my $i;
5562 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5563 my $node = $self->{open_elements}->[$_];
5564 if ($node->[1] & TABLE_ROW_GROUP_EL) {
5565 !!!cp ('t237');
5566 $i = $_;
5567 last INSCOPE;
5568 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5569 !!!cp ('t238');
5570 last INSCOPE;
5571 }
5572 } # INSCOPE
5573 unless (defined $i) {
5574 !!!cp ('t239');
5575 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
5576 ## Ignore the token
5577 !!!nack ('t239.1');
5578 !!!next-token;
5579 next B;
5580 }
5581
5582 ## Clear back to table body context
5583 while (not ($self->{open_elements}->[-1]->[1]
5584 & TABLE_ROWS_SCOPING_EL)) {
5585 !!!cp ('t240');
5586 pop @{$self->{open_elements}};
5587 }
5588
5589 ## As if <{current node}>
5590 ## have an element in table scope
5591 ## true by definition
5592
5593 ## Clear back to table body context
5594 ## nop by definition
5595
5596 pop @{$self->{open_elements}};
5597 $self->{insertion_mode} = IN_TABLE_IM;
5598 ## reprocess in the "in table" insertion mode...
5599 }
5600
5601 ## NOTE: </table> in the "in table" insertion mode.
5602 ## When you edit the code fragment below, please ensure that
5603 ## the code for <table> in the "in table" insertion mode
5604 ## is synced with it.
5605
5606 ## have a table element in table scope
5607 my $i;
5608 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5609 my $node = $self->{open_elements}->[$_];
5610 if ($node->[1] & TABLE_EL) {
5611 !!!cp ('t241');
5612 $i = $_;
5613 last INSCOPE;
5614 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5615 !!!cp ('t242');
5616 last INSCOPE;
5617 }
5618 } # INSCOPE
5619 unless (defined $i) {
5620 !!!cp ('t243');
5621 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
5622 ## Ignore the token
5623 !!!nack ('t243.1');
5624 !!!next-token;
5625 next B;
5626 }
5627
5628 splice @{$self->{open_elements}}, $i;
5629 pop @{$open_tables};
5630
5631 $self->_reset_insertion_mode;
5632
5633 !!!next-token;
5634 next B;
5635 } elsif ({
5636 tbody => 1, tfoot => 1, thead => 1,
5637 }->{$token->{tag_name}} and
5638 $self->{insertion_mode} & ROW_IMS) {
5639 if ($self->{insertion_mode} == IN_ROW_IM) {
5640 ## have an element in table scope
5641 my $i;
5642 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5643 my $node = $self->{open_elements}->[$_];
5644 if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
5645 !!!cp ('t247');
5646 $i = $_;
5647 last INSCOPE;
5648 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5649 !!!cp ('t248');
5650 last INSCOPE;
5651 }
5652 } # INSCOPE
5653 unless (defined $i) {
5654 !!!cp ('t249');
5655 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
5656 ## Ignore the token
5657 !!!nack ('t249.1');
5658 !!!next-token;
5659 next B;
5660 }
5661
5662 ## As if </tr>
5663 ## have an element in table scope
5664 my $i;
5665 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5666 my $node = $self->{open_elements}->[$_];
5667 if ($node->[1] & TABLE_ROW_EL) {
5668 !!!cp ('t250');
5669 $i = $_;
5670 last INSCOPE;
5671 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5672 !!!cp ('t251');
5673 last INSCOPE;
5674 }
5675 } # INSCOPE
5676 unless (defined $i) {
5677 !!!cp ('t252');
5678 !!!parse-error (type => 'unmatched end tag:tr', token => $token);
5679 ## Ignore the token
5680 !!!nack ('t252.1');
5681 !!!next-token;
5682 next B;
5683 }
5684
5685 ## Clear back to table row context
5686 while (not ($self->{open_elements}->[-1]->[1]
5687 & TABLE_ROW_SCOPING_EL)) {
5688 !!!cp ('t253');
5689 ## ISSUE: Can this case be reached?
5690 pop @{$self->{open_elements}};
5691 }
5692
5693 pop @{$self->{open_elements}}; # tr
5694 $self->{insertion_mode} = IN_TABLE_BODY_IM;
5695 ## reprocess in the "in table body" insertion mode...
5696 }
5697
5698 ## have an element in table scope
5699 my $i;
5700 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5701 my $node = $self->{open_elements}->[$_];
5702 if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
5703 !!!cp ('t254');
5704 $i = $_;
5705 last INSCOPE;
5706 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5707 !!!cp ('t255');
5708 last INSCOPE;
5709 }
5710 } # INSCOPE
5711 unless (defined $i) {
5712 !!!cp ('t256');
5713 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
5714 ## Ignore the token
5715 !!!nack ('t256.1');
5716 !!!next-token;
5717 next B;
5718 }
5719
5720 ## Clear back to table body context
5721 while (not ($self->{open_elements}->[-1]->[1]
5722 & TABLE_ROWS_SCOPING_EL)) {
5723 !!!cp ('t257');
5724 ## ISSUE: Can this case be reached?
5725 pop @{$self->{open_elements}};
5726 }
5727
5728 pop @{$self->{open_elements}};
5729 $self->{insertion_mode} = IN_TABLE_IM;
5730 !!!nack ('t257.1');
5731 !!!next-token;
5732 next B;
5733 } elsif ({
5734 body => 1, caption => 1, col => 1, colgroup => 1,
5735 html => 1, td => 1, th => 1,
5736 tr => 1, # $self->{insertion_mode} == IN_ROW_IM
5737 tbody => 1, tfoot => 1, thead => 1, # $self->{insertion_mode} == IN_TABLE_IM
5738 }->{$token->{tag_name}}) {
5739 !!!cp ('t258');
5740 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
5741 ## Ignore the token
5742 !!!nack ('t258.1');
5743 !!!next-token;
5744 next B;
5745 } else {
5746 !!!cp ('t259');
5747 !!!parse-error (type => 'in table:/'.$token->{tag_name}, token => $token);
5748
5749 $insert = $insert_to_foster;
5750 #
5751 }
5752 } elsif ($token->{type} == END_OF_FILE_TOKEN) {
5753 unless ($self->{open_elements}->[-1]->[1] & HTML_EL and
5754 @{$self->{open_elements}} == 1) { # redundant, maybe
5755 !!!parse-error (type => 'in body:#eof', token => $token);
5756 !!!cp ('t259.1');
5757 #
5758 } else {
5759 !!!cp ('t259.2');
5760 #
5761 }
5762
5763 ## Stop parsing
5764 last B;
5765 } else {
5766 die "$0: $token->{type}: Unknown token type";
5767 }
5768 } elsif ($self->{insertion_mode} == IN_COLUMN_GROUP_IM) {
5769 if ($token->{type} == CHARACTER_TOKEN) {
5770 if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {
5771 $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
5772 unless (length $token->{data}) {
5773 !!!cp ('t260');
5774 !!!next-token;
5775 next B;
5776 }
5777 }
5778
5779 !!!cp ('t261');
5780 #
5781 } elsif ($token->{type} == START_TAG_TOKEN) {
5782 if ($token->{tag_name} eq 'col') {
5783 !!!cp ('t262');
5784 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5785 pop @{$self->{open_elements}};
5786 !!!ack ('t262.1');
5787 !!!next-token;
5788 next B;
5789 } else {
5790 !!!cp ('t263');
5791 #
5792 }
5793 } elsif ($token->{type} == END_TAG_TOKEN) {
5794 if ($token->{tag_name} eq 'colgroup') {
5795 if ($self->{open_elements}->[-1]->[1] & HTML_EL) {
5796 !!!cp ('t264');
5797 !!!parse-error (type => 'unmatched end tag:colgroup', token => $token);
5798 ## Ignore the token
5799 !!!next-token;
5800 next B;
5801 } else {
5802 !!!cp ('t265');
5803 pop @{$self->{open_elements}}; # colgroup
5804 $self->{insertion_mode} = IN_TABLE_IM;
5805 !!!next-token;
5806 next B;
5807 }
5808 } elsif ($token->{tag_name} eq 'col') {
5809 !!!cp ('t266');
5810 !!!parse-error (type => 'unmatched end tag:col', token => $token);
5811 ## Ignore the token
5812 !!!next-token;
5813 next B;
5814 } else {
5815 !!!cp ('t267');
5816 #
5817 }
5818 } elsif ($token->{type} == END_OF_FILE_TOKEN) {
5819 if ($self->{open_elements}->[-1]->[1] & HTML_EL and
5820 @{$self->{open_elements}} == 1) { # redundant, maybe
5821 !!!cp ('t270.2');
5822 ## Stop parsing.
5823 last B;
5824 } else {
5825 ## NOTE: As if </colgroup>.
5826 !!!cp ('t270.1');
5827 pop @{$self->{open_elements}}; # colgroup
5828 $self->{insertion_mode} = IN_TABLE_IM;
5829 ## Reprocess.
5830 next B;
5831 }
5832 } else {
5833 die "$0: $token->{type}: Unknown token type";
5834 }
5835
5836 ## As if </colgroup>
5837 if ($self->{open_elements}->[-1]->[1] & HTML_EL) {
5838 !!!cp ('t269');
5839 ## TODO: Wrong error type?
5840 !!!parse-error (type => 'unmatched end tag:colgroup', token => $token);
5841 ## Ignore the token
5842 !!!nack ('t269.1');
5843 !!!next-token;
5844 next B;
5845 } else {
5846 !!!cp ('t270');
5847 pop @{$self->{open_elements}}; # colgroup
5848 $self->{insertion_mode} = IN_TABLE_IM;
5849 !!!ack-later;
5850 ## reprocess
5851 next B;
5852 }
5853 } elsif ($self->{insertion_mode} & SELECT_IMS) {
5854 if ($token->{type} == CHARACTER_TOKEN) {
5855 !!!cp ('t271');
5856 $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});
5857 !!!next-token;
5858 next B;
5859 } elsif ($token->{type} == START_TAG_TOKEN) {
5860 if ($token->{tag_name} eq 'option') {
5861 if ($self->{open_elements}->[-1]->[1] & OPTION_EL) {
5862 !!!cp ('t272');
5863 ## As if </option>
5864 pop @{$self->{open_elements}};
5865 } else {
5866 !!!cp ('t273');
5867 }
5868
5869 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5870 !!!nack ('t273.1');
5871 !!!next-token;
5872 next B;
5873 } elsif ($token->{tag_name} eq 'optgroup') {
5874 if ($self->{open_elements}->[-1]->[1] & OPTION_EL) {
5875 !!!cp ('t274');
5876 ## As if </option>
5877 pop @{$self->{open_elements}};
5878 } else {
5879 !!!cp ('t275');
5880 }
5881
5882 if ($self->{open_elements}->[-1]->[1] & OPTGROUP_EL) {
5883 !!!cp ('t276');
5884 ## As if </optgroup>
5885 pop @{$self->{open_elements}};
5886 } else {
5887 !!!cp ('t277');
5888 }
5889
5890 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5891 !!!nack ('t277.1');
5892 !!!next-token;
5893 next B;
5894 } elsif ({
5895 select => 1, input => 1, textarea => 1,
5896 }->{$token->{tag_name}} or
5897 ($self->{insertion_mode} == IN_SELECT_IN_TABLE_IM and
5898 {
5899 caption => 1, table => 1,
5900 tbody => 1, tfoot => 1, thead => 1,
5901 tr => 1, td => 1, th => 1,
5902 }->{$token->{tag_name}})) {
5903 ## TODO: The type below is not good - <select> is replaced by </select>
5904 !!!parse-error (type => 'not closed:select', token => $token);
5905 ## NOTE: As if the token were </select> (<select> case) or
5906 ## as if there were </select> (otherwise).
5907 ## have an element in table scope
5908 my $i;
5909 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5910 my $node = $self->{open_elements}->[$_];
5911 if ($node->[1] & SELECT_EL) {
5912 !!!cp ('t278');
5913 $i = $_;
5914 last INSCOPE;
5915 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5916 !!!cp ('t279');
5917 last INSCOPE;
5918 }
5919 } # INSCOPE
5920 unless (defined $i) {
5921 !!!cp ('t280');
5922 !!!parse-error (type => 'unmatched end tag:select', token => $token);
5923 ## Ignore the token
5924 !!!nack ('t280.1');
5925 !!!next-token;
5926 next B;
5927 }
5928
5929 !!!cp ('t281');
5930 splice @{$self->{open_elements}}, $i;
5931
5932 $self->_reset_insertion_mode;
5933
5934 if ($token->{tag_name} eq 'select') {
5935 !!!nack ('t281.2');
5936 !!!next-token;
5937 next B;
5938 } else {
5939 !!!cp ('t281.1');
5940 !!!ack-later;
5941 ## Reprocess the token.
5942 next B;
5943 }
5944 } else {
5945 !!!cp ('t282');
5946 !!!parse-error (type => 'in select:'.$token->{tag_name}, token => $token);
5947 ## Ignore the token
5948 !!!nack ('t282.1');
5949 !!!next-token;
5950 next B;
5951 }
5952 } elsif ($token->{type} == END_TAG_TOKEN) {
5953 if ($token->{tag_name} eq 'optgroup') {
5954 if ($self->{open_elements}->[-1]->[1] & OPTION_EL and
5955 $self->{open_elements}->[-2]->[1] & OPTGROUP_EL) {
5956 !!!cp ('t283');
5957 ## As if </option>
5958 splice @{$self->{open_elements}}, -2;
5959 } elsif ($self->{open_elements}->[-1]->[1] & OPTGROUP_EL) {
5960 !!!cp ('t284');
5961 pop @{$self->{open_elements}};
5962 } else {
5963 !!!cp ('t285');
5964 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
5965 ## Ignore the token
5966 }
5967 !!!nack ('t285.1');
5968 !!!next-token;
5969 next B;
5970 } elsif ($token->{tag_name} eq 'option') {
5971 if ($self->{open_elements}->[-1]->[1] & OPTION_EL) {
5972 !!!cp ('t286');
5973 pop @{$self->{open_elements}};
5974 } else {
5975 !!!cp ('t287');
5976 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
5977 ## Ignore the token
5978 }
5979 !!!nack ('t287.1');
5980 !!!next-token;
5981 next B;
5982 } elsif ($token->{tag_name} eq 'select') {
5983 ## have an element in table scope
5984 my $i;
5985 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5986 my $node = $self->{open_elements}->[$_];
5987 if ($node->[1] & SELECT_EL) {
5988 !!!cp ('t288');
5989 $i = $_;
5990 last INSCOPE;
5991 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5992 !!!cp ('t289');
5993 last INSCOPE;
5994 }
5995 } # INSCOPE
5996 unless (defined $i) {
5997 !!!cp ('t290');
5998 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
5999 ## Ignore the token
6000 !!!nack ('t290.1');
6001 !!!next-token;
6002 next B;
6003 }
6004
6005 !!!cp ('t291');
6006 splice @{$self->{open_elements}}, $i;
6007
6008 $self->_reset_insertion_mode;
6009
6010 !!!nack ('t291.1');
6011 !!!next-token;
6012 next B;
6013 } elsif ($self->{insertion_mode} == IN_SELECT_IN_TABLE_IM and
6014 {
6015 caption => 1, table => 1, tbody => 1,
6016 tfoot => 1, thead => 1, tr => 1, td => 1, th => 1,
6017 }->{$token->{tag_name}}) {
6018 ## TODO: The following is wrong?
6019 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
6020
6021 ## have an element in table scope
6022 my $i;
6023 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6024 my $node = $self->{open_elements}->[$_];
6025 if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
6026 !!!cp ('t292');
6027 $i = $_;
6028 last INSCOPE;
6029 } elsif ($node->[1] & TABLE_SCOPING_EL) {
6030 !!!cp ('t293');
6031 last INSCOPE;
6032 }
6033 } # INSCOPE
6034 unless (defined $i) {
6035 !!!cp ('t294');
6036 ## Ignore the token
6037 !!!nack ('t294.1');
6038 !!!next-token;
6039 next B;
6040 }
6041
6042 ## As if </select>
6043 ## have an element in table scope
6044 undef $i;
6045 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6046 my $node = $self->{open_elements}->[$_];
6047 if ($node->[1] & SELECT_EL) {
6048 !!!cp ('t295');
6049 $i = $_;
6050 last INSCOPE;
6051 } elsif ($node->[1] & TABLE_SCOPING_EL) {
6052 ## ISSUE: Can this state be reached?
6053 !!!cp ('t296');
6054 last INSCOPE;
6055 }
6056 } # INSCOPE
6057 unless (defined $i) {
6058 !!!cp ('t297');
6059 ## TODO: The following error type is correct?
6060 !!!parse-error (type => 'unmatched end tag:select', token => $token);
6061 ## Ignore the </select> token
6062 !!!nack ('t297.1');
6063 !!!next-token; ## TODO: ok?
6064 next B;
6065 }
6066
6067 !!!cp ('t298');
6068 splice @{$self->{open_elements}}, $i;
6069
6070 $self->_reset_insertion_mode;
6071
6072 !!!ack-later;
6073 ## reprocess
6074 next B;
6075 } else {
6076 !!!cp ('t299');
6077 !!!parse-error (type => 'in select:/'.$token->{tag_name}, token => $token);
6078 ## Ignore the token
6079 !!!nack ('t299.3');
6080 !!!next-token;
6081 next B;
6082 }
6083 } elsif ($token->{type} == END_OF_FILE_TOKEN) {
6084 unless ($self->{open_elements}->[-1]->[1] & HTML_EL and
6085 @{$self->{open_elements}} == 1) { # redundant, maybe
6086 !!!cp ('t299.1');
6087 !!!parse-error (type => 'in body:#eof', token => $token);
6088 } else {
6089 !!!cp ('t299.2');
6090 }
6091
6092 ## Stop parsing.
6093 last B;
6094 } else {
6095 die "$0: $token->{type}: Unknown token type";
6096 }
6097 } elsif ($self->{insertion_mode} & BODY_AFTER_IMS) {
6098 if ($token->{type} == CHARACTER_TOKEN) {
6099 if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {
6100 my $data = $1;
6101 ## As if in body
6102 $reconstruct_active_formatting_elements->($insert_to_current);
6103
6104 $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
6105
6106 unless (length $token->{data}) {
6107 !!!cp ('t300');
6108 !!!next-token;
6109 next B;
6110 }
6111 }
6112
6113 if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
6114 !!!cp ('t301');
6115 !!!parse-error (type => 'after html:#character', token => $token);
6116
6117 ## Reprocess in the "after body" insertion mode.
6118 } else {
6119 !!!cp ('t302');
6120 }
6121
6122 ## "after body" insertion mode
6123 !!!parse-error (type => 'after body:#character', token => $token);
6124
6125 $self->{insertion_mode} = IN_BODY_IM;
6126 ## reprocess
6127 next B;
6128 } elsif ($token->{type} == START_TAG_TOKEN) {
6129 if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
6130 !!!cp ('t303');
6131 !!!parse-error (type => 'after html:'.$token->{tag_name}, token => $token);
6132
6133 ## Reprocess in the "after body" insertion mode.
6134 } else {
6135 !!!cp ('t304');
6136 }
6137
6138 ## "after body" insertion mode
6139 !!!parse-error (type => 'after body:'.$token->{tag_name}, token => $token);
6140
6141 $self->{insertion_mode} = IN_BODY_IM;
6142 !!!ack-later;
6143 ## reprocess
6144 next B;
6145 } elsif ($token->{type} == END_TAG_TOKEN) {
6146 if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
6147 !!!cp ('t305');
6148 !!!parse-error (type => 'after html:/'.$token->{tag_name}, token => $token);
6149
6150 $self->{insertion_mode} = AFTER_BODY_IM;
6151 ## Reprocess in the "after body" insertion mode.
6152 } else {
6153 !!!cp ('t306');
6154 }
6155
6156 ## "after body" insertion mode
6157 if ($token->{tag_name} eq 'html') {
6158 if (defined $self->{inner_html_node}) {
6159 !!!cp ('t307');
6160 !!!parse-error (type => 'unmatched end tag:html', token => $token);
6161 ## Ignore the token
6162 !!!next-token;
6163 next B;
6164 } else {
6165 !!!cp ('t308');
6166 $self->{insertion_mode} = AFTER_HTML_BODY_IM;
6167 !!!next-token;
6168 next B;
6169 }
6170 } else {
6171 !!!cp ('t309');
6172 !!!parse-error (type => 'after body:/'.$token->{tag_name}, token => $token);
6173
6174 $self->{insertion_mode} = IN_BODY_IM;
6175 ## reprocess
6176 next B;
6177 }
6178 } elsif ($token->{type} == END_OF_FILE_TOKEN) {
6179 !!!cp ('t309.2');
6180 ## Stop parsing
6181 last B;
6182 } else {
6183 die "$0: $token->{type}: Unknown token type";
6184 }
6185 } elsif ($self->{insertion_mode} & FRAME_IMS) {
6186 if ($token->{type} == CHARACTER_TOKEN) {
6187 if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {
6188 $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
6189
6190 unless (length $token->{data}) {
6191 !!!cp ('t310');
6192 !!!next-token;
6193 next B;
6194 }
6195 }
6196
6197 if ($token->{data} =~ s/^[^\x09\x0A\x0B\x0C\x20]+//) {
6198 if ($self->{insertion_mode} == IN_FRAMESET_IM) {
6199 !!!cp ('t311');
6200 !!!parse-error (type => 'in frameset:#character', token => $token);
6201 } elsif ($self->{insertion_mode} == AFTER_FRAMESET_IM) {
6202 !!!cp ('t312');
6203 !!!parse-error (type => 'after frameset:#character', token => $token);
6204 } else { # "after html frameset"
6205 !!!cp ('t313');
6206 !!!parse-error (type => 'after html:#character', token => $token);
6207
6208 $self->{insertion_mode} = AFTER_FRAMESET_IM;
6209 ## Reprocess in the "after frameset" insertion mode.
6210 !!!parse-error (type => 'after frameset:#character', token => $token);
6211 }
6212
6213 ## Ignore the token.
6214 if (length $token->{data}) {
6215 !!!cp ('t314');
6216 ## reprocess the rest of characters
6217 } else {
6218 !!!cp ('t315');
6219 !!!next-token;
6220 }
6221 next B;
6222 }
6223
6224 die qq[$0: Character "$token->{data}"];
6225 } elsif ($token->{type} == START_TAG_TOKEN) {
6226 if ($self->{insertion_mode} == AFTER_HTML_FRAMESET_IM) {
6227 !!!cp ('t316');
6228 !!!parse-error (type => 'after html:'.$token->{tag_name}, token => $token);
6229
6230 $self->{insertion_mode} = AFTER_FRAMESET_IM;
6231 ## Process in the "after frameset" insertion mode.
6232 } else {
6233 !!!cp ('t317');
6234 }
6235
6236 if ($token->{tag_name} eq 'frameset' and
6237 $self->{insertion_mode} == IN_FRAMESET_IM) {
6238 !!!cp ('t318');
6239 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
6240 !!!nack ('t318.1');
6241 !!!next-token;
6242 next B;
6243 } elsif ($token->{tag_name} eq 'frame' and
6244 $self->{insertion_mode} == IN_FRAMESET_IM) {
6245 !!!cp ('t319');
6246 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
6247 pop @{$self->{open_elements}};
6248 !!!ack ('t319.1');
6249 !!!next-token;
6250 next B;
6251 } elsif ($token->{tag_name} eq 'noframes') {
6252 !!!cp ('t320');
6253 ## NOTE: As if in head.
6254 $parse_rcdata->(CDATA_CONTENT_MODEL);
6255 next B;
6256 } else {
6257 if ($self->{insertion_mode} == IN_FRAMESET_IM) {
6258 !!!cp ('t321');
6259 !!!parse-error (type => 'in frameset:'.$token->{tag_name}, token => $token);
6260 } else {
6261 !!!cp ('t322');
6262 !!!parse-error (type => 'after frameset:'.$token->{tag_name}, token => $token);
6263 }
6264 ## Ignore the token
6265 !!!nack ('t322.1');
6266 !!!next-token;
6267 next B;
6268 }
6269 } elsif ($token->{type} == END_TAG_TOKEN) {
6270 if ($self->{insertion_mode} == AFTER_HTML_FRAMESET_IM) {
6271 !!!cp ('t323');
6272 !!!parse-error (type => 'after html:/'.$token->{tag_name}, token => $token);
6273
6274 $self->{insertion_mode} = AFTER_FRAMESET_IM;
6275 ## Process in the "after frameset" insertion mode.
6276 } else {
6277 !!!cp ('t324');
6278 }
6279
6280 if ($token->{tag_name} eq 'frameset' and
6281 $self->{insertion_mode} == IN_FRAMESET_IM) {
6282 if ($self->{open_elements}->[-1]->[1] & HTML_EL and
6283 @{$self->{open_elements}} == 1) {
6284 !!!cp ('t325');
6285 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
6286 ## Ignore the token
6287 !!!next-token;
6288 } else {
6289 !!!cp ('t326');
6290 pop @{$self->{open_elements}};
6291 !!!next-token;
6292 }
6293
6294 if (not defined $self->{inner_html_node} and
6295 not ($self->{open_elements}->[-1]->[1] & FRAMESET_EL)) {
6296 !!!cp ('t327');
6297 $self->{insertion_mode} = AFTER_FRAMESET_IM;
6298 } else {
6299 !!!cp ('t328');
6300 }
6301 next B;
6302 } elsif ($token->{tag_name} eq 'html' and
6303 $self->{insertion_mode} == AFTER_FRAMESET_IM) {
6304 !!!cp ('t329');
6305 $self->{insertion_mode} = AFTER_HTML_FRAMESET_IM;
6306 !!!next-token;
6307 next B;
6308 } else {
6309 if ($self->{insertion_mode} == IN_FRAMESET_IM) {
6310 !!!cp ('t330');
6311 !!!parse-error (type => 'in frameset:/'.$token->{tag_name}, token => $token);
6312 } else {
6313 !!!cp ('t331');
6314 !!!parse-error (type => 'after frameset:/'.$token->{tag_name}, token => $token);
6315 }
6316 ## Ignore the token
6317 !!!next-token;
6318 next B;
6319 }
6320 } elsif ($token->{type} == END_OF_FILE_TOKEN) {
6321 unless ($self->{open_elements}->[-1]->[1] & HTML_EL and
6322 @{$self->{open_elements}} == 1) { # redundant, maybe
6323 !!!cp ('t331.1');
6324 !!!parse-error (type => 'in body:#eof', token => $token);
6325 } else {
6326 !!!cp ('t331.2');
6327 }
6328
6329 ## Stop parsing
6330 last B;
6331 } else {
6332 die "$0: $token->{type}: Unknown token type";
6333 }
6334
6335 ## ISSUE: An issue in spec here
6336 } else {
6337 die "$0: $self->{insertion_mode}: Unknown insertion mode";
6338 }
6339
6340 ## "in body" insertion mode
6341 if ($token->{type} == START_TAG_TOKEN) {
6342 if ($token->{tag_name} eq 'script') {
6343 !!!cp ('t332');
6344 ## NOTE: This is an "as if in head" code clone
6345 $script_start_tag->();
6346 next B;
6347 } elsif ($token->{tag_name} eq 'style') {
6348 !!!cp ('t333');
6349 ## NOTE: This is an "as if in head" code clone
6350 $parse_rcdata->(CDATA_CONTENT_MODEL);
6351 next B;
6352 } elsif ({
6353 base => 1, link => 1,
6354 }->{$token->{tag_name}}) {
6355 !!!cp ('t334');
6356 ## NOTE: This is an "as if in head" code clone, only "-t" differs
6357 !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
6358 pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.
6359 !!!ack ('t334.1');
6360 !!!next-token;
6361 next B;
6362 } elsif ($token->{tag_name} eq 'meta') {
6363 ## NOTE: This is an "as if in head" code clone, only "-t" differs
6364 !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
6365 my $meta_el = pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.
6366
6367 unless ($self->{confident}) {
6368 if ($token->{attributes}->{charset}) {
6369 !!!cp ('t335');
6370 ## NOTE: Whether the encoding is supported or not is handled
6371 ## in the {change_encoding} callback.
6372 $self->{change_encoding}
6373 ->($self, $token->{attributes}->{charset}->{value}, $token);
6374
6375 $meta_el->[0]->get_attribute_node_ns (undef, 'charset')
6376 ->set_user_data (manakai_has_reference =>
6377 $token->{attributes}->{charset}
6378 ->{has_reference});
6379 } elsif ($token->{attributes}->{content}) {
6380 if ($token->{attributes}->{content}->{value}
6381 =~ /[Cc][Hh][Aa][Rr][Ss][Ee][Tt]
6382 [\x09-\x0D\x20]*=
6383 [\x09-\x0D\x20]*(?>"([^"]*)"|'([^']*)'|
6384 ([^"'\x09-\x0D\x20][^\x09-\x0D\x20\x3B]*))/x) {
6385 !!!cp ('t336');
6386 ## NOTE: Whether the encoding is supported or not is handled
6387 ## in the {change_encoding} callback.
6388 $self->{change_encoding}
6389 ->($self, defined $1 ? $1 : defined $2 ? $2 : $3, $token);
6390 $meta_el->[0]->get_attribute_node_ns (undef, 'content')
6391 ->set_user_data (manakai_has_reference =>
6392 $token->{attributes}->{content}
6393 ->{has_reference});
6394 }
6395 }
6396 } else {
6397 if ($token->{attributes}->{charset}) {
6398 !!!cp ('t337');
6399 $meta_el->[0]->get_attribute_node_ns (undef, 'charset')
6400 ->set_user_data (manakai_has_reference =>
6401 $token->{attributes}->{charset}
6402 ->{has_reference});
6403 }
6404 if ($token->{attributes}->{content}) {
6405 !!!cp ('t338');
6406 $meta_el->[0]->get_attribute_node_ns (undef, 'content')
6407 ->set_user_data (manakai_has_reference =>
6408 $token->{attributes}->{content}
6409 ->{has_reference});
6410 }
6411 }
6412
6413 !!!ack ('t338.1');
6414 !!!next-token;
6415 next B;
6416 } elsif ($token->{tag_name} eq 'title') {
6417 !!!cp ('t341');
6418 ## NOTE: This is an "as if in head" code clone
6419 $parse_rcdata->(RCDATA_CONTENT_MODEL);
6420 next B;
6421 } elsif ($token->{tag_name} eq 'body') {
6422 !!!parse-error (type => 'in body:body', token => $token);
6423
6424 if (@{$self->{open_elements}} == 1 or
6425 not ($self->{open_elements}->[1]->[1] & BODY_EL)) {
6426 !!!cp ('t342');
6427 ## Ignore the token
6428 } else {
6429 my $body_el = $self->{open_elements}->[1]->[0];
6430 for my $attr_name (keys %{$token->{attributes}}) {
6431 unless ($body_el->has_attribute_ns (undef, $attr_name)) {
6432 !!!cp ('t343');
6433 $body_el->set_attribute_ns
6434 (undef, [undef, $attr_name],
6435 $token->{attributes}->{$attr_name}->{value});
6436 }
6437 }
6438 }
6439 !!!nack ('t343.1');
6440 !!!next-token;
6441 next B;
6442 } elsif ({
6443 address => 1, blockquote => 1, center => 1, dir => 1,
6444 div => 1, dl => 1, fieldset => 1,
6445 h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,
6446 menu => 1, ol => 1, p => 1, ul => 1,
6447 pre => 1, listing => 1,
6448 form => 1,
6449 table => 1,
6450 hr => 1,
6451 }->{$token->{tag_name}}) {
6452 if ($token->{tag_name} eq 'form' and defined $self->{form_element}) {
6453 !!!cp ('t350');
6454 !!!parse-error (type => 'in form:form', token => $token);
6455 ## Ignore the token
6456 !!!nack ('t350.1');
6457 !!!next-token;
6458 next B;
6459 }
6460
6461 ## has a p element in scope
6462 INSCOPE: for (reverse @{$self->{open_elements}}) {
6463 if ($_->[1] & P_EL) {
6464 !!!cp ('t344');
6465 !!!back-token; # <form>
6466 $token = {type => END_TAG_TOKEN, tag_name => 'p',
6467 line => $token->{line}, column => $token->{column}};
6468 next B;
6469 } elsif ($_->[1] & SCOPING_EL) {
6470 !!!cp ('t345');
6471 last INSCOPE;
6472 }
6473 } # INSCOPE
6474
6475 !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
6476 if ($token->{tag_name} eq 'pre' or $token->{tag_name} eq 'listing') {
6477 !!!nack ('t346.1');
6478 !!!next-token;
6479 if ($token->{type} == CHARACTER_TOKEN) {
6480 $token->{data} =~ s/^\x0A//;
6481 unless (length $token->{data}) {
6482 !!!cp ('t346');
6483 !!!next-token;
6484 } else {
6485 !!!cp ('t349');
6486 }
6487 } else {
6488 !!!cp ('t348');
6489 }
6490 } elsif ($token->{tag_name} eq 'form') {
6491 !!!cp ('t347.1');
6492 $self->{form_element} = $self->{open_elements}->[-1]->[0];
6493
6494 !!!nack ('t347.2');
6495 !!!next-token;
6496 } elsif ($token->{tag_name} eq 'table') {
6497 !!!cp ('t382');
6498 push @{$open_tables}, [$self->{open_elements}->[-1]->[0]];
6499
6500 $self->{insertion_mode} = IN_TABLE_IM;
6501
6502 !!!nack ('t382.1');
6503 !!!next-token;
6504 } elsif ($token->{tag_name} eq 'hr') {
6505 !!!cp ('t386');
6506 pop @{$self->{open_elements}};
6507
6508 !!!nack ('t386.1');
6509 !!!next-token;
6510 } else {
6511 !!!nack ('t347.1');
6512 !!!next-token;
6513 }
6514 next B;
6515 } elsif ({li => 1, dt => 1, dd => 1}->{$token->{tag_name}}) {
6516 ## has a p element in scope
6517 INSCOPE: for (reverse @{$self->{open_elements}}) {
6518 if ($_->[1] & P_EL) {
6519 !!!cp ('t353');
6520 !!!back-token; # <x>
6521 $token = {type => END_TAG_TOKEN, tag_name => 'p',
6522 line => $token->{line}, column => $token->{column}};
6523 next B;
6524 } elsif ($_->[1] & SCOPING_EL) {
6525 !!!cp ('t354');
6526 last INSCOPE;
6527 }
6528 } # INSCOPE
6529
6530 ## Step 1
6531 my $i = -1;
6532 my $node = $self->{open_elements}->[$i];
6533 my $li_or_dtdd = {li => {li => 1},
6534 dt => {dt => 1, dd => 1},
6535 dd => {dt => 1, dd => 1}}->{$token->{tag_name}};
6536 LI: {
6537 ## Step 2
6538 if ($li_or_dtdd->{$node->[0]->manakai_local_name}) {
6539 if ($i != -1) {
6540 !!!cp ('t355');
6541 !!!parse-error (type => 'not closed',
6542 value => $self->{open_elements}->[-1]->[0]
6543 ->manakai_local_name,
6544 token => $token);
6545 } else {
6546 !!!cp ('t356');
6547 }
6548 splice @{$self->{open_elements}}, $i;
6549 last LI;
6550 } else {
6551 !!!cp ('t357');
6552 }
6553
6554 ## Step 3
6555 if (not ($node->[1] & FORMATTING_EL) and
6556 #not $phrasing_category->{$node->[1]} and
6557 ($node->[1] & SPECIAL_EL or
6558 $node->[1] & SCOPING_EL) and
6559 not ($node->[1] & ADDRESS_EL) and
6560 not ($node->[1] & DIV_EL)) {
6561 !!!cp ('t358');
6562 last LI;
6563 }
6564
6565 !!!cp ('t359');
6566 ## Step 4
6567 $i--;
6568 $node = $self->{open_elements}->[$i];
6569 redo LI;
6570 } # LI
6571
6572 !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
6573 !!!nack ('t359.1');
6574 !!!next-token;
6575 next B;
6576 } elsif ($token->{tag_name} eq 'plaintext') {
6577 ## has a p element in scope
6578 INSCOPE: for (reverse @{$self->{open_elements}}) {
6579 if ($_->[1] & P_EL) {
6580 !!!cp ('t367');
6581 !!!back-token; # <plaintext>
6582 $token = {type => END_TAG_TOKEN, tag_name => 'p',
6583 line => $token->{line}, column => $token->{column}};
6584 next B;
6585 } elsif ($_->[1] & SCOPING_EL) {
6586 !!!cp ('t368');
6587 last INSCOPE;
6588 }
6589 } # INSCOPE
6590
6591 !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
6592
6593 $self->{content_model} = PLAINTEXT_CONTENT_MODEL;
6594
6595 !!!nack ('t368.1');
6596 !!!next-token;
6597 next B;
6598 } elsif ($token->{tag_name} eq 'a') {
6599 AFE: for my $i (reverse 0..$#$active_formatting_elements) {
6600 my $node = $active_formatting_elements->[$i];
6601 if ($node->[1] & A_EL) {
6602 !!!cp ('t371');
6603 !!!parse-error (type => 'in a:a', token => $token);
6604
6605 !!!back-token; # <a>
6606 $token = {type => END_TAG_TOKEN, tag_name => 'a',
6607 line => $token->{line}, column => $token->{column}};
6608 $formatting_end_tag->($token);
6609
6610 AFE2: for (reverse 0..$#$active_formatting_elements) {
6611 if ($active_formatting_elements->[$_]->[0] eq $node->[0]) {
6612 !!!cp ('t372');
6613 splice @$active_formatting_elements, $_, 1;
6614 last AFE2;
6615 }
6616 } # AFE2
6617 OE: for (reverse 0..$#{$self->{open_elements}}) {
6618 if ($self->{open_elements}->[$_]->[0] eq $node->[0]) {
6619 !!!cp ('t373');
6620 splice @{$self->{open_elements}}, $_, 1;
6621 last OE;
6622 }
6623 } # OE
6624 last AFE;
6625 } elsif ($node->[0] eq '#marker') {
6626 !!!cp ('t374');
6627 last AFE;
6628 }
6629 } # AFE
6630
6631 $reconstruct_active_formatting_elements->($insert_to_current);
6632
6633 !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
6634 push @$active_formatting_elements, $self->{open_elements}->[-1];
6635
6636 !!!nack ('t374.1');
6637 !!!next-token;
6638 next B;
6639 } elsif ($token->{tag_name} eq 'nobr') {
6640 $reconstruct_active_formatting_elements->($insert_to_current);
6641
6642 ## has a |nobr| element in scope
6643 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6644 my $node = $self->{open_elements}->[$_];
6645 if ($node->[1] & NOBR_EL) {
6646 !!!cp ('t376');
6647 !!!parse-error (type => 'in nobr:nobr', token => $token);
6648 !!!back-token; # <nobr>
6649 $token = {type => END_TAG_TOKEN, tag_name => 'nobr',
6650 line => $token->{line}, column => $token->{column}};
6651 next B;
6652 } elsif ($node->[1] & SCOPING_EL) {
6653 !!!cp ('t377');
6654 last INSCOPE;
6655 }
6656 } # INSCOPE
6657
6658 !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
6659 push @$active_formatting_elements, $self->{open_elements}->[-1];
6660
6661 !!!nack ('t377.1');
6662 !!!next-token;
6663 next B;
6664 } elsif ($token->{tag_name} eq 'button') {
6665 ## has a button element in scope
6666 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6667 my $node = $self->{open_elements}->[$_];
6668 if ($node->[1] & BUTTON_EL) {
6669 !!!cp ('t378');
6670 !!!parse-error (type => 'in button:button', token => $token);
6671 !!!back-token; # <button>
6672 $token = {type => END_TAG_TOKEN, tag_name => 'button',
6673 line => $token->{line}, column => $token->{column}};
6674 next B;
6675 } elsif ($node->[1] & SCOPING_EL) {
6676 !!!cp ('t379');
6677 last INSCOPE;
6678 }
6679 } # INSCOPE
6680
6681 $reconstruct_active_formatting_elements->($insert_to_current);
6682
6683 !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
6684
6685 ## TODO: associate with $self->{form_element} if defined
6686
6687 push @$active_formatting_elements, ['#marker', ''];
6688
6689 !!!nack ('t379.1');
6690 !!!next-token;
6691 next B;
6692 } elsif ({
6693 xmp => 1,
6694 iframe => 1,
6695 noembed => 1,
6696 noframes => 1, ## NOTE: This is an "as if in head" code clone.
6697 noscript => 0, ## TODO: 1 if scripting is enabled
6698 }->{$token->{tag_name}}) {
6699 if ($token->{tag_name} eq 'xmp') {
6700 !!!cp ('t381');
6701 $reconstruct_active_formatting_elements->($insert_to_current);
6702 } else {
6703 !!!cp ('t399');
6704 }
6705 ## NOTE: There is an "as if in body" code clone.
6706 $parse_rcdata->(CDATA_CONTENT_MODEL);
6707 next B;
6708 } elsif ($token->{tag_name} eq 'isindex') {
6709 !!!parse-error (type => 'isindex', token => $token);
6710
6711 if (defined $self->{form_element}) {
6712 !!!cp ('t389');
6713 ## Ignore the token
6714 !!!nack ('t389'); ## NOTE: Not acknowledged.
6715 !!!next-token;
6716 next B;
6717 } else {
6718 !!!ack ('t391.1');
6719
6720 my $at = $token->{attributes};
6721 my $form_attrs;
6722 $form_attrs->{action} = $at->{action} if $at->{action};
6723 my $prompt_attr = $at->{prompt};
6724 $at->{name} = {name => 'name', value => 'isindex'};
6725 delete $at->{action};
6726 delete $at->{prompt};
6727 my @tokens = (
6728 {type => START_TAG_TOKEN, tag_name => 'form',
6729 attributes => $form_attrs,
6730 line => $token->{line}, column => $token->{column}},
6731 {type => START_TAG_TOKEN, tag_name => 'hr',
6732 line => $token->{line}, column => $token->{column}},
6733 {type => START_TAG_TOKEN, tag_name => 'p',
6734 line => $token->{line}, column => $token->{column}},
6735 {type => START_TAG_TOKEN, tag_name => 'label',
6736 line => $token->{line}, column => $token->{column}},
6737 );
6738 if ($prompt_attr) {
6739 !!!cp ('t390');
6740 push @tokens, {type => CHARACTER_TOKEN, data => $prompt_attr->{value},
6741 #line => $token->{line}, column => $token->{column},
6742 };
6743 } else {
6744 !!!cp ('t391');
6745 push @tokens, {type => CHARACTER_TOKEN,
6746 data => 'This is a searchable index. Insert your search keywords here: ',
6747 #line => $token->{line}, column => $token->{column},
6748 }; # SHOULD
6749 ## TODO: make this configurable
6750 }
6751 push @tokens,
6752 {type => START_TAG_TOKEN, tag_name => 'input', attributes => $at,
6753 line => $token->{line}, column => $token->{column}},
6754 #{type => CHARACTER_TOKEN, data => ''}, # SHOULD
6755 {type => END_TAG_TOKEN, tag_name => 'label',
6756 line => $token->{line}, column => $token->{column}},
6757 {type => END_TAG_TOKEN, tag_name => 'p',
6758 line => $token->{line}, column => $token->{column}},
6759 {type => START_TAG_TOKEN, tag_name => 'hr',
6760 line => $token->{line}, column => $token->{column}},
6761 {type => END_TAG_TOKEN, tag_name => 'form',
6762 line => $token->{line}, column => $token->{column}};
6763 !!!back-token (@tokens);
6764 !!!next-token;
6765 next B;
6766 }
6767 } elsif ($token->{tag_name} eq 'textarea') {
6768 my $tag_name = $token->{tag_name};
6769 my $el;
6770 !!!create-element ($el, $HTML_NS, $token->{tag_name}, $token->{attributes}, $token);
6771
6772 ## TODO: $self->{form_element} if defined
6773 $self->{content_model} = RCDATA_CONTENT_MODEL;
6774 delete $self->{escape}; # MUST
6775
6776 $insert->($el);
6777
6778 my $text = '';
6779 !!!nack ('t392.1');
6780 !!!next-token;
6781 if ($token->{type} == CHARACTER_TOKEN) {
6782 $token->{data} =~ s/^\x0A//;
6783 unless (length $token->{data}) {
6784 !!!cp ('t392');
6785 !!!next-token;
6786 } else {
6787 !!!cp ('t393');
6788 }
6789 } else {
6790 !!!cp ('t394');
6791 }
6792 while ($token->{type} == CHARACTER_TOKEN) {
6793 !!!cp ('t395');
6794 $text .= $token->{data};
6795 !!!next-token;
6796 }
6797 if (length $text) {
6798 !!!cp ('t396');
6799 $el->manakai_append_text ($text);
6800 }
6801
6802 $self->{content_model} = PCDATA_CONTENT_MODEL;
6803
6804 if ($token->{type} == END_TAG_TOKEN and
6805 $token->{tag_name} eq $tag_name) {
6806 !!!cp ('t397');
6807 ## Ignore the token
6808 } else {
6809 !!!cp ('t398');
6810 !!!parse-error (type => 'in RCDATA:#'.$token->{type}, token => $token);
6811 }
6812 !!!next-token;
6813 next B;
6814 } elsif ($token->{tag_name} eq 'rt' or
6815 $token->{tag_name} eq 'rp') {
6816 ## has a |ruby| element in scope
6817 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6818 my $node = $self->{open_elements}->[$_];
6819 if ($node->[1] & RUBY_EL) {
6820 !!!cp ('t398.1');
6821 ## generate implied end tags
6822 while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
6823 !!!cp ('t398.2');
6824 pop @{$self->{open_elements}};
6825 }
6826 unless ($self->{open_elements}->[-1]->[1] & RUBY_EL) {
6827 !!!cp ('t398.3');
6828 !!!parse-error (type => 'not closed',
6829 value => $self->{open_elements}->[-1]->[0]
6830 ->manakai_local_name,
6831 token => $token);
6832 pop @{$self->{open_elements}}
6833 while not $self->{open_elements}->[-1]->[1] & RUBY_EL;
6834 }
6835 last INSCOPE;
6836 } elsif ($node->[1] & SCOPING_EL) {
6837 !!!cp ('t398.4');
6838 last INSCOPE;
6839 }
6840 } # INSCOPE
6841
6842 !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
6843
6844 !!!nack ('t398.5');
6845 !!!next-token;
6846 redo B;
6847 } elsif ($token->{tag_name} eq 'math' or
6848 $token->{tag_name} eq 'svg') {
6849 $reconstruct_active_formatting_elements->($insert_to_current);
6850
6851 ## "adjust SVG attributes" ('svg' only) - done in insert-element-f
6852
6853 ## "adjust foreign attributes" - done in insert-element-f
6854
6855 !!!insert-element-f ($token->{tag_name} eq 'math' ? $MML_NS : $SVG_NS, $token->{tag_name}, $token->{attributes}, $token);
6856
6857 if ($self->{self_closing}) {
6858 pop @{$self->{open_elements}};
6859 !!!ack ('t398.1');
6860 } else {
6861 !!!cp ('t398.2');
6862 $self->{insertion_mode} |= IN_FOREIGN_CONTENT_IM;
6863 ## NOTE: |<body><math><mi><svg>| -> "in foreign content" insertion
6864 ## mode, "in body" (not "in foreign content") secondary insertion
6865 ## mode, maybe.
6866 }
6867
6868 !!!next-token;
6869 next B;
6870 } elsif ({
6871 caption => 1, col => 1, colgroup => 1, frame => 1,
6872 frameset => 1, head => 1, option => 1, optgroup => 1,
6873 tbody => 1, td => 1, tfoot => 1, th => 1,
6874 thead => 1, tr => 1,
6875 }->{$token->{tag_name}}) {
6876 !!!cp ('t401');
6877 !!!parse-error (type => 'in body:'.$token->{tag_name}, token => $token);
6878 ## Ignore the token
6879 !!!nack ('t401.1'); ## NOTE: |<col/>| or |<frame/>| here is an error.
6880 !!!next-token;
6881 next B;
6882
6883 ## ISSUE: An issue on HTML5 new elements in the spec.
6884 } else {
6885 if ($token->{tag_name} eq 'image') {
6886 !!!cp ('t384');
6887 !!!parse-error (type => 'image', token => $token);
6888 $token->{tag_name} = 'img';
6889 } else {
6890 !!!cp ('t385');
6891 }
6892
6893 ## NOTE: There is an "as if <br>" code clone.
6894 $reconstruct_active_formatting_elements->($insert_to_current);
6895
6896 !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
6897
6898 if ({
6899 applet => 1, marquee => 1, object => 1,
6900 }->{$token->{tag_name}}) {
6901 !!!cp ('t380');
6902 push @$active_formatting_elements, ['#marker', ''];
6903 !!!nack ('t380.1');
6904 } elsif ({
6905 b => 1, big => 1, em => 1, font => 1, i => 1,
6906 s => 1, small => 1, strile => 1,
6907 strong => 1, tt => 1, u => 1,
6908 }->{$token->{tag_name}}) {
6909 !!!cp ('t375');
6910 push @$active_formatting_elements, $self->{open_elements}->[-1];
6911 !!!nack ('t375.1');
6912 } elsif ($token->{tag_name} eq 'input') {
6913 !!!cp ('t388');
6914 ## TODO: associate with $self->{form_element} if defined
6915 pop @{$self->{open_elements}};
6916 !!!ack ('t388.2');
6917 } elsif ({
6918 area => 1, basefont => 1, bgsound => 1, br => 1,
6919 embed => 1, img => 1, param => 1, spacer => 1, wbr => 1,
6920 #image => 1,
6921 }->{$token->{tag_name}}) {
6922 !!!cp ('t388.1');
6923 pop @{$self->{open_elements}};
6924 !!!ack ('t388.3');
6925 } elsif ($token->{tag_name} eq 'select') {
6926 ## TODO: associate with $self->{form_element} if defined
6927
6928 if ($self->{insertion_mode} & TABLE_IMS or
6929 $self->{insertion_mode} & BODY_TABLE_IMS or
6930 $self->{insertion_mode} == IN_COLUMN_GROUP_IM) {
6931 !!!cp ('t400.1');
6932 $self->{insertion_mode} = IN_SELECT_IN_TABLE_IM;
6933 } else {
6934 !!!cp ('t400.2');
6935 $self->{insertion_mode} = IN_SELECT_IM;
6936 }
6937 !!!nack ('t400.3');
6938 } else {
6939 !!!nack ('t402');
6940 }
6941
6942 !!!next-token;
6943 next B;
6944 }
6945 } elsif ($token->{type} == END_TAG_TOKEN) {
6946 if ($token->{tag_name} eq 'body') {
6947 ## has a |body| element in scope
6948 my $i;
6949 INSCOPE: {
6950 for (reverse @{$self->{open_elements}}) {
6951 if ($_->[1] & BODY_EL) {
6952 !!!cp ('t405');
6953 $i = $_;
6954 last INSCOPE;
6955 } elsif ($_->[1] & SCOPING_EL) {
6956 !!!cp ('t405.1');
6957 last;
6958 }
6959 }
6960
6961 !!!parse-error (type => 'start tag not allowed',
6962 value => $token->{tag_name}, token => $token);
6963 ## NOTE: Ignore the token.
6964 !!!next-token;
6965 next B;
6966 } # INSCOPE
6967
6968 for (@{$self->{open_elements}}) {
6969 unless ($_->[1] & ALL_END_TAG_OPTIONAL_EL) {
6970 !!!cp ('t403');
6971 !!!parse-error (type => 'not closed',
6972 value => $_->[0]->manakai_local_name,
6973 token => $token);
6974 last;
6975 } else {
6976 !!!cp ('t404');
6977 }
6978 }
6979
6980 $self->{insertion_mode} = AFTER_BODY_IM;
6981 !!!next-token;
6982 next B;
6983 } elsif ($token->{tag_name} eq 'html') {
6984 ## TODO: Update this code. It seems that the code below is not
6985 ## up-to-date, though it has same effect as speced.
6986 if (@{$self->{open_elements}} > 1 and
6987 $self->{open_elements}->[1]->[1] & BODY_EL) {
6988 ## ISSUE: There is an issue in the spec.
6989 unless ($self->{open_elements}->[-1]->[1] & BODY_EL) {
6990 !!!cp ('t406');
6991 !!!parse-error (type => 'not closed',
6992 value => $self->{open_elements}->[1]->[0]
6993 ->manakai_local_name,
6994 token => $token);
6995 } else {
6996 !!!cp ('t407');
6997 }
6998 $self->{insertion_mode} = AFTER_BODY_IM;
6999 ## reprocess
7000 next B;
7001 } else {
7002 !!!cp ('t408');
7003 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
7004 ## Ignore the token
7005 !!!next-token;
7006 next B;
7007 }
7008 } elsif ({
7009 address => 1, blockquote => 1, center => 1, dir => 1,
7010 div => 1, dl => 1, fieldset => 1, listing => 1,
7011 menu => 1, ol => 1, pre => 1, ul => 1,
7012 dd => 1, dt => 1, li => 1,
7013 applet => 1, button => 1, marquee => 1, object => 1,
7014 }->{$token->{tag_name}}) {
7015 ## has an element in scope
7016 my $i;
7017 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7018 my $node = $self->{open_elements}->[$_];
7019 if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
7020 !!!cp ('t410');
7021 $i = $_;
7022 last INSCOPE;
7023 } elsif ($node->[1] & SCOPING_EL) {
7024 !!!cp ('t411');
7025 last INSCOPE;
7026 }
7027 } # INSCOPE
7028
7029 unless (defined $i) { # has an element in scope
7030 !!!cp ('t413');
7031 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
7032 } else {
7033 ## Step 1. generate implied end tags
7034 while ({
7035 ## END_TAG_OPTIONAL_EL
7036 dd => ($token->{tag_name} ne 'dd'),
7037 dt => ($token->{tag_name} ne 'dt'),
7038 li => ($token->{tag_name} ne 'li'),
7039 p => 1,
7040 rt => 1,
7041 rp => 1,
7042 }->{$self->{open_elements}->[-1]->[0]->manakai_local_name}) {
7043 !!!cp ('t409');
7044 pop @{$self->{open_elements}};
7045 }
7046
7047 ## Step 2.
7048 if ($self->{open_elements}->[-1]->[0]->manakai_local_name
7049 ne $token->{tag_name}) {
7050 !!!cp ('t412');
7051 !!!parse-error (type => 'not closed',
7052 value => $self->{open_elements}->[-1]->[0]
7053 ->manakai_local_name,
7054 token => $token);
7055 } else {
7056 !!!cp ('t414');
7057 }
7058
7059 ## Step 3.
7060 splice @{$self->{open_elements}}, $i;
7061
7062 ## Step 4.
7063 $clear_up_to_marker->()
7064 if {
7065 applet => 1, button => 1, marquee => 1, object => 1,
7066 }->{$token->{tag_name}};
7067 }
7068 !!!next-token;
7069 next B;
7070 } elsif ($token->{tag_name} eq 'form') {
7071 undef $self->{form_element};
7072
7073 ## has an element in scope
7074 my $i;
7075 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7076 my $node = $self->{open_elements}->[$_];
7077 if ($node->[1] & FORM_EL) {
7078 !!!cp ('t418');
7079 $i = $_;
7080 last INSCOPE;
7081 } elsif ($node->[1] & SCOPING_EL) {
7082 !!!cp ('t419');
7083 last INSCOPE;
7084 }
7085 } # INSCOPE
7086
7087 unless (defined $i) { # has an element in scope
7088 !!!cp ('t421');
7089 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
7090 } else {
7091 ## Step 1. generate implied end tags
7092 while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
7093 !!!cp ('t417');
7094 pop @{$self->{open_elements}};
7095 }
7096
7097 ## Step 2.
7098 if ($self->{open_elements}->[-1]->[0]->manakai_local_name
7099 ne $token->{tag_name}) {
7100 !!!cp ('t417.1');
7101 !!!parse-error (type => 'not closed',
7102 value => $self->{open_elements}->[-1]->[0]
7103 ->manakai_local_name,
7104 token => $token);
7105 } else {
7106 !!!cp ('t420');
7107 }
7108
7109 ## Step 3.
7110 splice @{$self->{open_elements}}, $i;
7111 }
7112
7113 !!!next-token;
7114 next B;
7115 } elsif ({
7116 h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,
7117 }->{$token->{tag_name}}) {
7118 ## has an element in scope
7119 my $i;
7120 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7121 my $node = $self->{open_elements}->[$_];
7122 if ($node->[1] & HEADING_EL) {
7123 !!!cp ('t423');
7124 $i = $_;
7125 last INSCOPE;
7126 } elsif ($node->[1] & SCOPING_EL) {
7127 !!!cp ('t424');
7128 last INSCOPE;
7129 }
7130 } # INSCOPE
7131
7132 unless (defined $i) { # has an element in scope
7133 !!!cp ('t425.1');
7134 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
7135 } else {
7136 ## Step 1. generate implied end tags
7137 while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
7138 !!!cp ('t422');
7139 pop @{$self->{open_elements}};
7140 }
7141
7142 ## Step 2.
7143 if ($self->{open_elements}->[-1]->[0]->manakai_local_name
7144 ne $token->{tag_name}) {
7145 !!!cp ('t425');
7146 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
7147 } else {
7148 !!!cp ('t426');
7149 }
7150
7151 ## Step 3.
7152 splice @{$self->{open_elements}}, $i;
7153 }
7154
7155 !!!next-token;
7156 next B;
7157 } elsif ($token->{tag_name} eq 'p') {
7158 ## has an element in scope
7159 my $i;
7160 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7161 my $node = $self->{open_elements}->[$_];
7162 if ($node->[1] & P_EL) {
7163 !!!cp ('t410.1');
7164 $i = $_;
7165 last INSCOPE;
7166 } elsif ($node->[1] & SCOPING_EL) {
7167 !!!cp ('t411.1');
7168 last INSCOPE;
7169 }
7170 } # INSCOPE
7171
7172 if (defined $i) {
7173 if ($self->{open_elements}->[-1]->[0]->manakai_local_name
7174 ne $token->{tag_name}) {
7175 !!!cp ('t412.1');
7176 !!!parse-error (type => 'not closed',
7177 value => $self->{open_elements}->[-1]->[0]
7178 ->manakai_local_name,
7179 token => $token);
7180 } else {
7181 !!!cp ('t414.1');
7182 }
7183
7184 splice @{$self->{open_elements}}, $i;
7185 } else {
7186 !!!cp ('t413.1');
7187 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
7188
7189 !!!cp ('t415.1');
7190 ## As if <p>, then reprocess the current token
7191 my $el;
7192 !!!create-element ($el, $HTML_NS, 'p',, $token);
7193 $insert->($el);
7194 ## NOTE: Not inserted into |$self->{open_elements}|.
7195 }
7196
7197 !!!next-token;
7198 next B;
7199 } elsif ({
7200 a => 1,
7201 b => 1, big => 1, em => 1, font => 1, i => 1,
7202 nobr => 1, s => 1, small => 1, strile => 1,
7203 strong => 1, tt => 1, u => 1,
7204 }->{$token->{tag_name}}) {
7205 !!!cp ('t427');
7206 $formatting_end_tag->($token);
7207 next B;
7208 } elsif ($token->{tag_name} eq 'br') {
7209 !!!cp ('t428');
7210 !!!parse-error (type => 'unmatched end tag:br', token => $token);
7211
7212 ## As if <br>
7213 $reconstruct_active_formatting_elements->($insert_to_current);
7214
7215 my $el;
7216 !!!create-element ($el, $HTML_NS, 'br',, $token);
7217 $insert->($el);
7218
7219 ## Ignore the token.
7220 !!!next-token;
7221 next B;
7222 } elsif ({
7223 caption => 1, col => 1, colgroup => 1, frame => 1,
7224 frameset => 1, head => 1, option => 1, optgroup => 1,
7225 tbody => 1, td => 1, tfoot => 1, th => 1,
7226 thead => 1, tr => 1,
7227 area => 1, basefont => 1, bgsound => 1,
7228 embed => 1, hr => 1, iframe => 1, image => 1,
7229 img => 1, input => 1, isindex => 1, noembed => 1,
7230 noframes => 1, param => 1, select => 1, spacer => 1,
7231 table => 1, textarea => 1, wbr => 1,
7232 noscript => 0, ## TODO: if scripting is enabled
7233 }->{$token->{tag_name}}) {
7234 !!!cp ('t429');
7235 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
7236 ## Ignore the token
7237 !!!next-token;
7238 next B;
7239
7240 ## ISSUE: Issue on HTML5 new elements in spec
7241
7242 } else {
7243 ## Step 1
7244 my $node_i = -1;
7245 my $node = $self->{open_elements}->[$node_i];
7246
7247 ## Step 2
7248 S2: {
7249 if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
7250 ## Step 1
7251 ## generate implied end tags
7252 while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
7253 !!!cp ('t430');
7254 ## NOTE: |<ruby><rt></ruby>|.
7255 ## ISSUE: <ruby><rt></rt> will also take this code path,
7256 ## which seems wrong.
7257 pop @{$self->{open_elements}};
7258 $node_i++;
7259 }
7260
7261 ## Step 2
7262 if ($self->{open_elements}->[-1]->[0]->manakai_local_name
7263 ne $token->{tag_name}) {
7264 !!!cp ('t431');
7265 ## NOTE: <x><y></x>
7266 !!!parse-error (type => 'not closed',
7267 value => $self->{open_elements}->[-1]->[0]
7268 ->manakai_local_name,
7269 token => $token);
7270 } else {
7271 !!!cp ('t432');
7272 }
7273
7274 ## Step 3
7275 splice @{$self->{open_elements}}, $node_i if $node_i < 0;
7276
7277 !!!next-token;
7278 last S2;
7279 } else {
7280 ## Step 3
7281 if (not ($node->[1] & FORMATTING_EL) and
7282 #not $phrasing_category->{$node->[1]} and
7283 ($node->[1] & SPECIAL_EL or
7284 $node->[1] & SCOPING_EL)) {
7285 !!!cp ('t433');
7286 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
7287 ## Ignore the token
7288 !!!next-token;
7289 last S2;
7290 }
7291
7292 !!!cp ('t434');
7293 }
7294
7295 ## Step 4
7296 $node_i--;
7297 $node = $self->{open_elements}->[$node_i];
7298
7299 ## Step 5;
7300 redo S2;
7301 } # S2
7302 next B;
7303 }
7304 }
7305 next B;
7306 } continue { # B
7307 if ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM) {
7308 ## NOTE: The code below is executed in cases where it does not have
7309 ## to be, but it it is harmless even in those cases.
7310 ## has an element in scope
7311 INSCOPE: {
7312 for (reverse 0..$#{$self->{open_elements}}) {
7313 my $node = $self->{open_elements}->[$_];
7314 if ($node->[1] & FOREIGN_EL) {
7315 last INSCOPE;
7316 } elsif ($node->[1] & SCOPING_EL) {
7317 last;
7318 }
7319 }
7320
7321 ## NOTE: No foreign element in scope.
7322 $self->{insertion_mode} &= ~ IN_FOREIGN_CONTENT_IM;
7323 } # INSCOPE
7324 }
7325 } # B
7326
7327 ## Stop parsing # MUST
7328
7329 ## TODO: script stuffs
7330 } # _tree_construct_main
7331
7332 sub set_inner_html ($$$) {
7333 my $class = shift;
7334 my $node = shift;
7335 my $s = \$_[0];
7336 my $onerror = $_[1];
7337
7338 ## ISSUE: Should {confident} be true?
7339
7340 my $nt = $node->node_type;
7341 if ($nt == 9) {
7342 # MUST
7343
7344 ## Step 1 # MUST
7345 ## TODO: If the document has an active parser, ...
7346 ## ISSUE: There is an issue in the spec.
7347
7348 ## Step 2 # MUST
7349 my @cn = @{$node->child_nodes};
7350 for (@cn) {
7351 $node->remove_child ($_);
7352 }
7353
7354 ## Step 3, 4, 5 # MUST
7355 $class->parse_string ($$s => $node, $onerror);
7356 } elsif ($nt == 1) {
7357 ## TODO: If non-html element
7358
7359 ## NOTE: Most of this code is copied from |parse_string|
7360
7361 ## Step 1 # MUST
7362 my $this_doc = $node->owner_document;
7363 my $doc = $this_doc->implementation->create_document;
7364 $doc->manakai_is_html (1);
7365 my $p = $class->new;
7366 $p->{document} = $doc;
7367
7368 ## Step 8 # MUST
7369 my $i = 0;
7370 $p->{line_prev} = $p->{line} = 1;
7371 $p->{column_prev} = $p->{column} = 0;
7372 $p->{set_next_char} = sub {
7373 my $self = shift;
7374
7375 pop @{$self->{prev_char}};
7376 unshift @{$self->{prev_char}}, $self->{next_char};
7377
7378 $self->{next_char} = -1 and return if $i >= length $$s;
7379 $self->{next_char} = ord substr $$s, $i++, 1;
7380
7381 ($p->{line_prev}, $p->{column_prev}) = ($p->{line}, $p->{column});
7382 $p->{column}++;
7383
7384 if ($self->{next_char} == 0x000A) { # LF
7385 $p->{line}++;
7386 $p->{column} = 0;
7387 !!!cp ('i1');
7388 } elsif ($self->{next_char} == 0x000D) { # CR
7389 $i++ if substr ($$s, $i, 1) eq "\x0A";
7390 $self->{next_char} = 0x000A; # LF # MUST
7391 $p->{line}++;
7392 $p->{column} = 0;
7393 !!!cp ('i2');
7394 } elsif ($self->{next_char} > 0x10FFFF) {
7395 $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
7396 !!!cp ('i3');
7397 } elsif ($self->{next_char} == 0x0000) { # NULL
7398 !!!cp ('i4');
7399 !!!parse-error (type => 'NULL');
7400 $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
7401 } elsif ($self->{next_char} <= 0x0008 or
7402 (0x000E <= $self->{next_char} and
7403 $self->{next_char} <= 0x001F) or
7404 (0x007F <= $self->{next_char} and
7405 $self->{next_char} <= 0x009F) or
7406 (0xD800 <= $self->{next_char} and
7407 $self->{next_char} <= 0xDFFF) or
7408 (0xFDD0 <= $self->{next_char} and
7409 $self->{next_char} <= 0xFDDF) or
7410 {
7411 0xFFFE => 1, 0xFFFF => 1, 0x1FFFE => 1, 0x1FFFF => 1,
7412 0x2FFFE => 1, 0x2FFFF => 1, 0x3FFFE => 1, 0x3FFFF => 1,
7413 0x4FFFE => 1, 0x4FFFF => 1, 0x5FFFE => 1, 0x5FFFF => 1,
7414 0x6FFFE => 1, 0x6FFFF => 1, 0x7FFFE => 1, 0x7FFFF => 1,
7415 0x8FFFE => 1, 0x8FFFF => 1, 0x9FFFE => 1, 0x9FFFF => 1,
7416 0xAFFFE => 1, 0xAFFFF => 1, 0xBFFFE => 1, 0xBFFFF => 1,
7417 0xCFFFE => 1, 0xCFFFF => 1, 0xDFFFE => 1, 0xDFFFF => 1,
7418 0xEFFFE => 1, 0xEFFFF => 1, 0xFFFFE => 1, 0xFFFFF => 1,
7419 0x10FFFE => 1, 0x10FFFF => 1,
7420 }->{$self->{next_char}}) {
7421 !!!cp ('i4.1');
7422 !!!parse-error (type => 'control char', level => $self->{must_level});
7423 ## TODO: error type documentation
7424 }
7425 };
7426 $p->{prev_char} = [-1, -1, -1];
7427 $p->{next_char} = -1;
7428
7429 my $ponerror = $onerror || sub {
7430 my (%opt) = @_;
7431 my $line = $opt{line};
7432 my $column = $opt{column};
7433 if (defined $opt{token} and defined $opt{token}->{line}) {
7434 $line = $opt{token}->{line};
7435 $column = $opt{token}->{column};
7436 }
7437 warn "Parse error ($opt{type}) at line $line column $column\n";
7438 };
7439 $p->{parse_error} = sub {
7440 $ponerror->(line => $p->{line}, column => $p->{column}, @_);
7441 };
7442
7443 $p->_initialize_tokenizer;
7444 $p->_initialize_tree_constructor;
7445
7446 ## Step 2
7447 my $node_ln = $node->manakai_local_name;
7448 $p->{content_model} = {
7449 title => RCDATA_CONTENT_MODEL,
7450 textarea => RCDATA_CONTENT_MODEL,
7451 style => CDATA_CONTENT_MODEL,
7452 script => CDATA_CONTENT_MODEL,
7453 xmp => CDATA_CONTENT_MODEL,
7454 iframe => CDATA_CONTENT_MODEL,
7455 noembed => CDATA_CONTENT_MODEL,
7456 noframes => CDATA_CONTENT_MODEL,
7457 noscript => CDATA_CONTENT_MODEL,
7458 plaintext => PLAINTEXT_CONTENT_MODEL,
7459 }->{$node_ln};
7460 $p->{content_model} = PCDATA_CONTENT_MODEL
7461 unless defined $p->{content_model};
7462 ## ISSUE: What is "the name of the element"? local name?
7463
7464 $p->{inner_html_node} = [$node, $el_category->{$node_ln}];
7465 ## TODO: Foreign element OK?
7466
7467 ## Step 3
7468 my $root = $doc->create_element_ns
7469 ('http://www.w3.org/1999/xhtml', [undef, 'html']);
7470
7471 ## Step 4 # MUST
7472 $doc->append_child ($root);
7473
7474 ## Step 5 # MUST
7475 push @{$p->{open_elements}}, [$root, $el_category->{html}];
7476
7477 undef $p->{head_element};
7478
7479 ## Step 6 # MUST
7480 $p->_reset_insertion_mode;
7481
7482 ## Step 7 # MUST
7483 my $anode = $node;
7484 AN: while (defined $anode) {
7485 if ($anode->node_type == 1) {
7486 my $nsuri = $anode->namespace_uri;
7487 if (defined $nsuri and $nsuri eq 'http://www.w3.org/1999/xhtml') {
7488 if ($anode->manakai_local_name eq 'form') {
7489 !!!cp ('i5');
7490 $p->{form_element} = $anode;
7491 last AN;
7492 }
7493 }
7494 }
7495 $anode = $anode->parent_node;
7496 } # AN
7497
7498 ## Step 9 # MUST
7499 {
7500 my $self = $p;
7501 !!!next-token;
7502 }
7503 $p->_tree_construction_main;
7504
7505 ## Step 10 # MUST
7506 my @cn = @{$node->child_nodes};
7507 for (@cn) {
7508 $node->remove_child ($_);
7509 }
7510 ## ISSUE: mutation events? read-only?
7511
7512 ## Step 11 # MUST
7513 @cn = @{$root->child_nodes};
7514 for (@cn) {
7515 $this_doc->adopt_node ($_);
7516 $node->append_child ($_);
7517 }
7518 ## ISSUE: mutation events?
7519
7520 $p->_terminate_tree_constructor;
7521
7522 delete $p->{parse_error}; # delete loop
7523 } else {
7524 die "$0: |set_inner_html| is not defined for node of type $nt";
7525 }
7526 } # set_inner_html
7527
7528 } # tree construction stage
7529
7530 package Whatpm::HTML::RestartParser;
7531 push our @ISA, 'Error';
7532
7533 1;
7534 # $Date: 2008/06/01 06:47:08 $

[email protected]
ViewVC Help
Powered by ViewVC 1.1.24