/[suikacvs]/markup/html/whatpm/Whatpm/HTML.pm.src
Suika

Contents of /markup/html/whatpm/Whatpm/HTML.pm.src

Parent Directory Parent Directory | Revision Log Revision Log


Revision 1.143 - (show annotations) (download) (as text)
Sat May 24 10:48:57 2008 UTC (18 years, 4 months ago) by wakaba
Branch: MAIN
Changes since 1.142: +76 -81 lines
File MIME type: application/x-wais-source
++ whatpm/Whatpm/ChangeLog	24 May 2008 10:48:45 -0000
	* HTML.pm.src: Ignore language part of public identifiers for
	quriks mode detection (HTML5 revision 1679).

2008-05-24  Wakaba  <wakaba@suika.fam.cx>

1 package Whatpm::HTML;
2 use strict;
3 our $VERSION=do{my @r=(q$Revision: 1.142 $=~/\d+/g);sprintf "%d."."%02d" x $#r,@r};
4 use Error qw(:try);
5
6 ## ISSUE:
7 ## var doc = implementation.createDocument (null, null, null);
8 ## doc.write ('');
9 ## alert (doc.compatMode);
10
11 ## TODO: 1252 parse error (revision 1264)
12 ## TODO: 8859-11 = 874 (revision 1271)
13
14 require IO::Handle;
15
16 my $HTML_NS = q<http://www.w3.org/1999/xhtml>;
17 my $MML_NS = q<http://www.w3.org/1998/Math/MathML>;
18 my $SVG_NS = q<http://www.w3.org/2000/svg>;
19 my $XLINK_NS = q<http://www.w3.org/1999/xlink>;
20 my $XML_NS = q<http://www.w3.org/XML/1998/namespace>;
21 my $XMLNS_NS = q<http://www.w3.org/2000/xmlns/>;
22
23 sub A_EL () { 0b1 }
24 sub ADDRESS_EL () { 0b10 }
25 sub BODY_EL () { 0b100 }
26 sub BUTTON_EL () { 0b1000 }
27 sub CAPTION_EL () { 0b10000 }
28 sub DD_EL () { 0b100000 }
29 sub DIV_EL () { 0b1000000 }
30 sub DT_EL () { 0b10000000 }
31 sub FORM_EL () { 0b100000000 }
32 sub FORMATTING_EL () { 0b1000000000 }
33 sub FRAMESET_EL () { 0b10000000000 }
34 sub HEADING_EL () { 0b100000000000 }
35 sub HTML_EL () { 0b1000000000000 }
36 sub LI_EL () { 0b10000000000000 }
37 sub NOBR_EL () { 0b100000000000000 }
38 sub OPTION_EL () { 0b1000000000000000 }
39 sub OPTGROUP_EL () { 0b10000000000000000 }
40 sub P_EL () { 0b100000000000000000 }
41 sub SELECT_EL () { 0b1000000000000000000 }
42 sub TABLE_EL () { 0b10000000000000000000 }
43 sub TABLE_CELL_EL () { 0b100000000000000000000 }
44 sub TABLE_ROW_EL () { 0b1000000000000000000000 }
45 sub TABLE_ROW_GROUP_EL () { 0b10000000000000000000000 }
46 sub MISC_SCOPING_EL () { 0b100000000000000000000000 }
47 sub MISC_SPECIAL_EL () { 0b1000000000000000000000000 }
48 sub FOREIGN_EL () { 0b10000000000000000000000000 }
49 sub FOREIGN_FLOW_CONTENT_EL () { 0b100000000000000000000000000 }
50 sub MML_AXML_EL () { 0b1000000000000000000000000000 }
51
52 sub TABLE_ROWS_EL () {
53 TABLE_EL |
54 TABLE_ROW_EL |
55 TABLE_ROW_GROUP_EL
56 }
57
58 sub END_TAG_OPTIONAL_EL () {
59 DD_EL |
60 DT_EL |
61 LI_EL |
62 P_EL
63 }
64
65 sub ALL_END_TAG_OPTIONAL_EL () {
66 END_TAG_OPTIONAL_EL |
67 BODY_EL |
68 HTML_EL |
69 TABLE_CELL_EL |
70 TABLE_ROW_EL |
71 TABLE_ROW_GROUP_EL
72 }
73
74 sub SCOPING_EL () {
75 BUTTON_EL |
76 CAPTION_EL |
77 HTML_EL |
78 TABLE_EL |
79 TABLE_CELL_EL |
80 MISC_SCOPING_EL
81 }
82
83 sub TABLE_SCOPING_EL () {
84 HTML_EL |
85 TABLE_EL
86 }
87
88 sub TABLE_ROWS_SCOPING_EL () {
89 HTML_EL |
90 TABLE_ROW_GROUP_EL
91 }
92
93 sub TABLE_ROW_SCOPING_EL () {
94 HTML_EL |
95 TABLE_ROW_EL
96 }
97
98 sub SPECIAL_EL () {
99 ADDRESS_EL |
100 BODY_EL |
101 DIV_EL |
102 END_TAG_OPTIONAL_EL |
103 FORM_EL |
104 FRAMESET_EL |
105 HEADING_EL |
106 OPTION_EL |
107 OPTGROUP_EL |
108 SELECT_EL |
109 TABLE_ROW_EL |
110 TABLE_ROW_GROUP_EL |
111 MISC_SPECIAL_EL
112 }
113
114 my $el_category = {
115 a => A_EL | FORMATTING_EL,
116 address => ADDRESS_EL,
117 applet => MISC_SCOPING_EL,
118 area => MISC_SPECIAL_EL,
119 b => FORMATTING_EL,
120 base => MISC_SPECIAL_EL,
121 basefont => MISC_SPECIAL_EL,
122 bgsound => MISC_SPECIAL_EL,
123 big => FORMATTING_EL,
124 blockquote => MISC_SPECIAL_EL,
125 body => BODY_EL,
126 br => MISC_SPECIAL_EL,
127 button => BUTTON_EL,
128 caption => CAPTION_EL,
129 center => MISC_SPECIAL_EL,
130 col => MISC_SPECIAL_EL,
131 colgroup => MISC_SPECIAL_EL,
132 dd => DD_EL,
133 dir => MISC_SPECIAL_EL,
134 div => DIV_EL,
135 dl => MISC_SPECIAL_EL,
136 dt => DT_EL,
137 em => FORMATTING_EL,
138 embed => MISC_SPECIAL_EL,
139 fieldset => MISC_SPECIAL_EL,
140 font => FORMATTING_EL,
141 form => FORM_EL,
142 frame => MISC_SPECIAL_EL,
143 frameset => FRAMESET_EL,
144 h1 => HEADING_EL,
145 h2 => HEADING_EL,
146 h3 => HEADING_EL,
147 h4 => HEADING_EL,
148 h5 => HEADING_EL,
149 h6 => HEADING_EL,
150 head => MISC_SPECIAL_EL,
151 hr => MISC_SPECIAL_EL,
152 html => HTML_EL,
153 i => FORMATTING_EL,
154 iframe => MISC_SPECIAL_EL,
155 img => MISC_SPECIAL_EL,
156 input => MISC_SPECIAL_EL,
157 isindex => MISC_SPECIAL_EL,
158 li => LI_EL,
159 link => MISC_SPECIAL_EL,
160 listing => MISC_SPECIAL_EL,
161 marquee => MISC_SCOPING_EL,
162 menu => MISC_SPECIAL_EL,
163 meta => MISC_SPECIAL_EL,
164 nobr => NOBR_EL | FORMATTING_EL,
165 noembed => MISC_SPECIAL_EL,
166 noframes => MISC_SPECIAL_EL,
167 noscript => MISC_SPECIAL_EL,
168 object => MISC_SCOPING_EL,
169 ol => MISC_SPECIAL_EL,
170 optgroup => OPTGROUP_EL,
171 option => OPTION_EL,
172 p => P_EL,
173 param => MISC_SPECIAL_EL,
174 plaintext => MISC_SPECIAL_EL,
175 pre => MISC_SPECIAL_EL,
176 s => FORMATTING_EL,
177 script => MISC_SPECIAL_EL,
178 select => SELECT_EL,
179 small => FORMATTING_EL,
180 spacer => MISC_SPECIAL_EL,
181 strike => FORMATTING_EL,
182 strong => FORMATTING_EL,
183 style => MISC_SPECIAL_EL,
184 table => TABLE_EL,
185 tbody => TABLE_ROW_GROUP_EL,
186 td => TABLE_CELL_EL,
187 textarea => MISC_SPECIAL_EL,
188 tfoot => TABLE_ROW_GROUP_EL,
189 th => TABLE_CELL_EL,
190 thead => TABLE_ROW_GROUP_EL,
191 title => MISC_SPECIAL_EL,
192 tr => TABLE_ROW_EL,
193 tt => FORMATTING_EL,
194 u => FORMATTING_EL,
195 ul => MISC_SPECIAL_EL,
196 wbr => MISC_SPECIAL_EL,
197 };
198
199 my $el_category_f = {
200 $MML_NS => {
201 'annotation-xml' => MML_AXML_EL,
202 mi => FOREIGN_FLOW_CONTENT_EL,
203 mo => FOREIGN_FLOW_CONTENT_EL,
204 mn => FOREIGN_FLOW_CONTENT_EL,
205 ms => FOREIGN_FLOW_CONTENT_EL,
206 mtext => FOREIGN_FLOW_CONTENT_EL,
207 },
208 $SVG_NS => {
209 foreignObject => FOREIGN_FLOW_CONTENT_EL,
210 desc => FOREIGN_FLOW_CONTENT_EL,
211 title => FOREIGN_FLOW_CONTENT_EL,
212 },
213 ## NOTE: In addition, FOREIGN_EL is set to non-HTML elements.
214 };
215
216 my $svg_attr_name = {
217 attributetype => 'attributeType',
218 basefrequency => 'baseFrequency',
219 baseprofile => 'baseProfile',
220 calcmode => 'calcMode',
221 clippathunits => 'clipPathUnits',
222 contentscripttype => 'contentScriptType',
223 contentstyletype => 'contentStyleType',
224 diffuseconstant => 'diffuseConstant',
225 edgemode => 'edgeMode',
226 externalresourcesrequired => 'externalResourcesRequired',
227 fecolormatrix => 'feColorMatrix',
228 fecomposite => 'feComposite',
229 fegaussianblur => 'feGaussianBlur',
230 femorphology => 'feMorphology',
231 fetile => 'feTile',
232 filterres => 'filterRes',
233 filterunits => 'filterUnits',
234 glyphref => 'glyphRef',
235 gradienttransform => 'gradientTransform',
236 gradientunits => 'gradientUnits',
237 kernelmatrix => 'kernelMatrix',
238 kernelunitlength => 'kernelUnitLength',
239 keypoints => 'keyPoints',
240 keysplines => 'keySplines',
241 keytimes => 'keyTimes',
242 lengthadjust => 'lengthAdjust',
243 limitingconeangle => 'limitingConeAngle',
244 markerheight => 'markerHeight',
245 markerunits => 'markerUnits',
246 markerwidth => 'markerWidth',
247 maskcontentunits => 'maskContentUnits',
248 maskunits => 'maskUnits',
249 numoctaves => 'numOctaves',
250 pathlength => 'pathLength',
251 patterncontentunits => 'patternContentUnits',
252 patterntransform => 'patternTransform',
253 patternunits => 'patternUnits',
254 pointsatx => 'pointsAtX',
255 pointsaty => 'pointsAtY',
256 pointsatz => 'pointsAtZ',
257 preservealpha => 'preserveAlpha',
258 preserveaspectratio => 'preserveAspectRatio',
259 primitiveunits => 'primitiveUnits',
260 refx => 'refX',
261 refy => 'refY',
262 repeatcount => 'repeatCount',
263 repeatdur => 'repeatDur',
264 requiredextensions => 'requiredExtensions',
265 specularconstant => 'specularConstant',
266 specularexponent => 'specularExponent',
267 spreadmethod => 'spreadMethod',
268 startoffset => 'startOffset',
269 stddeviation => 'stdDeviation',
270 stitchtiles => 'stitchTiles',
271 surfacescale => 'surfaceScale',
272 systemlanguage => 'systemLanguage',
273 tablevalues => 'tableValues',
274 targetx => 'targetX',
275 targety => 'targetY',
276 textlength => 'textLength',
277 viewbox => 'viewBox',
278 viewtarget => 'viewTarget',
279 xchannelselector => 'xChannelSelector',
280 ychannelselector => 'yChannelSelector',
281 zoomandpan => 'zoomAndPan',
282 };
283
284 my $foreign_attr_xname = {
285 'xlink:actuate' => [$XLINK_NS, ['xlink', 'actuate']],
286 'xlink:arcrole' => [$XLINK_NS, ['xlink', 'arcrole']],
287 'xlink:href' => [$XLINK_NS, ['xlink', 'href']],
288 'xlink:role' => [$XLINK_NS, ['xlink', 'role']],
289 'xlink:show' => [$XLINK_NS, ['xlink', 'show']],
290 'xlink:title' => [$XLINK_NS, ['xlink', 'title']],
291 'xlink:type' => [$XLINK_NS, ['xlink', 'type']],
292 'xml:base' => [$XML_NS, ['xml', 'base']],
293 'xml:lang' => [$XML_NS, ['xml', 'lang']],
294 'xml:space' => [$XML_NS, ['xml', 'space']],
295 'xmlns' => [$XMLNS_NS, [undef, 'xmlns']],
296 'xmlns:xlink' => [$XMLNS_NS, ['xmlns', 'xlink']],
297 };
298
299 ## ISSUE: xmlns:xlink="non-xlink-ns" is not an error.
300
301 my $c1_entity_char = {
302 0x80 => 0x20AC,
303 0x81 => 0xFFFD,
304 0x82 => 0x201A,
305 0x83 => 0x0192,
306 0x84 => 0x201E,
307 0x85 => 0x2026,
308 0x86 => 0x2020,
309 0x87 => 0x2021,
310 0x88 => 0x02C6,
311 0x89 => 0x2030,
312 0x8A => 0x0160,
313 0x8B => 0x2039,
314 0x8C => 0x0152,
315 0x8D => 0xFFFD,
316 0x8E => 0x017D,
317 0x8F => 0xFFFD,
318 0x90 => 0xFFFD,
319 0x91 => 0x2018,
320 0x92 => 0x2019,
321 0x93 => 0x201C,
322 0x94 => 0x201D,
323 0x95 => 0x2022,
324 0x96 => 0x2013,
325 0x97 => 0x2014,
326 0x98 => 0x02DC,
327 0x99 => 0x2122,
328 0x9A => 0x0161,
329 0x9B => 0x203A,
330 0x9C => 0x0153,
331 0x9D => 0xFFFD,
332 0x9E => 0x017E,
333 0x9F => 0x0178,
334 }; # $c1_entity_char
335
336 sub parse_byte_string ($$$$;$) {
337 my $self = shift;
338 my $charset_name = shift;
339 open my $input, '<', ref $_[0] ? $_[0] : \($_[0]);
340 return $self->parse_byte_stream ($charset_name, $input, @_[1..$#_]);
341 } # parse_byte_string
342
343 sub parse_byte_stream ($$$$;$) {
344 my $self = ref $_[0] ? shift : shift->new;
345 my $charset_name = shift;
346 my $byte_stream = $_[0];
347
348 my $onerror = $_[2] || sub {
349 my (%opt) = @_;
350 warn "Parse error ($opt{type})\n";
351 };
352 $self->{parse_error} = $onerror; # updated later by parse_char_string
353
354 ## HTML5 encoding sniffing algorithm
355 require Message::Charset::Info;
356 my $charset;
357 my $buffer;
358 my ($char_stream, $e_status);
359
360 SNIFFING: {
361
362 ## Step 1
363 if (defined $charset_name) {
364 $charset = Message::Charset::Info->get_by_iana_name ($charset_name);
365
366 ## ISSUE: Unsupported encoding is not ignored according to the spec.
367 ($char_stream, $e_status) = $charset->get_decode_handle
368 ($byte_stream, allow_error_reporting => 1,
369 allow_fallback => 1);
370 if ($char_stream) {
371 $self->{confident} = 1;
372 last SNIFFING;
373 } else {
374 ## TODO: unsupported error
375 }
376 }
377
378 ## Step 2
379 my $byte_buffer = '';
380 for (1..1024) {
381 my $char = $byte_stream->getc;
382 last unless defined $char;
383 $byte_buffer .= $char;
384 } ## TODO: timeout
385
386 ## Step 3
387 if ($byte_buffer =~ /^\xFE\xFF/) {
388 $charset = Message::Charset::Info->get_by_iana_name ('utf-16be');
389 ($char_stream, $e_status) = $charset->get_decode_handle
390 ($byte_stream, allow_error_reporting => 1,
391 allow_fallback => 1, byte_buffer => \$byte_buffer);
392 $self->{confident} = 1;
393 last SNIFFING;
394 } elsif ($byte_buffer =~ /^\xFF\xFE/) {
395 $charset = Message::Charset::Info->get_by_iana_name ('utf-16le');
396 ($char_stream, $e_status) = $charset->get_decode_handle
397 ($byte_stream, allow_error_reporting => 1,
398 allow_fallback => 1, byte_buffer => \$byte_buffer);
399 $self->{confident} = 1;
400 last SNIFFING;
401 } elsif ($byte_buffer =~ /^\xEF\xBB\xBF/) {
402 $charset = Message::Charset::Info->get_by_iana_name ('utf-8');
403 ($char_stream, $e_status) = $charset->get_decode_handle
404 ($byte_stream, allow_error_reporting => 1,
405 allow_fallback => 1, byte_buffer => \$byte_buffer);
406 $self->{confident} = 1;
407 last SNIFFING;
408 }
409
410 ## Step 4
411 ## TODO: <meta charset>
412
413 ## Step 5
414 ## TODO: from history
415
416 ## Step 6
417 require Whatpm::Charset::UniversalCharDet;
418 $charset_name = Whatpm::Charset::UniversalCharDet->detect_byte_string
419 ($byte_buffer);
420 if (defined $charset_name) {
421 $charset = Message::Charset::Info->get_by_iana_name ($charset_name);
422
423 ## ISSUE: Unsupported encoding is not ignored according to the spec.
424 require Whatpm::Charset::DecodeHandle;
425 $buffer = Whatpm::Charset::DecodeHandle::ByteBuffer->new
426 ($byte_stream);
427 ($char_stream, $e_status) = $charset->get_decode_handle
428 ($buffer, allow_error_reporting => 1,
429 allow_fallback => 1, byte_buffer => \$byte_buffer);
430 if ($char_stream) {
431 $buffer->{buffer} = $byte_buffer;
432 !!!parse-error (type => 'sniffing:chardet', ## TODO: type name
433 value => $charset_name,
434 level => $self->{info_level},
435 line => 1, column => 1);
436 $self->{confident} = 0;
437 last SNIFFING;
438 }
439 }
440
441 ## Step 7: default
442 ## TODO: Make this configurable.
443 $charset = Message::Charset::Info->get_by_iana_name ('windows-1252');
444 ## NOTE: We choose |windows-1252| here, since |utf-8| should be
445 ## detectable in the step 6.
446 require Whatpm::Charset::DecodeHandle;
447 $buffer = Whatpm::Charset::DecodeHandle::ByteBuffer->new
448 ($byte_stream);
449 ($char_stream, $e_status)
450 = $charset->get_decode_handle ($buffer,
451 allow_error_reporting => 1,
452 allow_fallback => 1,
453 byte_buffer => \$byte_buffer);
454 $buffer->{buffer} = $byte_buffer;
455 !!!parse-error (type => 'sniffing:default', ## TODO: type name
456 value => 'windows-1252',
457 level => $self->{info_level},
458 line => 1, column => 1);
459 $self->{confident} = 0;
460 } # SNIFFING
461
462 $self->{input_encoding} = $charset->get_iana_name;
463 if ($e_status & Message::Charset::Info::FALLBACK_ENCODING_IMPL ()) {
464 !!!parse-error (type => 'chardecode:fallback', ## TODO: type name
465 value => $self->{input_encoding},
466 level => $self->{unsupported_level},
467 line => 1, column => 1);
468 } elsif (not ($e_status &
469 Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL())) {
470 !!!parse-error (type => 'chardecode:no error', ## TODO: type name
471 value => $self->{input_encoding},
472 level => $self->{unsupported_level},
473 line => 1, column => 1);
474 }
475
476 $self->{change_encoding} = sub {
477 my $self = shift;
478 $charset_name = shift;
479 my $token = shift;
480
481 $charset = Message::Charset::Info->get_by_iana_name ($charset_name);
482 ($char_stream, $e_status) = $charset->get_decode_handle
483 ($byte_stream, allow_error_reporting => 1, allow_fallback => 1,
484 byte_buffer => \ $buffer->{buffer});
485
486 if ($char_stream) { # if supported
487 ## "Change the encoding" algorithm:
488
489 ## Step 1
490 if ($charset->{iana_names}->{'utf-16'}) { ## ISSUE: UTF-16BE -> UTF-8? UTF-16LE -> UTF-8?
491 $charset = Message::Charset::Info->get_by_iana_name ('utf-8');
492 ($char_stream, $e_status) = $charset->get_decode_handle
493 ($byte_stream,
494 byte_buffer => \ $buffer->{buffer});
495 }
496 $charset_name = $charset->get_iana_name;
497
498 ## Step 2
499 if (defined $self->{input_encoding} and
500 $self->{input_encoding} eq $charset_name) {
501 !!!parse-error (type => 'charset label:matching', ## TODO: type
502 value => $charset_name,
503 level => $self->{info_level});
504 $self->{confident} = 1;
505 return;
506 }
507
508 !!!parse-error (type => 'charset label detected:'.$self->{input_encoding}.
509 ':'.$charset_name, level => 'w', token => $token);
510
511 ## Step 3
512 # if (can) {
513 ## change the encoding on the fly.
514 #$self->{confident} = 1;
515 #return;
516 # }
517
518 ## Step 4
519 throw Whatpm::HTML::RestartParser ();
520 }
521 }; # $self->{change_encoding}
522
523 my $char_onerror = sub {
524 my (undef, $type, %opt) = @_;
525 !!!parse-error (%opt, type => $type,
526 line => $self->{line}, column => $self->{column} + 1);
527 if ($opt{octets}) {
528 ${$opt{octets}} = "\x{FFFD}"; # relacement character
529 }
530 };
531 $char_stream->onerror ($char_onerror);
532
533 my @args = @_; shift @args; # $s
534 my $return;
535 try {
536 $return = $self->parse_char_stream ($char_stream, @args);
537 } catch Whatpm::HTML::RestartParser with {
538 ## NOTE: Invoked after {change_encoding}.
539
540 $self->{input_encoding} = $charset->get_iana_name;
541 if ($e_status & Message::Charset::Info::FALLBACK_ENCODING_IMPL ()) {
542 !!!parse-error (type => 'chardecode:fallback', ## TODO: type name
543 value => $self->{input_encoding},
544 level => $self->{unsupported_level},
545 line => 1, column => 1);
546 } elsif (not ($e_status &
547 Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL())) {
548 !!!parse-error (type => 'chardecode:no error', ## TODO: type name
549 value => $self->{input_encoding},
550 level => $self->{unsupported_level},
551 line => 1, column => 1);
552 }
553 $self->{confident} = 1;
554 $char_stream->onerror ($char_onerror);
555 $return = $self->parse_char_stream ($char_stream, @args);
556 };
557 return $return;
558 } # parse_byte_stream
559
560 ## NOTE: HTML5 spec says that the encoding layer MUST NOT strip BOM
561 ## and the HTML layer MUST ignore it. However, we does strip BOM in
562 ## the encoding layer and the HTML layer does not ignore any U+FEFF,
563 ## because the core part of our HTML parser expects a string of character,
564 ## not a string of bytes or code units or anything which might contain a BOM.
565 ## Therefore, any parser interface that accepts a string of bytes,
566 ## such as |parse_byte_string| in this module, must ensure that it does
567 ## strip the BOM and never strip any ZWNBSP.
568
569 sub parse_char_string ($$$;$) {
570 my $self = shift;
571 require utf8;
572 my $s = ref $_[0] ? $_[0] : \($_[0]);
573 open my $input, '<' . (utf8::is_utf8 ($$s) ? ':utf8' : ''), $s;
574 return $self->parse_char_stream ($input, @_[1..$#_]);
575 } # parse_char_string
576 *parse_string = \&parse_char_string;
577
578 sub parse_char_stream ($$$;$) {
579 my $self = ref $_[0] ? shift : shift->new;
580 my $input = $_[0];
581 $self->{document} = $_[1];
582 @{$self->{document}->child_nodes} = ();
583
584 ## NOTE: |set_inner_html| copies most of this method's code
585
586 $self->{confident} = 1 unless exists $self->{confident};
587 $self->{document}->input_encoding ($self->{input_encoding})
588 if defined $self->{input_encoding};
589
590 my $i = 0;
591 $self->{line_prev} = $self->{line} = 1;
592 $self->{column_prev} = $self->{column} = 0;
593 $self->{set_next_char} = sub {
594 my $self = shift;
595
596 pop @{$self->{prev_char}};
597 unshift @{$self->{prev_char}}, $self->{next_char};
598
599 my $char;
600 if (defined $self->{next_next_char}) {
601 $char = $self->{next_next_char};
602 delete $self->{next_next_char};
603 } else {
604 $char = $input->getc;
605 }
606 $self->{next_char} = -1 and return unless defined $char;
607 $self->{next_char} = ord $char;
608
609 ($self->{line_prev}, $self->{column_prev})
610 = ($self->{line}, $self->{column});
611 $self->{column}++;
612
613 if ($self->{next_char} == 0x000A) { # LF
614 !!!cp ('j1');
615 $self->{line}++;
616 $self->{column} = 0;
617 } elsif ($self->{next_char} == 0x000D) { # CR
618 !!!cp ('j2');
619 my $next = $input->getc;
620 if (defined $next and $next ne "\x0A") {
621 $self->{next_next_char} = $next;
622 }
623 $self->{next_char} = 0x000A; # LF # MUST
624 $self->{line}++;
625 $self->{column} = 0;
626 } elsif ($self->{next_char} > 0x10FFFF) {
627 !!!cp ('j3');
628 $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
629 } elsif ($self->{next_char} == 0x0000) { # NULL
630 !!!cp ('j4');
631 !!!parse-error (type => 'NULL');
632 $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
633 } elsif ($self->{next_char} <= 0x0008 or
634 (0x000E <= $self->{next_char} and $self->{next_char} <= 0x001F) or
635 (0x007F <= $self->{next_char} and $self->{next_char} <= 0x009F) or
636 (0xD800 <= $self->{next_char} and $self->{next_char} <= 0xDFFF) or
637 (0xFDD0 <= $self->{next_char} and $self->{next_char} <= 0xFDDF) or
638 {
639 0xFFFE => 1, 0xFFFF => 1, 0x1FFFE => 1, 0x1FFFF => 1,
640 0x2FFFE => 1, 0x2FFFF => 1, 0x3FFFE => 1, 0x3FFFF => 1,
641 0x4FFFE => 1, 0x4FFFF => 1, 0x5FFFE => 1, 0x5FFFF => 1,
642 0x6FFFE => 1, 0x6FFFF => 1, 0x7FFFE => 1, 0x7FFFF => 1,
643 0x8FFFE => 1, 0x8FFFF => 1, 0x9FFFE => 1, 0x9FFFF => 1,
644 0xAFFFE => 1, 0xAFFFF => 1, 0xBFFFE => 1, 0xBFFFF => 1,
645 0xCFFFE => 1, 0xCFFFF => 1, 0xDFFFE => 1, 0xDFFFF => 1,
646 0xEFFFE => 1, 0xEFFFF => 1, 0xFFFFE => 1, 0xFFFFF => 1,
647 0x10FFFE => 1, 0x10FFFF => 1,
648 }->{$self->{next_char}}) {
649 !!!cp ('j5');
650 !!!parse-error (type => 'control char', level => $self->{must_level});
651 ## TODO: error type documentation
652 }
653 };
654 $self->{prev_char} = [-1, -1, -1];
655 $self->{next_char} = -1;
656
657 my $onerror = $_[2] || sub {
658 my (%opt) = @_;
659 my $line = $opt{token} ? $opt{token}->{line} : $opt{line};
660 my $column = $opt{token} ? $opt{token}->{column} : $opt{column};
661 warn "Parse error ($opt{type}) at line $line column $column\n";
662 };
663 $self->{parse_error} = sub {
664 $onerror->(line => $self->{line}, column => $self->{column}, @_);
665 };
666
667 $self->_initialize_tokenizer;
668 $self->_initialize_tree_constructor;
669 $self->_construct_tree;
670 $self->_terminate_tree_constructor;
671
672 delete $self->{parse_error}; # remove loop
673
674 return $self->{document};
675 } # parse_char_stream
676
677 sub new ($) {
678 my $class = shift;
679 my $self = bless {
680 must_level => 'm',
681 should_level => 's',
682 good_level => 'w',
683 warn_level => 'w',
684 info_level => 'i',
685 unsupported_level => 'u',
686 }, $class;
687 $self->{set_next_char} = sub {
688 $self->{next_char} = -1;
689 };
690 $self->{parse_error} = sub {
691 #
692 };
693 $self->{change_encoding} = sub {
694 # if ($_[0] is a supported encoding) {
695 # run "change the encoding" algorithm;
696 # throw Whatpm::HTML::RestartParser (charset => $new_encoding);
697 # }
698 };
699 $self->{application_cache_selection} = sub {
700 #
701 };
702 return $self;
703 } # new
704
705 sub CM_ENTITY () { 0b001 } # & markup in data
706 sub CM_LIMITED_MARKUP () { 0b010 } # < markup in data (limited)
707 sub CM_FULL_MARKUP () { 0b100 } # < markup in data (any)
708
709 sub PLAINTEXT_CONTENT_MODEL () { 0 }
710 sub CDATA_CONTENT_MODEL () { CM_LIMITED_MARKUP }
711 sub RCDATA_CONTENT_MODEL () { CM_ENTITY | CM_LIMITED_MARKUP }
712 sub PCDATA_CONTENT_MODEL () { CM_ENTITY | CM_FULL_MARKUP }
713
714 sub DATA_STATE () { 0 }
715 sub ENTITY_DATA_STATE () { 1 }
716 sub TAG_OPEN_STATE () { 2 }
717 sub CLOSE_TAG_OPEN_STATE () { 3 }
718 sub TAG_NAME_STATE () { 4 }
719 sub BEFORE_ATTRIBUTE_NAME_STATE () { 5 }
720 sub ATTRIBUTE_NAME_STATE () { 6 }
721 sub AFTER_ATTRIBUTE_NAME_STATE () { 7 }
722 sub BEFORE_ATTRIBUTE_VALUE_STATE () { 8 }
723 sub ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE () { 9 }
724 sub ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE () { 10 }
725 sub ATTRIBUTE_VALUE_UNQUOTED_STATE () { 11 }
726 sub ENTITY_IN_ATTRIBUTE_VALUE_STATE () { 12 }
727 sub MARKUP_DECLARATION_OPEN_STATE () { 13 }
728 sub COMMENT_START_STATE () { 14 }
729 sub COMMENT_START_DASH_STATE () { 15 }
730 sub COMMENT_STATE () { 16 }
731 sub COMMENT_END_STATE () { 17 }
732 sub COMMENT_END_DASH_STATE () { 18 }
733 sub BOGUS_COMMENT_STATE () { 19 }
734 sub DOCTYPE_STATE () { 20 }
735 sub BEFORE_DOCTYPE_NAME_STATE () { 21 }
736 sub DOCTYPE_NAME_STATE () { 22 }
737 sub AFTER_DOCTYPE_NAME_STATE () { 23 }
738 sub BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE () { 24 }
739 sub DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE () { 25 }
740 sub DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE () { 26 }
741 sub AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE () { 27 }
742 sub BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE () { 28 }
743 sub DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE () { 29 }
744 sub DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE () { 30 }
745 sub AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE () { 31 }
746 sub BOGUS_DOCTYPE_STATE () { 32 }
747 sub AFTER_ATTRIBUTE_VALUE_QUOTED_STATE () { 33 }
748 sub SELF_CLOSING_START_TAG_STATE () { 34 }
749 sub CDATA_BLOCK_STATE () { 35 }
750
751 sub DOCTYPE_TOKEN () { 1 }
752 sub COMMENT_TOKEN () { 2 }
753 sub START_TAG_TOKEN () { 3 }
754 sub END_TAG_TOKEN () { 4 }
755 sub END_OF_FILE_TOKEN () { 5 }
756 sub CHARACTER_TOKEN () { 6 }
757
758 sub AFTER_HTML_IMS () { 0b100 }
759 sub HEAD_IMS () { 0b1000 }
760 sub BODY_IMS () { 0b10000 }
761 sub BODY_TABLE_IMS () { 0b100000 }
762 sub TABLE_IMS () { 0b1000000 }
763 sub ROW_IMS () { 0b10000000 }
764 sub BODY_AFTER_IMS () { 0b100000000 }
765 sub FRAME_IMS () { 0b1000000000 }
766 sub SELECT_IMS () { 0b10000000000 }
767 sub IN_FOREIGN_CONTENT_IM () { 0b100000000000 }
768 ## NOTE: "in foreign content" insertion mode is special; it is combined
769 ## with the secondary insertion mode. In this parser, they are stored
770 ## together in the bit-or'ed form.
771
772 ## NOTE: "initial" and "before html" insertion modes have no constants.
773
774 ## NOTE: "after after body" insertion mode.
775 sub AFTER_HTML_BODY_IM () { AFTER_HTML_IMS | BODY_AFTER_IMS }
776
777 ## NOTE: "after after frameset" insertion mode.
778 sub AFTER_HTML_FRAMESET_IM () { AFTER_HTML_IMS | FRAME_IMS }
779
780 sub IN_HEAD_IM () { HEAD_IMS | 0b00 }
781 sub IN_HEAD_NOSCRIPT_IM () { HEAD_IMS | 0b01 }
782 sub AFTER_HEAD_IM () { HEAD_IMS | 0b10 }
783 sub BEFORE_HEAD_IM () { HEAD_IMS | 0b11 }
784 sub IN_BODY_IM () { BODY_IMS }
785 sub IN_CELL_IM () { BODY_IMS | BODY_TABLE_IMS | 0b01 }
786 sub IN_CAPTION_IM () { BODY_IMS | BODY_TABLE_IMS | 0b10 }
787 sub IN_ROW_IM () { TABLE_IMS | ROW_IMS | 0b01 }
788 sub IN_TABLE_BODY_IM () { TABLE_IMS | ROW_IMS | 0b10 }
789 sub IN_TABLE_IM () { TABLE_IMS }
790 sub AFTER_BODY_IM () { BODY_AFTER_IMS }
791 sub IN_FRAMESET_IM () { FRAME_IMS | 0b01 }
792 sub AFTER_FRAMESET_IM () { FRAME_IMS | 0b10 }
793 sub IN_SELECT_IM () { SELECT_IMS | 0b01 }
794 sub IN_SELECT_IN_TABLE_IM () { SELECT_IMS | 0b10 }
795 sub IN_COLUMN_GROUP_IM () { 0b10 }
796
797 ## Implementations MUST act as if state machine in the spec
798
799 sub _initialize_tokenizer ($) {
800 my $self = shift;
801 $self->{state} = DATA_STATE; # MUST
802 $self->{content_model} = PCDATA_CONTENT_MODEL; # be
803 undef $self->{current_token}; # start tag, end tag, comment, or DOCTYPE
804 undef $self->{current_attribute};
805 undef $self->{last_emitted_start_tag_name};
806 undef $self->{last_attribute_value_state};
807 delete $self->{self_closing};
808 $self->{char} = [];
809 # $self->{next_char}
810 !!!next-input-character;
811 $self->{token} = [];
812 # $self->{escape}
813 } # _initialize_tokenizer
814
815 ## A token has:
816 ## ->{type} == DOCTYPE_TOKEN, START_TAG_TOKEN, END_TAG_TOKEN, COMMENT_TOKEN,
817 ## CHARACTER_TOKEN, or END_OF_FILE_TOKEN
818 ## ->{name} (DOCTYPE_TOKEN)
819 ## ->{tag_name} (START_TAG_TOKEN, END_TAG_TOKEN)
820 ## ->{public_identifier} (DOCTYPE_TOKEN)
821 ## ->{system_identifier} (DOCTYPE_TOKEN)
822 ## ->{quirks} == 1 or 0 (DOCTYPE_TOKEN): "force-quirks" flag
823 ## ->{attributes} isa HASH (START_TAG_TOKEN, END_TAG_TOKEN)
824 ## ->{name}
825 ## ->{value}
826 ## ->{has_reference} == 1 or 0
827 ## ->{data} (COMMENT_TOKEN, CHARACTER_TOKEN)
828 ## NOTE: The "self-closing flag" is hold as |$self->{self_closing}|.
829 ## |->{self_closing}| is used to save the value of |$self->{self_closing}|
830 ## while the token is pushed back to the stack.
831
832 ## ISSUE: "When a DOCTYPE token is created, its
833 ## <i>self-closing flag</i> must be unset (its other state is that it
834 ## be set), and its attributes list must be empty.": Wrong subject?
835
836 ## Emitted token MUST immediately be handled by the tree construction state.
837
838 ## Before each step, UA MAY check to see if either one of the scripts in
839 ## "list of scripts that will execute as soon as possible" or the first
840 ## script in the "list of scripts that will execute asynchronously",
841 ## has completed loading. If one has, then it MUST be executed
842 ## and removed from the list.
843
844 ## NOTE: HTML5 "Writing HTML documents" section, applied to
845 ## documents and not to user agents and conformance checkers,
846 ## contains some requirements that are not detected by the
847 ## parsing algorithm:
848 ## - Some requirements on character encoding declarations. ## TODO
849 ## - "Elements MUST NOT contain content that their content model disallows."
850 ## ... Some are parse error, some are not (will be reported by c.c.).
851 ## - Polytheistic slash SHOULD NOT be used. (Applied only to atheists.) ## TODO
852 ## - Text (in elements, attributes, and comments) SHOULD NOT contain
853 ## control characters other than space characters. ## TODO: (what is control character? C0, C1 and DEL? Unicode control character?)
854
855 ## TODO: HTML5 poses authors two SHOULD-level requirements that cannot
856 ## be detected by the HTML5 parsing algorithm:
857 ## - Text,
858
859 sub _get_next_token ($) {
860 my $self = shift;
861
862 if ($self->{self_closing}) {
863 !!!parse-error (type => 'nestc', token => $self->{current_token});
864 ## NOTE: The |self_closing| flag is only set by start tag token.
865 ## In addition, when a start tag token is emitted, it is always set to
866 ## |current_token|.
867 delete $self->{self_closing};
868 }
869
870 if (@{$self->{token}}) {
871 $self->{self_closing} = $self->{token}->[0]->{self_closing};
872 return shift @{$self->{token}};
873 }
874
875 A: {
876 if ($self->{state} == DATA_STATE) {
877 if ($self->{next_char} == 0x0026) { # &
878 if ($self->{content_model} & CM_ENTITY and # PCDATA | RCDATA
879 not $self->{escape}) {
880 !!!cp (1);
881 $self->{state} = ENTITY_DATA_STATE;
882 !!!next-input-character;
883 redo A;
884 } else {
885 !!!cp (2);
886 #
887 }
888 } elsif ($self->{next_char} == 0x002D) { # -
889 if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
890 unless ($self->{escape}) {
891 if ($self->{prev_char}->[0] == 0x002D and # -
892 $self->{prev_char}->[1] == 0x0021 and # !
893 $self->{prev_char}->[2] == 0x003C) { # <
894 !!!cp (3);
895 $self->{escape} = 1;
896 } else {
897 !!!cp (4);
898 }
899 } else {
900 !!!cp (5);
901 }
902 }
903
904 #
905 } elsif ($self->{next_char} == 0x003C) { # <
906 if ($self->{content_model} & CM_FULL_MARKUP or # PCDATA
907 (($self->{content_model} & CM_LIMITED_MARKUP) and # CDATA | RCDATA
908 not $self->{escape})) {
909 !!!cp (6);
910 $self->{state} = TAG_OPEN_STATE;
911 !!!next-input-character;
912 redo A;
913 } else {
914 !!!cp (7);
915 #
916 }
917 } elsif ($self->{next_char} == 0x003E) { # >
918 if ($self->{escape} and
919 ($self->{content_model} & CM_LIMITED_MARKUP)) { # RCDATA | CDATA
920 if ($self->{prev_char}->[0] == 0x002D and # -
921 $self->{prev_char}->[1] == 0x002D) { # -
922 !!!cp (8);
923 delete $self->{escape};
924 } else {
925 !!!cp (9);
926 }
927 } else {
928 !!!cp (10);
929 }
930
931 #
932 } elsif ($self->{next_char} == -1) {
933 !!!cp (11);
934 !!!emit ({type => END_OF_FILE_TOKEN,
935 line => $self->{line}, column => $self->{column}});
936 last A; ## TODO: ok?
937 } else {
938 !!!cp (12);
939 }
940 # Anything else
941 my $token = {type => CHARACTER_TOKEN,
942 data => chr $self->{next_char},
943 line => $self->{line}, column => $self->{column},
944 };
945 ## Stay in the data state
946 !!!next-input-character;
947
948 !!!emit ($token);
949
950 redo A;
951 } elsif ($self->{state} == ENTITY_DATA_STATE) {
952 ## (cannot happen in CDATA state)
953
954 my ($l, $c) = ($self->{line_prev}, $self->{column_prev});
955
956 my $token = $self->_tokenize_attempt_to_consume_an_entity (0, -1);
957
958 $self->{state} = DATA_STATE;
959 # next-input-character is already done
960
961 unless (defined $token) {
962 !!!cp (13);
963 !!!emit ({type => CHARACTER_TOKEN, data => '&',
964 line => $l, column => $c,
965 });
966 } else {
967 !!!cp (14);
968 !!!emit ($token);
969 }
970
971 redo A;
972 } elsif ($self->{state} == TAG_OPEN_STATE) {
973 if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
974 if ($self->{next_char} == 0x002F) { # /
975 !!!cp (15);
976 !!!next-input-character;
977 $self->{state} = CLOSE_TAG_OPEN_STATE;
978 redo A;
979 } else {
980 !!!cp (16);
981 ## reconsume
982 $self->{state} = DATA_STATE;
983
984 !!!emit ({type => CHARACTER_TOKEN, data => '<',
985 line => $self->{line_prev},
986 column => $self->{column_prev},
987 });
988
989 redo A;
990 }
991 } elsif ($self->{content_model} & CM_FULL_MARKUP) { # PCDATA
992 if ($self->{next_char} == 0x0021) { # !
993 !!!cp (17);
994 $self->{state} = MARKUP_DECLARATION_OPEN_STATE;
995 !!!next-input-character;
996 redo A;
997 } elsif ($self->{next_char} == 0x002F) { # /
998 !!!cp (18);
999 $self->{state} = CLOSE_TAG_OPEN_STATE;
1000 !!!next-input-character;
1001 redo A;
1002 } elsif (0x0041 <= $self->{next_char} and
1003 $self->{next_char} <= 0x005A) { # A..Z
1004 !!!cp (19);
1005 $self->{current_token}
1006 = {type => START_TAG_TOKEN,
1007 tag_name => chr ($self->{next_char} + 0x0020),
1008 line => $self->{line_prev},
1009 column => $self->{column_prev}};
1010 $self->{state} = TAG_NAME_STATE;
1011 !!!next-input-character;
1012 redo A;
1013 } elsif (0x0061 <= $self->{next_char} and
1014 $self->{next_char} <= 0x007A) { # a..z
1015 !!!cp (20);
1016 $self->{current_token} = {type => START_TAG_TOKEN,
1017 tag_name => chr ($self->{next_char}),
1018 line => $self->{line_prev},
1019 column => $self->{column_prev}};
1020 $self->{state} = TAG_NAME_STATE;
1021 !!!next-input-character;
1022 redo A;
1023 } elsif ($self->{next_char} == 0x003E) { # >
1024 !!!cp (21);
1025 !!!parse-error (type => 'empty start tag',
1026 line => $self->{line_prev},
1027 column => $self->{column_prev});
1028 $self->{state} = DATA_STATE;
1029 !!!next-input-character;
1030
1031 !!!emit ({type => CHARACTER_TOKEN, data => '<>',
1032 line => $self->{line_prev},
1033 column => $self->{column_prev},
1034 });
1035
1036 redo A;
1037 } elsif ($self->{next_char} == 0x003F) { # ?
1038 !!!cp (22);
1039 !!!parse-error (type => 'pio',
1040 line => $self->{line_prev},
1041 column => $self->{column_prev});
1042 $self->{state} = BOGUS_COMMENT_STATE;
1043 $self->{current_token} = {type => COMMENT_TOKEN, data => '',
1044 line => $self->{line_prev},
1045 column => $self->{column_prev},
1046 };
1047 ## $self->{next_char} is intentionally left as is
1048 redo A;
1049 } else {
1050 !!!cp (23);
1051 !!!parse-error (type => 'bare stago',
1052 line => $self->{line_prev},
1053 column => $self->{column_prev});
1054 $self->{state} = DATA_STATE;
1055 ## reconsume
1056
1057 !!!emit ({type => CHARACTER_TOKEN, data => '<',
1058 line => $self->{line_prev},
1059 column => $self->{column_prev},
1060 });
1061
1062 redo A;
1063 }
1064 } else {
1065 die "$0: $self->{content_model} in tag open";
1066 }
1067 } elsif ($self->{state} == CLOSE_TAG_OPEN_STATE) {
1068 my ($l, $c) = ($self->{line_prev}, $self->{column_prev} - 1); # "<"of"</"
1069 if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
1070 if (defined $self->{last_emitted_start_tag_name}) {
1071
1072 ## NOTE: <http://krijnhoetmer.nl/irc-logs/whatwg/20070626#l-564>
1073 my @next_char;
1074 TAGNAME: for (my $i = 0; $i < length $self->{last_emitted_start_tag_name}; $i++) {
1075 push @next_char, $self->{next_char};
1076 my $c = ord substr ($self->{last_emitted_start_tag_name}, $i, 1);
1077 my $C = 0x0061 <= $c && $c <= 0x007A ? $c - 0x0020 : $c;
1078 if ($self->{next_char} == $c or $self->{next_char} == $C) {
1079 !!!cp (24);
1080 !!!next-input-character;
1081 next TAGNAME;
1082 } else {
1083 !!!cp (25);
1084 $self->{next_char} = shift @next_char; # reconsume
1085 !!!back-next-input-character (@next_char);
1086 $self->{state} = DATA_STATE;
1087
1088 !!!emit ({type => CHARACTER_TOKEN, data => '</',
1089 line => $l, column => $c,
1090 });
1091
1092 redo A;
1093 }
1094 }
1095 push @next_char, $self->{next_char};
1096
1097 unless ($self->{next_char} == 0x0009 or # HT
1098 $self->{next_char} == 0x000A or # LF
1099 $self->{next_char} == 0x000B or # VT
1100 $self->{next_char} == 0x000C or # FF
1101 $self->{next_char} == 0x0020 or # SP
1102 $self->{next_char} == 0x003E or # >
1103 $self->{next_char} == 0x002F or # /
1104 $self->{next_char} == -1) {
1105 !!!cp (26);
1106 $self->{next_char} = shift @next_char; # reconsume
1107 !!!back-next-input-character (@next_char);
1108 $self->{state} = DATA_STATE;
1109 !!!emit ({type => CHARACTER_TOKEN, data => '</',
1110 line => $l, column => $c,
1111 });
1112 redo A;
1113 } else {
1114 !!!cp (27);
1115 $self->{next_char} = shift @next_char;
1116 !!!back-next-input-character (@next_char);
1117 # and consume...
1118 }
1119 } else {
1120 ## No start tag token has ever been emitted
1121 !!!cp (28);
1122 # next-input-character is already done
1123 $self->{state} = DATA_STATE;
1124 !!!emit ({type => CHARACTER_TOKEN, data => '</',
1125 line => $l, column => $c,
1126 });
1127 redo A;
1128 }
1129 }
1130
1131 if (0x0041 <= $self->{next_char} and
1132 $self->{next_char} <= 0x005A) { # A..Z
1133 !!!cp (29);
1134 $self->{current_token}
1135 = {type => END_TAG_TOKEN,
1136 tag_name => chr ($self->{next_char} + 0x0020),
1137 line => $l, column => $c};
1138 $self->{state} = TAG_NAME_STATE;
1139 !!!next-input-character;
1140 redo A;
1141 } elsif (0x0061 <= $self->{next_char} and
1142 $self->{next_char} <= 0x007A) { # a..z
1143 !!!cp (30);
1144 $self->{current_token} = {type => END_TAG_TOKEN,
1145 tag_name => chr ($self->{next_char}),
1146 line => $l, column => $c};
1147 $self->{state} = TAG_NAME_STATE;
1148 !!!next-input-character;
1149 redo A;
1150 } elsif ($self->{next_char} == 0x003E) { # >
1151 !!!cp (31);
1152 !!!parse-error (type => 'empty end tag',
1153 line => $self->{line_prev}, ## "<" in "</>"
1154 column => $self->{column_prev} - 1);
1155 $self->{state} = DATA_STATE;
1156 !!!next-input-character;
1157 redo A;
1158 } elsif ($self->{next_char} == -1) {
1159 !!!cp (32);
1160 !!!parse-error (type => 'bare etago');
1161 $self->{state} = DATA_STATE;
1162 # reconsume
1163
1164 !!!emit ({type => CHARACTER_TOKEN, data => '</',
1165 line => $l, column => $c,
1166 });
1167
1168 redo A;
1169 } else {
1170 !!!cp (33);
1171 !!!parse-error (type => 'bogus end tag');
1172 $self->{state} = BOGUS_COMMENT_STATE;
1173 $self->{current_token} = {type => COMMENT_TOKEN, data => '',
1174 line => $self->{line_prev}, # "<" of "</"
1175 column => $self->{column_prev} - 1,
1176 };
1177 ## $self->{next_char} is intentionally left as is
1178 redo A;
1179 }
1180 } elsif ($self->{state} == TAG_NAME_STATE) {
1181 if ($self->{next_char} == 0x0009 or # HT
1182 $self->{next_char} == 0x000A or # LF
1183 $self->{next_char} == 0x000B or # VT
1184 $self->{next_char} == 0x000C or # FF
1185 $self->{next_char} == 0x0020) { # SP
1186 !!!cp (34);
1187 $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1188 !!!next-input-character;
1189 redo A;
1190 } elsif ($self->{next_char} == 0x003E) { # >
1191 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1192 !!!cp (35);
1193 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1194 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1195 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1196 #if ($self->{current_token}->{attributes}) {
1197 # ## NOTE: This should never be reached.
1198 # !!! cp (36);
1199 # !!! parse-error (type => 'end tag attribute');
1200 #} else {
1201 !!!cp (37);
1202 #}
1203 } else {
1204 die "$0: $self->{current_token}->{type}: Unknown token type";
1205 }
1206 $self->{state} = DATA_STATE;
1207 !!!next-input-character;
1208
1209 !!!emit ($self->{current_token}); # start tag or end tag
1210
1211 redo A;
1212 } elsif (0x0041 <= $self->{next_char} and
1213 $self->{next_char} <= 0x005A) { # A..Z
1214 !!!cp (38);
1215 $self->{current_token}->{tag_name} .= chr ($self->{next_char} + 0x0020);
1216 # start tag or end tag
1217 ## Stay in this state
1218 !!!next-input-character;
1219 redo A;
1220 } elsif ($self->{next_char} == -1) {
1221 !!!parse-error (type => 'unclosed tag');
1222 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1223 !!!cp (39);
1224 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1225 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1226 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1227 #if ($self->{current_token}->{attributes}) {
1228 # ## NOTE: This state should never be reached.
1229 # !!! cp (40);
1230 # !!! parse-error (type => 'end tag attribute');
1231 #} else {
1232 !!!cp (41);
1233 #}
1234 } else {
1235 die "$0: $self->{current_token}->{type}: Unknown token type";
1236 }
1237 $self->{state} = DATA_STATE;
1238 # reconsume
1239
1240 !!!emit ($self->{current_token}); # start tag or end tag
1241
1242 redo A;
1243 } elsif ($self->{next_char} == 0x002F) { # /
1244 !!!cp (42);
1245 $self->{state} = SELF_CLOSING_START_TAG_STATE;
1246 !!!next-input-character;
1247 redo A;
1248 } else {
1249 !!!cp (44);
1250 $self->{current_token}->{tag_name} .= chr $self->{next_char};
1251 # start tag or end tag
1252 ## Stay in the state
1253 !!!next-input-character;
1254 redo A;
1255 }
1256 } elsif ($self->{state} == BEFORE_ATTRIBUTE_NAME_STATE) {
1257 if ($self->{next_char} == 0x0009 or # HT
1258 $self->{next_char} == 0x000A or # LF
1259 $self->{next_char} == 0x000B or # VT
1260 $self->{next_char} == 0x000C or # FF
1261 $self->{next_char} == 0x0020) { # SP
1262 !!!cp (45);
1263 ## Stay in the state
1264 !!!next-input-character;
1265 redo A;
1266 } elsif ($self->{next_char} == 0x003E) { # >
1267 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1268 !!!cp (46);
1269 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1270 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1271 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1272 if ($self->{current_token}->{attributes}) {
1273 !!!cp (47);
1274 !!!parse-error (type => 'end tag attribute');
1275 } else {
1276 !!!cp (48);
1277 }
1278 } else {
1279 die "$0: $self->{current_token}->{type}: Unknown token type";
1280 }
1281 $self->{state} = DATA_STATE;
1282 !!!next-input-character;
1283
1284 !!!emit ($self->{current_token}); # start tag or end tag
1285
1286 redo A;
1287 } elsif (0x0041 <= $self->{next_char} and
1288 $self->{next_char} <= 0x005A) { # A..Z
1289 !!!cp (49);
1290 $self->{current_attribute}
1291 = {name => chr ($self->{next_char} + 0x0020),
1292 value => '',
1293 line => $self->{line}, column => $self->{column}};
1294 $self->{state} = ATTRIBUTE_NAME_STATE;
1295 !!!next-input-character;
1296 redo A;
1297 } elsif ($self->{next_char} == 0x002F) { # /
1298 !!!cp (50);
1299 $self->{state} = SELF_CLOSING_START_TAG_STATE;
1300 !!!next-input-character;
1301 redo A;
1302 } elsif ($self->{next_char} == -1) {
1303 !!!parse-error (type => 'unclosed tag');
1304 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1305 !!!cp (52);
1306 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1307 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1308 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1309 if ($self->{current_token}->{attributes}) {
1310 !!!cp (53);
1311 !!!parse-error (type => 'end tag attribute');
1312 } else {
1313 !!!cp (54);
1314 }
1315 } else {
1316 die "$0: $self->{current_token}->{type}: Unknown token type";
1317 }
1318 $self->{state} = DATA_STATE;
1319 # reconsume
1320
1321 !!!emit ($self->{current_token}); # start tag or end tag
1322
1323 redo A;
1324 } else {
1325 if ({
1326 0x0022 => 1, # "
1327 0x0027 => 1, # '
1328 0x003D => 1, # =
1329 }->{$self->{next_char}}) {
1330 !!!cp (55);
1331 !!!parse-error (type => 'bad attribute name');
1332 } else {
1333 !!!cp (56);
1334 }
1335 $self->{current_attribute}
1336 = {name => chr ($self->{next_char}),
1337 value => '',
1338 line => $self->{line}, column => $self->{column}};
1339 $self->{state} = ATTRIBUTE_NAME_STATE;
1340 !!!next-input-character;
1341 redo A;
1342 }
1343 } elsif ($self->{state} == ATTRIBUTE_NAME_STATE) {
1344 my $before_leave = sub {
1345 if (exists $self->{current_token}->{attributes} # start tag or end tag
1346 ->{$self->{current_attribute}->{name}}) { # MUST
1347 !!!cp (57);
1348 !!!parse-error (type => 'duplicate attribute:'.$self->{current_attribute}->{name}, line => $self->{current_attribute}->{line}, column => $self->{current_attribute}->{column});
1349 ## Discard $self->{current_attribute} # MUST
1350 } else {
1351 !!!cp (58);
1352 $self->{current_token}->{attributes}->{$self->{current_attribute}->{name}}
1353 = $self->{current_attribute};
1354 }
1355 }; # $before_leave
1356
1357 if ($self->{next_char} == 0x0009 or # HT
1358 $self->{next_char} == 0x000A or # LF
1359 $self->{next_char} == 0x000B or # VT
1360 $self->{next_char} == 0x000C or # FF
1361 $self->{next_char} == 0x0020) { # SP
1362 !!!cp (59);
1363 $before_leave->();
1364 $self->{state} = AFTER_ATTRIBUTE_NAME_STATE;
1365 !!!next-input-character;
1366 redo A;
1367 } elsif ($self->{next_char} == 0x003D) { # =
1368 !!!cp (60);
1369 $before_leave->();
1370 $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;
1371 !!!next-input-character;
1372 redo A;
1373 } elsif ($self->{next_char} == 0x003E) { # >
1374 $before_leave->();
1375 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1376 !!!cp (61);
1377 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1378 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1379 !!!cp (62);
1380 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1381 if ($self->{current_token}->{attributes}) {
1382 !!!parse-error (type => 'end tag attribute');
1383 }
1384 } else {
1385 die "$0: $self->{current_token}->{type}: Unknown token type";
1386 }
1387 $self->{state} = DATA_STATE;
1388 !!!next-input-character;
1389
1390 !!!emit ($self->{current_token}); # start tag or end tag
1391
1392 redo A;
1393 } elsif (0x0041 <= $self->{next_char} and
1394 $self->{next_char} <= 0x005A) { # A..Z
1395 !!!cp (63);
1396 $self->{current_attribute}->{name} .= chr ($self->{next_char} + 0x0020);
1397 ## Stay in the state
1398 !!!next-input-character;
1399 redo A;
1400 } elsif ($self->{next_char} == 0x002F) { # /
1401 !!!cp (64);
1402 $before_leave->();
1403 $self->{state} = SELF_CLOSING_START_TAG_STATE;
1404 !!!next-input-character;
1405 redo A;
1406 } elsif ($self->{next_char} == -1) {
1407 !!!parse-error (type => 'unclosed tag');
1408 $before_leave->();
1409 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1410 !!!cp (66);
1411 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1412 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1413 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1414 if ($self->{current_token}->{attributes}) {
1415 !!!cp (67);
1416 !!!parse-error (type => 'end tag attribute');
1417 } else {
1418 ## NOTE: This state should never be reached.
1419 !!!cp (68);
1420 }
1421 } else {
1422 die "$0: $self->{current_token}->{type}: Unknown token type";
1423 }
1424 $self->{state} = DATA_STATE;
1425 # reconsume
1426
1427 !!!emit ($self->{current_token}); # start tag or end tag
1428
1429 redo A;
1430 } else {
1431 if ($self->{next_char} == 0x0022 or # "
1432 $self->{next_char} == 0x0027) { # '
1433 !!!cp (69);
1434 !!!parse-error (type => 'bad attribute name');
1435 } else {
1436 !!!cp (70);
1437 }
1438 $self->{current_attribute}->{name} .= chr ($self->{next_char});
1439 ## Stay in the state
1440 !!!next-input-character;
1441 redo A;
1442 }
1443 } elsif ($self->{state} == AFTER_ATTRIBUTE_NAME_STATE) {
1444 if ($self->{next_char} == 0x0009 or # HT
1445 $self->{next_char} == 0x000A or # LF
1446 $self->{next_char} == 0x000B or # VT
1447 $self->{next_char} == 0x000C or # FF
1448 $self->{next_char} == 0x0020) { # SP
1449 !!!cp (71);
1450 ## Stay in the state
1451 !!!next-input-character;
1452 redo A;
1453 } elsif ($self->{next_char} == 0x003D) { # =
1454 !!!cp (72);
1455 $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;
1456 !!!next-input-character;
1457 redo A;
1458 } elsif ($self->{next_char} == 0x003E) { # >
1459 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1460 !!!cp (73);
1461 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1462 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1463 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1464 if ($self->{current_token}->{attributes}) {
1465 !!!cp (74);
1466 !!!parse-error (type => 'end tag attribute');
1467 } else {
1468 ## NOTE: This state should never be reached.
1469 !!!cp (75);
1470 }
1471 } else {
1472 die "$0: $self->{current_token}->{type}: Unknown token type";
1473 }
1474 $self->{state} = DATA_STATE;
1475 !!!next-input-character;
1476
1477 !!!emit ($self->{current_token}); # start tag or end tag
1478
1479 redo A;
1480 } elsif (0x0041 <= $self->{next_char} and
1481 $self->{next_char} <= 0x005A) { # A..Z
1482 !!!cp (76);
1483 $self->{current_attribute}
1484 = {name => chr ($self->{next_char} + 0x0020),
1485 value => '',
1486 line => $self->{line}, column => $self->{column}};
1487 $self->{state} = ATTRIBUTE_NAME_STATE;
1488 !!!next-input-character;
1489 redo A;
1490 } elsif ($self->{next_char} == 0x002F) { # /
1491 !!!cp (77);
1492 $self->{state} = SELF_CLOSING_START_TAG_STATE;
1493 !!!next-input-character;
1494 redo A;
1495 } elsif ($self->{next_char} == -1) {
1496 !!!parse-error (type => 'unclosed tag');
1497 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1498 !!!cp (79);
1499 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1500 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1501 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1502 if ($self->{current_token}->{attributes}) {
1503 !!!cp (80);
1504 !!!parse-error (type => 'end tag attribute');
1505 } else {
1506 ## NOTE: This state should never be reached.
1507 !!!cp (81);
1508 }
1509 } else {
1510 die "$0: $self->{current_token}->{type}: Unknown token type";
1511 }
1512 $self->{state} = DATA_STATE;
1513 # reconsume
1514
1515 !!!emit ($self->{current_token}); # start tag or end tag
1516
1517 redo A;
1518 } else {
1519 !!!cp (82);
1520 $self->{current_attribute}
1521 = {name => chr ($self->{next_char}),
1522 value => '',
1523 line => $self->{line}, column => $self->{column}};
1524 $self->{state} = ATTRIBUTE_NAME_STATE;
1525 !!!next-input-character;
1526 redo A;
1527 }
1528 } elsif ($self->{state} == BEFORE_ATTRIBUTE_VALUE_STATE) {
1529 if ($self->{next_char} == 0x0009 or # HT
1530 $self->{next_char} == 0x000A or # LF
1531 $self->{next_char} == 0x000B or # VT
1532 $self->{next_char} == 0x000C or # FF
1533 $self->{next_char} == 0x0020) { # SP
1534 !!!cp (83);
1535 ## Stay in the state
1536 !!!next-input-character;
1537 redo A;
1538 } elsif ($self->{next_char} == 0x0022) { # "
1539 !!!cp (84);
1540 $self->{state} = ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE;
1541 !!!next-input-character;
1542 redo A;
1543 } elsif ($self->{next_char} == 0x0026) { # &
1544 !!!cp (85);
1545 $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;
1546 ## reconsume
1547 redo A;
1548 } elsif ($self->{next_char} == 0x0027) { # '
1549 !!!cp (86);
1550 $self->{state} = ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE;
1551 !!!next-input-character;
1552 redo A;
1553 } elsif ($self->{next_char} == 0x003E) { # >
1554 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1555 !!!cp (87);
1556 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1557 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1558 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1559 if ($self->{current_token}->{attributes}) {
1560 !!!cp (88);
1561 !!!parse-error (type => 'end tag attribute');
1562 } else {
1563 ## NOTE: This state should never be reached.
1564 !!!cp (89);
1565 }
1566 } else {
1567 die "$0: $self->{current_token}->{type}: Unknown token type";
1568 }
1569 $self->{state} = DATA_STATE;
1570 !!!next-input-character;
1571
1572 !!!emit ($self->{current_token}); # start tag or end tag
1573
1574 redo A;
1575 } elsif ($self->{next_char} == -1) {
1576 !!!parse-error (type => 'unclosed tag');
1577 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1578 !!!cp (90);
1579 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1580 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1581 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1582 if ($self->{current_token}->{attributes}) {
1583 !!!cp (91);
1584 !!!parse-error (type => 'end tag attribute');
1585 } else {
1586 ## NOTE: This state should never be reached.
1587 !!!cp (92);
1588 }
1589 } else {
1590 die "$0: $self->{current_token}->{type}: Unknown token type";
1591 }
1592 $self->{state} = DATA_STATE;
1593 ## reconsume
1594
1595 !!!emit ($self->{current_token}); # start tag or end tag
1596
1597 redo A;
1598 } else {
1599 if ($self->{next_char} == 0x003D) { # =
1600 !!!cp (93);
1601 !!!parse-error (type => 'bad attribute value');
1602 } else {
1603 !!!cp (94);
1604 }
1605 $self->{current_attribute}->{value} .= chr ($self->{next_char});
1606 $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;
1607 !!!next-input-character;
1608 redo A;
1609 }
1610 } elsif ($self->{state} == ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE) {
1611 if ($self->{next_char} == 0x0022) { # "
1612 !!!cp (95);
1613 $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;
1614 !!!next-input-character;
1615 redo A;
1616 } elsif ($self->{next_char} == 0x0026) { # &
1617 !!!cp (96);
1618 $self->{last_attribute_value_state} = $self->{state};
1619 $self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE;
1620 !!!next-input-character;
1621 redo A;
1622 } elsif ($self->{next_char} == -1) {
1623 !!!parse-error (type => 'unclosed attribute value');
1624 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1625 !!!cp (97);
1626 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1627 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1628 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1629 if ($self->{current_token}->{attributes}) {
1630 !!!cp (98);
1631 !!!parse-error (type => 'end tag attribute');
1632 } else {
1633 ## NOTE: This state should never be reached.
1634 !!!cp (99);
1635 }
1636 } else {
1637 die "$0: $self->{current_token}->{type}: Unknown token type";
1638 }
1639 $self->{state} = DATA_STATE;
1640 ## reconsume
1641
1642 !!!emit ($self->{current_token}); # start tag or end tag
1643
1644 redo A;
1645 } else {
1646 !!!cp (100);
1647 $self->{current_attribute}->{value} .= chr ($self->{next_char});
1648 ## Stay in the state
1649 !!!next-input-character;
1650 redo A;
1651 }
1652 } elsif ($self->{state} == ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE) {
1653 if ($self->{next_char} == 0x0027) { # '
1654 !!!cp (101);
1655 $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;
1656 !!!next-input-character;
1657 redo A;
1658 } elsif ($self->{next_char} == 0x0026) { # &
1659 !!!cp (102);
1660 $self->{last_attribute_value_state} = $self->{state};
1661 $self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE;
1662 !!!next-input-character;
1663 redo A;
1664 } elsif ($self->{next_char} == -1) {
1665 !!!parse-error (type => 'unclosed attribute value');
1666 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1667 !!!cp (103);
1668 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1669 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1670 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1671 if ($self->{current_token}->{attributes}) {
1672 !!!cp (104);
1673 !!!parse-error (type => 'end tag attribute');
1674 } else {
1675 ## NOTE: This state should never be reached.
1676 !!!cp (105);
1677 }
1678 } else {
1679 die "$0: $self->{current_token}->{type}: Unknown token type";
1680 }
1681 $self->{state} = DATA_STATE;
1682 ## reconsume
1683
1684 !!!emit ($self->{current_token}); # start tag or end tag
1685
1686 redo A;
1687 } else {
1688 !!!cp (106);
1689 $self->{current_attribute}->{value} .= chr ($self->{next_char});
1690 ## Stay in the state
1691 !!!next-input-character;
1692 redo A;
1693 }
1694 } elsif ($self->{state} == ATTRIBUTE_VALUE_UNQUOTED_STATE) {
1695 if ($self->{next_char} == 0x0009 or # HT
1696 $self->{next_char} == 0x000A or # LF
1697 $self->{next_char} == 0x000B or # HT
1698 $self->{next_char} == 0x000C or # FF
1699 $self->{next_char} == 0x0020) { # SP
1700 !!!cp (107);
1701 $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1702 !!!next-input-character;
1703 redo A;
1704 } elsif ($self->{next_char} == 0x0026) { # &
1705 !!!cp (108);
1706 $self->{last_attribute_value_state} = $self->{state};
1707 $self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE;
1708 !!!next-input-character;
1709 redo A;
1710 } elsif ($self->{next_char} == 0x003E) { # >
1711 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1712 !!!cp (109);
1713 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1714 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1715 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1716 if ($self->{current_token}->{attributes}) {
1717 !!!cp (110);
1718 !!!parse-error (type => 'end tag attribute');
1719 } else {
1720 ## NOTE: This state should never be reached.
1721 !!!cp (111);
1722 }
1723 } else {
1724 die "$0: $self->{current_token}->{type}: Unknown token type";
1725 }
1726 $self->{state} = DATA_STATE;
1727 !!!next-input-character;
1728
1729 !!!emit ($self->{current_token}); # start tag or end tag
1730
1731 redo A;
1732 } elsif ($self->{next_char} == -1) {
1733 !!!parse-error (type => 'unclosed tag');
1734 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1735 !!!cp (112);
1736 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1737 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1738 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1739 if ($self->{current_token}->{attributes}) {
1740 !!!cp (113);
1741 !!!parse-error (type => 'end tag attribute');
1742 } else {
1743 ## NOTE: This state should never be reached.
1744 !!!cp (114);
1745 }
1746 } else {
1747 die "$0: $self->{current_token}->{type}: Unknown token type";
1748 }
1749 $self->{state} = DATA_STATE;
1750 ## reconsume
1751
1752 !!!emit ($self->{current_token}); # start tag or end tag
1753
1754 redo A;
1755 } else {
1756 if ({
1757 0x0022 => 1, # "
1758 0x0027 => 1, # '
1759 0x003D => 1, # =
1760 }->{$self->{next_char}}) {
1761 !!!cp (115);
1762 !!!parse-error (type => 'bad attribute value');
1763 } else {
1764 !!!cp (116);
1765 }
1766 $self->{current_attribute}->{value} .= chr ($self->{next_char});
1767 ## Stay in the state
1768 !!!next-input-character;
1769 redo A;
1770 }
1771 } elsif ($self->{state} == ENTITY_IN_ATTRIBUTE_VALUE_STATE) {
1772 my $token = $self->_tokenize_attempt_to_consume_an_entity
1773 (1,
1774 $self->{last_attribute_value_state}
1775 == ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE ? 0x0022 : # "
1776 $self->{last_attribute_value_state}
1777 == ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE ? 0x0027 : # '
1778 -1);
1779
1780 unless (defined $token) {
1781 !!!cp (117);
1782 $self->{current_attribute}->{value} .= '&';
1783 } else {
1784 !!!cp (118);
1785 $self->{current_attribute}->{value} .= $token->{data};
1786 $self->{current_attribute}->{has_reference} = $token->{has_reference};
1787 ## ISSUE: spec says "append the returned character token to the current attribute's value"
1788 }
1789
1790 $self->{state} = $self->{last_attribute_value_state};
1791 # next-input-character is already done
1792 redo A;
1793 } elsif ($self->{state} == AFTER_ATTRIBUTE_VALUE_QUOTED_STATE) {
1794 if ($self->{next_char} == 0x0009 or # HT
1795 $self->{next_char} == 0x000A or # LF
1796 $self->{next_char} == 0x000B or # VT
1797 $self->{next_char} == 0x000C or # FF
1798 $self->{next_char} == 0x0020) { # SP
1799 !!!cp (118);
1800 $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1801 !!!next-input-character;
1802 redo A;
1803 } elsif ($self->{next_char} == 0x003E) { # >
1804 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1805 !!!cp (119);
1806 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1807 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1808 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1809 if ($self->{current_token}->{attributes}) {
1810 !!!cp (120);
1811 !!!parse-error (type => 'end tag attribute');
1812 } else {
1813 ## NOTE: This state should never be reached.
1814 !!!cp (121);
1815 }
1816 } else {
1817 die "$0: $self->{current_token}->{type}: Unknown token type";
1818 }
1819 $self->{state} = DATA_STATE;
1820 !!!next-input-character;
1821
1822 !!!emit ($self->{current_token}); # start tag or end tag
1823
1824 redo A;
1825 } elsif ($self->{next_char} == 0x002F) { # /
1826 !!!cp (122);
1827 $self->{state} = SELF_CLOSING_START_TAG_STATE;
1828 !!!next-input-character;
1829 redo A;
1830 } elsif ($self->{next_char} == -1) {
1831 !!!parse-error (type => 'unclosed tag');
1832 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1833 !!!cp (122.3);
1834 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1835 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1836 if ($self->{current_token}->{attributes}) {
1837 !!!cp (122.1);
1838 !!!parse-error (type => 'end tag attribute');
1839 } else {
1840 ## NOTE: This state should never be reached.
1841 !!!cp (122.2);
1842 }
1843 } else {
1844 die "$0: $self->{current_token}->{type}: Unknown token type";
1845 }
1846 $self->{state} = DATA_STATE;
1847 ## Reconsume.
1848 !!!emit ($self->{current_token}); # start tag or end tag
1849 redo A;
1850 } else {
1851 !!!cp ('124.1');
1852 !!!parse-error (type => 'no space between attributes');
1853 $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1854 ## reconsume
1855 redo A;
1856 }
1857 } elsif ($self->{state} == SELF_CLOSING_START_TAG_STATE) {
1858 if ($self->{next_char} == 0x003E) { # >
1859 if ($self->{current_token}->{type} == END_TAG_TOKEN) {
1860 !!!cp ('124.2');
1861 !!!parse-error (type => 'nestc', token => $self->{current_token});
1862 ## TODO: Different type than slash in start tag
1863 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1864 if ($self->{current_token}->{attributes}) {
1865 !!!cp ('124.4');
1866 !!!parse-error (type => 'end tag attribute');
1867 } else {
1868 !!!cp ('124.5');
1869 }
1870 ## TODO: Test |<title></title/>|
1871 } else {
1872 !!!cp ('124.3');
1873 $self->{self_closing} = 1;
1874 }
1875
1876 $self->{state} = DATA_STATE;
1877 !!!next-input-character;
1878
1879 !!!emit ($self->{current_token}); # start tag or end tag
1880
1881 redo A;
1882 } elsif ($self->{next_char} == -1) {
1883 !!!parse-error (type => 'unclosed tag');
1884 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1885 !!!cp (124.7);
1886 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1887 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1888 if ($self->{current_token}->{attributes}) {
1889 !!!cp (124.5);
1890 !!!parse-error (type => 'end tag attribute');
1891 } else {
1892 ## NOTE: This state should never be reached.
1893 !!!cp (124.6);
1894 }
1895 } else {
1896 die "$0: $self->{current_token}->{type}: Unknown token type";
1897 }
1898 $self->{state} = DATA_STATE;
1899 ## Reconsume.
1900 !!!emit ($self->{current_token}); # start tag or end tag
1901 redo A;
1902 } else {
1903 !!!cp ('124.4');
1904 !!!parse-error (type => 'nestc');
1905 ## TODO: This error type is wrong.
1906 $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1907 ## Reconsume.
1908 redo A;
1909 }
1910 } elsif ($self->{state} == BOGUS_COMMENT_STATE) {
1911 ## (only happen if PCDATA state)
1912
1913 ## NOTE: Set by the previous state
1914 #my $token = {type => COMMENT_TOKEN, data => ''};
1915
1916 BC: {
1917 if ($self->{next_char} == 0x003E) { # >
1918 !!!cp (124);
1919 $self->{state} = DATA_STATE;
1920 !!!next-input-character;
1921
1922 !!!emit ($self->{current_token}); # comment
1923
1924 redo A;
1925 } elsif ($self->{next_char} == -1) {
1926 !!!cp (125);
1927 $self->{state} = DATA_STATE;
1928 ## reconsume
1929
1930 !!!emit ($self->{current_token}); # comment
1931
1932 redo A;
1933 } else {
1934 !!!cp (126);
1935 $self->{current_token}->{data} .= chr ($self->{next_char}); # comment
1936 !!!next-input-character;
1937 redo BC;
1938 }
1939 } # BC
1940
1941 die "$0: _get_next_token: unexpected case [BC]";
1942 } elsif ($self->{state} == MARKUP_DECLARATION_OPEN_STATE) {
1943 ## (only happen if PCDATA state)
1944
1945 my ($l, $c) = ($self->{line_prev}, $self->{column_prev} - 1);
1946
1947 my @next_char;
1948 push @next_char, $self->{next_char};
1949
1950 if ($self->{next_char} == 0x002D) { # -
1951 !!!next-input-character;
1952 push @next_char, $self->{next_char};
1953 if ($self->{next_char} == 0x002D) { # -
1954 !!!cp (127);
1955 $self->{current_token} = {type => COMMENT_TOKEN, data => '',
1956 line => $l, column => $c,
1957 };
1958 $self->{state} = COMMENT_START_STATE;
1959 !!!next-input-character;
1960 redo A;
1961 } else {
1962 !!!cp (128);
1963 }
1964 } elsif ($self->{next_char} == 0x0044 or # D
1965 $self->{next_char} == 0x0064) { # d
1966 !!!next-input-character;
1967 push @next_char, $self->{next_char};
1968 if ($self->{next_char} == 0x004F or # O
1969 $self->{next_char} == 0x006F) { # o
1970 !!!next-input-character;
1971 push @next_char, $self->{next_char};
1972 if ($self->{next_char} == 0x0043 or # C
1973 $self->{next_char} == 0x0063) { # c
1974 !!!next-input-character;
1975 push @next_char, $self->{next_char};
1976 if ($self->{next_char} == 0x0054 or # T
1977 $self->{next_char} == 0x0074) { # t
1978 !!!next-input-character;
1979 push @next_char, $self->{next_char};
1980 if ($self->{next_char} == 0x0059 or # Y
1981 $self->{next_char} == 0x0079) { # y
1982 !!!next-input-character;
1983 push @next_char, $self->{next_char};
1984 if ($self->{next_char} == 0x0050 or # P
1985 $self->{next_char} == 0x0070) { # p
1986 !!!next-input-character;
1987 push @next_char, $self->{next_char};
1988 if ($self->{next_char} == 0x0045 or # E
1989 $self->{next_char} == 0x0065) { # e
1990 !!!cp (129);
1991 ## TODO: What a stupid code this is!
1992 $self->{state} = DOCTYPE_STATE;
1993 $self->{current_token} = {type => DOCTYPE_TOKEN,
1994 quirks => 1,
1995 line => $l, column => $c,
1996 };
1997 !!!next-input-character;
1998 redo A;
1999 } else {
2000 !!!cp (130);
2001 }
2002 } else {
2003 !!!cp (131);
2004 }
2005 } else {
2006 !!!cp (132);
2007 }
2008 } else {
2009 !!!cp (133);
2010 }
2011 } else {
2012 !!!cp (134);
2013 }
2014 } else {
2015 !!!cp (135);
2016 }
2017 } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM and
2018 $self->{open_elements}->[-1]->[1] & FOREIGN_EL and
2019 $self->{next_char} == 0x005B) { # [
2020 !!!next-input-character;
2021 push @next_char, $self->{next_char};
2022 if ($self->{next_char} == 0x0043) { # C
2023 !!!next-input-character;
2024 push @next_char, $self->{next_char};
2025 if ($self->{next_char} == 0x0044) { # D
2026 !!!next-input-character;
2027 push @next_char, $self->{next_char};
2028 if ($self->{next_char} == 0x0041) { # A
2029 !!!next-input-character;
2030 push @next_char, $self->{next_char};
2031 if ($self->{next_char} == 0x0054) { # T
2032 !!!next-input-character;
2033 push @next_char, $self->{next_char};
2034 if ($self->{next_char} == 0x0041) { # A
2035 !!!next-input-character;
2036 push @next_char, $self->{next_char};
2037 if ($self->{next_char} == 0x005B) { # [
2038 !!!cp (135.1);
2039 $self->{state} = CDATA_BLOCK_STATE;
2040 !!!next-input-character;
2041 redo A;
2042 } else {
2043 !!!cp (135.2);
2044 }
2045 } else {
2046 !!!cp (135.3);
2047 }
2048 } else {
2049 !!!cp (135.4);
2050 }
2051 } else {
2052 !!!cp (135.5);
2053 }
2054 } else {
2055 !!!cp (135.6);
2056 }
2057 } else {
2058 !!!cp (135.7);
2059 }
2060 } else {
2061 !!!cp (136);
2062 }
2063
2064 !!!parse-error (type => 'bogus comment');
2065 $self->{next_char} = shift @next_char;
2066 !!!back-next-input-character (@next_char);
2067 $self->{state} = BOGUS_COMMENT_STATE;
2068 $self->{current_token} = {type => COMMENT_TOKEN, data => '',
2069 line => $l, column => $c,
2070 };
2071 redo A;
2072
2073 ## ISSUE: typos in spec: chacacters, is is a parse error
2074 ## ISSUE: spec is somewhat unclear on "is the first character that will be in the comment"; what is "that will be in the comment" is what the algorithm defines, isn't it?
2075 } elsif ($self->{state} == COMMENT_START_STATE) {
2076 if ($self->{next_char} == 0x002D) { # -
2077 !!!cp (137);
2078 $self->{state} = COMMENT_START_DASH_STATE;
2079 !!!next-input-character;
2080 redo A;
2081 } elsif ($self->{next_char} == 0x003E) { # >
2082 !!!cp (138);
2083 !!!parse-error (type => 'bogus comment');
2084 $self->{state} = DATA_STATE;
2085 !!!next-input-character;
2086
2087 !!!emit ($self->{current_token}); # comment
2088
2089 redo A;
2090 } elsif ($self->{next_char} == -1) {
2091 !!!cp (139);
2092 !!!parse-error (type => 'unclosed comment');
2093 $self->{state} = DATA_STATE;
2094 ## reconsume
2095
2096 !!!emit ($self->{current_token}); # comment
2097
2098 redo A;
2099 } else {
2100 !!!cp (140);
2101 $self->{current_token}->{data} # comment
2102 .= chr ($self->{next_char});
2103 $self->{state} = COMMENT_STATE;
2104 !!!next-input-character;
2105 redo A;
2106 }
2107 } elsif ($self->{state} == COMMENT_START_DASH_STATE) {
2108 if ($self->{next_char} == 0x002D) { # -
2109 !!!cp (141);
2110 $self->{state} = COMMENT_END_STATE;
2111 !!!next-input-character;
2112 redo A;
2113 } elsif ($self->{next_char} == 0x003E) { # >
2114 !!!cp (142);
2115 !!!parse-error (type => 'bogus comment');
2116 $self->{state} = DATA_STATE;
2117 !!!next-input-character;
2118
2119 !!!emit ($self->{current_token}); # comment
2120
2121 redo A;
2122 } elsif ($self->{next_char} == -1) {
2123 !!!cp (143);
2124 !!!parse-error (type => 'unclosed comment');
2125 $self->{state} = DATA_STATE;
2126 ## reconsume
2127
2128 !!!emit ($self->{current_token}); # comment
2129
2130 redo A;
2131 } else {
2132 !!!cp (144);
2133 $self->{current_token}->{data} # comment
2134 .= '-' . chr ($self->{next_char});
2135 $self->{state} = COMMENT_STATE;
2136 !!!next-input-character;
2137 redo A;
2138 }
2139 } elsif ($self->{state} == COMMENT_STATE) {
2140 if ($self->{next_char} == 0x002D) { # -
2141 !!!cp (145);
2142 $self->{state} = COMMENT_END_DASH_STATE;
2143 !!!next-input-character;
2144 redo A;
2145 } elsif ($self->{next_char} == -1) {
2146 !!!cp (146);
2147 !!!parse-error (type => 'unclosed comment');
2148 $self->{state} = DATA_STATE;
2149 ## reconsume
2150
2151 !!!emit ($self->{current_token}); # comment
2152
2153 redo A;
2154 } else {
2155 !!!cp (147);
2156 $self->{current_token}->{data} .= chr ($self->{next_char}); # comment
2157 ## Stay in the state
2158 !!!next-input-character;
2159 redo A;
2160 }
2161 } elsif ($self->{state} == COMMENT_END_DASH_STATE) {
2162 if ($self->{next_char} == 0x002D) { # -
2163 !!!cp (148);
2164 $self->{state} = COMMENT_END_STATE;
2165 !!!next-input-character;
2166 redo A;
2167 } elsif ($self->{next_char} == -1) {
2168 !!!cp (149);
2169 !!!parse-error (type => 'unclosed comment');
2170 $self->{state} = DATA_STATE;
2171 ## reconsume
2172
2173 !!!emit ($self->{current_token}); # comment
2174
2175 redo A;
2176 } else {
2177 !!!cp (150);
2178 $self->{current_token}->{data} .= '-' . chr ($self->{next_char}); # comment
2179 $self->{state} = COMMENT_STATE;
2180 !!!next-input-character;
2181 redo A;
2182 }
2183 } elsif ($self->{state} == COMMENT_END_STATE) {
2184 if ($self->{next_char} == 0x003E) { # >
2185 !!!cp (151);
2186 $self->{state} = DATA_STATE;
2187 !!!next-input-character;
2188
2189 !!!emit ($self->{current_token}); # comment
2190
2191 redo A;
2192 } elsif ($self->{next_char} == 0x002D) { # -
2193 !!!cp (152);
2194 !!!parse-error (type => 'dash in comment',
2195 line => $self->{line_prev},
2196 column => $self->{column_prev});
2197 $self->{current_token}->{data} .= '-'; # comment
2198 ## Stay in the state
2199 !!!next-input-character;
2200 redo A;
2201 } elsif ($self->{next_char} == -1) {
2202 !!!cp (153);
2203 !!!parse-error (type => 'unclosed comment');
2204 $self->{state} = DATA_STATE;
2205 ## reconsume
2206
2207 !!!emit ($self->{current_token}); # comment
2208
2209 redo A;
2210 } else {
2211 !!!cp (154);
2212 !!!parse-error (type => 'dash in comment',
2213 line => $self->{line_prev},
2214 column => $self->{column_prev});
2215 $self->{current_token}->{data} .= '--' . chr ($self->{next_char}); # comment
2216 $self->{state} = COMMENT_STATE;
2217 !!!next-input-character;
2218 redo A;
2219 }
2220 } elsif ($self->{state} == DOCTYPE_STATE) {
2221 if ($self->{next_char} == 0x0009 or # HT
2222 $self->{next_char} == 0x000A or # LF
2223 $self->{next_char} == 0x000B or # VT
2224 $self->{next_char} == 0x000C or # FF
2225 $self->{next_char} == 0x0020) { # SP
2226 !!!cp (155);
2227 $self->{state} = BEFORE_DOCTYPE_NAME_STATE;
2228 !!!next-input-character;
2229 redo A;
2230 } else {
2231 !!!cp (156);
2232 !!!parse-error (type => 'no space before DOCTYPE name');
2233 $self->{state} = BEFORE_DOCTYPE_NAME_STATE;
2234 ## reconsume
2235 redo A;
2236 }
2237 } elsif ($self->{state} == BEFORE_DOCTYPE_NAME_STATE) {
2238 if ($self->{next_char} == 0x0009 or # HT
2239 $self->{next_char} == 0x000A or # LF
2240 $self->{next_char} == 0x000B or # VT
2241 $self->{next_char} == 0x000C or # FF
2242 $self->{next_char} == 0x0020) { # SP
2243 !!!cp (157);
2244 ## Stay in the state
2245 !!!next-input-character;
2246 redo A;
2247 } elsif ($self->{next_char} == 0x003E) { # >
2248 !!!cp (158);
2249 !!!parse-error (type => 'no DOCTYPE name');
2250 $self->{state} = DATA_STATE;
2251 !!!next-input-character;
2252
2253 !!!emit ($self->{current_token}); # DOCTYPE (quirks)
2254
2255 redo A;
2256 } elsif ($self->{next_char} == -1) {
2257 !!!cp (159);
2258 !!!parse-error (type => 'no DOCTYPE name');
2259 $self->{state} = DATA_STATE;
2260 ## reconsume
2261
2262 !!!emit ($self->{current_token}); # DOCTYPE (quirks)
2263
2264 redo A;
2265 } else {
2266 !!!cp (160);
2267 $self->{current_token}->{name} = chr $self->{next_char};
2268 delete $self->{current_token}->{quirks};
2269 ## ISSUE: "Set the token's name name to the" in the spec
2270 $self->{state} = DOCTYPE_NAME_STATE;
2271 !!!next-input-character;
2272 redo A;
2273 }
2274 } elsif ($self->{state} == DOCTYPE_NAME_STATE) {
2275 ## ISSUE: Redundant "First," in the spec.
2276 if ($self->{next_char} == 0x0009 or # HT
2277 $self->{next_char} == 0x000A or # LF
2278 $self->{next_char} == 0x000B or # VT
2279 $self->{next_char} == 0x000C or # FF
2280 $self->{next_char} == 0x0020) { # SP
2281 !!!cp (161);
2282 $self->{state} = AFTER_DOCTYPE_NAME_STATE;
2283 !!!next-input-character;
2284 redo A;
2285 } elsif ($self->{next_char} == 0x003E) { # >
2286 !!!cp (162);
2287 $self->{state} = DATA_STATE;
2288 !!!next-input-character;
2289
2290 !!!emit ($self->{current_token}); # DOCTYPE
2291
2292 redo A;
2293 } elsif ($self->{next_char} == -1) {
2294 !!!cp (163);
2295 !!!parse-error (type => 'unclosed DOCTYPE');
2296 $self->{state} = DATA_STATE;
2297 ## reconsume
2298
2299 $self->{current_token}->{quirks} = 1;
2300 !!!emit ($self->{current_token}); # DOCTYPE
2301
2302 redo A;
2303 } else {
2304 !!!cp (164);
2305 $self->{current_token}->{name}
2306 .= chr ($self->{next_char}); # DOCTYPE
2307 ## Stay in the state
2308 !!!next-input-character;
2309 redo A;
2310 }
2311 } elsif ($self->{state} == AFTER_DOCTYPE_NAME_STATE) {
2312 if ($self->{next_char} == 0x0009 or # HT
2313 $self->{next_char} == 0x000A or # LF
2314 $self->{next_char} == 0x000B or # VT
2315 $self->{next_char} == 0x000C or # FF
2316 $self->{next_char} == 0x0020) { # SP
2317 !!!cp (165);
2318 ## Stay in the state
2319 !!!next-input-character;
2320 redo A;
2321 } elsif ($self->{next_char} == 0x003E) { # >
2322 !!!cp (166);
2323 $self->{state} = DATA_STATE;
2324 !!!next-input-character;
2325
2326 !!!emit ($self->{current_token}); # DOCTYPE
2327
2328 redo A;
2329 } elsif ($self->{next_char} == -1) {
2330 !!!cp (167);
2331 !!!parse-error (type => 'unclosed DOCTYPE');
2332 $self->{state} = DATA_STATE;
2333 ## reconsume
2334
2335 $self->{current_token}->{quirks} = 1;
2336 !!!emit ($self->{current_token}); # DOCTYPE
2337
2338 redo A;
2339 } elsif ($self->{next_char} == 0x0050 or # P
2340 $self->{next_char} == 0x0070) { # p
2341 !!!next-input-character;
2342 if ($self->{next_char} == 0x0055 or # U
2343 $self->{next_char} == 0x0075) { # u
2344 !!!next-input-character;
2345 if ($self->{next_char} == 0x0042 or # B
2346 $self->{next_char} == 0x0062) { # b
2347 !!!next-input-character;
2348 if ($self->{next_char} == 0x004C or # L
2349 $self->{next_char} == 0x006C) { # l
2350 !!!next-input-character;
2351 if ($self->{next_char} == 0x0049 or # I
2352 $self->{next_char} == 0x0069) { # i
2353 !!!next-input-character;
2354 if ($self->{next_char} == 0x0043 or # C
2355 $self->{next_char} == 0x0063) { # c
2356 !!!cp (168);
2357 $self->{state} = BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2358 !!!next-input-character;
2359 redo A;
2360 } else {
2361 !!!cp (169);
2362 }
2363 } else {
2364 !!!cp (170);
2365 }
2366 } else {
2367 !!!cp (171);
2368 }
2369 } else {
2370 !!!cp (172);
2371 }
2372 } else {
2373 !!!cp (173);
2374 }
2375
2376 #
2377 } elsif ($self->{next_char} == 0x0053 or # S
2378 $self->{next_char} == 0x0073) { # s
2379 !!!next-input-character;
2380 if ($self->{next_char} == 0x0059 or # Y
2381 $self->{next_char} == 0x0079) { # y
2382 !!!next-input-character;
2383 if ($self->{next_char} == 0x0053 or # S
2384 $self->{next_char} == 0x0073) { # s
2385 !!!next-input-character;
2386 if ($self->{next_char} == 0x0054 or # T
2387 $self->{next_char} == 0x0074) { # t
2388 !!!next-input-character;
2389 if ($self->{next_char} == 0x0045 or # E
2390 $self->{next_char} == 0x0065) { # e
2391 !!!next-input-character;
2392 if ($self->{next_char} == 0x004D or # M
2393 $self->{next_char} == 0x006D) { # m
2394 !!!cp (174);
2395 $self->{state} = BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2396 !!!next-input-character;
2397 redo A;
2398 } else {
2399 !!!cp (175);
2400 }
2401 } else {
2402 !!!cp (176);
2403 }
2404 } else {
2405 !!!cp (177);
2406 }
2407 } else {
2408 !!!cp (178);
2409 }
2410 } else {
2411 !!!cp (179);
2412 }
2413
2414 #
2415 } else {
2416 !!!cp (180);
2417 !!!next-input-character;
2418 #
2419 }
2420
2421 !!!parse-error (type => 'string after DOCTYPE name');
2422 $self->{current_token}->{quirks} = 1;
2423
2424 $self->{state} = BOGUS_DOCTYPE_STATE;
2425 # next-input-character is already done
2426 redo A;
2427 } elsif ($self->{state} == BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {
2428 if ({
2429 0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,
2430 #0x000D => 1, # HT, LF, VT, FF, SP, CR
2431 }->{$self->{next_char}}) {
2432 !!!cp (181);
2433 ## Stay in the state
2434 !!!next-input-character;
2435 redo A;
2436 } elsif ($self->{next_char} eq 0x0022) { # "
2437 !!!cp (182);
2438 $self->{current_token}->{public_identifier} = ''; # DOCTYPE
2439 $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE;
2440 !!!next-input-character;
2441 redo A;
2442 } elsif ($self->{next_char} eq 0x0027) { # '
2443 !!!cp (183);
2444 $self->{current_token}->{public_identifier} = ''; # DOCTYPE
2445 $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE;
2446 !!!next-input-character;
2447 redo A;
2448 } elsif ($self->{next_char} eq 0x003E) { # >
2449 !!!cp (184);
2450 !!!parse-error (type => 'no PUBLIC literal');
2451
2452 $self->{state} = DATA_STATE;
2453 !!!next-input-character;
2454
2455 $self->{current_token}->{quirks} = 1;
2456 !!!emit ($self->{current_token}); # DOCTYPE
2457
2458 redo A;
2459 } elsif ($self->{next_char} == -1) {
2460 !!!cp (185);
2461 !!!parse-error (type => 'unclosed DOCTYPE');
2462
2463 $self->{state} = DATA_STATE;
2464 ## reconsume
2465
2466 $self->{current_token}->{quirks} = 1;
2467 !!!emit ($self->{current_token}); # DOCTYPE
2468
2469 redo A;
2470 } else {
2471 !!!cp (186);
2472 !!!parse-error (type => 'string after PUBLIC');
2473 $self->{current_token}->{quirks} = 1;
2474
2475 $self->{state} = BOGUS_DOCTYPE_STATE;
2476 !!!next-input-character;
2477 redo A;
2478 }
2479 } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE) {
2480 if ($self->{next_char} == 0x0022) { # "
2481 !!!cp (187);
2482 $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2483 !!!next-input-character;
2484 redo A;
2485 } elsif ($self->{next_char} == 0x003E) { # >
2486 !!!cp (188);
2487 !!!parse-error (type => 'unclosed PUBLIC literal');
2488
2489 $self->{state} = DATA_STATE;
2490 !!!next-input-character;
2491
2492 $self->{current_token}->{quirks} = 1;
2493 !!!emit ($self->{current_token}); # DOCTYPE
2494
2495 redo A;
2496 } elsif ($self->{next_char} == -1) {
2497 !!!cp (189);
2498 !!!parse-error (type => 'unclosed PUBLIC literal');
2499
2500 $self->{state} = DATA_STATE;
2501 ## reconsume
2502
2503 $self->{current_token}->{quirks} = 1;
2504 !!!emit ($self->{current_token}); # DOCTYPE
2505
2506 redo A;
2507 } else {
2508 !!!cp (190);
2509 $self->{current_token}->{public_identifier} # DOCTYPE
2510 .= chr $self->{next_char};
2511 ## Stay in the state
2512 !!!next-input-character;
2513 redo A;
2514 }
2515 } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE) {
2516 if ($self->{next_char} == 0x0027) { # '
2517 !!!cp (191);
2518 $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2519 !!!next-input-character;
2520 redo A;
2521 } elsif ($self->{next_char} == 0x003E) { # >
2522 !!!cp (192);
2523 !!!parse-error (type => 'unclosed PUBLIC literal');
2524
2525 $self->{state} = DATA_STATE;
2526 !!!next-input-character;
2527
2528 $self->{current_token}->{quirks} = 1;
2529 !!!emit ($self->{current_token}); # DOCTYPE
2530
2531 redo A;
2532 } elsif ($self->{next_char} == -1) {
2533 !!!cp (193);
2534 !!!parse-error (type => 'unclosed PUBLIC literal');
2535
2536 $self->{state} = DATA_STATE;
2537 ## reconsume
2538
2539 $self->{current_token}->{quirks} = 1;
2540 !!!emit ($self->{current_token}); # DOCTYPE
2541
2542 redo A;
2543 } else {
2544 !!!cp (194);
2545 $self->{current_token}->{public_identifier} # DOCTYPE
2546 .= chr $self->{next_char};
2547 ## Stay in the state
2548 !!!next-input-character;
2549 redo A;
2550 }
2551 } elsif ($self->{state} == AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {
2552 if ({
2553 0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,
2554 #0x000D => 1, # HT, LF, VT, FF, SP, CR
2555 }->{$self->{next_char}}) {
2556 !!!cp (195);
2557 ## Stay in the state
2558 !!!next-input-character;
2559 redo A;
2560 } elsif ($self->{next_char} == 0x0022) { # "
2561 !!!cp (196);
2562 $self->{current_token}->{system_identifier} = ''; # DOCTYPE
2563 $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;
2564 !!!next-input-character;
2565 redo A;
2566 } elsif ($self->{next_char} == 0x0027) { # '
2567 !!!cp (197);
2568 $self->{current_token}->{system_identifier} = ''; # DOCTYPE
2569 $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;
2570 !!!next-input-character;
2571 redo A;
2572 } elsif ($self->{next_char} == 0x003E) { # >
2573 !!!cp (198);
2574 $self->{state} = DATA_STATE;
2575 !!!next-input-character;
2576
2577 !!!emit ($self->{current_token}); # DOCTYPE
2578
2579 redo A;
2580 } elsif ($self->{next_char} == -1) {
2581 !!!cp (199);
2582 !!!parse-error (type => 'unclosed DOCTYPE');
2583
2584 $self->{state} = DATA_STATE;
2585 ## reconsume
2586
2587 $self->{current_token}->{quirks} = 1;
2588 !!!emit ($self->{current_token}); # DOCTYPE
2589
2590 redo A;
2591 } else {
2592 !!!cp (200);
2593 !!!parse-error (type => 'string after PUBLIC literal');
2594 $self->{current_token}->{quirks} = 1;
2595
2596 $self->{state} = BOGUS_DOCTYPE_STATE;
2597 !!!next-input-character;
2598 redo A;
2599 }
2600 } elsif ($self->{state} == BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {
2601 if ({
2602 0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,
2603 #0x000D => 1, # HT, LF, VT, FF, SP, CR
2604 }->{$self->{next_char}}) {
2605 !!!cp (201);
2606 ## Stay in the state
2607 !!!next-input-character;
2608 redo A;
2609 } elsif ($self->{next_char} == 0x0022) { # "
2610 !!!cp (202);
2611 $self->{current_token}->{system_identifier} = ''; # DOCTYPE
2612 $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;
2613 !!!next-input-character;
2614 redo A;
2615 } elsif ($self->{next_char} == 0x0027) { # '
2616 !!!cp (203);
2617 $self->{current_token}->{system_identifier} = ''; # DOCTYPE
2618 $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;
2619 !!!next-input-character;
2620 redo A;
2621 } elsif ($self->{next_char} == 0x003E) { # >
2622 !!!cp (204);
2623 !!!parse-error (type => 'no SYSTEM literal');
2624 $self->{state} = DATA_STATE;
2625 !!!next-input-character;
2626
2627 $self->{current_token}->{quirks} = 1;
2628 !!!emit ($self->{current_token}); # DOCTYPE
2629
2630 redo A;
2631 } elsif ($self->{next_char} == -1) {
2632 !!!cp (205);
2633 !!!parse-error (type => 'unclosed DOCTYPE');
2634
2635 $self->{state} = DATA_STATE;
2636 ## reconsume
2637
2638 $self->{current_token}->{quirks} = 1;
2639 !!!emit ($self->{current_token}); # DOCTYPE
2640
2641 redo A;
2642 } else {
2643 !!!cp (206);
2644 !!!parse-error (type => 'string after SYSTEM');
2645 $self->{current_token}->{quirks} = 1;
2646
2647 $self->{state} = BOGUS_DOCTYPE_STATE;
2648 !!!next-input-character;
2649 redo A;
2650 }
2651 } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE) {
2652 if ($self->{next_char} == 0x0022) { # "
2653 !!!cp (207);
2654 $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2655 !!!next-input-character;
2656 redo A;
2657 } elsif ($self->{next_char} == 0x003E) { # >
2658 !!!cp (208);
2659 !!!parse-error (type => 'unclosed PUBLIC literal');
2660
2661 $self->{state} = DATA_STATE;
2662 !!!next-input-character;
2663
2664 $self->{current_token}->{quirks} = 1;
2665 !!!emit ($self->{current_token}); # DOCTYPE
2666
2667 redo A;
2668 } elsif ($self->{next_char} == -1) {
2669 !!!cp (209);
2670 !!!parse-error (type => 'unclosed SYSTEM literal');
2671
2672 $self->{state} = DATA_STATE;
2673 ## reconsume
2674
2675 $self->{current_token}->{quirks} = 1;
2676 !!!emit ($self->{current_token}); # DOCTYPE
2677
2678 redo A;
2679 } else {
2680 !!!cp (210);
2681 $self->{current_token}->{system_identifier} # DOCTYPE
2682 .= chr $self->{next_char};
2683 ## Stay in the state
2684 !!!next-input-character;
2685 redo A;
2686 }
2687 } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE) {
2688 if ($self->{next_char} == 0x0027) { # '
2689 !!!cp (211);
2690 $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2691 !!!next-input-character;
2692 redo A;
2693 } elsif ($self->{next_char} == 0x003E) { # >
2694 !!!cp (212);
2695 !!!parse-error (type => 'unclosed PUBLIC literal');
2696
2697 $self->{state} = DATA_STATE;
2698 !!!next-input-character;
2699
2700 $self->{current_token}->{quirks} = 1;
2701 !!!emit ($self->{current_token}); # DOCTYPE
2702
2703 redo A;
2704 } elsif ($self->{next_char} == -1) {
2705 !!!cp (213);
2706 !!!parse-error (type => 'unclosed SYSTEM literal');
2707
2708 $self->{state} = DATA_STATE;
2709 ## reconsume
2710
2711 $self->{current_token}->{quirks} = 1;
2712 !!!emit ($self->{current_token}); # DOCTYPE
2713
2714 redo A;
2715 } else {
2716 !!!cp (214);
2717 $self->{current_token}->{system_identifier} # DOCTYPE
2718 .= chr $self->{next_char};
2719 ## Stay in the state
2720 !!!next-input-character;
2721 redo A;
2722 }
2723 } elsif ($self->{state} == AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {
2724 if ({
2725 0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,
2726 #0x000D => 1, # HT, LF, VT, FF, SP, CR
2727 }->{$self->{next_char}}) {
2728 !!!cp (215);
2729 ## Stay in the state
2730 !!!next-input-character;
2731 redo A;
2732 } elsif ($self->{next_char} == 0x003E) { # >
2733 !!!cp (216);
2734 $self->{state} = DATA_STATE;
2735 !!!next-input-character;
2736
2737 !!!emit ($self->{current_token}); # DOCTYPE
2738
2739 redo A;
2740 } elsif ($self->{next_char} == -1) {
2741 !!!cp (217);
2742
2743 $self->{state} = DATA_STATE;
2744 ## reconsume
2745
2746 $self->{current_token}->{quirks} = 1;
2747 !!!emit ($self->{current_token}); # DOCTYPE
2748
2749 redo A;
2750 } else {
2751 !!!cp (218);
2752 !!!parse-error (type => 'string after SYSTEM literal');
2753 #$self->{current_token}->{quirks} = 1;
2754
2755 $self->{state} = BOGUS_DOCTYPE_STATE;
2756 !!!next-input-character;
2757 redo A;
2758 }
2759 } elsif ($self->{state} == BOGUS_DOCTYPE_STATE) {
2760 if ($self->{next_char} == 0x003E) { # >
2761 !!!cp (219);
2762 $self->{state} = DATA_STATE;
2763 !!!next-input-character;
2764
2765 !!!emit ($self->{current_token}); # DOCTYPE
2766
2767 redo A;
2768 } elsif ($self->{next_char} == -1) {
2769 !!!cp (220);
2770 !!!parse-error (type => 'unclosed DOCTYPE');
2771 $self->{state} = DATA_STATE;
2772 ## reconsume
2773
2774 !!!emit ($self->{current_token}); # DOCTYPE
2775
2776 redo A;
2777 } else {
2778 !!!cp (221);
2779 ## Stay in the state
2780 !!!next-input-character;
2781 redo A;
2782 }
2783 } elsif ($self->{state} == CDATA_BLOCK_STATE) {
2784 my $s = '';
2785
2786 my ($l, $c) = ($self->{line}, $self->{column});
2787
2788 CS: while ($self->{next_char} != -1) {
2789 if ($self->{next_char} == 0x005D) { # ]
2790 !!!next-input-character;
2791 if ($self->{next_char} == 0x005D) { # ]
2792 !!!next-input-character;
2793 MDC: {
2794 if ($self->{next_char} == 0x003E) { # >
2795 !!!cp (221.1);
2796 !!!next-input-character;
2797 last CS;
2798 } elsif ($self->{next_char} == 0x005D) { # ]
2799 !!!cp (221.2);
2800 $s .= ']';
2801 !!!next-input-character;
2802 redo MDC;
2803 } else {
2804 !!!cp (221.3);
2805 $s .= ']]';
2806 #
2807 }
2808 } # MDC
2809 } else {
2810 !!!cp (221.4);
2811 $s .= ']';
2812 #
2813 }
2814 } else {
2815 !!!cp (221.5);
2816 #
2817 }
2818 $s .= chr $self->{next_char};
2819 !!!next-input-character;
2820 } # CS
2821
2822 $self->{state} = DATA_STATE;
2823 ## next-input-character done or EOF, which is reconsumed.
2824
2825 if (length $s) {
2826 !!!cp (221.6);
2827 !!!emit ({type => CHARACTER_TOKEN, data => $s,
2828 line => $l, column => $c});
2829 } else {
2830 !!!cp (221.7);
2831 }
2832
2833 redo A;
2834
2835 ## ISSUE: "text tokens" in spec.
2836 ## TODO: Streaming support
2837 } else {
2838 die "$0: $self->{state}: Unknown state";
2839 }
2840 } # A
2841
2842 die "$0: _get_next_token: unexpected case";
2843 } # _get_next_token
2844
2845 sub _tokenize_attempt_to_consume_an_entity ($$$) {
2846 my ($self, $in_attr, $additional) = @_;
2847
2848 my ($l, $c) = ($self->{line_prev}, $self->{column_prev});
2849
2850 if ({
2851 0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, # HT, LF, VT, FF,
2852 0x0020 => 1, 0x003C => 1, 0x0026 => 1, -1 => 1, # SP, <, & # 0x000D # CR
2853 $additional => 1,
2854 }->{$self->{next_char}}) {
2855 !!!cp (1001);
2856 ## Don't consume
2857 ## No error
2858 return undef;
2859 } elsif ($self->{next_char} == 0x0023) { # #
2860 !!!next-input-character;
2861 if ($self->{next_char} == 0x0078 or # x
2862 $self->{next_char} == 0x0058) { # X
2863 my $code;
2864 X: {
2865 my $x_char = $self->{next_char};
2866 !!!next-input-character;
2867 if (0x0030 <= $self->{next_char} and
2868 $self->{next_char} <= 0x0039) { # 0..9
2869 !!!cp (1002);
2870 $code ||= 0;
2871 $code *= 0x10;
2872 $code += $self->{next_char} - 0x0030;
2873 redo X;
2874 } elsif (0x0061 <= $self->{next_char} and
2875 $self->{next_char} <= 0x0066) { # a..f
2876 !!!cp (1003);
2877 $code ||= 0;
2878 $code *= 0x10;
2879 $code += $self->{next_char} - 0x0060 + 9;
2880 redo X;
2881 } elsif (0x0041 <= $self->{next_char} and
2882 $self->{next_char} <= 0x0046) { # A..F
2883 !!!cp (1004);
2884 $code ||= 0;
2885 $code *= 0x10;
2886 $code += $self->{next_char} - 0x0040 + 9;
2887 redo X;
2888 } elsif (not defined $code) { # no hexadecimal digit
2889 !!!cp (1005);
2890 !!!parse-error (type => 'bare hcro', line => $l, column => $c);
2891 !!!back-next-input-character ($x_char, $self->{next_char});
2892 $self->{next_char} = 0x0023; # #
2893 return undef;
2894 } elsif ($self->{next_char} == 0x003B) { # ;
2895 !!!cp (1006);
2896 !!!next-input-character;
2897 } else {
2898 !!!cp (1007);
2899 !!!parse-error (type => 'no refc', line => $l, column => $c);
2900 }
2901
2902 if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) {
2903 !!!cp (1008);
2904 !!!parse-error (type => (sprintf 'invalid character reference:U+%04X', $code), line => $l, column => $c);
2905 $code = 0xFFFD;
2906 } elsif ($code > 0x10FFFF) {
2907 !!!cp (1009);
2908 !!!parse-error (type => (sprintf 'invalid character reference:U-%08X', $code), line => $l, column => $c);
2909 $code = 0xFFFD;
2910 } elsif ($code == 0x000D) {
2911 !!!cp (1010);
2912 !!!parse-error (type => 'CR character reference', line => $l, column => $c);
2913 $code = 0x000A;
2914 } elsif (0x80 <= $code and $code <= 0x9F) {
2915 !!!cp (1011);
2916 !!!parse-error (type => (sprintf 'C1 character reference:U+%04X', $code), line => $l, column => $c);
2917 $code = $c1_entity_char->{$code};
2918 }
2919
2920 return {type => CHARACTER_TOKEN, data => chr $code,
2921 has_reference => 1,
2922 line => $l, column => $c,
2923 };
2924 } # X
2925 } elsif (0x0030 <= $self->{next_char} and
2926 $self->{next_char} <= 0x0039) { # 0..9
2927 my $code = $self->{next_char} - 0x0030;
2928 !!!next-input-character;
2929
2930 while (0x0030 <= $self->{next_char} and
2931 $self->{next_char} <= 0x0039) { # 0..9
2932 !!!cp (1012);
2933 $code *= 10;
2934 $code += $self->{next_char} - 0x0030;
2935
2936 !!!next-input-character;
2937 }
2938
2939 if ($self->{next_char} == 0x003B) { # ;
2940 !!!cp (1013);
2941 !!!next-input-character;
2942 } else {
2943 !!!cp (1014);
2944 !!!parse-error (type => 'no refc', line => $l, column => $c);
2945 }
2946
2947 if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) {
2948 !!!cp (1015);
2949 !!!parse-error (type => (sprintf 'invalid character reference:U+%04X', $code), line => $l, column => $c);
2950 $code = 0xFFFD;
2951 } elsif ($code > 0x10FFFF) {
2952 !!!cp (1016);
2953 !!!parse-error (type => (sprintf 'invalid character reference:U-%08X', $code), line => $l, column => $c);
2954 $code = 0xFFFD;
2955 } elsif ($code == 0x000D) {
2956 !!!cp (1017);
2957 !!!parse-error (type => 'CR character reference', line => $l, column => $c);
2958 $code = 0x000A;
2959 } elsif (0x80 <= $code and $code <= 0x9F) {
2960 !!!cp (1018);
2961 !!!parse-error (type => (sprintf 'C1 character reference:U+%04X', $code), line => $l, column => $c);
2962 $code = $c1_entity_char->{$code};
2963 }
2964
2965 return {type => CHARACTER_TOKEN, data => chr $code, has_reference => 1,
2966 line => $l, column => $c,
2967 };
2968 } else {
2969 !!!cp (1019);
2970 !!!parse-error (type => 'bare nero', line => $l, column => $c);
2971 !!!back-next-input-character ($self->{next_char});
2972 $self->{next_char} = 0x0023; # #
2973 return undef;
2974 }
2975 } elsif ((0x0041 <= $self->{next_char} and
2976 $self->{next_char} <= 0x005A) or
2977 (0x0061 <= $self->{next_char} and
2978 $self->{next_char} <= 0x007A)) {
2979 my $entity_name = chr $self->{next_char};
2980 !!!next-input-character;
2981
2982 my $value = $entity_name;
2983 my $match = 0;
2984 require Whatpm::_NamedEntityList;
2985 our $EntityChar;
2986
2987 while (length $entity_name < 30 and
2988 ## NOTE: Some number greater than the maximum length of entity name
2989 ((0x0041 <= $self->{next_char} and # a
2990 $self->{next_char} <= 0x005A) or # x
2991 (0x0061 <= $self->{next_char} and # a
2992 $self->{next_char} <= 0x007A) or # z
2993 (0x0030 <= $self->{next_char} and # 0
2994 $self->{next_char} <= 0x0039) or # 9
2995 $self->{next_char} == 0x003B)) { # ;
2996 $entity_name .= chr $self->{next_char};
2997 if (defined $EntityChar->{$entity_name}) {
2998 if ($self->{next_char} == 0x003B) { # ;
2999 !!!cp (1020);
3000 $value = $EntityChar->{$entity_name};
3001 $match = 1;
3002 !!!next-input-character;
3003 last;
3004 } else {
3005 !!!cp (1021);
3006 $value = $EntityChar->{$entity_name};
3007 $match = -1;
3008 !!!next-input-character;
3009 }
3010 } else {
3011 !!!cp (1022);
3012 $value .= chr $self->{next_char};
3013 $match *= 2;
3014 !!!next-input-character;
3015 }
3016 }
3017
3018 if ($match > 0) {
3019 !!!cp (1023);
3020 return {type => CHARACTER_TOKEN, data => $value, has_reference => 1,
3021 line => $l, column => $c,
3022 };
3023 } elsif ($match < 0) {
3024 !!!parse-error (type => 'no refc', line => $l, column => $c);
3025 if ($in_attr and $match < -1) {
3026 !!!cp (1024);
3027 return {type => CHARACTER_TOKEN, data => '&'.$entity_name,
3028 line => $l, column => $c,
3029 };
3030 } else {
3031 !!!cp (1025);
3032 return {type => CHARACTER_TOKEN, data => $value, has_reference => 1,
3033 line => $l, column => $c,
3034 };
3035 }
3036 } else {
3037 !!!cp (1026);
3038 !!!parse-error (type => 'bare ero', line => $l, column => $c);
3039 ## NOTE: "No characters are consumed" in the spec.
3040 return {type => CHARACTER_TOKEN, data => '&'.$value,
3041 line => $l, column => $c,
3042 };
3043 }
3044 } else {
3045 !!!cp (1027);
3046 ## no characters are consumed
3047 !!!parse-error (type => 'bare ero', line => $l, column => $c);
3048 return undef;
3049 }
3050 } # _tokenize_attempt_to_consume_an_entity
3051
3052 sub _initialize_tree_constructor ($) {
3053 my $self = shift;
3054 ## NOTE: $self->{document} MUST be specified before this method is called
3055 $self->{document}->strict_error_checking (0);
3056 ## TODO: Turn mutation events off # MUST
3057 ## TODO: Turn loose Document option (manakai extension) on
3058 $self->{document}->manakai_is_html (1); # MUST
3059 } # _initialize_tree_constructor
3060
3061 sub _terminate_tree_constructor ($) {
3062 my $self = shift;
3063 $self->{document}->strict_error_checking (1);
3064 ## TODO: Turn mutation events on
3065 } # _terminate_tree_constructor
3066
3067 ## ISSUE: Should append_child (for example) in script executed in tree construction stage fire mutation events?
3068
3069 { # tree construction stage
3070 my $token;
3071
3072 sub _construct_tree ($) {
3073 my ($self) = @_;
3074
3075 ## When an interactive UA render the $self->{document} available
3076 ## to the user, or when it begin accepting user input, are
3077 ## not defined.
3078
3079 ## Append a character: collect it and all subsequent consecutive
3080 ## characters and insert one Text node whose data is concatenation
3081 ## of all those characters. # MUST
3082
3083 !!!next-token;
3084
3085 undef $self->{form_element};
3086 undef $self->{head_element};
3087 $self->{open_elements} = [];
3088 undef $self->{inner_html_node};
3089
3090 ## NOTE: The "initial" insertion mode.
3091 $self->_tree_construction_initial; # MUST
3092
3093 ## NOTE: The "before html" insertion mode.
3094 $self->_tree_construction_root_element;
3095 $self->{insertion_mode} = BEFORE_HEAD_IM;
3096
3097 ## NOTE: The "before head" insertion mode and so on.
3098 $self->_tree_construction_main;
3099 } # _construct_tree
3100
3101 sub _tree_construction_initial ($) {
3102 my $self = shift;
3103
3104 ## NOTE: "initial" insertion mode
3105
3106 INITIAL: {
3107 if ($token->{type} == DOCTYPE_TOKEN) {
3108 ## NOTE: Conformance checkers MAY, instead of reporting "not HTML5"
3109 ## error, switch to a conformance checking mode for another
3110 ## language.
3111 my $doctype_name = $token->{name};
3112 $doctype_name = '' unless defined $doctype_name;
3113 $doctype_name =~ tr/a-z/A-Z/;
3114 if (not defined $token->{name} or # <!DOCTYPE>
3115 defined $token->{public_identifier} or
3116 defined $token->{system_identifier}) {
3117 !!!cp ('t1');
3118 !!!parse-error (type => 'not HTML5', token => $token);
3119 } elsif ($doctype_name ne 'HTML') {
3120 !!!cp ('t2');
3121 ## ISSUE: ASCII case-insensitive? (in fact it does not matter)
3122 !!!parse-error (type => 'not HTML5', token => $token);
3123 } else {
3124 !!!cp ('t3');
3125 }
3126
3127 my $doctype = $self->{document}->create_document_type_definition
3128 ($token->{name}); ## ISSUE: If name is missing (e.g. <!DOCTYPE>)?
3129 ## NOTE: Default value for both |public_id| and |system_id| attributes
3130 ## are empty strings, so that we don't set any value in missing cases.
3131 $doctype->public_id ($token->{public_identifier})
3132 if defined $token->{public_identifier};
3133 $doctype->system_id ($token->{system_identifier})
3134 if defined $token->{system_identifier};
3135 ## NOTE: Other DocumentType attributes are null or empty lists.
3136 ## ISSUE: internalSubset = null??
3137 $self->{document}->append_child ($doctype);
3138
3139 if ($token->{quirks} or $doctype_name ne 'HTML') {
3140 !!!cp ('t4');
3141 $self->{document}->manakai_compat_mode ('quirks');
3142 } elsif (defined $token->{public_identifier}) {
3143 my $pubid = $token->{public_identifier};
3144 $pubid =~ tr/a-z/A-z/;
3145 my $prefix = [
3146 "+//SILMARIL//DTD HTML PRO V0R11 19970101//",
3147 "-//ADVASOFT LTD//DTD HTML 3.0 ASWEDIT + EXTENSIONS//",
3148 "-//AS//DTD HTML 3.0 ASWEDIT + EXTENSIONS//",
3149 "-//IETF//DTD HTML 2.0 LEVEL 1//",
3150 "-//IETF//DTD HTML 2.0 LEVEL 2//",
3151 "-//IETF//DTD HTML 2.0 STRICT LEVEL 1//",
3152 "-//IETF//DTD HTML 2.0 STRICT LEVEL 2//",
3153 "-//IETF//DTD HTML 2.0 STRICT//",
3154 "-//IETF//DTD HTML 2.0//",
3155 "-//IETF//DTD HTML 2.1E//",
3156 "-//IETF//DTD HTML 3.0//",
3157 "-//IETF//DTD HTML 3.2 FINAL//",
3158 "-//IETF//DTD HTML 3.2//",
3159 "-//IETF//DTD HTML 3//",
3160 "-//IETF//DTD HTML LEVEL 0//",
3161 "-//IETF//DTD HTML LEVEL 1//",
3162 "-//IETF//DTD HTML LEVEL 2//",
3163 "-//IETF//DTD HTML LEVEL 3//",
3164 "-//IETF//DTD HTML STRICT LEVEL 0//",
3165 "-//IETF//DTD HTML STRICT LEVEL 1//",
3166 "-//IETF//DTD HTML STRICT LEVEL 2//",
3167 "-//IETF//DTD HTML STRICT LEVEL 3//",
3168 "-//IETF//DTD HTML STRICT//",
3169 "-//IETF//DTD HTML//",
3170 "-//METRIUS//DTD METRIUS PRESENTATIONAL//",
3171 "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 HTML STRICT//",
3172 "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 HTML//",
3173 "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 TABLES//",
3174 "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 HTML STRICT//",
3175 "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 HTML//",
3176 "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 TABLES//",
3177 "-//NETSCAPE COMM. CORP.//DTD HTML//",
3178 "-//NETSCAPE COMM. CORP.//DTD STRICT HTML//",
3179 "-//O'REILLY AND ASSOCIATES//DTD HTML 2.0//",
3180 "-//O'REILLY AND ASSOCIATES//DTD HTML EXTENDED 1.0//",
3181 "-//O'REILLY AND ASSOCIATES//DTD HTML EXTENDED RELAXED 1.0//",
3182 "-//SOFTQUAD SOFTWARE//DTD HOTMETAL PRO 6.0::19990601::EXTENSIONS TO HTML 4.0//",
3183 "-//SOFTQUAD//DTD HOTMETAL PRO 4.0::19971010::EXTENSIONS TO HTML 4.0//",
3184 "-//SPYGLASS//DTD HTML 2.0 EXTENDED//",
3185 "-//SQ//DTD HTML 2.0 HOTMETAL + EXTENSIONS//",
3186 "-//SUN MICROSYSTEMS CORP.//DTD HOTJAVA HTML//",
3187 "-//SUN MICROSYSTEMS CORP.//DTD HOTJAVA STRICT HTML//",
3188 "-//W3C//DTD HTML 3 1995-03-24//",
3189 "-//W3C//DTD HTML 3.2 DRAFT//",
3190 "-//W3C//DTD HTML 3.2 FINAL//",
3191 "-//W3C//DTD HTML 3.2//",
3192 "-//W3C//DTD HTML 3.2S DRAFT//",
3193 "-//W3C//DTD HTML 4.0 FRAMESET//",
3194 "-//W3C//DTD HTML 4.0 TRANSITIONAL//",
3195 "-//W3C//DTD HTML EXPERIMETNAL 19960712//",
3196 "-//W3C//DTD HTML EXPERIMENTAL 970421//",
3197 "-//W3C//DTD W3 HTML//",
3198 "-//W3O//DTD W3 HTML 3.0//",
3199 "-//WEBTECHS//DTD MOZILLA HTML 2.0//",
3200 "-//WEBTECHS//DTD MOZILLA HTML//",
3201 ]; # $prefix
3202 my $match;
3203 for (@$prefix) {
3204 if (substr ($prefix, 0, length $_) eq $_) {
3205 $match = 1;
3206 last;
3207 }
3208 }
3209 if ($match or
3210 $pubid eq "-//W3O//DTD W3 HTML STRICT 3.0//EN//" or
3211 $pubid eq "-/W3C/DTD HTML 4.0 TRANSITIONAL/EN" or
3212 $pubid eq "HTML") {
3213 !!!cp ('t5');
3214 $self->{document}->manakai_compat_mode ('quirks');
3215 } elsif ($pubid =~ m[^-//W3C//DTD HTML 4.01 FRAMESET//] or
3216 $pubid =~ m[^-//W3C//DTD HTML 4.01 TRANSITIONAL//]) {
3217 if (defined $token->{system_identifier}) {
3218 !!!cp ('t6');
3219 $self->{document}->manakai_compat_mode ('quirks');
3220 } else {
3221 !!!cp ('t7');
3222 $self->{document}->manakai_compat_mode ('limited quirks');
3223 }
3224 } elsif ($pubid =~ m[^-//W3C//DTD XHTML 1.0 FRAMESET//] or
3225 $pubid =~ m[^-//W3C//DTD XHTML 1.0 TRANSITIONAL//]) {
3226 !!!cp ('t8');
3227 $self->{document}->manakai_compat_mode ('limited quirks');
3228 } else {
3229 !!!cp ('t9');
3230 }
3231 } else {
3232 !!!cp ('t10');
3233 }
3234 if (defined $token->{system_identifier}) {
3235 my $sysid = $token->{system_identifier};
3236 $sysid =~ tr/A-Z/a-z/;
3237 if ($sysid eq "http://www.ibm.com/data/dtd/v11/ibmxhtml1-transitional.dtd") {
3238 ## NOTE: Ensure that |PUBLIC "(limited quirks)" "(quirks)"| is
3239 ## marked as quirks.
3240 $self->{document}->manakai_compat_mode ('quirks');
3241 !!!cp ('t11');
3242 } else {
3243 !!!cp ('t12');
3244 }
3245 } else {
3246 !!!cp ('t13');
3247 }
3248
3249 ## Go to the "before html" insertion mode.
3250 !!!next-token;
3251 return;
3252 } elsif ({
3253 START_TAG_TOKEN, 1,
3254 END_TAG_TOKEN, 1,
3255 END_OF_FILE_TOKEN, 1,
3256 }->{$token->{type}}) {
3257 !!!cp ('t14');
3258 !!!parse-error (type => 'no DOCTYPE', token => $token);
3259 $self->{document}->manakai_compat_mode ('quirks');
3260 ## Go to the "before html" insertion mode.
3261 ## reprocess
3262 !!!ack-later;
3263 return;
3264 } elsif ($token->{type} == CHARACTER_TOKEN) {
3265 if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D
3266 ## Ignore the token
3267
3268 unless (length $token->{data}) {
3269 !!!cp ('t15');
3270 ## Stay in the insertion mode.
3271 !!!next-token;
3272 redo INITIAL;
3273 } else {
3274 !!!cp ('t16');
3275 }
3276 } else {
3277 !!!cp ('t17');
3278 }
3279
3280 !!!parse-error (type => 'no DOCTYPE', token => $token);
3281 $self->{document}->manakai_compat_mode ('quirks');
3282 ## Go to the "before html" insertion mode.
3283 ## reprocess
3284 return;
3285 } elsif ($token->{type} == COMMENT_TOKEN) {
3286 !!!cp ('t18');
3287 my $comment = $self->{document}->create_comment ($token->{data});
3288 $self->{document}->append_child ($comment);
3289
3290 ## Stay in the insertion mode.
3291 !!!next-token;
3292 redo INITIAL;
3293 } else {
3294 die "$0: $token->{type}: Unknown token type";
3295 }
3296 } # INITIAL
3297
3298 die "$0: _tree_construction_initial: This should be never reached";
3299 } # _tree_construction_initial
3300
3301 sub _tree_construction_root_element ($) {
3302 my $self = shift;
3303
3304 ## NOTE: "before html" insertion mode.
3305
3306 B: {
3307 if ($token->{type} == DOCTYPE_TOKEN) {
3308 !!!cp ('t19');
3309 !!!parse-error (type => 'in html:#DOCTYPE', token => $token);
3310 ## Ignore the token
3311 ## Stay in the insertion mode.
3312 !!!next-token;
3313 redo B;
3314 } elsif ($token->{type} == COMMENT_TOKEN) {
3315 !!!cp ('t20');
3316 my $comment = $self->{document}->create_comment ($token->{data});
3317 $self->{document}->append_child ($comment);
3318 ## Stay in the insertion mode.
3319 !!!next-token;
3320 redo B;
3321 } elsif ($token->{type} == CHARACTER_TOKEN) {
3322 if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D
3323 ## Ignore the token.
3324
3325 unless (length $token->{data}) {
3326 !!!cp ('t21');
3327 ## Stay in the insertion mode.
3328 !!!next-token;
3329 redo B;
3330 } else {
3331 !!!cp ('t22');
3332 }
3333 } else {
3334 !!!cp ('t23');
3335 }
3336
3337 $self->{application_cache_selection}->(undef);
3338
3339 #
3340 } elsif ($token->{type} == START_TAG_TOKEN) {
3341 if ($token->{tag_name} eq 'html') {
3342 my $root_element;
3343 !!!create-element ($root_element, $HTML_NS, $token->{tag_name}, $token->{attributes}, $token);
3344 $self->{document}->append_child ($root_element);
3345 push @{$self->{open_elements}},
3346 [$root_element, $el_category->{html}];
3347
3348 if ($token->{attributes}->{manifest}) {
3349 !!!cp ('t24');
3350 $self->{application_cache_selection}
3351 ->($token->{attributes}->{manifest}->{value});
3352 ## ISSUE: Spec is unclear on relative references.
3353 ## According to Hixie (#whatwg 2008-03-19), it should be
3354 ## resolved against the base URI of the document in HTML
3355 ## or xml:base of the element in XHTML.
3356 } else {
3357 !!!cp ('t25');
3358 $self->{application_cache_selection}->(undef);
3359 }
3360
3361 !!!nack ('t25c');
3362
3363 !!!next-token;
3364 return; ## Go to the "before head" insertion mode.
3365 } else {
3366 !!!cp ('t25.1');
3367 #
3368 }
3369 } elsif ({
3370 END_TAG_TOKEN, 1,
3371 END_OF_FILE_TOKEN, 1,
3372 }->{$token->{type}}) {
3373 !!!cp ('t26');
3374 #
3375 } else {
3376 die "$0: $token->{type}: Unknown token type";
3377 }
3378
3379 my $root_element;
3380 !!!create-element ($root_element, $HTML_NS, 'html',, $token);
3381 $self->{document}->append_child ($root_element);
3382 push @{$self->{open_elements}}, [$root_element, $el_category->{html}];
3383
3384 $self->{application_cache_selection}->(undef);
3385
3386 ## NOTE: Reprocess the token.
3387 !!!ack-later;
3388 return; ## Go to the "before head" insertion mode.
3389
3390 ## ISSUE: There is an issue in the spec
3391 } # B
3392
3393 die "$0: _tree_construction_root_element: This should never be reached";
3394 } # _tree_construction_root_element
3395
3396 sub _reset_insertion_mode ($) {
3397 my $self = shift;
3398
3399 ## Step 1
3400 my $last;
3401
3402 ## Step 2
3403 my $i = -1;
3404 my $node = $self->{open_elements}->[$i];
3405
3406 ## Step 3
3407 S3: {
3408 if ($self->{open_elements}->[0]->[0] eq $node->[0]) {
3409 $last = 1;
3410 if (defined $self->{inner_html_node}) {
3411 !!!cp ('t28');
3412 $node = $self->{inner_html_node};
3413 } else {
3414 die "_reset_insertion_mode: t27";
3415 }
3416 }
3417
3418 ## Step 4..14
3419 my $new_mode;
3420 if ($node->[1] & FOREIGN_EL) {
3421 !!!cp ('t28.1');
3422 ## NOTE: Strictly spaking, the line below only applies to MathML and
3423 ## SVG elements. Currently the HTML syntax supports only MathML and
3424 ## SVG elements as foreigners.
3425 $new_mode = $self->{insertion_mode} | IN_FOREIGN_CONTENT_IM;
3426 ## ISSUE: What is set as the secondary insertion mode?
3427 } elsif ($node->[1] & TABLE_CELL_EL) {
3428 if ($last) {
3429 !!!cp ('t28.2');
3430 #
3431 } else {
3432 !!!cp ('t28.3');
3433 $new_mode = IN_CELL_IM;
3434 }
3435 } else {
3436 !!!cp ('t28.4');
3437 $new_mode = {
3438 select => IN_SELECT_IM,
3439 ## NOTE: |option| and |optgroup| do not set
3440 ## insertion mode to "in select" by themselves.
3441 tr => IN_ROW_IM,
3442 tbody => IN_TABLE_BODY_IM,
3443 thead => IN_TABLE_BODY_IM,
3444 tfoot => IN_TABLE_BODY_IM,
3445 caption => IN_CAPTION_IM,
3446 colgroup => IN_COLUMN_GROUP_IM,
3447 table => IN_TABLE_IM,
3448 head => IN_BODY_IM, # not in head!
3449 body => IN_BODY_IM,
3450 frameset => IN_FRAMESET_IM,
3451 }->{$node->[0]->manakai_local_name};
3452 }
3453 $self->{insertion_mode} = $new_mode and return if defined $new_mode;
3454
3455 ## Step 15
3456 if ($node->[1] & HTML_EL) {
3457 unless (defined $self->{head_element}) {
3458 !!!cp ('t29');
3459 $self->{insertion_mode} = BEFORE_HEAD_IM;
3460 } else {
3461 ## ISSUE: Can this state be reached?
3462 !!!cp ('t30');
3463 $self->{insertion_mode} = AFTER_HEAD_IM;
3464 }
3465 return;
3466 } else {
3467 !!!cp ('t31');
3468 }
3469
3470 ## Step 16
3471 $self->{insertion_mode} = IN_BODY_IM and return if $last;
3472
3473 ## Step 17
3474 $i--;
3475 $node = $self->{open_elements}->[$i];
3476
3477 ## Step 18
3478 redo S3;
3479 } # S3
3480
3481 die "$0: _reset_insertion_mode: This line should never be reached";
3482 } # _reset_insertion_mode
3483
3484 sub _tree_construction_main ($) {
3485 my $self = shift;
3486
3487 my $active_formatting_elements = [];
3488
3489 my $reconstruct_active_formatting_elements = sub { # MUST
3490 my $insert = shift;
3491
3492 ## Step 1
3493 return unless @$active_formatting_elements;
3494
3495 ## Step 3
3496 my $i = -1;
3497 my $entry = $active_formatting_elements->[$i];
3498
3499 ## Step 2
3500 return if $entry->[0] eq '#marker';
3501 for (@{$self->{open_elements}}) {
3502 if ($entry->[0] eq $_->[0]) {
3503 !!!cp ('t32');
3504 return;
3505 }
3506 }
3507
3508 S4: {
3509 ## Step 4
3510 last S4 if $active_formatting_elements->[0]->[0] eq $entry->[0];
3511
3512 ## Step 5
3513 $i--;
3514 $entry = $active_formatting_elements->[$i];
3515
3516 ## Step 6
3517 if ($entry->[0] eq '#marker') {
3518 !!!cp ('t33_1');
3519 #
3520 } else {
3521 my $in_open_elements;
3522 OE: for (@{$self->{open_elements}}) {
3523 if ($entry->[0] eq $_->[0]) {
3524 !!!cp ('t33');
3525 $in_open_elements = 1;
3526 last OE;
3527 }
3528 }
3529 if ($in_open_elements) {
3530 !!!cp ('t34');
3531 #
3532 } else {
3533 ## NOTE: <!DOCTYPE HTML><p><b><i><u></p> <p>X
3534 !!!cp ('t35');
3535 redo S4;
3536 }
3537 }
3538
3539 ## Step 7
3540 $i++;
3541 $entry = $active_formatting_elements->[$i];
3542 } # S4
3543
3544 S7: {
3545 ## Step 8
3546 my $clone = [$entry->[0]->clone_node (0), $entry->[1]];
3547
3548 ## Step 9
3549 $insert->($clone->[0]);
3550 push @{$self->{open_elements}}, $clone;
3551
3552 ## Step 10
3553 $active_formatting_elements->[$i] = $self->{open_elements}->[-1];
3554
3555 ## Step 11
3556 unless ($clone->[0] eq $active_formatting_elements->[-1]->[0]) {
3557 !!!cp ('t36');
3558 ## Step 7'
3559 $i++;
3560 $entry = $active_formatting_elements->[$i];
3561
3562 redo S7;
3563 }
3564
3565 !!!cp ('t37');
3566 } # S7
3567 }; # $reconstruct_active_formatting_elements
3568
3569 my $clear_up_to_marker = sub {
3570 for (reverse 0..$#$active_formatting_elements) {
3571 if ($active_formatting_elements->[$_]->[0] eq '#marker') {
3572 !!!cp ('t38');
3573 splice @$active_formatting_elements, $_;
3574 return;
3575 }
3576 }
3577
3578 !!!cp ('t39');
3579 }; # $clear_up_to_marker
3580
3581 my $insert;
3582
3583 my $parse_rcdata = sub ($) {
3584 my ($content_model_flag) = @_;
3585
3586 ## Step 1
3587 my $start_tag_name = $token->{tag_name};
3588 my $el;
3589 !!!create-element ($el, $HTML_NS, $start_tag_name, $token->{attributes}, $token);
3590
3591 ## Step 2
3592 $insert->($el);
3593
3594 ## Step 3
3595 $self->{content_model} = $content_model_flag; # CDATA or RCDATA
3596 delete $self->{escape}; # MUST
3597
3598 ## Step 4
3599 my $text = '';
3600 !!!nack ('t40.1');
3601 !!!next-token;
3602 while ($token->{type} == CHARACTER_TOKEN) { # or until stop tokenizing
3603 !!!cp ('t40');
3604 $text .= $token->{data};
3605 !!!next-token;
3606 }
3607
3608 ## Step 5
3609 if (length $text) {
3610 !!!cp ('t41');
3611 my $text = $self->{document}->create_text_node ($text);
3612 $el->append_child ($text);
3613 }
3614
3615 ## Step 6
3616 $self->{content_model} = PCDATA_CONTENT_MODEL;
3617
3618 ## Step 7
3619 if ($token->{type} == END_TAG_TOKEN and
3620 $token->{tag_name} eq $start_tag_name) {
3621 !!!cp ('t42');
3622 ## Ignore the token
3623 } else {
3624 ## NOTE: An end-of-file token.
3625 if ($content_model_flag == CDATA_CONTENT_MODEL) {
3626 !!!cp ('t43');
3627 !!!parse-error (type => 'in CDATA:#'.$token->{type}, token => $token);
3628 } elsif ($content_model_flag == RCDATA_CONTENT_MODEL) {
3629 !!!cp ('t44');
3630 !!!parse-error (type => 'in RCDATA:#'.$token->{type}, token => $token);
3631 } else {
3632 die "$0: $content_model_flag in parse_rcdata";
3633 }
3634 }
3635 !!!next-token;
3636 }; # $parse_rcdata
3637
3638 my $script_start_tag = sub () {
3639 my $script_el;
3640 !!!create-element ($script_el, $HTML_NS, 'script', $token->{attributes}, $token);
3641 ## TODO: mark as "parser-inserted"
3642
3643 $self->{content_model} = CDATA_CONTENT_MODEL;
3644 delete $self->{escape}; # MUST
3645
3646 my $text = '';
3647 !!!nack ('t45.1');
3648 !!!next-token;
3649 while ($token->{type} == CHARACTER_TOKEN) {
3650 !!!cp ('t45');
3651 $text .= $token->{data};
3652 !!!next-token;
3653 } # stop if non-character token or tokenizer stops tokenising
3654 if (length $text) {
3655 !!!cp ('t46');
3656 $script_el->manakai_append_text ($text);
3657 }
3658
3659 $self->{content_model} = PCDATA_CONTENT_MODEL;
3660
3661 if ($token->{type} == END_TAG_TOKEN and
3662 $token->{tag_name} eq 'script') {
3663 !!!cp ('t47');
3664 ## Ignore the token
3665 } else {
3666 !!!cp ('t48');
3667 !!!parse-error (type => 'in CDATA:#'.$token->{type}, token => $token);
3668 ## ISSUE: And ignore?
3669 ## TODO: mark as "already executed"
3670 }
3671
3672 if (defined $self->{inner_html_node}) {
3673 !!!cp ('t49');
3674 ## TODO: mark as "already executed"
3675 } else {
3676 !!!cp ('t50');
3677 ## TODO: $old_insertion_point = current insertion point
3678 ## TODO: insertion point = just before the next input character
3679
3680 $insert->($script_el);
3681
3682 ## TODO: insertion point = $old_insertion_point (might be "undefined")
3683
3684 ## TODO: if there is a script that will execute as soon as the parser resume, then...
3685 }
3686
3687 !!!next-token;
3688 }; # $script_start_tag
3689
3690 ## NOTE: $open_tables->[-1]->[0] is the "current table" element node.
3691 ## NOTE: $open_tables->[-1]->[1] is the "tainted" flag.
3692 my $open_tables = [[$self->{open_elements}->[0]->[0]]];
3693
3694 my $formatting_end_tag = sub {
3695 my $end_tag_token = shift;
3696 my $tag_name = $end_tag_token->{tag_name};
3697
3698 ## NOTE: The adoption agency algorithm (AAA).
3699
3700 FET: {
3701 ## Step 1
3702 my $formatting_element;
3703 my $formatting_element_i_in_active;
3704 AFE: for (reverse 0..$#$active_formatting_elements) {
3705 if ($active_formatting_elements->[$_]->[0] eq '#marker') {
3706 !!!cp ('t52');
3707 last AFE;
3708 } elsif ($active_formatting_elements->[$_]->[0]->manakai_local_name
3709 eq $tag_name) {
3710 !!!cp ('t51');
3711 $formatting_element = $active_formatting_elements->[$_];
3712 $formatting_element_i_in_active = $_;
3713 last AFE;
3714 }
3715 } # AFE
3716 unless (defined $formatting_element) {
3717 !!!cp ('t53');
3718 !!!parse-error (type => 'unmatched end tag:'.$tag_name, token => $end_tag_token);
3719 ## Ignore the token
3720 !!!next-token;
3721 return;
3722 }
3723 ## has an element in scope
3724 my $in_scope = 1;
3725 my $formatting_element_i_in_open;
3726 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3727 my $node = $self->{open_elements}->[$_];
3728 if ($node->[0] eq $formatting_element->[0]) {
3729 if ($in_scope) {
3730 !!!cp ('t54');
3731 $formatting_element_i_in_open = $_;
3732 last INSCOPE;
3733 } else { # in open elements but not in scope
3734 !!!cp ('t55');
3735 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name},
3736 token => $end_tag_token);
3737 ## Ignore the token
3738 !!!next-token;
3739 return;
3740 }
3741 } elsif ($node->[1] & SCOPING_EL) {
3742 !!!cp ('t56');
3743 $in_scope = 0;
3744 }
3745 } # INSCOPE
3746 unless (defined $formatting_element_i_in_open) {
3747 !!!cp ('t57');
3748 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name},
3749 token => $end_tag_token);
3750 pop @$active_formatting_elements; # $formatting_element
3751 !!!next-token; ## TODO: ok?
3752 return;
3753 }
3754 if (not $self->{open_elements}->[-1]->[0] eq $formatting_element->[0]) {
3755 !!!cp ('t58');
3756 !!!parse-error (type => 'not closed',
3757 value => $self->{open_elements}->[-1]->[0]
3758 ->manakai_local_name,
3759 token => $end_tag_token);
3760 }
3761
3762 ## Step 2
3763 my $furthest_block;
3764 my $furthest_block_i_in_open;
3765 OE: for (reverse 0..$#{$self->{open_elements}}) {
3766 my $node = $self->{open_elements}->[$_];
3767 if (not ($node->[1] & FORMATTING_EL) and
3768 #not $phrasing_category->{$node->[1]} and
3769 ($node->[1] & SPECIAL_EL or
3770 $node->[1] & SCOPING_EL)) { ## Scoping is redundant, maybe
3771 !!!cp ('t59');
3772 $furthest_block = $node;
3773 $furthest_block_i_in_open = $_;
3774 } elsif ($node->[0] eq $formatting_element->[0]) {
3775 !!!cp ('t60');
3776 last OE;
3777 }
3778 } # OE
3779
3780 ## Step 3
3781 unless (defined $furthest_block) { # MUST
3782 !!!cp ('t61');
3783 splice @{$self->{open_elements}}, $formatting_element_i_in_open;
3784 splice @$active_formatting_elements, $formatting_element_i_in_active, 1;
3785 !!!next-token;
3786 return;
3787 }
3788
3789 ## Step 4
3790 my $common_ancestor_node = $self->{open_elements}->[$formatting_element_i_in_open - 1];
3791
3792 ## Step 5
3793 my $furthest_block_parent = $furthest_block->[0]->parent_node;
3794 if (defined $furthest_block_parent) {
3795 !!!cp ('t62');
3796 $furthest_block_parent->remove_child ($furthest_block->[0]);
3797 }
3798
3799 ## Step 6
3800 my $bookmark_prev_el
3801 = $active_formatting_elements->[$formatting_element_i_in_active - 1]
3802 ->[0];
3803
3804 ## Step 7
3805 my $node = $furthest_block;
3806 my $node_i_in_open = $furthest_block_i_in_open;
3807 my $last_node = $furthest_block;
3808 S7: {
3809 ## Step 1
3810 $node_i_in_open--;
3811 $node = $self->{open_elements}->[$node_i_in_open];
3812
3813 ## Step 2
3814 my $node_i_in_active;
3815 S7S2: {
3816 for (reverse 0..$#$active_formatting_elements) {
3817 if ($active_formatting_elements->[$_]->[0] eq $node->[0]) {
3818 !!!cp ('t63');
3819 $node_i_in_active = $_;
3820 last S7S2;
3821 }
3822 }
3823 splice @{$self->{open_elements}}, $node_i_in_open, 1;
3824 redo S7;
3825 } # S7S2
3826
3827 ## Step 3
3828 last S7 if $node->[0] eq $formatting_element->[0];
3829
3830 ## Step 4
3831 if ($last_node->[0] eq $furthest_block->[0]) {
3832 !!!cp ('t64');
3833 $bookmark_prev_el = $node->[0];
3834 }
3835
3836 ## Step 5
3837 if ($node->[0]->has_child_nodes ()) {
3838 !!!cp ('t65');
3839 my $clone = [$node->[0]->clone_node (0), $node->[1]];
3840 $active_formatting_elements->[$node_i_in_active] = $clone;
3841 $self->{open_elements}->[$node_i_in_open] = $clone;
3842 $node = $clone;
3843 }
3844
3845 ## Step 6
3846 $node->[0]->append_child ($last_node->[0]);
3847
3848 ## Step 7
3849 $last_node = $node;
3850
3851 ## Step 8
3852 redo S7;
3853 } # S7
3854
3855 ## Step 8
3856 if ($common_ancestor_node->[1] & TABLE_ROWS_EL) {
3857 my $foster_parent_element;
3858 my $next_sibling;
3859 OE: for (reverse 0..$#{$self->{open_elements}}) {
3860 if ($self->{open_elements}->[$_]->[1] & TABLE_EL) {
3861 my $parent = $self->{open_elements}->[$_]->[0]->parent_node;
3862 if (defined $parent and $parent->node_type == 1) {
3863 !!!cp ('t65.1');
3864 $foster_parent_element = $parent;
3865 $next_sibling = $self->{open_elements}->[$_]->[0];
3866 } else {
3867 !!!cp ('t65.2');
3868 $foster_parent_element
3869 = $self->{open_elements}->[$_ - 1]->[0];
3870 }
3871 last OE;
3872 }
3873 } # OE
3874 $foster_parent_element = $self->{open_elements}->[0]->[0]
3875 unless defined $foster_parent_element;
3876 $foster_parent_element->insert_before ($last_node->[0], $next_sibling);
3877 $open_tables->[-1]->[1] = 1; # tainted
3878 } else {
3879 !!!cp ('t65.3');
3880 $common_ancestor_node->[0]->append_child ($last_node->[0]);
3881 }
3882
3883 ## Step 9
3884 my $clone = [$formatting_element->[0]->clone_node (0),
3885 $formatting_element->[1]];
3886
3887 ## Step 10
3888 my @cn = @{$furthest_block->[0]->child_nodes};
3889 $clone->[0]->append_child ($_) for @cn;
3890
3891 ## Step 11
3892 $furthest_block->[0]->append_child ($clone->[0]);
3893
3894 ## Step 12
3895 my $i;
3896 AFE: for (reverse 0..$#$active_formatting_elements) {
3897 if ($active_formatting_elements->[$_]->[0] eq $formatting_element->[0]) {
3898 !!!cp ('t66');
3899 splice @$active_formatting_elements, $_, 1;
3900 $i-- and last AFE if defined $i;
3901 } elsif ($active_formatting_elements->[$_]->[0] eq $bookmark_prev_el) {
3902 !!!cp ('t67');
3903 $i = $_;
3904 }
3905 } # AFE
3906 splice @$active_formatting_elements, $i + 1, 0, $clone;
3907
3908 ## Step 13
3909 undef $i;
3910 OE: for (reverse 0..$#{$self->{open_elements}}) {
3911 if ($self->{open_elements}->[$_]->[0] eq $formatting_element->[0]) {
3912 !!!cp ('t68');
3913 splice @{$self->{open_elements}}, $_, 1;
3914 $i-- and last OE if defined $i;
3915 } elsif ($self->{open_elements}->[$_]->[0] eq $furthest_block->[0]) {
3916 !!!cp ('t69');
3917 $i = $_;
3918 }
3919 } # OE
3920 splice @{$self->{open_elements}}, $i + 1, 1, $clone;
3921
3922 ## Step 14
3923 redo FET;
3924 } # FET
3925 }; # $formatting_end_tag
3926
3927 $insert = my $insert_to_current = sub {
3928 $self->{open_elements}->[-1]->[0]->append_child ($_[0]);
3929 }; # $insert_to_current
3930
3931 my $insert_to_foster = sub {
3932 my $child = shift;
3933 if ($self->{open_elements}->[-1]->[1] & TABLE_ROWS_EL) {
3934 # MUST
3935 my $foster_parent_element;
3936 my $next_sibling;
3937 OE: for (reverse 0..$#{$self->{open_elements}}) {
3938 if ($self->{open_elements}->[$_]->[1] & TABLE_EL) {
3939 my $parent = $self->{open_elements}->[$_]->[0]->parent_node;
3940 if (defined $parent and $parent->node_type == 1) {
3941 !!!cp ('t70');
3942 $foster_parent_element = $parent;
3943 $next_sibling = $self->{open_elements}->[$_]->[0];
3944 } else {
3945 !!!cp ('t71');
3946 $foster_parent_element
3947 = $self->{open_elements}->[$_ - 1]->[0];
3948 }
3949 last OE;
3950 }
3951 } # OE
3952 $foster_parent_element = $self->{open_elements}->[0]->[0]
3953 unless defined $foster_parent_element;
3954 $foster_parent_element->insert_before
3955 ($child, $next_sibling);
3956 $open_tables->[-1]->[1] = 1; # tainted
3957 } else {
3958 !!!cp ('t72');
3959 $self->{open_elements}->[-1]->[0]->append_child ($child);
3960 }
3961 }; # $insert_to_foster
3962
3963 B: while (1) {
3964 if ($token->{type} == DOCTYPE_TOKEN) {
3965 !!!cp ('t73');
3966 !!!parse-error (type => 'DOCTYPE in the middle', token => $token);
3967 ## Ignore the token
3968 ## Stay in the phase
3969 !!!next-token;
3970 next B;
3971 } elsif ($token->{type} == START_TAG_TOKEN and
3972 $token->{tag_name} eq 'html') {
3973 if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
3974 !!!cp ('t79');
3975 !!!parse-error (type => 'after html:html', token => $token);
3976 $self->{insertion_mode} = AFTER_BODY_IM;
3977 } elsif ($self->{insertion_mode} == AFTER_HTML_FRAMESET_IM) {
3978 !!!cp ('t80');
3979 !!!parse-error (type => 'after html:html', token => $token);
3980 $self->{insertion_mode} = AFTER_FRAMESET_IM;
3981 } else {
3982 !!!cp ('t81');
3983 }
3984
3985 !!!cp ('t82');
3986 !!!parse-error (type => 'not first start tag', token => $token);
3987 my $top_el = $self->{open_elements}->[0]->[0];
3988 for my $attr_name (keys %{$token->{attributes}}) {
3989 unless ($top_el->has_attribute_ns (undef, $attr_name)) {
3990 !!!cp ('t84');
3991 $top_el->set_attribute_ns
3992 (undef, [undef, $attr_name],
3993 $token->{attributes}->{$attr_name}->{value});
3994 }
3995 }
3996 !!!nack ('t84.1');
3997 !!!next-token;
3998 next B;
3999 } elsif ($token->{type} == COMMENT_TOKEN) {
4000 my $comment = $self->{document}->create_comment ($token->{data});
4001 if ($self->{insertion_mode} & AFTER_HTML_IMS) {
4002 !!!cp ('t85');
4003 $self->{document}->append_child ($comment);
4004 } elsif ($self->{insertion_mode} == AFTER_BODY_IM) {
4005 !!!cp ('t86');
4006 $self->{open_elements}->[0]->[0]->append_child ($comment);
4007 } else {
4008 !!!cp ('t87');
4009 $self->{open_elements}->[-1]->[0]->append_child ($comment);
4010 }
4011 !!!next-token;
4012 next B;
4013 } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM) {
4014 if ($token->{type} == CHARACTER_TOKEN) {
4015 !!!cp ('t87.1');
4016 $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});
4017 !!!next-token;
4018 next B;
4019 } elsif ($token->{type} == START_TAG_TOKEN) {
4020 if ((not {mglyph => 1, malignmark => 1}->{$token->{tag_name}} and
4021 $self->{open_elements}->[-1]->[1] & FOREIGN_FLOW_CONTENT_EL) or
4022 not ($self->{open_elements}->[-1]->[1] & FOREIGN_EL) or
4023 ($token->{tag_name} eq 'svg' and
4024 $self->{open_elements}->[-1]->[1] & MML_AXML_EL)) {
4025 ## NOTE: "using the rules for secondary insertion mode"then"continue"
4026 !!!cp ('t87.2');
4027 #
4028 } elsif ({
4029 b => 1, big => 1, blockquote => 1, body => 1, br => 1,
4030 center => 1, code => 1, dd => 1, div => 1, dl => 1, em => 1,
4031 embed => 1, font => 1, h1 => 1, h2 => 1, h3 => 1, ## No h4!
4032 h5 => 1, h6 => 1, head => 1, hr => 1, i => 1, img => 1,
4033 li => 1, menu => 1, meta => 1, nobr => 1, p => 1, pre => 1,
4034 ruby => 1, s => 1, small => 1, span => 1, strong => 1,
4035 sub => 1, sup => 1, table => 1, tt => 1, u => 1, ul => 1,
4036 var => 1,
4037 }->{$token->{tag_name}}) {
4038 !!!cp ('t87.2');
4039 !!!parse-error (type => 'not closed',
4040 value => $self->{open_elements}->[-1]->[0]
4041 ->manakai_local_name,
4042 token => $token);
4043
4044 pop @{$self->{open_elements}}
4045 while $self->{open_elements}->[-1]->[1] & FOREIGN_EL;
4046
4047 $self->{insertion_mode} &= ~ IN_FOREIGN_CONTENT_IM;
4048 ## Reprocess.
4049 next B;
4050 } else {
4051 my $nsuri = $self->{open_elements}->[-1]->[0]->namespace_uri;
4052 my $tag_name = $token->{tag_name};
4053 if ($nsuri eq $SVG_NS) {
4054 $tag_name = {
4055 altglyph => 'altGlyph',
4056 altglyphdef => 'altGlyphDef',
4057 altglyphitem => 'altGlyphItem',
4058 animatecolor => 'animateColor',
4059 animatemotion => 'animateMotion',
4060 animatetransform => 'animateTransform',
4061 clippath => 'clipPath',
4062 feblend => 'feBlend',
4063 fecolormatrix => 'feColorMatrix',
4064 fecomponenttransfer => 'feComponentTransfer',
4065 fecomposite => 'feComposite',
4066 feconvolvematrix => 'feConvolveMatrix',
4067 fediffuselighting => 'feDiffuseLighting',
4068 fedisplacementmap => 'feDisplacementMap',
4069 fedistantlight => 'feDistantLight',
4070 feflood => 'feFlood',
4071 fefunca => 'feFuncA',
4072 fefuncb => 'feFuncB',
4073 fefuncg => 'feFuncG',
4074 fefuncr => 'feFuncR',
4075 fegaussianblur => 'feGaussianBlur',
4076 feimage => 'feImage',
4077 femerge => 'feMerge',
4078 femergenode => 'feMergeNode',
4079 femorphology => 'feMorphology',
4080 feoffset => 'feOffset',
4081 fepointlight => 'fePointLight',
4082 fespecularlighting => 'feSpecularLighting',
4083 fespotlight => 'feSpotLight',
4084 fetile => 'feTile',
4085 feturbulence => 'feTurbulence',
4086 foreignobject => 'foreignObject',
4087 glyphref => 'glyphRef',
4088 lineargradient => 'linearGradient',
4089 radialgradient => 'radialGradient',
4090 #solidcolor => 'solidColor', ## NOTE: Commented in spec (SVG1.2)
4091 textpath => 'textPath',
4092 }->{$tag_name} || $tag_name;
4093 }
4094
4095 ## "adjust SVG attributes" (SVG only) - done in insert-element-f
4096
4097 ## "adjust foreign attributes" - done in insert-element-f
4098
4099 !!!insert-element-f ($nsuri, $tag_name, $token->{attributes}, $token);
4100
4101 if ($self->{self_closing}) {
4102 pop @{$self->{open_elements}};
4103 !!!ack ('t87.3');
4104 } else {
4105 !!!cp ('t87.4');
4106 }
4107
4108 !!!next-token;
4109 next B;
4110 }
4111 } elsif ($token->{type} == END_TAG_TOKEN) {
4112 ## NOTE: "using the rules for secondary insertion mode" then "continue"
4113 !!!cp ('t87.5');
4114 #
4115 } elsif ($token->{type} == END_OF_FILE_TOKEN) {
4116 ## NOTE: "using the rules for secondary insertion mode" then "continue"
4117 !!!cp ('t87.6');
4118 #
4119 ## TODO: ...
4120 } else {
4121 die "$0: $token->{type}: Unknown token type";
4122 }
4123 }
4124
4125 if ($self->{insertion_mode} & HEAD_IMS) {
4126 if ($token->{type} == CHARACTER_TOKEN) {
4127 if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {
4128 unless ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4129 !!!cp ('t88.2');
4130 $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
4131 } else {
4132 !!!cp ('t88.1');
4133 ## Ignore the token.
4134 !!!next-token;
4135 next B;
4136 }
4137 unless (length $token->{data}) {
4138 !!!cp ('t88');
4139 !!!next-token;
4140 next B;
4141 }
4142 }
4143
4144 if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4145 !!!cp ('t89');
4146 ## As if <head>
4147 !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);
4148 $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
4149 push @{$self->{open_elements}},
4150 [$self->{head_element}, $el_category->{head}];
4151
4152 ## Reprocess in the "in head" insertion mode...
4153 pop @{$self->{open_elements}};
4154
4155 ## Reprocess in the "after head" insertion mode...
4156 } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4157 !!!cp ('t90');
4158 ## As if </noscript>
4159 pop @{$self->{open_elements}};
4160 !!!parse-error (type => 'in noscript:#character', token => $token);
4161
4162 ## Reprocess in the "in head" insertion mode...
4163 ## As if </head>
4164 pop @{$self->{open_elements}};
4165
4166 ## Reprocess in the "after head" insertion mode...
4167 } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
4168 !!!cp ('t91');
4169 pop @{$self->{open_elements}};
4170
4171 ## Reprocess in the "after head" insertion mode...
4172 } else {
4173 !!!cp ('t92');
4174 }
4175
4176 ## "after head" insertion mode
4177 ## As if <body>
4178 !!!insert-element ('body',, $token);
4179 $self->{insertion_mode} = IN_BODY_IM;
4180 ## reprocess
4181 next B;
4182 } elsif ($token->{type} == START_TAG_TOKEN) {
4183 if ($token->{tag_name} eq 'head') {
4184 if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4185 !!!cp ('t93');
4186 !!!create-element ($self->{head_element}, $HTML_NS, $token->{tag_name}, $token->{attributes}, $token);
4187 $self->{open_elements}->[-1]->[0]->append_child
4188 ($self->{head_element});
4189 push @{$self->{open_elements}},
4190 [$self->{head_element}, $el_category->{head}];
4191 $self->{insertion_mode} = IN_HEAD_IM;
4192 !!!nack ('t93.1');
4193 !!!next-token;
4194 next B;
4195 } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4196 !!!cp ('t93.2');
4197 !!!parse-error (type => 'after head:head', token => $token); ## TODO: error type
4198 ## Ignore the token
4199 !!!nack ('t93.3');
4200 !!!next-token;
4201 next B;
4202 } else {
4203 !!!cp ('t95');
4204 !!!parse-error (type => 'in head:head', token => $token); # or in head noscript
4205 ## Ignore the token
4206 !!!nack ('t95.1');
4207 !!!next-token;
4208 next B;
4209 }
4210 } elsif ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4211 !!!cp ('t96');
4212 ## As if <head>
4213 !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);
4214 $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
4215 push @{$self->{open_elements}},
4216 [$self->{head_element}, $el_category->{head}];
4217
4218 $self->{insertion_mode} = IN_HEAD_IM;
4219 ## Reprocess in the "in head" insertion mode...
4220 } else {
4221 !!!cp ('t97');
4222 }
4223
4224 if ($token->{tag_name} eq 'base') {
4225 if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4226 !!!cp ('t98');
4227 ## As if </noscript>
4228 pop @{$self->{open_elements}};
4229 !!!parse-error (type => 'in noscript:base', token => $token);
4230
4231 $self->{insertion_mode} = IN_HEAD_IM;
4232 ## Reprocess in the "in head" insertion mode...
4233 } else {
4234 !!!cp ('t99');
4235 }
4236
4237 ## NOTE: There is a "as if in head" code clone.
4238 if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4239 !!!cp ('t100');
4240 !!!parse-error (type => 'after head:'.$token->{tag_name}, token => $token);
4241 push @{$self->{open_elements}},
4242 [$self->{head_element}, $el_category->{head}];
4243 } else {
4244 !!!cp ('t101');
4245 }
4246 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4247 pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.
4248 pop @{$self->{open_elements}} # <head>
4249 if $self->{insertion_mode} == AFTER_HEAD_IM;
4250 !!!nack ('t101.1');
4251 !!!next-token;
4252 next B;
4253 } elsif ($token->{tag_name} eq 'link') {
4254 ## NOTE: There is a "as if in head" code clone.
4255 if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4256 !!!cp ('t102');
4257 !!!parse-error (type => 'after head:'.$token->{tag_name}, token => $token);
4258 push @{$self->{open_elements}},
4259 [$self->{head_element}, $el_category->{head}];
4260 } else {
4261 !!!cp ('t103');
4262 }
4263 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4264 pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.
4265 pop @{$self->{open_elements}} # <head>
4266 if $self->{insertion_mode} == AFTER_HEAD_IM;
4267 !!!ack ('t103.1');
4268 !!!next-token;
4269 next B;
4270 } elsif ($token->{tag_name} eq 'meta') {
4271 ## NOTE: There is a "as if in head" code clone.
4272 if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4273 !!!cp ('t104');
4274 !!!parse-error (type => 'after head:'.$token->{tag_name}, token => $token);
4275 push @{$self->{open_elements}},
4276 [$self->{head_element}, $el_category->{head}];
4277 } else {
4278 !!!cp ('t105');
4279 }
4280 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4281 my $meta_el = pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.
4282
4283 unless ($self->{confident}) {
4284 if ($token->{attributes}->{charset}) {
4285 !!!cp ('t106');
4286 ## NOTE: Whether the encoding is supported or not is handled
4287 ## in the {change_encoding} callback.
4288 $self->{change_encoding}
4289 ->($self, $token->{attributes}->{charset}->{value},
4290 $token);
4291
4292 $meta_el->[0]->get_attribute_node_ns (undef, 'charset')
4293 ->set_user_data (manakai_has_reference =>
4294 $token->{attributes}->{charset}
4295 ->{has_reference});
4296 } elsif ($token->{attributes}->{content}) {
4297 if ($token->{attributes}->{content}->{value}
4298 =~ /\A[^;]*;[\x09-\x0D\x20]*[Cc][Hh][Aa][Rr][Ss][Ee][Tt]
4299 [\x09-\x0D\x20]*=
4300 [\x09-\x0D\x20]*(?>"([^"]*)"|'([^']*)'|
4301 ([^"'\x09-\x0D\x20][^\x09-\x0D\x20]*))/x) {
4302 !!!cp ('t107');
4303 ## NOTE: Whether the encoding is supported or not is handled
4304 ## in the {change_encoding} callback.
4305 $self->{change_encoding}
4306 ->($self, defined $1 ? $1 : defined $2 ? $2 : $3,
4307 $token);
4308 $meta_el->[0]->get_attribute_node_ns (undef, 'content')
4309 ->set_user_data (manakai_has_reference =>
4310 $token->{attributes}->{content}
4311 ->{has_reference});
4312 } else {
4313 !!!cp ('t108');
4314 }
4315 }
4316 } else {
4317 if ($token->{attributes}->{charset}) {
4318 !!!cp ('t109');
4319 $meta_el->[0]->get_attribute_node_ns (undef, 'charset')
4320 ->set_user_data (manakai_has_reference =>
4321 $token->{attributes}->{charset}
4322 ->{has_reference});
4323 }
4324 if ($token->{attributes}->{content}) {
4325 !!!cp ('t110');
4326 $meta_el->[0]->get_attribute_node_ns (undef, 'content')
4327 ->set_user_data (manakai_has_reference =>
4328 $token->{attributes}->{content}
4329 ->{has_reference});
4330 }
4331 }
4332
4333 pop @{$self->{open_elements}} # <head>
4334 if $self->{insertion_mode} == AFTER_HEAD_IM;
4335 !!!ack ('t110.1');
4336 !!!next-token;
4337 next B;
4338 } elsif ($token->{tag_name} eq 'title') {
4339 if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4340 !!!cp ('t111');
4341 ## As if </noscript>
4342 pop @{$self->{open_elements}};
4343 !!!parse-error (type => 'in noscript:title', token => $token);
4344
4345 $self->{insertion_mode} = IN_HEAD_IM;
4346 ## Reprocess in the "in head" insertion mode...
4347 } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4348 !!!cp ('t112');
4349 !!!parse-error (type => 'after head:'.$token->{tag_name}, token => $token);
4350 push @{$self->{open_elements}},
4351 [$self->{head_element}, $el_category->{head}];
4352 } else {
4353 !!!cp ('t113');
4354 }
4355
4356 ## NOTE: There is a "as if in head" code clone.
4357 my $parent = defined $self->{head_element} ? $self->{head_element}
4358 : $self->{open_elements}->[-1]->[0];
4359 $parse_rcdata->(RCDATA_CONTENT_MODEL);
4360 pop @{$self->{open_elements}} # <head>
4361 if $self->{insertion_mode} == AFTER_HEAD_IM;
4362 next B;
4363 } elsif ($token->{tag_name} eq 'style') {
4364 ## NOTE: Or (scripting is enabled and tag_name eq 'noscript' and
4365 ## insertion mode IN_HEAD_IM)
4366 ## NOTE: There is a "as if in head" code clone.
4367 if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4368 !!!cp ('t114');
4369 !!!parse-error (type => 'after head:'.$token->{tag_name}, token => $token);
4370 push @{$self->{open_elements}},
4371 [$self->{head_element}, $el_category->{head}];
4372 } else {
4373 !!!cp ('t115');
4374 }
4375 $parse_rcdata->(CDATA_CONTENT_MODEL);
4376 pop @{$self->{open_elements}} # <head>
4377 if $self->{insertion_mode} == AFTER_HEAD_IM;
4378 next B;
4379 } elsif ($token->{tag_name} eq 'noscript') {
4380 if ($self->{insertion_mode} == IN_HEAD_IM) {
4381 !!!cp ('t116');
4382 ## NOTE: and scripting is disalbed
4383 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4384 $self->{insertion_mode} = IN_HEAD_NOSCRIPT_IM;
4385 !!!nack ('t116.1');
4386 !!!next-token;
4387 next B;
4388 } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4389 !!!cp ('t117');
4390 !!!parse-error (type => 'in noscript:noscript', token => $token);
4391 ## Ignore the token
4392 !!!nack ('t117.1');
4393 !!!next-token;
4394 next B;
4395 } else {
4396 !!!cp ('t118');
4397 #
4398 }
4399 } elsif ($token->{tag_name} eq 'script') {
4400 if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4401 !!!cp ('t119');
4402 ## As if </noscript>
4403 pop @{$self->{open_elements}};
4404 !!!parse-error (type => 'in noscript:script', token => $token);
4405
4406 $self->{insertion_mode} = IN_HEAD_IM;
4407 ## Reprocess in the "in head" insertion mode...
4408 } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4409 !!!cp ('t120');
4410 !!!parse-error (type => 'after head:'.$token->{tag_name}, token => $token);
4411 push @{$self->{open_elements}},
4412 [$self->{head_element}, $el_category->{head}];
4413 } else {
4414 !!!cp ('t121');
4415 }
4416
4417 ## NOTE: There is a "as if in head" code clone.
4418 $script_start_tag->();
4419 pop @{$self->{open_elements}} # <head>
4420 if $self->{insertion_mode} == AFTER_HEAD_IM;
4421 next B;
4422 } elsif ($token->{tag_name} eq 'body' or
4423 $token->{tag_name} eq 'frameset') {
4424 if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4425 !!!cp ('t122');
4426 ## As if </noscript>
4427 pop @{$self->{open_elements}};
4428 !!!parse-error (type => 'in noscript:'.$token->{tag_name}, token => $token);
4429
4430 ## Reprocess in the "in head" insertion mode...
4431 ## As if </head>
4432 pop @{$self->{open_elements}};
4433
4434 ## Reprocess in the "after head" insertion mode...
4435 } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
4436 !!!cp ('t124');
4437 pop @{$self->{open_elements}};
4438
4439 ## Reprocess in the "after head" insertion mode...
4440 } else {
4441 !!!cp ('t125');
4442 }
4443
4444 ## "after head" insertion mode
4445 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4446 if ($token->{tag_name} eq 'body') {
4447 !!!cp ('t126');
4448 $self->{insertion_mode} = IN_BODY_IM;
4449 } elsif ($token->{tag_name} eq 'frameset') {
4450 !!!cp ('t127');
4451 $self->{insertion_mode} = IN_FRAMESET_IM;
4452 } else {
4453 die "$0: tag name: $self->{tag_name}";
4454 }
4455 !!!nack ('t127.1');
4456 !!!next-token;
4457 next B;
4458 } else {
4459 !!!cp ('t128');
4460 #
4461 }
4462
4463 if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4464 !!!cp ('t129');
4465 ## As if </noscript>
4466 pop @{$self->{open_elements}};
4467 !!!parse-error (type => 'in noscript:/'.$token->{tag_name}, token => $token);
4468
4469 ## Reprocess in the "in head" insertion mode...
4470 ## As if </head>
4471 pop @{$self->{open_elements}};
4472
4473 ## Reprocess in the "after head" insertion mode...
4474 } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
4475 !!!cp ('t130');
4476 ## As if </head>
4477 pop @{$self->{open_elements}};
4478
4479 ## Reprocess in the "after head" insertion mode...
4480 } else {
4481 !!!cp ('t131');
4482 }
4483
4484 ## "after head" insertion mode
4485 ## As if <body>
4486 !!!insert-element ('body',, $token);
4487 $self->{insertion_mode} = IN_BODY_IM;
4488 ## reprocess
4489 !!!ack-later;
4490 next B;
4491 } elsif ($token->{type} == END_TAG_TOKEN) {
4492 if ($token->{tag_name} eq 'head') {
4493 if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4494 !!!cp ('t132');
4495 ## As if <head>
4496 !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);
4497 $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
4498 push @{$self->{open_elements}},
4499 [$self->{head_element}, $el_category->{head}];
4500
4501 ## Reprocess in the "in head" insertion mode...
4502 pop @{$self->{open_elements}};
4503 $self->{insertion_mode} = AFTER_HEAD_IM;
4504 !!!next-token;
4505 next B;
4506 } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4507 !!!cp ('t133');
4508 ## As if </noscript>
4509 pop @{$self->{open_elements}};
4510 !!!parse-error (type => 'in noscript:/head', token => $token);
4511
4512 ## Reprocess in the "in head" insertion mode...
4513 pop @{$self->{open_elements}};
4514 $self->{insertion_mode} = AFTER_HEAD_IM;
4515 !!!next-token;
4516 next B;
4517 } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
4518 !!!cp ('t134');
4519 pop @{$self->{open_elements}};
4520 $self->{insertion_mode} = AFTER_HEAD_IM;
4521 !!!next-token;
4522 next B;
4523 } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4524 !!!cp ('t134.1');
4525 !!!parse-error (type => 'unmatched end tag:head', token => $token);
4526 ## Ignore the token
4527 !!!next-token;
4528 next B;
4529 } else {
4530 die "$0: $self->{insertion_mode}: Unknown insertion mode";
4531 }
4532 } elsif ($token->{tag_name} eq 'noscript') {
4533 if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4534 !!!cp ('t136');
4535 pop @{$self->{open_elements}};
4536 $self->{insertion_mode} = IN_HEAD_IM;
4537 !!!next-token;
4538 next B;
4539 } elsif ($self->{insertion_mode} == BEFORE_HEAD_IM or
4540 $self->{insertion_mode} == AFTER_HEAD_IM) {
4541 !!!cp ('t137');
4542 !!!parse-error (type => 'unmatched end tag:noscript', token => $token);
4543 ## Ignore the token ## ISSUE: An issue in the spec.
4544 !!!next-token;
4545 next B;
4546 } else {
4547 !!!cp ('t138');
4548 #
4549 }
4550 } elsif ({
4551 body => 1, html => 1,
4552 }->{$token->{tag_name}}) {
4553 if ($self->{insertion_mode} == BEFORE_HEAD_IM or
4554 $self->{insertion_mode} == IN_HEAD_IM or
4555 $self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4556 !!!cp ('t140');
4557 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
4558 ## Ignore the token
4559 !!!next-token;
4560 next B;
4561 } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4562 !!!cp ('t140.1');
4563 !!!parse-error (type => 'unmatched end tag:' . $token->{tag_name}, token => $token);
4564 ## Ignore the token
4565 !!!next-token;
4566 next B;
4567 } else {
4568 die "$0: $self->{insertion_mode}: Unknown insertion mode";
4569 }
4570 } elsif ($token->{tag_name} eq 'p') {
4571 !!!cp ('t142');
4572 !!!parse-error (type => 'unmatched end tag:p', token => $token);
4573 ## Ignore the token
4574 !!!next-token;
4575 next B;
4576 } elsif ($token->{tag_name} eq 'br') {
4577 if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4578 !!!cp ('t142.2');
4579 ## (before head) as if <head>, (in head) as if </head>
4580 !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);
4581 $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
4582 $self->{insertion_mode} = AFTER_HEAD_IM;
4583
4584 ## Reprocess in the "after head" insertion mode...
4585 } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
4586 !!!cp ('t143.2');
4587 ## As if </head>
4588 pop @{$self->{open_elements}};
4589 $self->{insertion_mode} = AFTER_HEAD_IM;
4590
4591 ## Reprocess in the "after head" insertion mode...
4592 } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4593 !!!cp ('t143.3');
4594 ## ISSUE: Two parse errors for <head><noscript></br>
4595 !!!parse-error (type => 'unmatched end tag:br', token => $token);
4596 ## As if </noscript>
4597 pop @{$self->{open_elements}};
4598 $self->{insertion_mode} = IN_HEAD_IM;
4599
4600 ## Reprocess in the "in head" insertion mode...
4601 ## As if </head>
4602 pop @{$self->{open_elements}};
4603 $self->{insertion_mode} = AFTER_HEAD_IM;
4604
4605 ## Reprocess in the "after head" insertion mode...
4606 } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4607 !!!cp ('t143.4');
4608 #
4609 } else {
4610 die "$0: $self->{insertion_mode}: Unknown insertion mode";
4611 }
4612
4613 ## ISSUE: does not agree with IE7 - it doesn't ignore </br>.
4614 !!!parse-error (type => 'unmatched end tag:br', token => $token);
4615 ## Ignore the token
4616 !!!next-token;
4617 next B;
4618 } else {
4619 !!!cp ('t145');
4620 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
4621 ## Ignore the token
4622 !!!next-token;
4623 next B;
4624 }
4625
4626 if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4627 !!!cp ('t146');
4628 ## As if </noscript>
4629 pop @{$self->{open_elements}};
4630 !!!parse-error (type => 'in noscript:/'.$token->{tag_name}, token => $token);
4631
4632 ## Reprocess in the "in head" insertion mode...
4633 ## As if </head>
4634 pop @{$self->{open_elements}};
4635
4636 ## Reprocess in the "after head" insertion mode...
4637 } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
4638 !!!cp ('t147');
4639 ## As if </head>
4640 pop @{$self->{open_elements}};
4641
4642 ## Reprocess in the "after head" insertion mode...
4643 } elsif ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4644 ## ISSUE: This case cannot be reached?
4645 !!!cp ('t148');
4646 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
4647 ## Ignore the token ## ISSUE: An issue in the spec.
4648 !!!next-token;
4649 next B;
4650 } else {
4651 !!!cp ('t149');
4652 }
4653
4654 ## "after head" insertion mode
4655 ## As if <body>
4656 !!!insert-element ('body',, $token);
4657 $self->{insertion_mode} = IN_BODY_IM;
4658 ## reprocess
4659 next B;
4660 } elsif ($token->{type} == END_OF_FILE_TOKEN) {
4661 if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4662 !!!cp ('t149.1');
4663
4664 ## NOTE: As if <head>
4665 !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);
4666 $self->{open_elements}->[-1]->[0]->append_child
4667 ($self->{head_element});
4668 #push @{$self->{open_elements}},
4669 # [$self->{head_element}, $el_category->{head}];
4670 #$self->{insertion_mode} = IN_HEAD_IM;
4671 ## NOTE: Reprocess.
4672
4673 ## NOTE: As if </head>
4674 #pop @{$self->{open_elements}};
4675 #$self->{insertion_mode} = IN_AFTER_HEAD_IM;
4676 ## NOTE: Reprocess.
4677
4678 #
4679 } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
4680 !!!cp ('t149.2');
4681
4682 ## NOTE: As if </head>
4683 pop @{$self->{open_elements}};
4684 #$self->{insertion_mode} = IN_AFTER_HEAD_IM;
4685 ## NOTE: Reprocess.
4686
4687 #
4688 } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4689 !!!cp ('t149.3');
4690
4691 !!!parse-error (type => 'in noscript:#eof', token => $token);
4692
4693 ## As if </noscript>
4694 pop @{$self->{open_elements}};
4695 #$self->{insertion_mode} = IN_HEAD_IM;
4696 ## NOTE: Reprocess.
4697
4698 ## NOTE: As if </head>
4699 pop @{$self->{open_elements}};
4700 #$self->{insertion_mode} = IN_AFTER_HEAD_IM;
4701 ## NOTE: Reprocess.
4702
4703 #
4704 } else {
4705 !!!cp ('t149.4');
4706 #
4707 }
4708
4709 ## NOTE: As if <body>
4710 !!!insert-element ('body',, $token);
4711 $self->{insertion_mode} = IN_BODY_IM;
4712 ## NOTE: Reprocess.
4713 next B;
4714 } else {
4715 die "$0: $token->{type}: Unknown token type";
4716 }
4717
4718 ## ISSUE: An issue in the spec.
4719 } elsif ($self->{insertion_mode} & BODY_IMS) {
4720 if ($token->{type} == CHARACTER_TOKEN) {
4721 !!!cp ('t150');
4722 ## NOTE: There is a code clone of "character in body".
4723 $reconstruct_active_formatting_elements->($insert_to_current);
4724
4725 $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});
4726
4727 !!!next-token;
4728 next B;
4729 } elsif ($token->{type} == START_TAG_TOKEN) {
4730 if ({
4731 caption => 1, col => 1, colgroup => 1, tbody => 1,
4732 td => 1, tfoot => 1, th => 1, thead => 1, tr => 1,
4733 }->{$token->{tag_name}}) {
4734 if ($self->{insertion_mode} == IN_CELL_IM) {
4735 ## have an element in table scope
4736 for (reverse 0..$#{$self->{open_elements}}) {
4737 my $node = $self->{open_elements}->[$_];
4738 if ($node->[1] & TABLE_CELL_EL) {
4739 !!!cp ('t151');
4740
4741 ## Close the cell
4742 !!!back-token; # <x>
4743 $token = {type => END_TAG_TOKEN,
4744 tag_name => $node->[0]->manakai_local_name,
4745 line => $token->{line},
4746 column => $token->{column}};
4747 next B;
4748 } elsif ($node->[1] & TABLE_SCOPING_EL) {
4749 !!!cp ('t152');
4750 ## ISSUE: This case can never be reached, maybe.
4751 last;
4752 }
4753 }
4754
4755 !!!cp ('t153');
4756 !!!parse-error (type => 'start tag not allowed',
4757 value => $token->{tag_name}, token => $token);
4758 ## Ignore the token
4759 !!!nack ('t153.1');
4760 !!!next-token;
4761 next B;
4762 } elsif ($self->{insertion_mode} == IN_CAPTION_IM) {
4763 !!!parse-error (type => 'not closed:caption', token => $token);
4764
4765 ## NOTE: As if </caption>.
4766 ## have a table element in table scope
4767 my $i;
4768 INSCOPE: {
4769 for (reverse 0..$#{$self->{open_elements}}) {
4770 my $node = $self->{open_elements}->[$_];
4771 if ($node->[1] & CAPTION_EL) {
4772 !!!cp ('t155');
4773 $i = $_;
4774 last INSCOPE;
4775 } elsif ($node->[1] & TABLE_SCOPING_EL) {
4776 !!!cp ('t156');
4777 last;
4778 }
4779 }
4780
4781 !!!cp ('t157');
4782 !!!parse-error (type => 'start tag not allowed',
4783 value => $token->{tag_name}, token => $token);
4784 ## Ignore the token
4785 !!!nack ('t157.1');
4786 !!!next-token;
4787 next B;
4788 } # INSCOPE
4789
4790 ## generate implied end tags
4791 while ($self->{open_elements}->[-1]->[1]
4792 & END_TAG_OPTIONAL_EL) {
4793 !!!cp ('t158');
4794 pop @{$self->{open_elements}};
4795 }
4796
4797 unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {
4798 !!!cp ('t159');
4799 !!!parse-error (type => 'not closed',
4800 value => $self->{open_elements}->[-1]->[0]
4801 ->manakai_local_name,
4802 token => $token);
4803 } else {
4804 !!!cp ('t160');
4805 }
4806
4807 splice @{$self->{open_elements}}, $i;
4808
4809 $clear_up_to_marker->();
4810
4811 $self->{insertion_mode} = IN_TABLE_IM;
4812
4813 ## reprocess
4814 !!!ack-later;
4815 next B;
4816 } else {
4817 !!!cp ('t161');
4818 #
4819 }
4820 } else {
4821 !!!cp ('t162');
4822 #
4823 }
4824 } elsif ($token->{type} == END_TAG_TOKEN) {
4825 if ($token->{tag_name} eq 'td' or $token->{tag_name} eq 'th') {
4826 if ($self->{insertion_mode} == IN_CELL_IM) {
4827 ## have an element in table scope
4828 my $i;
4829 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
4830 my $node = $self->{open_elements}->[$_];
4831 if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
4832 !!!cp ('t163');
4833 $i = $_;
4834 last INSCOPE;
4835 } elsif ($node->[1] & TABLE_SCOPING_EL) {
4836 !!!cp ('t164');
4837 last INSCOPE;
4838 }
4839 } # INSCOPE
4840 unless (defined $i) {
4841 !!!cp ('t165');
4842 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
4843 ## Ignore the token
4844 !!!next-token;
4845 next B;
4846 }
4847
4848 ## generate implied end tags
4849 while ($self->{open_elements}->[-1]->[1]
4850 & END_TAG_OPTIONAL_EL) {
4851 !!!cp ('t166');
4852 pop @{$self->{open_elements}};
4853 }
4854
4855 if ($self->{open_elements}->[-1]->[0]->manakai_local_name
4856 ne $token->{tag_name}) {
4857 !!!cp ('t167');
4858 !!!parse-error (type => 'not closed',
4859 value => $self->{open_elements}->[-1]->[0]
4860 ->manakai_local_name,
4861 token => $token);
4862 } else {
4863 !!!cp ('t168');
4864 }
4865
4866 splice @{$self->{open_elements}}, $i;
4867
4868 $clear_up_to_marker->();
4869
4870 $self->{insertion_mode} = IN_ROW_IM;
4871
4872 !!!next-token;
4873 next B;
4874 } elsif ($self->{insertion_mode} == IN_CAPTION_IM) {
4875 !!!cp ('t169');
4876 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
4877 ## Ignore the token
4878 !!!next-token;
4879 next B;
4880 } else {
4881 !!!cp ('t170');
4882 #
4883 }
4884 } elsif ($token->{tag_name} eq 'caption') {
4885 if ($self->{insertion_mode} == IN_CAPTION_IM) {
4886 ## have a table element in table scope
4887 my $i;
4888 INSCOPE: {
4889 for (reverse 0..$#{$self->{open_elements}}) {
4890 my $node = $self->{open_elements}->[$_];
4891 if ($node->[1] & CAPTION_EL) {
4892 !!!cp ('t171');
4893 $i = $_;
4894 last INSCOPE;
4895 } elsif ($node->[1] & TABLE_SCOPING_EL) {
4896 !!!cp ('t172');
4897 last;
4898 }
4899 }
4900
4901 !!!cp ('t173');
4902 !!!parse-error (type => 'unmatched end tag',
4903 value => $token->{tag_name}, token => $token);
4904 ## Ignore the token
4905 !!!next-token;
4906 next B;
4907 } # INSCOPE
4908
4909 ## generate implied end tags
4910 while ($self->{open_elements}->[-1]->[1]
4911 & END_TAG_OPTIONAL_EL) {
4912 !!!cp ('t174');
4913 pop @{$self->{open_elements}};
4914 }
4915
4916 unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {
4917 !!!cp ('t175');
4918 !!!parse-error (type => 'not closed',
4919 value => $self->{open_elements}->[-1]->[0]
4920 ->manakai_local_name,
4921 token => $token);
4922 } else {
4923 !!!cp ('t176');
4924 }
4925
4926 splice @{$self->{open_elements}}, $i;
4927
4928 $clear_up_to_marker->();
4929
4930 $self->{insertion_mode} = IN_TABLE_IM;
4931
4932 !!!next-token;
4933 next B;
4934 } elsif ($self->{insertion_mode} == IN_CELL_IM) {
4935 !!!cp ('t177');
4936 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
4937 ## Ignore the token
4938 !!!next-token;
4939 next B;
4940 } else {
4941 !!!cp ('t178');
4942 #
4943 }
4944 } elsif ({
4945 table => 1, tbody => 1, tfoot => 1,
4946 thead => 1, tr => 1,
4947 }->{$token->{tag_name}} and
4948 $self->{insertion_mode} == IN_CELL_IM) {
4949 ## have an element in table scope
4950 my $i;
4951 my $tn;
4952 INSCOPE: {
4953 for (reverse 0..$#{$self->{open_elements}}) {
4954 my $node = $self->{open_elements}->[$_];
4955 if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
4956 !!!cp ('t179');
4957 $i = $_;
4958
4959 ## Close the cell
4960 !!!back-token; # </x>
4961 $token = {type => END_TAG_TOKEN, tag_name => $tn,
4962 line => $token->{line},
4963 column => $token->{column}};
4964 next B;
4965 } elsif ($node->[1] & TABLE_CELL_EL) {
4966 !!!cp ('t180');
4967 $tn = $node->[0]->manakai_local_name;
4968 ## NOTE: There is exactly one |td| or |th| element
4969 ## in scope in the stack of open elements by definition.
4970 } elsif ($node->[1] & TABLE_SCOPING_EL) {
4971 ## ISSUE: Can this be reached?
4972 !!!cp ('t181');
4973 last;
4974 }
4975 }
4976
4977 !!!cp ('t182');
4978 !!!parse-error (type => 'unmatched end tag',
4979 value => $token->{tag_name}, token => $token);
4980 ## Ignore the token
4981 !!!next-token;
4982 next B;
4983 } # INSCOPE
4984 } elsif ($token->{tag_name} eq 'table' and
4985 $self->{insertion_mode} == IN_CAPTION_IM) {
4986 !!!parse-error (type => 'not closed:caption', token => $token);
4987
4988 ## As if </caption>
4989 ## have a table element in table scope
4990 my $i;
4991 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
4992 my $node = $self->{open_elements}->[$_];
4993 if ($node->[1] & CAPTION_EL) {
4994 !!!cp ('t184');
4995 $i = $_;
4996 last INSCOPE;
4997 } elsif ($node->[1] & TABLE_SCOPING_EL) {
4998 !!!cp ('t185');
4999 last INSCOPE;
5000 }
5001 } # INSCOPE
5002 unless (defined $i) {
5003 !!!cp ('t186');
5004 !!!parse-error (type => 'unmatched end tag:caption', token => $token);
5005 ## Ignore the token
5006 !!!next-token;
5007 next B;
5008 }
5009
5010 ## generate implied end tags
5011 while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
5012 !!!cp ('t187');
5013 pop @{$self->{open_elements}};
5014 }
5015
5016 unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {
5017 !!!cp ('t188');
5018 !!!parse-error (type => 'not closed',
5019 value => $self->{open_elements}->[-1]->[0]
5020 ->manakai_local_name,
5021 token => $token);
5022 } else {
5023 !!!cp ('t189');
5024 }
5025
5026 splice @{$self->{open_elements}}, $i;
5027
5028 $clear_up_to_marker->();
5029
5030 $self->{insertion_mode} = IN_TABLE_IM;
5031
5032 ## reprocess
5033 next B;
5034 } elsif ({
5035 body => 1, col => 1, colgroup => 1, html => 1,
5036 }->{$token->{tag_name}}) {
5037 if ($self->{insertion_mode} & BODY_TABLE_IMS) {
5038 !!!cp ('t190');
5039 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
5040 ## Ignore the token
5041 !!!next-token;
5042 next B;
5043 } else {
5044 !!!cp ('t191');
5045 #
5046 }
5047 } elsif ({
5048 tbody => 1, tfoot => 1,
5049 thead => 1, tr => 1,
5050 }->{$token->{tag_name}} and
5051 $self->{insertion_mode} == IN_CAPTION_IM) {
5052 !!!cp ('t192');
5053 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
5054 ## Ignore the token
5055 !!!next-token;
5056 next B;
5057 } else {
5058 !!!cp ('t193');
5059 #
5060 }
5061 } elsif ($token->{type} == END_OF_FILE_TOKEN) {
5062 for my $entry (@{$self->{open_elements}}) {
5063 unless ($entry->[1] & ALL_END_TAG_OPTIONAL_EL) {
5064 !!!cp ('t75');
5065 !!!parse-error (type => 'in body:#eof', token => $token);
5066 last;
5067 }
5068 }
5069
5070 ## Stop parsing.
5071 last B;
5072 } else {
5073 die "$0: $token->{type}: Unknown token type";
5074 }
5075
5076 $insert = $insert_to_current;
5077 #
5078 } elsif ($self->{insertion_mode} & TABLE_IMS) {
5079 if ($token->{type} == CHARACTER_TOKEN) {
5080 if (not $open_tables->[-1]->[1] and # tainted
5081 $token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {
5082 $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
5083
5084 unless (length $token->{data}) {
5085 !!!cp ('t194');
5086 !!!next-token;
5087 next B;
5088 } else {
5089 !!!cp ('t195');
5090 }
5091 }
5092
5093 !!!parse-error (type => 'in table:#character', token => $token);
5094
5095 ## As if in body, but insert into foster parent element
5096 ## ISSUE: Spec says that "whenever a node would be inserted
5097 ## into the current node" while characters might not be
5098 ## result in a new Text node.
5099 $reconstruct_active_formatting_elements->($insert_to_foster);
5100
5101 if ($self->{open_elements}->[-1]->[1] & TABLE_ROWS_EL) {
5102 # MUST
5103 my $foster_parent_element;
5104 my $next_sibling;
5105 my $prev_sibling;
5106 OE: for (reverse 0..$#{$self->{open_elements}}) {
5107 if ($self->{open_elements}->[$_]->[1] & TABLE_EL) {
5108 my $parent = $self->{open_elements}->[$_]->[0]->parent_node;
5109 if (defined $parent and $parent->node_type == 1) {
5110 !!!cp ('t196');
5111 $foster_parent_element = $parent;
5112 $next_sibling = $self->{open_elements}->[$_]->[0];
5113 $prev_sibling = $next_sibling->previous_sibling;
5114 } else {
5115 !!!cp ('t197');
5116 $foster_parent_element = $self->{open_elements}->[$_ - 1]->[0];
5117 $prev_sibling = $foster_parent_element->last_child;
5118 }
5119 last OE;
5120 }
5121 } # OE
5122 $foster_parent_element = $self->{open_elements}->[0]->[0] and
5123 $prev_sibling = $foster_parent_element->last_child
5124 unless defined $foster_parent_element;
5125 if (defined $prev_sibling and
5126 $prev_sibling->node_type == 3) {
5127 !!!cp ('t198');
5128 $prev_sibling->manakai_append_text ($token->{data});
5129 } else {
5130 !!!cp ('t199');
5131 $foster_parent_element->insert_before
5132 ($self->{document}->create_text_node ($token->{data}),
5133 $next_sibling);
5134 }
5135 $open_tables->[-1]->[1] = 1; # tainted
5136 } else {
5137 !!!cp ('t200');
5138 $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});
5139 }
5140
5141 !!!next-token;
5142 next B;
5143 } elsif ($token->{type} == START_TAG_TOKEN) {
5144 if ({
5145 tr => ($self->{insertion_mode} != IN_ROW_IM),
5146 th => 1, td => 1,
5147 }->{$token->{tag_name}}) {
5148 if ($self->{insertion_mode} == IN_TABLE_IM) {
5149 ## Clear back to table context
5150 while (not ($self->{open_elements}->[-1]->[1]
5151 & TABLE_SCOPING_EL)) {
5152 !!!cp ('t201');
5153 pop @{$self->{open_elements}};
5154 }
5155
5156 !!!insert-element ('tbody',, $token);
5157 $self->{insertion_mode} = IN_TABLE_BODY_IM;
5158 ## reprocess in the "in table body" insertion mode...
5159 }
5160
5161 if ($self->{insertion_mode} == IN_TABLE_BODY_IM) {
5162 unless ($token->{tag_name} eq 'tr') {
5163 !!!cp ('t202');
5164 !!!parse-error (type => 'missing start tag:tr', token => $token);
5165 }
5166
5167 ## Clear back to table body context
5168 while (not ($self->{open_elements}->[-1]->[1]
5169 & TABLE_ROWS_SCOPING_EL)) {
5170 !!!cp ('t203');
5171 ## ISSUE: Can this case be reached?
5172 pop @{$self->{open_elements}};
5173 }
5174
5175 $self->{insertion_mode} = IN_ROW_IM;
5176 if ($token->{tag_name} eq 'tr') {
5177 !!!cp ('t204');
5178 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5179 !!!nack ('t204');
5180 !!!next-token;
5181 next B;
5182 } else {
5183 !!!cp ('t205');
5184 !!!insert-element ('tr',, $token);
5185 ## reprocess in the "in row" insertion mode
5186 }
5187 } else {
5188 !!!cp ('t206');
5189 }
5190
5191 ## Clear back to table row context
5192 while (not ($self->{open_elements}->[-1]->[1]
5193 & TABLE_ROW_SCOPING_EL)) {
5194 !!!cp ('t207');
5195 pop @{$self->{open_elements}};
5196 }
5197
5198 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5199 $self->{insertion_mode} = IN_CELL_IM;
5200
5201 push @$active_formatting_elements, ['#marker', ''];
5202
5203 !!!nack ('t207.1');
5204 !!!next-token;
5205 next B;
5206 } elsif ({
5207 caption => 1, col => 1, colgroup => 1,
5208 tbody => 1, tfoot => 1, thead => 1,
5209 tr => 1, # $self->{insertion_mode} == IN_ROW_IM
5210 }->{$token->{tag_name}}) {
5211 if ($self->{insertion_mode} == IN_ROW_IM) {
5212 ## As if </tr>
5213 ## have an element in table scope
5214 my $i;
5215 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5216 my $node = $self->{open_elements}->[$_];
5217 if ($node->[1] & TABLE_ROW_EL) {
5218 !!!cp ('t208');
5219 $i = $_;
5220 last INSCOPE;
5221 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5222 !!!cp ('t209');
5223 last INSCOPE;
5224 }
5225 } # INSCOPE
5226 unless (defined $i) {
5227 !!!cp ('t210');
5228 ## TODO: This type is wrong.
5229 !!!parse-error (type => 'unmacthed end tag:'.$token->{tag_name}, token => $token);
5230 ## Ignore the token
5231 !!!nack ('t210.1');
5232 !!!next-token;
5233 next B;
5234 }
5235
5236 ## Clear back to table row context
5237 while (not ($self->{open_elements}->[-1]->[1]
5238 & TABLE_ROW_SCOPING_EL)) {
5239 !!!cp ('t211');
5240 ## ISSUE: Can this case be reached?
5241 pop @{$self->{open_elements}};
5242 }
5243
5244 pop @{$self->{open_elements}}; # tr
5245 $self->{insertion_mode} = IN_TABLE_BODY_IM;
5246 if ($token->{tag_name} eq 'tr') {
5247 !!!cp ('t212');
5248 ## reprocess
5249 !!!ack-later;
5250 next B;
5251 } else {
5252 !!!cp ('t213');
5253 ## reprocess in the "in table body" insertion mode...
5254 }
5255 }
5256
5257 if ($self->{insertion_mode} == IN_TABLE_BODY_IM) {
5258 ## have an element in table scope
5259 my $i;
5260 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5261 my $node = $self->{open_elements}->[$_];
5262 if ($node->[1] & TABLE_ROW_GROUP_EL) {
5263 !!!cp ('t214');
5264 $i = $_;
5265 last INSCOPE;
5266 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5267 !!!cp ('t215');
5268 last INSCOPE;
5269 }
5270 } # INSCOPE
5271 unless (defined $i) {
5272 !!!cp ('t216');
5273 ## TODO: This erorr type ios wrong.
5274 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
5275 ## Ignore the token
5276 !!!nack ('t216.1');
5277 !!!next-token;
5278 next B;
5279 }
5280
5281 ## Clear back to table body context
5282 while (not ($self->{open_elements}->[-1]->[1]
5283 & TABLE_ROWS_SCOPING_EL)) {
5284 !!!cp ('t217');
5285 ## ISSUE: Can this state be reached?
5286 pop @{$self->{open_elements}};
5287 }
5288
5289 ## As if <{current node}>
5290 ## have an element in table scope
5291 ## true by definition
5292
5293 ## Clear back to table body context
5294 ## nop by definition
5295
5296 pop @{$self->{open_elements}};
5297 $self->{insertion_mode} = IN_TABLE_IM;
5298 ## reprocess in "in table" insertion mode...
5299 } else {
5300 !!!cp ('t218');
5301 }
5302
5303 if ($token->{tag_name} eq 'col') {
5304 ## Clear back to table context
5305 while (not ($self->{open_elements}->[-1]->[1]
5306 & TABLE_SCOPING_EL)) {
5307 !!!cp ('t219');
5308 ## ISSUE: Can this state be reached?
5309 pop @{$self->{open_elements}};
5310 }
5311
5312 !!!insert-element ('colgroup',, $token);
5313 $self->{insertion_mode} = IN_COLUMN_GROUP_IM;
5314 ## reprocess
5315 !!!ack-later;
5316 next B;
5317 } elsif ({
5318 caption => 1,
5319 colgroup => 1,
5320 tbody => 1, tfoot => 1, thead => 1,
5321 }->{$token->{tag_name}}) {
5322 ## Clear back to table context
5323 while (not ($self->{open_elements}->[-1]->[1]
5324 & TABLE_SCOPING_EL)) {
5325 !!!cp ('t220');
5326 ## ISSUE: Can this state be reached?
5327 pop @{$self->{open_elements}};
5328 }
5329
5330 push @$active_formatting_elements, ['#marker', '']
5331 if $token->{tag_name} eq 'caption';
5332
5333 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5334 $self->{insertion_mode} = {
5335 caption => IN_CAPTION_IM,
5336 colgroup => IN_COLUMN_GROUP_IM,
5337 tbody => IN_TABLE_BODY_IM,
5338 tfoot => IN_TABLE_BODY_IM,
5339 thead => IN_TABLE_BODY_IM,
5340 }->{$token->{tag_name}};
5341 !!!next-token;
5342 !!!nack ('t220.1');
5343 next B;
5344 } else {
5345 die "$0: in table: <>: $token->{tag_name}";
5346 }
5347 } elsif ($token->{tag_name} eq 'table') {
5348 !!!parse-error (type => 'not closed',
5349 value => $self->{open_elements}->[-1]->[0]
5350 ->manakai_local_name,
5351 token => $token);
5352
5353 ## As if </table>
5354 ## have a table element in table scope
5355 my $i;
5356 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5357 my $node = $self->{open_elements}->[$_];
5358 if ($node->[1] & TABLE_EL) {
5359 !!!cp ('t221');
5360 $i = $_;
5361 last INSCOPE;
5362 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5363 !!!cp ('t222');
5364 last INSCOPE;
5365 }
5366 } # INSCOPE
5367 unless (defined $i) {
5368 !!!cp ('t223');
5369 ## TODO: The following is wrong, maybe.
5370 !!!parse-error (type => 'unmatched end tag:table', token => $token);
5371 ## Ignore tokens </table><table>
5372 !!!nack ('t223.1');
5373 !!!next-token;
5374 next B;
5375 }
5376
5377 ## TODO: Followings are removed from the latest spec.
5378 ## generate implied end tags
5379 while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
5380 !!!cp ('t224');
5381 pop @{$self->{open_elements}};
5382 }
5383
5384 unless ($self->{open_elements}->[-1]->[1] & TABLE_EL) {
5385 !!!cp ('t225');
5386 ## NOTE: |<table><tr><table>|
5387 !!!parse-error (type => 'not closed',
5388 value => $self->{open_elements}->[-1]->[0]
5389 ->manakai_local_name,
5390 token => $token);
5391 } else {
5392 !!!cp ('t226');
5393 }
5394
5395 splice @{$self->{open_elements}}, $i;
5396 pop @{$open_tables};
5397
5398 $self->_reset_insertion_mode;
5399
5400 ## reprocess
5401 !!!ack-later;
5402 next B;
5403 } elsif ($token->{tag_name} eq 'style') {
5404 if (not $open_tables->[-1]->[1]) { # tainted
5405 !!!cp ('t227.8');
5406 ## NOTE: This is a "as if in head" code clone.
5407 $parse_rcdata->(CDATA_CONTENT_MODEL);
5408 next B;
5409 } else {
5410 !!!cp ('t227.7');
5411 #
5412 }
5413 } elsif ($token->{tag_name} eq 'script') {
5414 if (not $open_tables->[-1]->[1]) { # tainted
5415 !!!cp ('t227.6');
5416 ## NOTE: This is a "as if in head" code clone.
5417 $script_start_tag->();
5418 next B;
5419 } else {
5420 !!!cp ('t227.5');
5421 #
5422 }
5423 } elsif ($token->{tag_name} eq 'input') {
5424 if (not $open_tables->[-1]->[1]) { # tainted
5425 if ($token->{attributes}->{type}) { ## TODO: case
5426 my $type = lc $token->{attributes}->{type}->{value};
5427 if ($type eq 'hidden') {
5428 !!!cp ('t227.3');
5429 !!!parse-error (type => 'in table:'.$token->{tag_name}, token => $token);
5430
5431 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5432
5433 ## TODO: form element pointer
5434
5435 pop @{$self->{open_elements}};
5436
5437 !!!next-token;
5438 !!!ack ('t227.2.1');
5439 next B;
5440 } else {
5441 !!!cp ('t227.2');
5442 #
5443 }
5444 } else {
5445 !!!cp ('t227.1');
5446 #
5447 }
5448 } else {
5449 !!!cp ('t227.4');
5450 #
5451 }
5452 } else {
5453 !!!cp ('t227');
5454 #
5455 }
5456
5457 !!!parse-error (type => 'in table:'.$token->{tag_name}, token => $token);
5458
5459 $insert = $insert_to_foster;
5460 #
5461 } elsif ($token->{type} == END_TAG_TOKEN) {
5462 if ($token->{tag_name} eq 'tr' and
5463 $self->{insertion_mode} == IN_ROW_IM) {
5464 ## have an element in table scope
5465 my $i;
5466 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5467 my $node = $self->{open_elements}->[$_];
5468 if ($node->[1] & TABLE_ROW_EL) {
5469 !!!cp ('t228');
5470 $i = $_;
5471 last INSCOPE;
5472 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5473 !!!cp ('t229');
5474 last INSCOPE;
5475 }
5476 } # INSCOPE
5477 unless (defined $i) {
5478 !!!cp ('t230');
5479 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
5480 ## Ignore the token
5481 !!!nack ('t230.1');
5482 !!!next-token;
5483 next B;
5484 } else {
5485 !!!cp ('t232');
5486 }
5487
5488 ## Clear back to table row context
5489 while (not ($self->{open_elements}->[-1]->[1]
5490 & TABLE_ROW_SCOPING_EL)) {
5491 !!!cp ('t231');
5492 ## ISSUE: Can this state be reached?
5493 pop @{$self->{open_elements}};
5494 }
5495
5496 pop @{$self->{open_elements}}; # tr
5497 $self->{insertion_mode} = IN_TABLE_BODY_IM;
5498 !!!next-token;
5499 !!!nack ('t231.1');
5500 next B;
5501 } elsif ($token->{tag_name} eq 'table') {
5502 if ($self->{insertion_mode} == IN_ROW_IM) {
5503 ## As if </tr>
5504 ## have an element in table scope
5505 my $i;
5506 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5507 my $node = $self->{open_elements}->[$_];
5508 if ($node->[1] & TABLE_ROW_EL) {
5509 !!!cp ('t233');
5510 $i = $_;
5511 last INSCOPE;
5512 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5513 !!!cp ('t234');
5514 last INSCOPE;
5515 }
5516 } # INSCOPE
5517 unless (defined $i) {
5518 !!!cp ('t235');
5519 ## TODO: The following is wrong.
5520 !!!parse-error (type => 'unmatched end tag:'.$token->{type}, token => $token);
5521 ## Ignore the token
5522 !!!nack ('t236.1');
5523 !!!next-token;
5524 next B;
5525 }
5526
5527 ## Clear back to table row context
5528 while (not ($self->{open_elements}->[-1]->[1]
5529 & TABLE_ROW_SCOPING_EL)) {
5530 !!!cp ('t236');
5531 ## ISSUE: Can this state be reached?
5532 pop @{$self->{open_elements}};
5533 }
5534
5535 pop @{$self->{open_elements}}; # tr
5536 $self->{insertion_mode} = IN_TABLE_BODY_IM;
5537 ## reprocess in the "in table body" insertion mode...
5538 }
5539
5540 if ($self->{insertion_mode} == IN_TABLE_BODY_IM) {
5541 ## have an element in table scope
5542 my $i;
5543 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5544 my $node = $self->{open_elements}->[$_];
5545 if ($node->[1] & TABLE_ROW_GROUP_EL) {
5546 !!!cp ('t237');
5547 $i = $_;
5548 last INSCOPE;
5549 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5550 !!!cp ('t238');
5551 last INSCOPE;
5552 }
5553 } # INSCOPE
5554 unless (defined $i) {
5555 !!!cp ('t239');
5556 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
5557 ## Ignore the token
5558 !!!nack ('t239.1');
5559 !!!next-token;
5560 next B;
5561 }
5562
5563 ## Clear back to table body context
5564 while (not ($self->{open_elements}->[-1]->[1]
5565 & TABLE_ROWS_SCOPING_EL)) {
5566 !!!cp ('t240');
5567 pop @{$self->{open_elements}};
5568 }
5569
5570 ## As if <{current node}>
5571 ## have an element in table scope
5572 ## true by definition
5573
5574 ## Clear back to table body context
5575 ## nop by definition
5576
5577 pop @{$self->{open_elements}};
5578 $self->{insertion_mode} = IN_TABLE_IM;
5579 ## reprocess in the "in table" insertion mode...
5580 }
5581
5582 ## NOTE: </table> in the "in table" insertion mode.
5583 ## When you edit the code fragment below, please ensure that
5584 ## the code for <table> in the "in table" insertion mode
5585 ## is synced with it.
5586
5587 ## have a table element in table scope
5588 my $i;
5589 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5590 my $node = $self->{open_elements}->[$_];
5591 if ($node->[1] & TABLE_EL) {
5592 !!!cp ('t241');
5593 $i = $_;
5594 last INSCOPE;
5595 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5596 !!!cp ('t242');
5597 last INSCOPE;
5598 }
5599 } # INSCOPE
5600 unless (defined $i) {
5601 !!!cp ('t243');
5602 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
5603 ## Ignore the token
5604 !!!nack ('t243.1');
5605 !!!next-token;
5606 next B;
5607 }
5608
5609 splice @{$self->{open_elements}}, $i;
5610 pop @{$open_tables};
5611
5612 $self->_reset_insertion_mode;
5613
5614 !!!next-token;
5615 next B;
5616 } elsif ({
5617 tbody => 1, tfoot => 1, thead => 1,
5618 }->{$token->{tag_name}} and
5619 $self->{insertion_mode} & ROW_IMS) {
5620 if ($self->{insertion_mode} == IN_ROW_IM) {
5621 ## have an element in table scope
5622 my $i;
5623 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5624 my $node = $self->{open_elements}->[$_];
5625 if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
5626 !!!cp ('t247');
5627 $i = $_;
5628 last INSCOPE;
5629 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5630 !!!cp ('t248');
5631 last INSCOPE;
5632 }
5633 } # INSCOPE
5634 unless (defined $i) {
5635 !!!cp ('t249');
5636 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
5637 ## Ignore the token
5638 !!!nack ('t249.1');
5639 !!!next-token;
5640 next B;
5641 }
5642
5643 ## As if </tr>
5644 ## have an element in table scope
5645 my $i;
5646 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5647 my $node = $self->{open_elements}->[$_];
5648 if ($node->[1] & TABLE_ROW_EL) {
5649 !!!cp ('t250');
5650 $i = $_;
5651 last INSCOPE;
5652 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5653 !!!cp ('t251');
5654 last INSCOPE;
5655 }
5656 } # INSCOPE
5657 unless (defined $i) {
5658 !!!cp ('t252');
5659 !!!parse-error (type => 'unmatched end tag:tr', token => $token);
5660 ## Ignore the token
5661 !!!nack ('t252.1');
5662 !!!next-token;
5663 next B;
5664 }
5665
5666 ## Clear back to table row context
5667 while (not ($self->{open_elements}->[-1]->[1]
5668 & TABLE_ROW_SCOPING_EL)) {
5669 !!!cp ('t253');
5670 ## ISSUE: Can this case be reached?
5671 pop @{$self->{open_elements}};
5672 }
5673
5674 pop @{$self->{open_elements}}; # tr
5675 $self->{insertion_mode} = IN_TABLE_BODY_IM;
5676 ## reprocess in the "in table body" insertion mode...
5677 }
5678
5679 ## have an element in table scope
5680 my $i;
5681 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5682 my $node = $self->{open_elements}->[$_];
5683 if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
5684 !!!cp ('t254');
5685 $i = $_;
5686 last INSCOPE;
5687 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5688 !!!cp ('t255');
5689 last INSCOPE;
5690 }
5691 } # INSCOPE
5692 unless (defined $i) {
5693 !!!cp ('t256');
5694 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
5695 ## Ignore the token
5696 !!!nack ('t256.1');
5697 !!!next-token;
5698 next B;
5699 }
5700
5701 ## Clear back to table body context
5702 while (not ($self->{open_elements}->[-1]->[1]
5703 & TABLE_ROWS_SCOPING_EL)) {
5704 !!!cp ('t257');
5705 ## ISSUE: Can this case be reached?
5706 pop @{$self->{open_elements}};
5707 }
5708
5709 pop @{$self->{open_elements}};
5710 $self->{insertion_mode} = IN_TABLE_IM;
5711 !!!nack ('t257.1');
5712 !!!next-token;
5713 next B;
5714 } elsif ({
5715 body => 1, caption => 1, col => 1, colgroup => 1,
5716 html => 1, td => 1, th => 1,
5717 tr => 1, # $self->{insertion_mode} == IN_ROW_IM
5718 tbody => 1, tfoot => 1, thead => 1, # $self->{insertion_mode} == IN_TABLE_IM
5719 }->{$token->{tag_name}}) {
5720 !!!cp ('t258');
5721 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
5722 ## Ignore the token
5723 !!!nack ('t258.1');
5724 !!!next-token;
5725 next B;
5726 } else {
5727 !!!cp ('t259');
5728 !!!parse-error (type => 'in table:/'.$token->{tag_name}, token => $token);
5729
5730 $insert = $insert_to_foster;
5731 #
5732 }
5733 } elsif ($token->{type} == END_OF_FILE_TOKEN) {
5734 unless ($self->{open_elements}->[-1]->[1] & HTML_EL and
5735 @{$self->{open_elements}} == 1) { # redundant, maybe
5736 !!!parse-error (type => 'in body:#eof', token => $token);
5737 !!!cp ('t259.1');
5738 #
5739 } else {
5740 !!!cp ('t259.2');
5741 #
5742 }
5743
5744 ## Stop parsing
5745 last B;
5746 } else {
5747 die "$0: $token->{type}: Unknown token type";
5748 }
5749 } elsif ($self->{insertion_mode} == IN_COLUMN_GROUP_IM) {
5750 if ($token->{type} == CHARACTER_TOKEN) {
5751 if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {
5752 $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
5753 unless (length $token->{data}) {
5754 !!!cp ('t260');
5755 !!!next-token;
5756 next B;
5757 }
5758 }
5759
5760 !!!cp ('t261');
5761 #
5762 } elsif ($token->{type} == START_TAG_TOKEN) {
5763 if ($token->{tag_name} eq 'col') {
5764 !!!cp ('t262');
5765 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5766 pop @{$self->{open_elements}};
5767 !!!ack ('t262.1');
5768 !!!next-token;
5769 next B;
5770 } else {
5771 !!!cp ('t263');
5772 #
5773 }
5774 } elsif ($token->{type} == END_TAG_TOKEN) {
5775 if ($token->{tag_name} eq 'colgroup') {
5776 if ($self->{open_elements}->[-1]->[1] & HTML_EL) {
5777 !!!cp ('t264');
5778 !!!parse-error (type => 'unmatched end tag:colgroup', token => $token);
5779 ## Ignore the token
5780 !!!next-token;
5781 next B;
5782 } else {
5783 !!!cp ('t265');
5784 pop @{$self->{open_elements}}; # colgroup
5785 $self->{insertion_mode} = IN_TABLE_IM;
5786 !!!next-token;
5787 next B;
5788 }
5789 } elsif ($token->{tag_name} eq 'col') {
5790 !!!cp ('t266');
5791 !!!parse-error (type => 'unmatched end tag:col', token => $token);
5792 ## Ignore the token
5793 !!!next-token;
5794 next B;
5795 } else {
5796 !!!cp ('t267');
5797 #
5798 }
5799 } elsif ($token->{type} == END_OF_FILE_TOKEN) {
5800 if ($self->{open_elements}->[-1]->[1] & HTML_EL and
5801 @{$self->{open_elements}} == 1) { # redundant, maybe
5802 !!!cp ('t270.2');
5803 ## Stop parsing.
5804 last B;
5805 } else {
5806 ## NOTE: As if </colgroup>.
5807 !!!cp ('t270.1');
5808 pop @{$self->{open_elements}}; # colgroup
5809 $self->{insertion_mode} = IN_TABLE_IM;
5810 ## Reprocess.
5811 next B;
5812 }
5813 } else {
5814 die "$0: $token->{type}: Unknown token type";
5815 }
5816
5817 ## As if </colgroup>
5818 if ($self->{open_elements}->[-1]->[1] & HTML_EL) {
5819 !!!cp ('t269');
5820 ## TODO: Wrong error type?
5821 !!!parse-error (type => 'unmatched end tag:colgroup', token => $token);
5822 ## Ignore the token
5823 !!!nack ('t269.1');
5824 !!!next-token;
5825 next B;
5826 } else {
5827 !!!cp ('t270');
5828 pop @{$self->{open_elements}}; # colgroup
5829 $self->{insertion_mode} = IN_TABLE_IM;
5830 !!!ack-later;
5831 ## reprocess
5832 next B;
5833 }
5834 } elsif ($self->{insertion_mode} & SELECT_IMS) {
5835 if ($token->{type} == CHARACTER_TOKEN) {
5836 !!!cp ('t271');
5837 $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});
5838 !!!next-token;
5839 next B;
5840 } elsif ($token->{type} == START_TAG_TOKEN) {
5841 if ($token->{tag_name} eq 'option') {
5842 if ($self->{open_elements}->[-1]->[1] & OPTION_EL) {
5843 !!!cp ('t272');
5844 ## As if </option>
5845 pop @{$self->{open_elements}};
5846 } else {
5847 !!!cp ('t273');
5848 }
5849
5850 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5851 !!!nack ('t273.1');
5852 !!!next-token;
5853 next B;
5854 } elsif ($token->{tag_name} eq 'optgroup') {
5855 if ($self->{open_elements}->[-1]->[1] & OPTION_EL) {
5856 !!!cp ('t274');
5857 ## As if </option>
5858 pop @{$self->{open_elements}};
5859 } else {
5860 !!!cp ('t275');
5861 }
5862
5863 if ($self->{open_elements}->[-1]->[1] & OPTGROUP_EL) {
5864 !!!cp ('t276');
5865 ## As if </optgroup>
5866 pop @{$self->{open_elements}};
5867 } else {
5868 !!!cp ('t277');
5869 }
5870
5871 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5872 !!!nack ('t277.1');
5873 !!!next-token;
5874 next B;
5875 } elsif ($token->{tag_name} eq 'select' or
5876 $token->{tag_name} eq 'input' or
5877 ($self->{insertion_mode} == IN_SELECT_IN_TABLE_IM and
5878 {
5879 caption => 1, table => 1,
5880 tbody => 1, tfoot => 1, thead => 1,
5881 tr => 1, td => 1, th => 1,
5882 }->{$token->{tag_name}})) {
5883 ## TODO: The type below is not good - <select> is replaced by </select>
5884 !!!parse-error (type => 'not closed:select', token => $token);
5885 ## NOTE: As if the token were </select> (<select> case) or
5886 ## as if there were </select> (otherwise).
5887 ## have an element in table scope
5888 my $i;
5889 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5890 my $node = $self->{open_elements}->[$_];
5891 if ($node->[1] & SELECT_EL) {
5892 !!!cp ('t278');
5893 $i = $_;
5894 last INSCOPE;
5895 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5896 !!!cp ('t279');
5897 last INSCOPE;
5898 }
5899 } # INSCOPE
5900 unless (defined $i) {
5901 !!!cp ('t280');
5902 !!!parse-error (type => 'unmatched end tag:select', token => $token);
5903 ## Ignore the token
5904 !!!nack ('t280.1');
5905 !!!next-token;
5906 next B;
5907 }
5908
5909 !!!cp ('t281');
5910 splice @{$self->{open_elements}}, $i;
5911
5912 $self->_reset_insertion_mode;
5913
5914 if ($token->{tag_name} eq 'select') {
5915 !!!nack ('t281.2');
5916 !!!next-token;
5917 next B;
5918 } else {
5919 !!!cp ('t281.1');
5920 !!!ack-later;
5921 ## Reprocess the token.
5922 next B;
5923 }
5924 } else {
5925 !!!cp ('t282');
5926 !!!parse-error (type => 'in select:'.$token->{tag_name}, token => $token);
5927 ## Ignore the token
5928 !!!nack ('t282.1');
5929 !!!next-token;
5930 next B;
5931 }
5932 } elsif ($token->{type} == END_TAG_TOKEN) {
5933 if ($token->{tag_name} eq 'optgroup') {
5934 if ($self->{open_elements}->[-1]->[1] & OPTION_EL and
5935 $self->{open_elements}->[-2]->[1] & OPTGROUP_EL) {
5936 !!!cp ('t283');
5937 ## As if </option>
5938 splice @{$self->{open_elements}}, -2;
5939 } elsif ($self->{open_elements}->[-1]->[1] & OPTGROUP_EL) {
5940 !!!cp ('t284');
5941 pop @{$self->{open_elements}};
5942 } else {
5943 !!!cp ('t285');
5944 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
5945 ## Ignore the token
5946 }
5947 !!!nack ('t285.1');
5948 !!!next-token;
5949 next B;
5950 } elsif ($token->{tag_name} eq 'option') {
5951 if ($self->{open_elements}->[-1]->[1] & OPTION_EL) {
5952 !!!cp ('t286');
5953 pop @{$self->{open_elements}};
5954 } else {
5955 !!!cp ('t287');
5956 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
5957 ## Ignore the token
5958 }
5959 !!!nack ('t287.1');
5960 !!!next-token;
5961 next B;
5962 } elsif ($token->{tag_name} eq 'select') {
5963 ## have an element in table scope
5964 my $i;
5965 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5966 my $node = $self->{open_elements}->[$_];
5967 if ($node->[1] & SELECT_EL) {
5968 !!!cp ('t288');
5969 $i = $_;
5970 last INSCOPE;
5971 } elsif ($node->[1] & TABLE_SCOPING_EL) {
5972 !!!cp ('t289');
5973 last INSCOPE;
5974 }
5975 } # INSCOPE
5976 unless (defined $i) {
5977 !!!cp ('t290');
5978 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
5979 ## Ignore the token
5980 !!!nack ('t290.1');
5981 !!!next-token;
5982 next B;
5983 }
5984
5985 !!!cp ('t291');
5986 splice @{$self->{open_elements}}, $i;
5987
5988 $self->_reset_insertion_mode;
5989
5990 !!!nack ('t291.1');
5991 !!!next-token;
5992 next B;
5993 } elsif ($self->{insertion_mode} == IN_SELECT_IN_TABLE_IM and
5994 {
5995 caption => 1, table => 1, tbody => 1,
5996 tfoot => 1, thead => 1, tr => 1, td => 1, th => 1,
5997 }->{$token->{tag_name}}) {
5998 ## TODO: The following is wrong?
5999 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
6000
6001 ## have an element in table scope
6002 my $i;
6003 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6004 my $node = $self->{open_elements}->[$_];
6005 if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
6006 !!!cp ('t292');
6007 $i = $_;
6008 last INSCOPE;
6009 } elsif ($node->[1] & TABLE_SCOPING_EL) {
6010 !!!cp ('t293');
6011 last INSCOPE;
6012 }
6013 } # INSCOPE
6014 unless (defined $i) {
6015 !!!cp ('t294');
6016 ## Ignore the token
6017 !!!nack ('t294.1');
6018 !!!next-token;
6019 next B;
6020 }
6021
6022 ## As if </select>
6023 ## have an element in table scope
6024 undef $i;
6025 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6026 my $node = $self->{open_elements}->[$_];
6027 if ($node->[1] & SELECT_EL) {
6028 !!!cp ('t295');
6029 $i = $_;
6030 last INSCOPE;
6031 } elsif ($node->[1] & TABLE_SCOPING_EL) {
6032 ## ISSUE: Can this state be reached?
6033 !!!cp ('t296');
6034 last INSCOPE;
6035 }
6036 } # INSCOPE
6037 unless (defined $i) {
6038 !!!cp ('t297');
6039 ## TODO: The following error type is correct?
6040 !!!parse-error (type => 'unmatched end tag:select', token => $token);
6041 ## Ignore the </select> token
6042 !!!nack ('t297.1');
6043 !!!next-token; ## TODO: ok?
6044 next B;
6045 }
6046
6047 !!!cp ('t298');
6048 splice @{$self->{open_elements}}, $i;
6049
6050 $self->_reset_insertion_mode;
6051
6052 !!!ack-later;
6053 ## reprocess
6054 next B;
6055 } else {
6056 !!!cp ('t299');
6057 !!!parse-error (type => 'in select:/'.$token->{tag_name}, token => $token);
6058 ## Ignore the token
6059 !!!nack ('t299.3');
6060 !!!next-token;
6061 next B;
6062 }
6063 } elsif ($token->{type} == END_OF_FILE_TOKEN) {
6064 unless ($self->{open_elements}->[-1]->[1] & HTML_EL and
6065 @{$self->{open_elements}} == 1) { # redundant, maybe
6066 !!!cp ('t299.1');
6067 !!!parse-error (type => 'in body:#eof', token => $token);
6068 } else {
6069 !!!cp ('t299.2');
6070 }
6071
6072 ## Stop parsing.
6073 last B;
6074 } else {
6075 die "$0: $token->{type}: Unknown token type";
6076 }
6077 } elsif ($self->{insertion_mode} & BODY_AFTER_IMS) {
6078 if ($token->{type} == CHARACTER_TOKEN) {
6079 if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {
6080 my $data = $1;
6081 ## As if in body
6082 $reconstruct_active_formatting_elements->($insert_to_current);
6083
6084 $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
6085
6086 unless (length $token->{data}) {
6087 !!!cp ('t300');
6088 !!!next-token;
6089 next B;
6090 }
6091 }
6092
6093 if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
6094 !!!cp ('t301');
6095 !!!parse-error (type => 'after html:#character', token => $token);
6096
6097 ## Reprocess in the "after body" insertion mode.
6098 } else {
6099 !!!cp ('t302');
6100 }
6101
6102 ## "after body" insertion mode
6103 !!!parse-error (type => 'after body:#character', token => $token);
6104
6105 $self->{insertion_mode} = IN_BODY_IM;
6106 ## reprocess
6107 next B;
6108 } elsif ($token->{type} == START_TAG_TOKEN) {
6109 if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
6110 !!!cp ('t303');
6111 !!!parse-error (type => 'after html:'.$token->{tag_name}, token => $token);
6112
6113 ## Reprocess in the "after body" insertion mode.
6114 } else {
6115 !!!cp ('t304');
6116 }
6117
6118 ## "after body" insertion mode
6119 !!!parse-error (type => 'after body:'.$token->{tag_name}, token => $token);
6120
6121 $self->{insertion_mode} = IN_BODY_IM;
6122 !!!ack-later;
6123 ## reprocess
6124 next B;
6125 } elsif ($token->{type} == END_TAG_TOKEN) {
6126 if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
6127 !!!cp ('t305');
6128 !!!parse-error (type => 'after html:/'.$token->{tag_name}, token => $token);
6129
6130 $self->{insertion_mode} = AFTER_BODY_IM;
6131 ## Reprocess in the "after body" insertion mode.
6132 } else {
6133 !!!cp ('t306');
6134 }
6135
6136 ## "after body" insertion mode
6137 if ($token->{tag_name} eq 'html') {
6138 if (defined $self->{inner_html_node}) {
6139 !!!cp ('t307');
6140 !!!parse-error (type => 'unmatched end tag:html', token => $token);
6141 ## Ignore the token
6142 !!!next-token;
6143 next B;
6144 } else {
6145 !!!cp ('t308');
6146 $self->{insertion_mode} = AFTER_HTML_BODY_IM;
6147 !!!next-token;
6148 next B;
6149 }
6150 } else {
6151 !!!cp ('t309');
6152 !!!parse-error (type => 'after body:/'.$token->{tag_name}, token => $token);
6153
6154 $self->{insertion_mode} = IN_BODY_IM;
6155 ## reprocess
6156 next B;
6157 }
6158 } elsif ($token->{type} == END_OF_FILE_TOKEN) {
6159 !!!cp ('t309.2');
6160 ## Stop parsing
6161 last B;
6162 } else {
6163 die "$0: $token->{type}: Unknown token type";
6164 }
6165 } elsif ($self->{insertion_mode} & FRAME_IMS) {
6166 if ($token->{type} == CHARACTER_TOKEN) {
6167 if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {
6168 $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
6169
6170 unless (length $token->{data}) {
6171 !!!cp ('t310');
6172 !!!next-token;
6173 next B;
6174 }
6175 }
6176
6177 if ($token->{data} =~ s/^[^\x09\x0A\x0B\x0C\x20]+//) {
6178 if ($self->{insertion_mode} == IN_FRAMESET_IM) {
6179 !!!cp ('t311');
6180 !!!parse-error (type => 'in frameset:#character', token => $token);
6181 } elsif ($self->{insertion_mode} == AFTER_FRAMESET_IM) {
6182 !!!cp ('t312');
6183 !!!parse-error (type => 'after frameset:#character', token => $token);
6184 } else { # "after html frameset"
6185 !!!cp ('t313');
6186 !!!parse-error (type => 'after html:#character', token => $token);
6187
6188 $self->{insertion_mode} = AFTER_FRAMESET_IM;
6189 ## Reprocess in the "after frameset" insertion mode.
6190 !!!parse-error (type => 'after frameset:#character', token => $token);
6191 }
6192
6193 ## Ignore the token.
6194 if (length $token->{data}) {
6195 !!!cp ('t314');
6196 ## reprocess the rest of characters
6197 } else {
6198 !!!cp ('t315');
6199 !!!next-token;
6200 }
6201 next B;
6202 }
6203
6204 die qq[$0: Character "$token->{data}"];
6205 } elsif ($token->{type} == START_TAG_TOKEN) {
6206 if ($self->{insertion_mode} == AFTER_HTML_FRAMESET_IM) {
6207 !!!cp ('t316');
6208 !!!parse-error (type => 'after html:'.$token->{tag_name}, token => $token);
6209
6210 $self->{insertion_mode} = AFTER_FRAMESET_IM;
6211 ## Process in the "after frameset" insertion mode.
6212 } else {
6213 !!!cp ('t317');
6214 }
6215
6216 if ($token->{tag_name} eq 'frameset' and
6217 $self->{insertion_mode} == IN_FRAMESET_IM) {
6218 !!!cp ('t318');
6219 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
6220 !!!nack ('t318.1');
6221 !!!next-token;
6222 next B;
6223 } elsif ($token->{tag_name} eq 'frame' and
6224 $self->{insertion_mode} == IN_FRAMESET_IM) {
6225 !!!cp ('t319');
6226 !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
6227 pop @{$self->{open_elements}};
6228 !!!ack ('t319.1');
6229 !!!next-token;
6230 next B;
6231 } elsif ($token->{tag_name} eq 'noframes') {
6232 !!!cp ('t320');
6233 ## NOTE: As if in body.
6234 $parse_rcdata->(CDATA_CONTENT_MODEL);
6235 next B;
6236 } else {
6237 if ($self->{insertion_mode} == IN_FRAMESET_IM) {
6238 !!!cp ('t321');
6239 !!!parse-error (type => 'in frameset:'.$token->{tag_name}, token => $token);
6240 } else {
6241 !!!cp ('t322');
6242 !!!parse-error (type => 'after frameset:'.$token->{tag_name}, token => $token);
6243 }
6244 ## Ignore the token
6245 !!!nack ('t322.1');
6246 !!!next-token;
6247 next B;
6248 }
6249 } elsif ($token->{type} == END_TAG_TOKEN) {
6250 if ($self->{insertion_mode} == AFTER_HTML_FRAMESET_IM) {
6251 !!!cp ('t323');
6252 !!!parse-error (type => 'after html:/'.$token->{tag_name}, token => $token);
6253
6254 $self->{insertion_mode} = AFTER_FRAMESET_IM;
6255 ## Process in the "after frameset" insertion mode.
6256 } else {
6257 !!!cp ('t324');
6258 }
6259
6260 if ($token->{tag_name} eq 'frameset' and
6261 $self->{insertion_mode} == IN_FRAMESET_IM) {
6262 if ($self->{open_elements}->[-1]->[1] & HTML_EL and
6263 @{$self->{open_elements}} == 1) {
6264 !!!cp ('t325');
6265 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
6266 ## Ignore the token
6267 !!!next-token;
6268 } else {
6269 !!!cp ('t326');
6270 pop @{$self->{open_elements}};
6271 !!!next-token;
6272 }
6273
6274 if (not defined $self->{inner_html_node} and
6275 not ($self->{open_elements}->[-1]->[1] & FRAMESET_EL)) {
6276 !!!cp ('t327');
6277 $self->{insertion_mode} = AFTER_FRAMESET_IM;
6278 } else {
6279 !!!cp ('t328');
6280 }
6281 next B;
6282 } elsif ($token->{tag_name} eq 'html' and
6283 $self->{insertion_mode} == AFTER_FRAMESET_IM) {
6284 !!!cp ('t329');
6285 $self->{insertion_mode} = AFTER_HTML_FRAMESET_IM;
6286 !!!next-token;
6287 next B;
6288 } else {
6289 if ($self->{insertion_mode} == IN_FRAMESET_IM) {
6290 !!!cp ('t330');
6291 !!!parse-error (type => 'in frameset:/'.$token->{tag_name}, token => $token);
6292 } else {
6293 !!!cp ('t331');
6294 !!!parse-error (type => 'after frameset:/'.$token->{tag_name}, token => $token);
6295 }
6296 ## Ignore the token
6297 !!!next-token;
6298 next B;
6299 }
6300 } elsif ($token->{type} == END_OF_FILE_TOKEN) {
6301 unless ($self->{open_elements}->[-1]->[1] & HTML_EL and
6302 @{$self->{open_elements}} == 1) { # redundant, maybe
6303 !!!cp ('t331.1');
6304 !!!parse-error (type => 'in body:#eof', token => $token);
6305 } else {
6306 !!!cp ('t331.2');
6307 }
6308
6309 ## Stop parsing
6310 last B;
6311 } else {
6312 die "$0: $token->{type}: Unknown token type";
6313 }
6314
6315 ## ISSUE: An issue in spec here
6316 } else {
6317 die "$0: $self->{insertion_mode}: Unknown insertion mode";
6318 }
6319
6320 ## "in body" insertion mode
6321 if ($token->{type} == START_TAG_TOKEN) {
6322 if ($token->{tag_name} eq 'script') {
6323 !!!cp ('t332');
6324 ## NOTE: This is an "as if in head" code clone
6325 $script_start_tag->();
6326 next B;
6327 } elsif ($token->{tag_name} eq 'style') {
6328 !!!cp ('t333');
6329 ## NOTE: This is an "as if in head" code clone
6330 $parse_rcdata->(CDATA_CONTENT_MODEL);
6331 next B;
6332 } elsif ({
6333 base => 1, link => 1,
6334 }->{$token->{tag_name}}) {
6335 !!!cp ('t334');
6336 ## NOTE: This is an "as if in head" code clone, only "-t" differs
6337 !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
6338 pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.
6339 !!!ack ('t334.1');
6340 !!!next-token;
6341 next B;
6342 } elsif ($token->{tag_name} eq 'meta') {
6343 ## NOTE: This is an "as if in head" code clone, only "-t" differs
6344 !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
6345 my $meta_el = pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.
6346
6347 unless ($self->{confident}) {
6348 if ($token->{attributes}->{charset}) {
6349 !!!cp ('t335');
6350 ## NOTE: Whether the encoding is supported or not is handled
6351 ## in the {change_encoding} callback.
6352 $self->{change_encoding}
6353 ->($self, $token->{attributes}->{charset}->{value}, $token);
6354
6355 $meta_el->[0]->get_attribute_node_ns (undef, 'charset')
6356 ->set_user_data (manakai_has_reference =>
6357 $token->{attributes}->{charset}
6358 ->{has_reference});
6359 } elsif ($token->{attributes}->{content}) {
6360 if ($token->{attributes}->{content}->{value}
6361 =~ /\A[^;]*;[\x09-\x0D\x20]*[Cc][Hh][Aa][Rr][Ss][Ee][Tt]
6362 [\x09-\x0D\x20]*=
6363 [\x09-\x0D\x20]*(?>"([^"]*)"|'([^']*)'|
6364 ([^"'\x09-\x0D\x20][^\x09-\x0D\x20]*))/x) {
6365 !!!cp ('t336');
6366 ## NOTE: Whether the encoding is supported or not is handled
6367 ## in the {change_encoding} callback.
6368 $self->{change_encoding}
6369 ->($self, defined $1 ? $1 : defined $2 ? $2 : $3, $token);
6370 $meta_el->[0]->get_attribute_node_ns (undef, 'content')
6371 ->set_user_data (manakai_has_reference =>
6372 $token->{attributes}->{content}
6373 ->{has_reference});
6374 }
6375 }
6376 } else {
6377 if ($token->{attributes}->{charset}) {
6378 !!!cp ('t337');
6379 $meta_el->[0]->get_attribute_node_ns (undef, 'charset')
6380 ->set_user_data (manakai_has_reference =>
6381 $token->{attributes}->{charset}
6382 ->{has_reference});
6383 }
6384 if ($token->{attributes}->{content}) {
6385 !!!cp ('t338');
6386 $meta_el->[0]->get_attribute_node_ns (undef, 'content')
6387 ->set_user_data (manakai_has_reference =>
6388 $token->{attributes}->{content}
6389 ->{has_reference});
6390 }
6391 }
6392
6393 !!!ack ('t338.1');
6394 !!!next-token;
6395 next B;
6396 } elsif ($token->{tag_name} eq 'title') {
6397 !!!cp ('t341');
6398 ## NOTE: This is an "as if in head" code clone
6399 $parse_rcdata->(RCDATA_CONTENT_MODEL);
6400 next B;
6401 } elsif ($token->{tag_name} eq 'body') {
6402 !!!parse-error (type => 'in body:body', token => $token);
6403
6404 if (@{$self->{open_elements}} == 1 or
6405 not ($self->{open_elements}->[1]->[1] & BODY_EL)) {
6406 !!!cp ('t342');
6407 ## Ignore the token
6408 } else {
6409 my $body_el = $self->{open_elements}->[1]->[0];
6410 for my $attr_name (keys %{$token->{attributes}}) {
6411 unless ($body_el->has_attribute_ns (undef, $attr_name)) {
6412 !!!cp ('t343');
6413 $body_el->set_attribute_ns
6414 (undef, [undef, $attr_name],
6415 $token->{attributes}->{$attr_name}->{value});
6416 }
6417 }
6418 }
6419 !!!nack ('t343.1');
6420 !!!next-token;
6421 next B;
6422 } elsif ({
6423 address => 1, blockquote => 1, center => 1, dir => 1,
6424 div => 1, dl => 1, fieldset => 1,
6425 h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,
6426 menu => 1, ol => 1, p => 1, ul => 1,
6427 pre => 1, listing => 1,
6428 form => 1,
6429 table => 1,
6430 hr => 1,
6431 }->{$token->{tag_name}}) {
6432 if ($token->{tag_name} eq 'form' and defined $self->{form_element}) {
6433 !!!cp ('t350');
6434 !!!parse-error (type => 'in form:form', token => $token);
6435 ## Ignore the token
6436 !!!nack ('t350.1');
6437 !!!next-token;
6438 next B;
6439 }
6440
6441 ## has a p element in scope
6442 INSCOPE: for (reverse @{$self->{open_elements}}) {
6443 if ($_->[1] & P_EL) {
6444 !!!cp ('t344');
6445 !!!back-token; # <form>
6446 $token = {type => END_TAG_TOKEN, tag_name => 'p',
6447 line => $token->{line}, column => $token->{column}};
6448 next B;
6449 } elsif ($_->[1] & SCOPING_EL) {
6450 !!!cp ('t345');
6451 last INSCOPE;
6452 }
6453 } # INSCOPE
6454
6455 !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
6456 if ($token->{tag_name} eq 'pre' or $token->{tag_name} eq 'listing') {
6457 !!!nack ('t346.1');
6458 !!!next-token;
6459 if ($token->{type} == CHARACTER_TOKEN) {
6460 $token->{data} =~ s/^\x0A//;
6461 unless (length $token->{data}) {
6462 !!!cp ('t346');
6463 !!!next-token;
6464 } else {
6465 !!!cp ('t349');
6466 }
6467 } else {
6468 !!!cp ('t348');
6469 }
6470 } elsif ($token->{tag_name} eq 'form') {
6471 !!!cp ('t347.1');
6472 $self->{form_element} = $self->{open_elements}->[-1]->[0];
6473
6474 !!!nack ('t347.2');
6475 !!!next-token;
6476 } elsif ($token->{tag_name} eq 'table') {
6477 !!!cp ('t382');
6478 push @{$open_tables}, [$self->{open_elements}->[-1]->[0]];
6479
6480 $self->{insertion_mode} = IN_TABLE_IM;
6481
6482 !!!nack ('t382.1');
6483 !!!next-token;
6484 } elsif ($token->{tag_name} eq 'hr') {
6485 !!!cp ('t386');
6486 pop @{$self->{open_elements}};
6487
6488 !!!nack ('t386.1');
6489 !!!next-token;
6490 } else {
6491 !!!nack ('t347.1');
6492 !!!next-token;
6493 }
6494 next B;
6495 } elsif ({li => 1, dt => 1, dd => 1}->{$token->{tag_name}}) {
6496 ## has a p element in scope
6497 INSCOPE: for (reverse @{$self->{open_elements}}) {
6498 if ($_->[1] & P_EL) {
6499 !!!cp ('t353');
6500 !!!back-token; # <x>
6501 $token = {type => END_TAG_TOKEN, tag_name => 'p',
6502 line => $token->{line}, column => $token->{column}};
6503 next B;
6504 } elsif ($_->[1] & SCOPING_EL) {
6505 !!!cp ('t354');
6506 last INSCOPE;
6507 }
6508 } # INSCOPE
6509
6510 ## Step 1
6511 my $i = -1;
6512 my $node = $self->{open_elements}->[$i];
6513 my $li_or_dtdd = {li => {li => 1},
6514 dt => {dt => 1, dd => 1},
6515 dd => {dt => 1, dd => 1}}->{$token->{tag_name}};
6516 LI: {
6517 ## Step 2
6518 if ($li_or_dtdd->{$node->[0]->manakai_local_name}) {
6519 if ($i != -1) {
6520 !!!cp ('t355');
6521 !!!parse-error (type => 'not closed',
6522 value => $self->{open_elements}->[-1]->[0]
6523 ->manakai_local_name,
6524 token => $token);
6525 } else {
6526 !!!cp ('t356');
6527 }
6528 splice @{$self->{open_elements}}, $i;
6529 last LI;
6530 } else {
6531 !!!cp ('t357');
6532 }
6533
6534 ## Step 3
6535 if (not ($node->[1] & FORMATTING_EL) and
6536 #not $phrasing_category->{$node->[1]} and
6537 ($node->[1] & SPECIAL_EL or
6538 $node->[1] & SCOPING_EL) and
6539 not ($node->[1] & ADDRESS_EL) and
6540 not ($node->[1] & DIV_EL)) {
6541 !!!cp ('t358');
6542 last LI;
6543 }
6544
6545 !!!cp ('t359');
6546 ## Step 4
6547 $i--;
6548 $node = $self->{open_elements}->[$i];
6549 redo LI;
6550 } # LI
6551
6552 !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
6553 !!!nack ('t359.1');
6554 !!!next-token;
6555 next B;
6556 } elsif ($token->{tag_name} eq 'plaintext') {
6557 ## has a p element in scope
6558 INSCOPE: for (reverse @{$self->{open_elements}}) {
6559 if ($_->[1] & P_EL) {
6560 !!!cp ('t367');
6561 !!!back-token; # <plaintext>
6562 $token = {type => END_TAG_TOKEN, tag_name => 'p',
6563 line => $token->{line}, column => $token->{column}};
6564 next B;
6565 } elsif ($_->[1] & SCOPING_EL) {
6566 !!!cp ('t368');
6567 last INSCOPE;
6568 }
6569 } # INSCOPE
6570
6571 !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
6572
6573 $self->{content_model} = PLAINTEXT_CONTENT_MODEL;
6574
6575 !!!nack ('t368.1');
6576 !!!next-token;
6577 next B;
6578 } elsif ($token->{tag_name} eq 'a') {
6579 AFE: for my $i (reverse 0..$#$active_formatting_elements) {
6580 my $node = $active_formatting_elements->[$i];
6581 if ($node->[1] & A_EL) {
6582 !!!cp ('t371');
6583 !!!parse-error (type => 'in a:a', token => $token);
6584
6585 !!!back-token; # <a>
6586 $token = {type => END_TAG_TOKEN, tag_name => 'a',
6587 line => $token->{line}, column => $token->{column}};
6588 $formatting_end_tag->($token);
6589
6590 AFE2: for (reverse 0..$#$active_formatting_elements) {
6591 if ($active_formatting_elements->[$_]->[0] eq $node->[0]) {
6592 !!!cp ('t372');
6593 splice @$active_formatting_elements, $_, 1;
6594 last AFE2;
6595 }
6596 } # AFE2
6597 OE: for (reverse 0..$#{$self->{open_elements}}) {
6598 if ($self->{open_elements}->[$_]->[0] eq $node->[0]) {
6599 !!!cp ('t373');
6600 splice @{$self->{open_elements}}, $_, 1;
6601 last OE;
6602 }
6603 } # OE
6604 last AFE;
6605 } elsif ($node->[0] eq '#marker') {
6606 !!!cp ('t374');
6607 last AFE;
6608 }
6609 } # AFE
6610
6611 $reconstruct_active_formatting_elements->($insert_to_current);
6612
6613 !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
6614 push @$active_formatting_elements, $self->{open_elements}->[-1];
6615
6616 !!!nack ('t374.1');
6617 !!!next-token;
6618 next B;
6619 } elsif ($token->{tag_name} eq 'nobr') {
6620 $reconstruct_active_formatting_elements->($insert_to_current);
6621
6622 ## has a |nobr| element in scope
6623 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6624 my $node = $self->{open_elements}->[$_];
6625 if ($node->[1] & NOBR_EL) {
6626 !!!cp ('t376');
6627 !!!parse-error (type => 'in nobr:nobr', token => $token);
6628 !!!back-token; # <nobr>
6629 $token = {type => END_TAG_TOKEN, tag_name => 'nobr',
6630 line => $token->{line}, column => $token->{column}};
6631 next B;
6632 } elsif ($node->[1] & SCOPING_EL) {
6633 !!!cp ('t377');
6634 last INSCOPE;
6635 }
6636 } # INSCOPE
6637
6638 !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
6639 push @$active_formatting_elements, $self->{open_elements}->[-1];
6640
6641 !!!nack ('t377.1');
6642 !!!next-token;
6643 next B;
6644 } elsif ($token->{tag_name} eq 'button') {
6645 ## has a button element in scope
6646 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6647 my $node = $self->{open_elements}->[$_];
6648 if ($node->[1] & BUTTON_EL) {
6649 !!!cp ('t378');
6650 !!!parse-error (type => 'in button:button', token => $token);
6651 !!!back-token; # <button>
6652 $token = {type => END_TAG_TOKEN, tag_name => 'button',
6653 line => $token->{line}, column => $token->{column}};
6654 next B;
6655 } elsif ($node->[1] & SCOPING_EL) {
6656 !!!cp ('t379');
6657 last INSCOPE;
6658 }
6659 } # INSCOPE
6660
6661 $reconstruct_active_formatting_elements->($insert_to_current);
6662
6663 !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
6664
6665 ## TODO: associate with $self->{form_element} if defined
6666
6667 push @$active_formatting_elements, ['#marker', ''];
6668
6669 !!!nack ('t379.1');
6670 !!!next-token;
6671 next B;
6672 } elsif ({
6673 xmp => 1,
6674 iframe => 1,
6675 noembed => 1,
6676 noframes => 1,
6677 noscript => 0, ## TODO: 1 if scripting is enabled
6678 }->{$token->{tag_name}}) {
6679 if ($token->{tag_name} eq 'xmp') {
6680 !!!cp ('t381');
6681 $reconstruct_active_formatting_elements->($insert_to_current);
6682 } else {
6683 !!!cp ('t399');
6684 }
6685 ## NOTE: There is an "as if in body" code clone.
6686 $parse_rcdata->(CDATA_CONTENT_MODEL);
6687 next B;
6688 } elsif ($token->{tag_name} eq 'isindex') {
6689 !!!parse-error (type => 'isindex', token => $token);
6690
6691 if (defined $self->{form_element}) {
6692 !!!cp ('t389');
6693 ## Ignore the token
6694 !!!nack ('t389'); ## NOTE: Not acknowledged.
6695 !!!next-token;
6696 next B;
6697 } else {
6698 my $at = $token->{attributes};
6699 my $form_attrs;
6700 $form_attrs->{action} = $at->{action} if $at->{action};
6701 my $prompt_attr = $at->{prompt};
6702 $at->{name} = {name => 'name', value => 'isindex'};
6703 delete $at->{action};
6704 delete $at->{prompt};
6705 my @tokens = (
6706 {type => START_TAG_TOKEN, tag_name => 'form',
6707 attributes => $form_attrs,
6708 line => $token->{line}, column => $token->{column}},
6709 {type => START_TAG_TOKEN, tag_name => 'hr',
6710 line => $token->{line}, column => $token->{column}},
6711 {type => START_TAG_TOKEN, tag_name => 'p',
6712 line => $token->{line}, column => $token->{column}},
6713 {type => START_TAG_TOKEN, tag_name => 'label',
6714 line => $token->{line}, column => $token->{column}},
6715 );
6716 if ($prompt_attr) {
6717 !!!cp ('t390');
6718 push @tokens, {type => CHARACTER_TOKEN, data => $prompt_attr->{value},
6719 #line => $token->{line}, column => $token->{column},
6720 };
6721 } else {
6722 !!!cp ('t391');
6723 push @tokens, {type => CHARACTER_TOKEN,
6724 data => 'This is a searchable index. Insert your search keywords here: ',
6725 #line => $token->{line}, column => $token->{column},
6726 }; # SHOULD
6727 ## TODO: make this configurable
6728 }
6729 push @tokens,
6730 {type => START_TAG_TOKEN, tag_name => 'input', attributes => $at,
6731 line => $token->{line}, column => $token->{column}},
6732 #{type => CHARACTER_TOKEN, data => ''}, # SHOULD
6733 {type => END_TAG_TOKEN, tag_name => 'label',
6734 line => $token->{line}, column => $token->{column}},
6735 {type => END_TAG_TOKEN, tag_name => 'p',
6736 line => $token->{line}, column => $token->{column}},
6737 {type => START_TAG_TOKEN, tag_name => 'hr',
6738 line => $token->{line}, column => $token->{column}},
6739 {type => END_TAG_TOKEN, tag_name => 'form',
6740 line => $token->{line}, column => $token->{column}};
6741 !!!nack ('t391.1'); ## NOTE: Not acknowledged.
6742 !!!back-token (@tokens);
6743 !!!next-token;
6744 next B;
6745 }
6746 } elsif ($token->{tag_name} eq 'textarea') {
6747 my $tag_name = $token->{tag_name};
6748 my $el;
6749 !!!create-element ($el, $HTML_NS, $token->{tag_name}, $token->{attributes}, $token);
6750
6751 ## TODO: $self->{form_element} if defined
6752 $self->{content_model} = RCDATA_CONTENT_MODEL;
6753 delete $self->{escape}; # MUST
6754
6755 $insert->($el);
6756
6757 my $text = '';
6758 !!!nack ('t392.1');
6759 !!!next-token;
6760 if ($token->{type} == CHARACTER_TOKEN) {
6761 $token->{data} =~ s/^\x0A//;
6762 unless (length $token->{data}) {
6763 !!!cp ('t392');
6764 !!!next-token;
6765 } else {
6766 !!!cp ('t393');
6767 }
6768 } else {
6769 !!!cp ('t394');
6770 }
6771 while ($token->{type} == CHARACTER_TOKEN) {
6772 !!!cp ('t395');
6773 $text .= $token->{data};
6774 !!!next-token;
6775 }
6776 if (length $text) {
6777 !!!cp ('t396');
6778 $el->manakai_append_text ($text);
6779 }
6780
6781 $self->{content_model} = PCDATA_CONTENT_MODEL;
6782
6783 if ($token->{type} == END_TAG_TOKEN and
6784 $token->{tag_name} eq $tag_name) {
6785 !!!cp ('t397');
6786 ## Ignore the token
6787 } else {
6788 !!!cp ('t398');
6789 !!!parse-error (type => 'in RCDATA:#'.$token->{type}, token => $token);
6790 }
6791 !!!next-token;
6792 next B;
6793 } elsif ($token->{tag_name} eq 'math' or
6794 $token->{tag_name} eq 'svg') {
6795 $reconstruct_active_formatting_elements->($insert_to_current);
6796
6797 ## "adjust SVG attributes" ('svg' only) - done in insert-element-f
6798
6799 ## "adjust foreign attributes" - done in insert-element-f
6800
6801 !!!insert-element-f ($token->{tag_name} eq 'math' ? $MML_NS : $SVG_NS, $token->{tag_name}, $token->{attributes}, $token);
6802
6803 if ($self->{self_closing}) {
6804 pop @{$self->{open_elements}};
6805 !!!ack ('t398.1');
6806 } else {
6807 !!!cp ('t398.2');
6808 $self->{insertion_mode} |= IN_FOREIGN_CONTENT_IM;
6809 ## NOTE: |<body><math><mi><svg>| -> "in foreign content" insertion
6810 ## mode, "in body" (not "in foreign content") secondary insertion
6811 ## mode, maybe.
6812 }
6813
6814 !!!next-token;
6815 next B;
6816 } elsif ({
6817 caption => 1, col => 1, colgroup => 1, frame => 1,
6818 frameset => 1, head => 1, option => 1, optgroup => 1,
6819 tbody => 1, td => 1, tfoot => 1, th => 1,
6820 thead => 1, tr => 1,
6821 }->{$token->{tag_name}}) {
6822 !!!cp ('t401');
6823 !!!parse-error (type => 'in body:'.$token->{tag_name}, token => $token);
6824 ## Ignore the token
6825 !!!nack ('t401.1'); ## NOTE: |<col/>| or |<frame/>| here is an error.
6826 !!!next-token;
6827 next B;
6828
6829 ## ISSUE: An issue on HTML5 new elements in the spec.
6830 } else {
6831 if ($token->{tag_name} eq 'image') {
6832 !!!cp ('t384');
6833 !!!parse-error (type => 'image', token => $token);
6834 $token->{tag_name} = 'img';
6835 } else {
6836 !!!cp ('t385');
6837 }
6838
6839 ## NOTE: There is an "as if <br>" code clone.
6840 $reconstruct_active_formatting_elements->($insert_to_current);
6841
6842 !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
6843
6844 if ({
6845 applet => 1, marquee => 1, object => 1,
6846 }->{$token->{tag_name}}) {
6847 !!!cp ('t380');
6848 push @$active_formatting_elements, ['#marker', ''];
6849 !!!nack ('t380.1');
6850 } elsif ({
6851 b => 1, big => 1, em => 1, font => 1, i => 1,
6852 s => 1, small => 1, strile => 1,
6853 strong => 1, tt => 1, u => 1,
6854 }->{$token->{tag_name}}) {
6855 !!!cp ('t375');
6856 push @$active_formatting_elements, $self->{open_elements}->[-1];
6857 !!!nack ('t375.1');
6858 } elsif ($token->{tag_name} eq 'input') {
6859 !!!cp ('t388');
6860 ## TODO: associate with $self->{form_element} if defined
6861 pop @{$self->{open_elements}};
6862 !!!ack ('t388.2');
6863 } elsif ({
6864 area => 1, basefont => 1, bgsound => 1, br => 1,
6865 embed => 1, img => 1, param => 1, spacer => 1, wbr => 1,
6866 #image => 1,
6867 }->{$token->{tag_name}}) {
6868 !!!cp ('t388.1');
6869 pop @{$self->{open_elements}};
6870 !!!ack ('t388.3');
6871 } elsif ($token->{tag_name} eq 'select') {
6872 ## TODO: associate with $self->{form_element} if defined
6873
6874 if ($self->{insertion_mode} & TABLE_IMS or
6875 $self->{insertion_mode} & BODY_TABLE_IMS or
6876 $self->{insertion_mode} == IN_COLUMN_GROUP_IM) {
6877 !!!cp ('t400.1');
6878 $self->{insertion_mode} = IN_SELECT_IN_TABLE_IM;
6879 } else {
6880 !!!cp ('t400.2');
6881 $self->{insertion_mode} = IN_SELECT_IM;
6882 }
6883 !!!nack ('t400.3');
6884 } else {
6885 !!!nack ('t402');
6886 }
6887
6888 !!!next-token;
6889 next B;
6890 }
6891 } elsif ($token->{type} == END_TAG_TOKEN) {
6892 if ($token->{tag_name} eq 'body') {
6893 ## has a |body| element in scope
6894 my $i;
6895 INSCOPE: {
6896 for (reverse @{$self->{open_elements}}) {
6897 if ($_->[1] & BODY_EL) {
6898 !!!cp ('t405');
6899 $i = $_;
6900 last INSCOPE;
6901 } elsif ($_->[1] & SCOPING_EL) {
6902 !!!cp ('t405.1');
6903 last;
6904 }
6905 }
6906
6907 !!!parse-error (type => 'start tag not allowed',
6908 value => $token->{tag_name}, token => $token);
6909 ## NOTE: Ignore the token.
6910 !!!next-token;
6911 next B;
6912 } # INSCOPE
6913
6914 for (@{$self->{open_elements}}) {
6915 unless ($_->[1] & ALL_END_TAG_OPTIONAL_EL) {
6916 !!!cp ('t403');
6917 !!!parse-error (type => 'not closed',
6918 value => $_->[0]->manakai_local_name,
6919 token => $token);
6920 last;
6921 } else {
6922 !!!cp ('t404');
6923 }
6924 }
6925
6926 $self->{insertion_mode} = AFTER_BODY_IM;
6927 !!!next-token;
6928 next B;
6929 } elsif ($token->{tag_name} eq 'html') {
6930 ## TODO: Update this code. It seems that the code below is not
6931 ## up-to-date, though it has same effect as speced.
6932 if (@{$self->{open_elements}} > 1 and
6933 $self->{open_elements}->[1]->[1] & BODY_EL) {
6934 ## ISSUE: There is an issue in the spec.
6935 unless ($self->{open_elements}->[-1]->[1] & BODY_EL) {
6936 !!!cp ('t406');
6937 !!!parse-error (type => 'not closed',
6938 value => $self->{open_elements}->[1]->[0]
6939 ->manakai_local_name,
6940 token => $token);
6941 } else {
6942 !!!cp ('t407');
6943 }
6944 $self->{insertion_mode} = AFTER_BODY_IM;
6945 ## reprocess
6946 next B;
6947 } else {
6948 !!!cp ('t408');
6949 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
6950 ## Ignore the token
6951 !!!next-token;
6952 next B;
6953 }
6954 } elsif ({
6955 address => 1, blockquote => 1, center => 1, dir => 1,
6956 div => 1, dl => 1, fieldset => 1, listing => 1,
6957 menu => 1, ol => 1, pre => 1, ul => 1,
6958 dd => 1, dt => 1, li => 1,
6959 applet => 1, button => 1, marquee => 1, object => 1,
6960 }->{$token->{tag_name}}) {
6961 ## has an element in scope
6962 my $i;
6963 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6964 my $node = $self->{open_elements}->[$_];
6965 if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
6966 !!!cp ('t410');
6967 $i = $_;
6968 last INSCOPE;
6969 } elsif ($node->[1] & SCOPING_EL) {
6970 !!!cp ('t411');
6971 last INSCOPE;
6972 }
6973 } # INSCOPE
6974
6975 unless (defined $i) { # has an element in scope
6976 !!!cp ('t413');
6977 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
6978 } else {
6979 ## Step 1. generate implied end tags
6980 while ({
6981 dd => ($token->{tag_name} ne 'dd'),
6982 dt => ($token->{tag_name} ne 'dt'),
6983 li => ($token->{tag_name} ne 'li'),
6984 p => 1,
6985 }->{$self->{open_elements}->[-1]->[0]->manakai_local_name}) {
6986 !!!cp ('t409');
6987 pop @{$self->{open_elements}};
6988 }
6989
6990 ## Step 2.
6991 if ($self->{open_elements}->[-1]->[0]->manakai_local_name
6992 ne $token->{tag_name}) {
6993 !!!cp ('t412');
6994 !!!parse-error (type => 'not closed',
6995 value => $self->{open_elements}->[-1]->[0]
6996 ->manakai_local_name,
6997 token => $token);
6998 } else {
6999 !!!cp ('t414');
7000 }
7001
7002 ## Step 3.
7003 splice @{$self->{open_elements}}, $i;
7004
7005 ## Step 4.
7006 $clear_up_to_marker->()
7007 if {
7008 applet => 1, button => 1, marquee => 1, object => 1,
7009 }->{$token->{tag_name}};
7010 }
7011 !!!next-token;
7012 next B;
7013 } elsif ($token->{tag_name} eq 'form') {
7014 undef $self->{form_element};
7015
7016 ## has an element in scope
7017 my $i;
7018 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7019 my $node = $self->{open_elements}->[$_];
7020 if ($node->[1] & FORM_EL) {
7021 !!!cp ('t418');
7022 $i = $_;
7023 last INSCOPE;
7024 } elsif ($node->[1] & SCOPING_EL) {
7025 !!!cp ('t419');
7026 last INSCOPE;
7027 }
7028 } # INSCOPE
7029
7030 unless (defined $i) { # has an element in scope
7031 !!!cp ('t421');
7032 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
7033 } else {
7034 ## Step 1. generate implied end tags
7035 while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
7036 !!!cp ('t417');
7037 pop @{$self->{open_elements}};
7038 }
7039
7040 ## Step 2.
7041 if ($self->{open_elements}->[-1]->[0]->manakai_local_name
7042 ne $token->{tag_name}) {
7043 !!!cp ('t417.1');
7044 !!!parse-error (type => 'not closed',
7045 value => $self->{open_elements}->[-1]->[0]
7046 ->manakai_local_name,
7047 token => $token);
7048 } else {
7049 !!!cp ('t420');
7050 }
7051
7052 ## Step 3.
7053 splice @{$self->{open_elements}}, $i;
7054 }
7055
7056 !!!next-token;
7057 next B;
7058 } elsif ({
7059 h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,
7060 }->{$token->{tag_name}}) {
7061 ## has an element in scope
7062 my $i;
7063 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7064 my $node = $self->{open_elements}->[$_];
7065 if ($node->[1] & HEADING_EL) {
7066 !!!cp ('t423');
7067 $i = $_;
7068 last INSCOPE;
7069 } elsif ($node->[1] & SCOPING_EL) {
7070 !!!cp ('t424');
7071 last INSCOPE;
7072 }
7073 } # INSCOPE
7074
7075 unless (defined $i) { # has an element in scope
7076 !!!cp ('t425.1');
7077 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
7078 } else {
7079 ## Step 1. generate implied end tags
7080 while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
7081 !!!cp ('t422');
7082 pop @{$self->{open_elements}};
7083 }
7084
7085 ## Step 2.
7086 if ($self->{open_elements}->[-1]->[0]->manakai_local_name
7087 ne $token->{tag_name}) {
7088 !!!cp ('t425');
7089 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
7090 } else {
7091 !!!cp ('t426');
7092 }
7093
7094 ## Step 3.
7095 splice @{$self->{open_elements}}, $i;
7096 }
7097
7098 !!!next-token;
7099 next B;
7100 } elsif ($token->{tag_name} eq 'p') {
7101 ## has an element in scope
7102 my $i;
7103 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7104 my $node = $self->{open_elements}->[$_];
7105 if ($node->[1] & P_EL) {
7106 !!!cp ('t410.1');
7107 $i = $_;
7108 last INSCOPE;
7109 } elsif ($node->[1] & SCOPING_EL) {
7110 !!!cp ('t411.1');
7111 last INSCOPE;
7112 }
7113 } # INSCOPE
7114
7115 if (defined $i) {
7116 if ($self->{open_elements}->[-1]->[0]->manakai_local_name
7117 ne $token->{tag_name}) {
7118 !!!cp ('t412.1');
7119 !!!parse-error (type => 'not closed',
7120 value => $self->{open_elements}->[-1]->[0]
7121 ->manakai_local_name,
7122 token => $token);
7123 } else {
7124 !!!cp ('t414.1');
7125 }
7126
7127 splice @{$self->{open_elements}}, $i;
7128 } else {
7129 !!!cp ('t413.1');
7130 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
7131
7132 !!!cp ('t415.1');
7133 ## As if <p>, then reprocess the current token
7134 my $el;
7135 !!!create-element ($el, $HTML_NS, 'p',, $token);
7136 $insert->($el);
7137 ## NOTE: Not inserted into |$self->{open_elements}|.
7138 }
7139
7140 !!!next-token;
7141 next B;
7142 } elsif ({
7143 a => 1,
7144 b => 1, big => 1, em => 1, font => 1, i => 1,
7145 nobr => 1, s => 1, small => 1, strile => 1,
7146 strong => 1, tt => 1, u => 1,
7147 }->{$token->{tag_name}}) {
7148 !!!cp ('t427');
7149 $formatting_end_tag->($token);
7150 next B;
7151 } elsif ($token->{tag_name} eq 'br') {
7152 !!!cp ('t428');
7153 !!!parse-error (type => 'unmatched end tag:br', token => $token);
7154
7155 ## As if <br>
7156 $reconstruct_active_formatting_elements->($insert_to_current);
7157
7158 my $el;
7159 !!!create-element ($el, $HTML_NS, 'br',, $token);
7160 $insert->($el);
7161
7162 ## Ignore the token.
7163 !!!next-token;
7164 next B;
7165 } elsif ({
7166 caption => 1, col => 1, colgroup => 1, frame => 1,
7167 frameset => 1, head => 1, option => 1, optgroup => 1,
7168 tbody => 1, td => 1, tfoot => 1, th => 1,
7169 thead => 1, tr => 1,
7170 area => 1, basefont => 1, bgsound => 1,
7171 embed => 1, hr => 1, iframe => 1, image => 1,
7172 img => 1, input => 1, isindex => 1, noembed => 1,
7173 noframes => 1, param => 1, select => 1, spacer => 1,
7174 table => 1, textarea => 1, wbr => 1,
7175 noscript => 0, ## TODO: if scripting is enabled
7176 }->{$token->{tag_name}}) {
7177 !!!cp ('t429');
7178 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
7179 ## Ignore the token
7180 !!!next-token;
7181 next B;
7182
7183 ## ISSUE: Issue on HTML5 new elements in spec
7184
7185 } else {
7186 ## Step 1
7187 my $node_i = -1;
7188 my $node = $self->{open_elements}->[$node_i];
7189
7190 ## Step 2
7191 S2: {
7192 if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
7193 ## Step 1
7194 ## generate implied end tags
7195 while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
7196 !!!cp ('t430');
7197 ## ISSUE: Can this case be reached?
7198 pop @{$self->{open_elements}};
7199 }
7200
7201 ## Step 2
7202 if ($self->{open_elements}->[-1]->[0]->manakai_local_name
7203 ne $token->{tag_name}) {
7204 !!!cp ('t431');
7205 ## NOTE: <x><y></x>
7206 !!!parse-error (type => 'not closed',
7207 value => $self->{open_elements}->[-1]->[0]
7208 ->manakai_local_name,
7209 token => $token);
7210 } else {
7211 !!!cp ('t432');
7212 }
7213
7214 ## Step 3
7215 splice @{$self->{open_elements}}, $node_i;
7216
7217 !!!next-token;
7218 last S2;
7219 } else {
7220 ## Step 3
7221 if (not ($node->[1] & FORMATTING_EL) and
7222 #not $phrasing_category->{$node->[1]} and
7223 ($node->[1] & SPECIAL_EL or
7224 $node->[1] & SCOPING_EL)) {
7225 !!!cp ('t433');
7226 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
7227 ## Ignore the token
7228 !!!next-token;
7229 last S2;
7230 }
7231
7232 !!!cp ('t434');
7233 }
7234
7235 ## Step 4
7236 $node_i--;
7237 $node = $self->{open_elements}->[$node_i];
7238
7239 ## Step 5;
7240 redo S2;
7241 } # S2
7242 next B;
7243 }
7244 }
7245 next B;
7246 } continue { # B
7247 if ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM) {
7248 ## NOTE: The code below is executed in cases where it does not have
7249 ## to be, but it it is harmless even in those cases.
7250 ## has an element in scope
7251 INSCOPE: {
7252 for (reverse 0..$#{$self->{open_elements}}) {
7253 my $node = $self->{open_elements}->[$_];
7254 if ($node->[1] & FOREIGN_EL) {
7255 last INSCOPE;
7256 } elsif ($node->[1] & SCOPING_EL) {
7257 last;
7258 }
7259 }
7260
7261 ## NOTE: No foreign element in scope.
7262 $self->{insertion_mode} &= ~ IN_FOREIGN_CONTENT_IM;
7263 } # INSCOPE
7264 }
7265 } # B
7266
7267 ## Stop parsing # MUST
7268
7269 ## TODO: script stuffs
7270 } # _tree_construct_main
7271
7272 sub set_inner_html ($$$) {
7273 my $class = shift;
7274 my $node = shift;
7275 my $s = \$_[0];
7276 my $onerror = $_[1];
7277
7278 ## ISSUE: Should {confident} be true?
7279
7280 my $nt = $node->node_type;
7281 if ($nt == 9) {
7282 # MUST
7283
7284 ## Step 1 # MUST
7285 ## TODO: If the document has an active parser, ...
7286 ## ISSUE: There is an issue in the spec.
7287
7288 ## Step 2 # MUST
7289 my @cn = @{$node->child_nodes};
7290 for (@cn) {
7291 $node->remove_child ($_);
7292 }
7293
7294 ## Step 3, 4, 5 # MUST
7295 $class->parse_string ($$s => $node, $onerror);
7296 } elsif ($nt == 1) {
7297 ## TODO: If non-html element
7298
7299 ## NOTE: Most of this code is copied from |parse_string|
7300
7301 ## Step 1 # MUST
7302 my $this_doc = $node->owner_document;
7303 my $doc = $this_doc->implementation->create_document;
7304 $doc->manakai_is_html (1);
7305 my $p = $class->new;
7306 $p->{document} = $doc;
7307
7308 ## Step 8 # MUST
7309 my $i = 0;
7310 $p->{line_prev} = $p->{line} = 1;
7311 $p->{column_prev} = $p->{column} = 0;
7312 $p->{set_next_char} = sub {
7313 my $self = shift;
7314
7315 pop @{$self->{prev_char}};
7316 unshift @{$self->{prev_char}}, $self->{next_char};
7317
7318 $self->{next_char} = -1 and return if $i >= length $$s;
7319 $self->{next_char} = ord substr $$s, $i++, 1;
7320
7321 ($p->{line_prev}, $p->{column_prev}) = ($p->{line}, $p->{column});
7322 $p->{column}++;
7323
7324 if ($self->{next_char} == 0x000A) { # LF
7325 $p->{line}++;
7326 $p->{column} = 0;
7327 !!!cp ('i1');
7328 } elsif ($self->{next_char} == 0x000D) { # CR
7329 $i++ if substr ($$s, $i, 1) eq "\x0A";
7330 $self->{next_char} = 0x000A; # LF # MUST
7331 $p->{line}++;
7332 $p->{column} = 0;
7333 !!!cp ('i2');
7334 } elsif ($self->{next_char} > 0x10FFFF) {
7335 $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
7336 !!!cp ('i3');
7337 } elsif ($self->{next_char} == 0x0000) { # NULL
7338 !!!cp ('i4');
7339 !!!parse-error (type => 'NULL');
7340 $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
7341 } elsif ($self->{next_char} <= 0x0008 or
7342 (0x000E <= $self->{next_char} and
7343 $self->{next_char} <= 0x001F) or
7344 (0x007F <= $self->{next_char} and
7345 $self->{next_char} <= 0x009F) or
7346 (0xD800 <= $self->{next_char} and
7347 $self->{next_char} <= 0xDFFF) or
7348 (0xFDD0 <= $self->{next_char} and
7349 $self->{next_char} <= 0xFDDF) or
7350 {
7351 0xFFFE => 1, 0xFFFF => 1, 0x1FFFE => 1, 0x1FFFF => 1,
7352 0x2FFFE => 1, 0x2FFFF => 1, 0x3FFFE => 1, 0x3FFFF => 1,
7353 0x4FFFE => 1, 0x4FFFF => 1, 0x5FFFE => 1, 0x5FFFF => 1,
7354 0x6FFFE => 1, 0x6FFFF => 1, 0x7FFFE => 1, 0x7FFFF => 1,
7355 0x8FFFE => 1, 0x8FFFF => 1, 0x9FFFE => 1, 0x9FFFF => 1,
7356 0xAFFFE => 1, 0xAFFFF => 1, 0xBFFFE => 1, 0xBFFFF => 1,
7357 0xCFFFE => 1, 0xCFFFF => 1, 0xDFFFE => 1, 0xDFFFF => 1,
7358 0xEFFFE => 1, 0xEFFFF => 1, 0xFFFFE => 1, 0xFFFFF => 1,
7359 0x10FFFE => 1, 0x10FFFF => 1,
7360 }->{$self->{next_char}}) {
7361 !!!cp ('i4.1');
7362 !!!parse-error (type => 'control char', level => $self->{must_level});
7363 ## TODO: error type documentation
7364 }
7365 };
7366 $p->{prev_char} = [-1, -1, -1];
7367 $p->{next_char} = -1;
7368
7369 my $ponerror = $onerror || sub {
7370 my (%opt) = @_;
7371 my $line = $opt{line};
7372 my $column = $opt{column};
7373 if (defined $opt{token} and defined $opt{token}->{line}) {
7374 $line = $opt{token}->{line};
7375 $column = $opt{token}->{column};
7376 }
7377 warn "Parse error ($opt{type}) at line $line column $column\n";
7378 };
7379 $p->{parse_error} = sub {
7380 $ponerror->(line => $p->{line}, column => $p->{column}, @_);
7381 };
7382
7383 $p->_initialize_tokenizer;
7384 $p->_initialize_tree_constructor;
7385
7386 ## Step 2
7387 my $node_ln = $node->manakai_local_name;
7388 $p->{content_model} = {
7389 title => RCDATA_CONTENT_MODEL,
7390 textarea => RCDATA_CONTENT_MODEL,
7391 style => CDATA_CONTENT_MODEL,
7392 script => CDATA_CONTENT_MODEL,
7393 xmp => CDATA_CONTENT_MODEL,
7394 iframe => CDATA_CONTENT_MODEL,
7395 noembed => CDATA_CONTENT_MODEL,
7396 noframes => CDATA_CONTENT_MODEL,
7397 noscript => CDATA_CONTENT_MODEL,
7398 plaintext => PLAINTEXT_CONTENT_MODEL,
7399 }->{$node_ln};
7400 $p->{content_model} = PCDATA_CONTENT_MODEL
7401 unless defined $p->{content_model};
7402 ## ISSUE: What is "the name of the element"? local name?
7403
7404 $p->{inner_html_node} = [$node, $el_category->{$node_ln}];
7405 ## TODO: Foreign element OK?
7406
7407 ## Step 3
7408 my $root = $doc->create_element_ns
7409 ('http://www.w3.org/1999/xhtml', [undef, 'html']);
7410
7411 ## Step 4 # MUST
7412 $doc->append_child ($root);
7413
7414 ## Step 5 # MUST
7415 push @{$p->{open_elements}}, [$root, $el_category->{html}];
7416
7417 undef $p->{head_element};
7418
7419 ## Step 6 # MUST
7420 $p->_reset_insertion_mode;
7421
7422 ## Step 7 # MUST
7423 my $anode = $node;
7424 AN: while (defined $anode) {
7425 if ($anode->node_type == 1) {
7426 my $nsuri = $anode->namespace_uri;
7427 if (defined $nsuri and $nsuri eq 'http://www.w3.org/1999/xhtml') {
7428 if ($anode->manakai_local_name eq 'form') {
7429 !!!cp ('i5');
7430 $p->{form_element} = $anode;
7431 last AN;
7432 }
7433 }
7434 }
7435 $anode = $anode->parent_node;
7436 } # AN
7437
7438 ## Step 9 # MUST
7439 {
7440 my $self = $p;
7441 !!!next-token;
7442 }
7443 $p->_tree_construction_main;
7444
7445 ## Step 10 # MUST
7446 my @cn = @{$node->child_nodes};
7447 for (@cn) {
7448 $node->remove_child ($_);
7449 }
7450 ## ISSUE: mutation events? read-only?
7451
7452 ## Step 11 # MUST
7453 @cn = @{$root->child_nodes};
7454 for (@cn) {
7455 $this_doc->adopt_node ($_);
7456 $node->append_child ($_);
7457 }
7458 ## ISSUE: mutation events?
7459
7460 $p->_terminate_tree_constructor;
7461
7462 delete $p->{parse_error}; # delete loop
7463 } else {
7464 die "$0: |set_inner_html| is not defined for node of type $nt";
7465 }
7466 } # set_inner_html
7467
7468 } # tree construction stage
7469
7470 package Whatpm::HTML::RestartParser;
7471 push our @ISA, 'Error';
7472
7473 1;
7474 # $Date: 2008/05/24 10:32:29 $

[email protected]
ViewVC Help
Powered by ViewVC 1.1.24