/[suikacvs]/markup/html/whatpm/Whatpm/HTML.pm.src
Suika

Contents of /markup/html/whatpm/Whatpm/HTML.pm.src

Parent Directory Parent Directory | Revision Log Revision Log


Revision 1.55 - (show annotations) (download) (as text)
Sat Aug 11 06:53:38 2007 UTC (19 years, 1 month ago) by wakaba
Branch: MAIN
Changes since 1.54: +164 -156 lines
File MIME type: application/x-wais-source
++ whatpm/Whatpm/ChangeLog	11 Aug 2007 06:53:35 -0000
	* HTML.pm.src: Token types are now represented in number.

2007-08-11  Wakaba  <wakaba@suika.fam.cx>

1 package Whatpm::HTML;
2 use strict;
3 our $VERSION=do{my @r=(q$Revision: 1.54 $=~/\d+/g);sprintf "%d."."%02d" x $#r,@r};
4
5 ## ISSUE:
6 ## var doc = implementation.createDocument (null, null, null);
7 ## doc.write ('');
8 ## alert (doc.compatMode);
9
10 ## ISSUE: HTML5 revision 967 says that the encoding layer MUST NOT
11 ## strip BOM and the HTML layer MUST ignore it. Whether we can do it
12 ## is not yet clear.
13 ## "{U+FEFF}..." in UTF-16BE/UTF-16LE is three or four characters?
14 ## "{U+FEFF}..." in GB18030?
15
16 my $permitted_slash_tag_name = {
17 base => 1,
18 link => 1,
19 meta => 1,
20 hr => 1,
21 br => 1,
22 img=> 1,
23 embed => 1,
24 param => 1,
25 area => 1,
26 col => 1,
27 input => 1,
28 };
29
30 my $c1_entity_char = {
31 0x80 => 0x20AC,
32 0x81 => 0xFFFD,
33 0x82 => 0x201A,
34 0x83 => 0x0192,
35 0x84 => 0x201E,
36 0x85 => 0x2026,
37 0x86 => 0x2020,
38 0x87 => 0x2021,
39 0x88 => 0x02C6,
40 0x89 => 0x2030,
41 0x8A => 0x0160,
42 0x8B => 0x2039,
43 0x8C => 0x0152,
44 0x8D => 0xFFFD,
45 0x8E => 0x017D,
46 0x8F => 0xFFFD,
47 0x90 => 0xFFFD,
48 0x91 => 0x2018,
49 0x92 => 0x2019,
50 0x93 => 0x201C,
51 0x94 => 0x201D,
52 0x95 => 0x2022,
53 0x96 => 0x2013,
54 0x97 => 0x2014,
55 0x98 => 0x02DC,
56 0x99 => 0x2122,
57 0x9A => 0x0161,
58 0x9B => 0x203A,
59 0x9C => 0x0153,
60 0x9D => 0xFFFD,
61 0x9E => 0x017E,
62 0x9F => 0x0178,
63 }; # $c1_entity_char
64
65 my $special_category = {
66 address => 1, area => 1, base => 1, basefont => 1, bgsound => 1,
67 blockquote => 1, body => 1, br => 1, center => 1, col => 1, colgroup => 1,
68 dd => 1, dir => 1, div => 1, dl => 1, dt => 1, embed => 1, fieldset => 1,
69 form => 1, frame => 1, frameset => 1, h1 => 1, h2 => 1, h3 => 1,
70 h4 => 1, h5 => 1, h6 => 1, head => 1, hr => 1, iframe => 1, image => 1,
71 img => 1, input => 1, isindex => 1, li => 1, link => 1, listing => 1,
72 menu => 1, meta => 1, noembed => 1, noframes => 1, noscript => 1,
73 ol => 1, optgroup => 1, option => 1, p => 1, param => 1, plaintext => 1,
74 pre => 1, script => 1, select => 1, spacer => 1, style => 1, tbody => 1,
75 textarea => 1, tfoot => 1, thead => 1, title => 1, tr => 1, ul => 1, wbr => 1,
76 };
77 my $scoping_category = {
78 button => 1, caption => 1, html => 1, marquee => 1, object => 1,
79 table => 1, td => 1, th => 1,
80 };
81 my $formatting_category = {
82 a => 1, b => 1, big => 1, em => 1, font => 1, i => 1, nobr => 1,
83 s => 1, small => 1, strile => 1, strong => 1, tt => 1, u => 1,
84 };
85 # $phrasing_category: all other elements
86
87 sub parse_string ($$$;$) {
88 my $self = shift->new;
89 my $s = \$_[0];
90 $self->{document} = $_[1];
91
92 ## NOTE: |set_inner_html| copies most of this method's code
93
94 my $i = 0;
95 my $line = 1;
96 my $column = 0;
97 $self->{set_next_input_character} = sub {
98 my $self = shift;
99
100 pop @{$self->{prev_input_character}};
101 unshift @{$self->{prev_input_character}}, $self->{next_input_character};
102
103 $self->{next_input_character} = -1 and return if $i >= length $$s;
104 $self->{next_input_character} = ord substr $$s, $i++, 1;
105 $column++;
106
107 if ($self->{next_input_character} == 0x000A) { # LF
108 $line++;
109 $column = 0;
110 } elsif ($self->{next_input_character} == 0x000D) { # CR
111 $i++ if substr ($$s, $i, 1) eq "\x0A";
112 $self->{next_input_character} = 0x000A; # LF # MUST
113 $line++;
114 $column = 0;
115 } elsif ($self->{next_input_character} > 0x10FFFF) {
116 $self->{next_input_character} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
117 } elsif ($self->{next_input_character} == 0x0000) { # NULL
118 !!!parse-error (type => 'NULL');
119 $self->{next_input_character} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
120 }
121 };
122 $self->{prev_input_character} = [-1, -1, -1];
123 $self->{next_input_character} = -1;
124
125 my $onerror = $_[2] || sub {
126 my (%opt) = @_;
127 warn "Parse error ($opt{type}) at line $opt{line} column $opt{column}\n";
128 };
129 $self->{parse_error} = sub {
130 $onerror->(@_, line => $line, column => $column);
131 };
132
133 $self->_initialize_tokenizer;
134 $self->_initialize_tree_constructor;
135 $self->_construct_tree;
136 $self->_terminate_tree_constructor;
137
138 return $self->{document};
139 } # parse_string
140
141 sub new ($) {
142 my $class = shift;
143 my $self = bless {}, $class;
144 $self->{set_next_input_character} = sub {
145 $self->{next_input_character} = -1;
146 };
147 $self->{parse_error} = sub {
148 #
149 };
150 return $self;
151 } # new
152
153 sub CM_ENTITY () { 0b001 } # & markup in data
154 sub CM_LIMITED_MARKUP () { 0b010 } # < markup in data (limited)
155 sub CM_FULL_MARKUP () { 0b100 } # < markup in data (any)
156
157 sub PLAINTEXT_CONTENT_MODEL () { 0 }
158 sub CDATA_CONTENT_MODEL () { CM_LIMITED_MARKUP }
159 sub RCDATA_CONTENT_MODEL () { CM_ENTITY | CM_LIMITED_MARKUP }
160 sub PCDATA_CONTENT_MODEL () { CM_ENTITY | CM_FULL_MARKUP }
161
162 sub DOCTYPE_TOKEN () { 1 }
163 sub COMMENT_TOKEN () { 2 }
164 sub START_TAG_TOKEN () { 3 }
165 sub END_TAG_TOKEN () { 4 }
166 sub END_OF_FILE_TOKEN () { 5 }
167 sub CHARACTER_TOKEN () { 6 }
168
169 sub AFTER_HTML_IMS () { 0b100 }
170 sub HEAD_IMS () { 0b1000 }
171 sub BODY_IMS () { 0b10000 }
172 sub BODY_TABLE_IMS () { 0b100000 | BODY_IMS }
173 sub TABLE_IMS () { 0b1000000 }
174 sub ROW_IMS () { 0b10000000 | TABLE_IMS }
175 sub BODY_AFTER_IMS () { 0b100000000 }
176 sub FRAME_IMS () { 0b1000000000 }
177
178 sub AFTER_HTML_BODY_IM () { AFTER_HTML_IMS | BODY_AFTER_IMS }
179 sub AFTER_HTML_FRAMESET_IM () { AFTER_HTML_IMS | FRAME_IMS }
180 sub IN_HEAD_IM () { HEAD_IMS | 0b00 }
181 sub IN_HEAD_NOSCRIPT_IM () { HEAD_IMS | 0b01 }
182 sub AFTER_HEAD_IM () { HEAD_IMS | 0b10 }
183 sub BEFORE_HEAD_IM () { HEAD_IMS | 0b11 }
184 sub IN_BODY_IM () { BODY_IMS }
185 sub IN_CELL_IM () { BODY_TABLE_IMS | 0b01 }
186 sub IN_CAPTION_IM () { BODY_TABLE_IMS | 0b10 }
187 sub IN_ROW_IM () { ROW_IMS | 0b01 }
188 sub IN_TABLE_BODY_IM () { ROW_IMS | 0b10 }
189 sub IN_TABLE_IM () { TABLE_IMS }
190 sub AFTER_BODY_IM () { BODY_AFTER_IMS }
191 sub IN_FRAMESET_IM () { FRAME_IMS | 0b01 }
192 sub AFTER_FRAMESET_IM () { FRAME_IMS | 0b10 }
193 sub IN_SELECT_IM () { 0b01 }
194 sub IN_COLUMN_GROUP_IM () { 0b10 }
195
196 ## Implementations MUST act as if state machine in the spec
197
198 sub _initialize_tokenizer ($) {
199 my $self = shift;
200 $self->{state} = 'data'; # MUST
201 $self->{content_model} = PCDATA_CONTENT_MODEL; # be
202 undef $self->{current_token}; # start tag, end tag, comment, or DOCTYPE
203 undef $self->{current_attribute};
204 undef $self->{last_emitted_start_tag_name};
205 undef $self->{last_attribute_value_state};
206 $self->{char} = [];
207 # $self->{next_input_character}
208 !!!next-input-character;
209 $self->{token} = [];
210 # $self->{escape}
211 } # _initialize_tokenizer
212
213 ## A token has:
214 ## ->{type} == DOCTYPE_TOKEN, START_TAG_TOKEN, END_TAG_TOKEN, COMMENT_TOKEN,
215 ## CHARACTER_TOKEN, or END_OF_FILE_TOKEN
216 ## ->{name} (DOCTYPE_TOKEN)
217 ## ->{tag_name} (START_TAG_TOKEN, END_TAG_TOKEN)
218 ## ->{public_identifier} (DOCTYPE_TOKEN)
219 ## ->{system_identifier} (DOCTYPE_TOKEN)
220 ## ->{correct} == 1 or 0 (DOCTYPE_TOKEN)
221 ## ->{attributes} isa HASH (START_TAG_TOKEN, END_TAG_TOKEN)
222 ## ->{data} (COMMENT_TOKEN, CHARACTER_TOKEN)
223
224 ## Emitted token MUST immediately be handled by the tree construction state.
225
226 ## Before each step, UA MAY check to see if either one of the scripts in
227 ## "list of scripts that will execute as soon as possible" or the first
228 ## script in the "list of scripts that will execute asynchronously",
229 ## has completed loading. If one has, then it MUST be executed
230 ## and removed from the list.
231
232 sub _get_next_token ($) {
233 my $self = shift;
234 if (@{$self->{token}}) {
235 return shift @{$self->{token}};
236 }
237
238 A: {
239 if ($self->{state} eq 'data') {
240 if ($self->{next_input_character} == 0x0026) { # &
241 if ($self->{content_model} & CM_ENTITY) { # PCDATA | RCDATA
242 $self->{state} = 'entity data';
243 !!!next-input-character;
244 redo A;
245 } else {
246 #
247 }
248 } elsif ($self->{next_input_character} == 0x002D) { # -
249 if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
250 unless ($self->{escape}) {
251 if ($self->{prev_input_character}->[0] == 0x002D and # -
252 $self->{prev_input_character}->[1] == 0x0021 and # !
253 $self->{prev_input_character}->[2] == 0x003C) { # <
254 $self->{escape} = 1;
255 }
256 }
257 }
258
259 #
260 } elsif ($self->{next_input_character} == 0x003C) { # <
261 if ($self->{content_model} & CM_FULL_MARKUP or # PCDATA
262 (($self->{content_model} & CM_LIMITED_MARKUP) and # CDATA | RCDATA
263 not $self->{escape})) {
264 $self->{state} = 'tag open';
265 !!!next-input-character;
266 redo A;
267 } else {
268 #
269 }
270 } elsif ($self->{next_input_character} == 0x003E) { # >
271 if ($self->{escape} and
272 ($self->{content_model} & CM_LIMITED_MARKUP)) { # RCDATA | CDATA
273 if ($self->{prev_input_character}->[0] == 0x002D and # -
274 $self->{prev_input_character}->[1] == 0x002D) { # -
275 delete $self->{escape};
276 }
277 }
278
279 #
280 } elsif ($self->{next_input_character} == -1) {
281 !!!emit ({type => END_OF_FILE_TOKEN});
282 last A; ## TODO: ok?
283 }
284 # Anything else
285 my $token = {type => CHARACTER_TOKEN,
286 data => chr $self->{next_input_character}};
287 ## Stay in the data state
288 !!!next-input-character;
289
290 !!!emit ($token);
291
292 redo A;
293 } elsif ($self->{state} eq 'entity data') {
294 ## (cannot happen in CDATA state)
295
296 my $token = $self->_tokenize_attempt_to_consume_an_entity (0);
297
298 $self->{state} = 'data';
299 # next-input-character is already done
300
301 unless (defined $token) {
302 !!!emit ({type => CHARACTER_TOKEN, data => '&'});
303 } else {
304 !!!emit ($token);
305 }
306
307 redo A;
308 } elsif ($self->{state} eq 'tag open') {
309 if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
310 if ($self->{next_input_character} == 0x002F) { # /
311 !!!next-input-character;
312 $self->{state} = 'close tag open';
313 redo A;
314 } else {
315 ## reconsume
316 $self->{state} = 'data';
317
318 !!!emit ({type => CHARACTER_TOKEN, data => '<'});
319
320 redo A;
321 }
322 } elsif ($self->{content_model} & CM_FULL_MARKUP) { # PCDATA
323 if ($self->{next_input_character} == 0x0021) { # !
324 $self->{state} = 'markup declaration open';
325 !!!next-input-character;
326 redo A;
327 } elsif ($self->{next_input_character} == 0x002F) { # /
328 $self->{state} = 'close tag open';
329 !!!next-input-character;
330 redo A;
331 } elsif (0x0041 <= $self->{next_input_character} and
332 $self->{next_input_character} <= 0x005A) { # A..Z
333 $self->{current_token}
334 = {type => START_TAG_TOKEN,
335 tag_name => chr ($self->{next_input_character} + 0x0020)};
336 $self->{state} = 'tag name';
337 !!!next-input-character;
338 redo A;
339 } elsif (0x0061 <= $self->{next_input_character} and
340 $self->{next_input_character} <= 0x007A) { # a..z
341 $self->{current_token} = {type => START_TAG_TOKEN,
342 tag_name => chr ($self->{next_input_character})};
343 $self->{state} = 'tag name';
344 !!!next-input-character;
345 redo A;
346 } elsif ($self->{next_input_character} == 0x003E) { # >
347 !!!parse-error (type => 'empty start tag');
348 $self->{state} = 'data';
349 !!!next-input-character;
350
351 !!!emit ({type => CHARACTER_TOKEN, data => '<>'});
352
353 redo A;
354 } elsif ($self->{next_input_character} == 0x003F) { # ?
355 !!!parse-error (type => 'pio');
356 $self->{state} = 'bogus comment';
357 ## $self->{next_input_character} is intentionally left as is
358 redo A;
359 } else {
360 !!!parse-error (type => 'bare stago');
361 $self->{state} = 'data';
362 ## reconsume
363
364 !!!emit ({type => CHARACTER_TOKEN, data => '<'});
365
366 redo A;
367 }
368 } else {
369 die "$0: $self->{content_model} in tag open";
370 }
371 } elsif ($self->{state} eq 'close tag open') {
372 if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
373 if (defined $self->{last_emitted_start_tag_name}) {
374 ## NOTE: <http://krijnhoetmer.nl/irc-logs/whatwg/20070626#l-564>
375 my @next_char;
376 TAGNAME: for (my $i = 0; $i < length $self->{last_emitted_start_tag_name}; $i++) {
377 push @next_char, $self->{next_input_character};
378 my $c = ord substr ($self->{last_emitted_start_tag_name}, $i, 1);
379 my $C = 0x0061 <= $c && $c <= 0x007A ? $c - 0x0020 : $c;
380 if ($self->{next_input_character} == $c or $self->{next_input_character} == $C) {
381 !!!next-input-character;
382 next TAGNAME;
383 } else {
384 $self->{next_input_character} = shift @next_char; # reconsume
385 !!!back-next-input-character (@next_char);
386 $self->{state} = 'data';
387
388 !!!emit ({type => CHARACTER_TOKEN, data => '</'});
389
390 redo A;
391 }
392 }
393 push @next_char, $self->{next_input_character};
394
395 unless ($self->{next_input_character} == 0x0009 or # HT
396 $self->{next_input_character} == 0x000A or # LF
397 $self->{next_input_character} == 0x000B or # VT
398 $self->{next_input_character} == 0x000C or # FF
399 $self->{next_input_character} == 0x0020 or # SP
400 $self->{next_input_character} == 0x003E or # >
401 $self->{next_input_character} == 0x002F or # /
402 $self->{next_input_character} == -1) {
403 $self->{next_input_character} = shift @next_char; # reconsume
404 !!!back-next-input-character (@next_char);
405 $self->{state} = 'data';
406 !!!emit ({type => CHARACTER_TOKEN, data => '</'});
407 redo A;
408 } else {
409 $self->{next_input_character} = shift @next_char;
410 !!!back-next-input-character (@next_char);
411 # and consume...
412 }
413 } else {
414 ## No start tag token has ever been emitted
415 # next-input-character is already done
416 $self->{state} = 'data';
417 !!!emit ({type => CHARACTER_TOKEN, data => '</'});
418 redo A;
419 }
420 }
421
422 if (0x0041 <= $self->{next_input_character} and
423 $self->{next_input_character} <= 0x005A) { # A..Z
424 $self->{current_token} = {type => END_TAG_TOKEN,
425 tag_name => chr ($self->{next_input_character} + 0x0020)};
426 $self->{state} = 'tag name';
427 !!!next-input-character;
428 redo A;
429 } elsif (0x0061 <= $self->{next_input_character} and
430 $self->{next_input_character} <= 0x007A) { # a..z
431 $self->{current_token} = {type => END_TAG_TOKEN,
432 tag_name => chr ($self->{next_input_character})};
433 $self->{state} = 'tag name';
434 !!!next-input-character;
435 redo A;
436 } elsif ($self->{next_input_character} == 0x003E) { # >
437 !!!parse-error (type => 'empty end tag');
438 $self->{state} = 'data';
439 !!!next-input-character;
440 redo A;
441 } elsif ($self->{next_input_character} == -1) {
442 !!!parse-error (type => 'bare etago');
443 $self->{state} = 'data';
444 # reconsume
445
446 !!!emit ({type => CHARACTER_TOKEN, data => '</'});
447
448 redo A;
449 } else {
450 !!!parse-error (type => 'bogus end tag');
451 $self->{state} = 'bogus comment';
452 ## $self->{next_input_character} is intentionally left as is
453 redo A;
454 }
455 } elsif ($self->{state} eq 'tag name') {
456 if ($self->{next_input_character} == 0x0009 or # HT
457 $self->{next_input_character} == 0x000A or # LF
458 $self->{next_input_character} == 0x000B or # VT
459 $self->{next_input_character} == 0x000C or # FF
460 $self->{next_input_character} == 0x0020) { # SP
461 $self->{state} = 'before attribute name';
462 !!!next-input-character;
463 redo A;
464 } elsif ($self->{next_input_character} == 0x003E) { # >
465 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
466 $self->{current_token}->{first_start_tag}
467 = not defined $self->{last_emitted_start_tag_name};
468 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
469 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
470 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
471 if ($self->{current_token}->{attributes}) {
472 !!!parse-error (type => 'end tag attribute');
473 }
474 } else {
475 die "$0: $self->{current_token}->{type}: Unknown token type";
476 }
477 $self->{state} = 'data';
478 !!!next-input-character;
479
480 !!!emit ($self->{current_token}); # start tag or end tag
481
482 redo A;
483 } elsif (0x0041 <= $self->{next_input_character} and
484 $self->{next_input_character} <= 0x005A) { # A..Z
485 $self->{current_token}->{tag_name} .= chr ($self->{next_input_character} + 0x0020);
486 # start tag or end tag
487 ## Stay in this state
488 !!!next-input-character;
489 redo A;
490 } elsif ($self->{next_input_character} == -1) {
491 !!!parse-error (type => 'unclosed tag');
492 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
493 $self->{current_token}->{first_start_tag}
494 = not defined $self->{last_emitted_start_tag_name};
495 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
496 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
497 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
498 if ($self->{current_token}->{attributes}) {
499 !!!parse-error (type => 'end tag attribute');
500 }
501 } else {
502 die "$0: $self->{current_token}->{type}: Unknown token type";
503 }
504 $self->{state} = 'data';
505 # reconsume
506
507 !!!emit ($self->{current_token}); # start tag or end tag
508
509 redo A;
510 } elsif ($self->{next_input_character} == 0x002F) { # /
511 !!!next-input-character;
512 if ($self->{next_input_character} == 0x003E and # >
513 $self->{current_token}->{type} == START_TAG_TOKEN and
514 $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {
515 # permitted slash
516 #
517 } else {
518 !!!parse-error (type => 'nestc');
519 }
520 $self->{state} = 'before attribute name';
521 # next-input-character is already done
522 redo A;
523 } else {
524 $self->{current_token}->{tag_name} .= chr $self->{next_input_character};
525 # start tag or end tag
526 ## Stay in the state
527 !!!next-input-character;
528 redo A;
529 }
530 } elsif ($self->{state} eq 'before attribute name') {
531 if ($self->{next_input_character} == 0x0009 or # HT
532 $self->{next_input_character} == 0x000A or # LF
533 $self->{next_input_character} == 0x000B or # VT
534 $self->{next_input_character} == 0x000C or # FF
535 $self->{next_input_character} == 0x0020) { # SP
536 ## Stay in the state
537 !!!next-input-character;
538 redo A;
539 } elsif ($self->{next_input_character} == 0x003E) { # >
540 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
541 $self->{current_token}->{first_start_tag}
542 = not defined $self->{last_emitted_start_tag_name};
543 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
544 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
545 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
546 if ($self->{current_token}->{attributes}) {
547 !!!parse-error (type => 'end tag attribute');
548 }
549 } else {
550 die "$0: $self->{current_token}->{type}: Unknown token type";
551 }
552 $self->{state} = 'data';
553 !!!next-input-character;
554
555 !!!emit ($self->{current_token}); # start tag or end tag
556
557 redo A;
558 } elsif (0x0041 <= $self->{next_input_character} and
559 $self->{next_input_character} <= 0x005A) { # A..Z
560 $self->{current_attribute} = {name => chr ($self->{next_input_character} + 0x0020),
561 value => ''};
562 $self->{state} = 'attribute name';
563 !!!next-input-character;
564 redo A;
565 } elsif ($self->{next_input_character} == 0x002F) { # /
566 !!!next-input-character;
567 if ($self->{next_input_character} == 0x003E and # >
568 $self->{current_token}->{type} == START_TAG_TOKEN and
569 $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {
570 # permitted slash
571 #
572 } else {
573 !!!parse-error (type => 'nestc');
574 }
575 ## Stay in the state
576 # next-input-character is already done
577 redo A;
578 } elsif ($self->{next_input_character} == -1) {
579 !!!parse-error (type => 'unclosed tag');
580 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
581 $self->{current_token}->{first_start_tag}
582 = not defined $self->{last_emitted_start_tag_name};
583 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
584 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
585 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
586 if ($self->{current_token}->{attributes}) {
587 !!!parse-error (type => 'end tag attribute');
588 }
589 } else {
590 die "$0: $self->{current_token}->{type}: Unknown token type";
591 }
592 $self->{state} = 'data';
593 # reconsume
594
595 !!!emit ($self->{current_token}); # start tag or end tag
596
597 redo A;
598 } else {
599 $self->{current_attribute} = {name => chr ($self->{next_input_character}),
600 value => ''};
601 $self->{state} = 'attribute name';
602 !!!next-input-character;
603 redo A;
604 }
605 } elsif ($self->{state} eq 'attribute name') {
606 my $before_leave = sub {
607 if (exists $self->{current_token}->{attributes} # start tag or end tag
608 ->{$self->{current_attribute}->{name}}) { # MUST
609 !!!parse-error (type => 'duplicate attribute:'.$self->{current_attribute}->{name});
610 ## Discard $self->{current_attribute} # MUST
611 } else {
612 $self->{current_token}->{attributes}->{$self->{current_attribute}->{name}}
613 = $self->{current_attribute};
614 }
615 }; # $before_leave
616
617 if ($self->{next_input_character} == 0x0009 or # HT
618 $self->{next_input_character} == 0x000A or # LF
619 $self->{next_input_character} == 0x000B or # VT
620 $self->{next_input_character} == 0x000C or # FF
621 $self->{next_input_character} == 0x0020) { # SP
622 $before_leave->();
623 $self->{state} = 'after attribute name';
624 !!!next-input-character;
625 redo A;
626 } elsif ($self->{next_input_character} == 0x003D) { # =
627 $before_leave->();
628 $self->{state} = 'before attribute value';
629 !!!next-input-character;
630 redo A;
631 } elsif ($self->{next_input_character} == 0x003E) { # >
632 $before_leave->();
633 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
634 $self->{current_token}->{first_start_tag}
635 = not defined $self->{last_emitted_start_tag_name};
636 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
637 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
638 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
639 if ($self->{current_token}->{attributes}) {
640 !!!parse-error (type => 'end tag attribute');
641 }
642 } else {
643 die "$0: $self->{current_token}->{type}: Unknown token type";
644 }
645 $self->{state} = 'data';
646 !!!next-input-character;
647
648 !!!emit ($self->{current_token}); # start tag or end tag
649
650 redo A;
651 } elsif (0x0041 <= $self->{next_input_character} and
652 $self->{next_input_character} <= 0x005A) { # A..Z
653 $self->{current_attribute}->{name} .= chr ($self->{next_input_character} + 0x0020);
654 ## Stay in the state
655 !!!next-input-character;
656 redo A;
657 } elsif ($self->{next_input_character} == 0x002F) { # /
658 $before_leave->();
659 !!!next-input-character;
660 if ($self->{next_input_character} == 0x003E and # >
661 $self->{current_token}->{type} == START_TAG_TOKEN and
662 $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {
663 # permitted slash
664 #
665 } else {
666 !!!parse-error (type => 'nestc');
667 }
668 $self->{state} = 'before attribute name';
669 # next-input-character is already done
670 redo A;
671 } elsif ($self->{next_input_character} == -1) {
672 !!!parse-error (type => 'unclosed tag');
673 $before_leave->();
674 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
675 $self->{current_token}->{first_start_tag}
676 = not defined $self->{last_emitted_start_tag_name};
677 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
678 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
679 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
680 if ($self->{current_token}->{attributes}) {
681 !!!parse-error (type => 'end tag attribute');
682 }
683 } else {
684 die "$0: $self->{current_token}->{type}: Unknown token type";
685 }
686 $self->{state} = 'data';
687 # reconsume
688
689 !!!emit ($self->{current_token}); # start tag or end tag
690
691 redo A;
692 } else {
693 $self->{current_attribute}->{name} .= chr ($self->{next_input_character});
694 ## Stay in the state
695 !!!next-input-character;
696 redo A;
697 }
698 } elsif ($self->{state} eq 'after attribute name') {
699 if ($self->{next_input_character} == 0x0009 or # HT
700 $self->{next_input_character} == 0x000A or # LF
701 $self->{next_input_character} == 0x000B or # VT
702 $self->{next_input_character} == 0x000C or # FF
703 $self->{next_input_character} == 0x0020) { # SP
704 ## Stay in the state
705 !!!next-input-character;
706 redo A;
707 } elsif ($self->{next_input_character} == 0x003D) { # =
708 $self->{state} = 'before attribute value';
709 !!!next-input-character;
710 redo A;
711 } elsif ($self->{next_input_character} == 0x003E) { # >
712 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
713 $self->{current_token}->{first_start_tag}
714 = not defined $self->{last_emitted_start_tag_name};
715 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
716 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
717 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
718 if ($self->{current_token}->{attributes}) {
719 !!!parse-error (type => 'end tag attribute');
720 }
721 } else {
722 die "$0: $self->{current_token}->{type}: Unknown token type";
723 }
724 $self->{state} = 'data';
725 !!!next-input-character;
726
727 !!!emit ($self->{current_token}); # start tag or end tag
728
729 redo A;
730 } elsif (0x0041 <= $self->{next_input_character} and
731 $self->{next_input_character} <= 0x005A) { # A..Z
732 $self->{current_attribute} = {name => chr ($self->{next_input_character} + 0x0020),
733 value => ''};
734 $self->{state} = 'attribute name';
735 !!!next-input-character;
736 redo A;
737 } elsif ($self->{next_input_character} == 0x002F) { # /
738 !!!next-input-character;
739 if ($self->{next_input_character} == 0x003E and # >
740 $self->{current_token}->{type} == START_TAG_TOKEN and
741 $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {
742 # permitted slash
743 #
744 } else {
745 !!!parse-error (type => 'nestc');
746 ## TODO: Different error type for <aa / bb> than <aa/>
747 }
748 $self->{state} = 'before attribute name';
749 # next-input-character is already done
750 redo A;
751 } elsif ($self->{next_input_character} == -1) {
752 !!!parse-error (type => 'unclosed tag');
753 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
754 $self->{current_token}->{first_start_tag}
755 = not defined $self->{last_emitted_start_tag_name};
756 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
757 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
758 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
759 if ($self->{current_token}->{attributes}) {
760 !!!parse-error (type => 'end tag attribute');
761 }
762 } else {
763 die "$0: $self->{current_token}->{type}: Unknown token type";
764 }
765 $self->{state} = 'data';
766 # reconsume
767
768 !!!emit ($self->{current_token}); # start tag or end tag
769
770 redo A;
771 } else {
772 $self->{current_attribute} = {name => chr ($self->{next_input_character}),
773 value => ''};
774 $self->{state} = 'attribute name';
775 !!!next-input-character;
776 redo A;
777 }
778 } elsif ($self->{state} eq 'before attribute value') {
779 if ($self->{next_input_character} == 0x0009 or # HT
780 $self->{next_input_character} == 0x000A or # LF
781 $self->{next_input_character} == 0x000B or # VT
782 $self->{next_input_character} == 0x000C or # FF
783 $self->{next_input_character} == 0x0020) { # SP
784 ## Stay in the state
785 !!!next-input-character;
786 redo A;
787 } elsif ($self->{next_input_character} == 0x0022) { # "
788 $self->{state} = 'attribute value (double-quoted)';
789 !!!next-input-character;
790 redo A;
791 } elsif ($self->{next_input_character} == 0x0026) { # &
792 $self->{state} = 'attribute value (unquoted)';
793 ## reconsume
794 redo A;
795 } elsif ($self->{next_input_character} == 0x0027) { # '
796 $self->{state} = 'attribute value (single-quoted)';
797 !!!next-input-character;
798 redo A;
799 } elsif ($self->{next_input_character} == 0x003E) { # >
800 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
801 $self->{current_token}->{first_start_tag}
802 = not defined $self->{last_emitted_start_tag_name};
803 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
804 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
805 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
806 if ($self->{current_token}->{attributes}) {
807 !!!parse-error (type => 'end tag attribute');
808 }
809 } else {
810 die "$0: $self->{current_token}->{type}: Unknown token type";
811 }
812 $self->{state} = 'data';
813 !!!next-input-character;
814
815 !!!emit ($self->{current_token}); # start tag or end tag
816
817 redo A;
818 } elsif ($self->{next_input_character} == -1) {
819 !!!parse-error (type => 'unclosed tag');
820 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
821 $self->{current_token}->{first_start_tag}
822 = not defined $self->{last_emitted_start_tag_name};
823 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
824 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
825 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
826 if ($self->{current_token}->{attributes}) {
827 !!!parse-error (type => 'end tag attribute');
828 }
829 } else {
830 die "$0: $self->{current_token}->{type}: Unknown token type";
831 }
832 $self->{state} = 'data';
833 ## reconsume
834
835 !!!emit ($self->{current_token}); # start tag or end tag
836
837 redo A;
838 } else {
839 $self->{current_attribute}->{value} .= chr ($self->{next_input_character});
840 $self->{state} = 'attribute value (unquoted)';
841 !!!next-input-character;
842 redo A;
843 }
844 } elsif ($self->{state} eq 'attribute value (double-quoted)') {
845 if ($self->{next_input_character} == 0x0022) { # "
846 $self->{state} = 'before attribute name';
847 !!!next-input-character;
848 redo A;
849 } elsif ($self->{next_input_character} == 0x0026) { # &
850 $self->{last_attribute_value_state} = 'attribute value (double-quoted)';
851 $self->{state} = 'entity in attribute value';
852 !!!next-input-character;
853 redo A;
854 } elsif ($self->{next_input_character} == -1) {
855 !!!parse-error (type => 'unclosed attribute value');
856 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
857 $self->{current_token}->{first_start_tag}
858 = not defined $self->{last_emitted_start_tag_name};
859 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
860 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
861 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
862 if ($self->{current_token}->{attributes}) {
863 !!!parse-error (type => 'end tag attribute');
864 }
865 } else {
866 die "$0: $self->{current_token}->{type}: Unknown token type";
867 }
868 $self->{state} = 'data';
869 ## reconsume
870
871 !!!emit ($self->{current_token}); # start tag or end tag
872
873 redo A;
874 } else {
875 $self->{current_attribute}->{value} .= chr ($self->{next_input_character});
876 ## Stay in the state
877 !!!next-input-character;
878 redo A;
879 }
880 } elsif ($self->{state} eq 'attribute value (single-quoted)') {
881 if ($self->{next_input_character} == 0x0027) { # '
882 $self->{state} = 'before attribute name';
883 !!!next-input-character;
884 redo A;
885 } elsif ($self->{next_input_character} == 0x0026) { # &
886 $self->{last_attribute_value_state} = 'attribute value (single-quoted)';
887 $self->{state} = 'entity in attribute value';
888 !!!next-input-character;
889 redo A;
890 } elsif ($self->{next_input_character} == -1) {
891 !!!parse-error (type => 'unclosed attribute value');
892 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
893 $self->{current_token}->{first_start_tag}
894 = not defined $self->{last_emitted_start_tag_name};
895 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
896 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
897 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
898 if ($self->{current_token}->{attributes}) {
899 !!!parse-error (type => 'end tag attribute');
900 }
901 } else {
902 die "$0: $self->{current_token}->{type}: Unknown token type";
903 }
904 $self->{state} = 'data';
905 ## reconsume
906
907 !!!emit ($self->{current_token}); # start tag or end tag
908
909 redo A;
910 } else {
911 $self->{current_attribute}->{value} .= chr ($self->{next_input_character});
912 ## Stay in the state
913 !!!next-input-character;
914 redo A;
915 }
916 } elsif ($self->{state} eq 'attribute value (unquoted)') {
917 if ($self->{next_input_character} == 0x0009 or # HT
918 $self->{next_input_character} == 0x000A or # LF
919 $self->{next_input_character} == 0x000B or # HT
920 $self->{next_input_character} == 0x000C or # FF
921 $self->{next_input_character} == 0x0020) { # SP
922 $self->{state} = 'before attribute name';
923 !!!next-input-character;
924 redo A;
925 } elsif ($self->{next_input_character} == 0x0026) { # &
926 $self->{last_attribute_value_state} = 'attribute value (unquoted)';
927 $self->{state} = 'entity in attribute value';
928 !!!next-input-character;
929 redo A;
930 } elsif ($self->{next_input_character} == 0x003E) { # >
931 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
932 $self->{current_token}->{first_start_tag}
933 = not defined $self->{last_emitted_start_tag_name};
934 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
935 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
936 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
937 if ($self->{current_token}->{attributes}) {
938 !!!parse-error (type => 'end tag attribute');
939 }
940 } else {
941 die "$0: $self->{current_token}->{type}: Unknown token type";
942 }
943 $self->{state} = 'data';
944 !!!next-input-character;
945
946 !!!emit ($self->{current_token}); # start tag or end tag
947
948 redo A;
949 } elsif ($self->{next_input_character} == -1) {
950 !!!parse-error (type => 'unclosed tag');
951 if ($self->{current_token}->{type} == START_TAG_TOKEN) {
952 $self->{current_token}->{first_start_tag}
953 = not defined $self->{last_emitted_start_tag_name};
954 $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
955 } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
956 $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
957 if ($self->{current_token}->{attributes}) {
958 !!!parse-error (type => 'end tag attribute');
959 }
960 } else {
961 die "$0: $self->{current_token}->{type}: Unknown token type";
962 }
963 $self->{state} = 'data';
964 ## reconsume
965
966 !!!emit ($self->{current_token}); # start tag or end tag
967
968 redo A;
969 } else {
970 $self->{current_attribute}->{value} .= chr ($self->{next_input_character});
971 ## Stay in the state
972 !!!next-input-character;
973 redo A;
974 }
975 } elsif ($self->{state} eq 'entity in attribute value') {
976 my $token = $self->_tokenize_attempt_to_consume_an_entity (1);
977
978 unless (defined $token) {
979 $self->{current_attribute}->{value} .= '&';
980 } else {
981 $self->{current_attribute}->{value} .= $token->{data};
982 ## ISSUE: spec says "append the returned character token to the current attribute's value"
983 }
984
985 $self->{state} = $self->{last_attribute_value_state};
986 # next-input-character is already done
987 redo A;
988 } elsif ($self->{state} eq 'bogus comment') {
989 ## (only happen if PCDATA state)
990
991 my $token = {type => COMMENT_TOKEN, data => ''};
992
993 BC: {
994 if ($self->{next_input_character} == 0x003E) { # >
995 $self->{state} = 'data';
996 !!!next-input-character;
997
998 !!!emit ($token);
999
1000 redo A;
1001 } elsif ($self->{next_input_character} == -1) {
1002 $self->{state} = 'data';
1003 ## reconsume
1004
1005 !!!emit ($token);
1006
1007 redo A;
1008 } else {
1009 $token->{data} .= chr ($self->{next_input_character});
1010 !!!next-input-character;
1011 redo BC;
1012 }
1013 } # BC
1014 } elsif ($self->{state} eq 'markup declaration open') {
1015 ## (only happen if PCDATA state)
1016
1017 my @next_char;
1018 push @next_char, $self->{next_input_character};
1019
1020 if ($self->{next_input_character} == 0x002D) { # -
1021 !!!next-input-character;
1022 push @next_char, $self->{next_input_character};
1023 if ($self->{next_input_character} == 0x002D) { # -
1024 $self->{current_token} = {type => COMMENT_TOKEN, data => ''};
1025 $self->{state} = 'comment start';
1026 !!!next-input-character;
1027 redo A;
1028 }
1029 } elsif ($self->{next_input_character} == 0x0044 or # D
1030 $self->{next_input_character} == 0x0064) { # d
1031 !!!next-input-character;
1032 push @next_char, $self->{next_input_character};
1033 if ($self->{next_input_character} == 0x004F or # O
1034 $self->{next_input_character} == 0x006F) { # o
1035 !!!next-input-character;
1036 push @next_char, $self->{next_input_character};
1037 if ($self->{next_input_character} == 0x0043 or # C
1038 $self->{next_input_character} == 0x0063) { # c
1039 !!!next-input-character;
1040 push @next_char, $self->{next_input_character};
1041 if ($self->{next_input_character} == 0x0054 or # T
1042 $self->{next_input_character} == 0x0074) { # t
1043 !!!next-input-character;
1044 push @next_char, $self->{next_input_character};
1045 if ($self->{next_input_character} == 0x0059 or # Y
1046 $self->{next_input_character} == 0x0079) { # y
1047 !!!next-input-character;
1048 push @next_char, $self->{next_input_character};
1049 if ($self->{next_input_character} == 0x0050 or # P
1050 $self->{next_input_character} == 0x0070) { # p
1051 !!!next-input-character;
1052 push @next_char, $self->{next_input_character};
1053 if ($self->{next_input_character} == 0x0045 or # E
1054 $self->{next_input_character} == 0x0065) { # e
1055 ## ISSUE: What a stupid code this is!
1056 $self->{state} = 'DOCTYPE';
1057 !!!next-input-character;
1058 redo A;
1059 }
1060 }
1061 }
1062 }
1063 }
1064 }
1065 }
1066
1067 !!!parse-error (type => 'bogus comment');
1068 $self->{next_input_character} = shift @next_char;
1069 !!!back-next-input-character (@next_char);
1070 $self->{state} = 'bogus comment';
1071 redo A;
1072
1073 ## ISSUE: typos in spec: chacacters, is is a parse error
1074 ## ISSUE: spec is somewhat unclear on "is the first character that will be in the comment"; what is "that will be in the comment" is what the algorithm defines, isn't it?
1075 } elsif ($self->{state} eq 'comment start') {
1076 if ($self->{next_input_character} == 0x002D) { # -
1077 $self->{state} = 'comment start dash';
1078 !!!next-input-character;
1079 redo A;
1080 } elsif ($self->{next_input_character} == 0x003E) { # >
1081 !!!parse-error (type => 'bogus comment');
1082 $self->{state} = 'data';
1083 !!!next-input-character;
1084
1085 !!!emit ($self->{current_token}); # comment
1086
1087 redo A;
1088 } elsif ($self->{next_input_character} == -1) {
1089 !!!parse-error (type => 'unclosed comment');
1090 $self->{state} = 'data';
1091 ## reconsume
1092
1093 !!!emit ($self->{current_token}); # comment
1094
1095 redo A;
1096 } else {
1097 $self->{current_token}->{data} # comment
1098 .= chr ($self->{next_input_character});
1099 $self->{state} = 'comment';
1100 !!!next-input-character;
1101 redo A;
1102 }
1103 } elsif ($self->{state} eq 'comment start dash') {
1104 if ($self->{next_input_character} == 0x002D) { # -
1105 $self->{state} = 'comment end';
1106 !!!next-input-character;
1107 redo A;
1108 } elsif ($self->{next_input_character} == 0x003E) { # >
1109 !!!parse-error (type => 'bogus comment');
1110 $self->{state} = 'data';
1111 !!!next-input-character;
1112
1113 !!!emit ($self->{current_token}); # comment
1114
1115 redo A;
1116 } elsif ($self->{next_input_character} == -1) {
1117 !!!parse-error (type => 'unclosed comment');
1118 $self->{state} = 'data';
1119 ## reconsume
1120
1121 !!!emit ($self->{current_token}); # comment
1122
1123 redo A;
1124 } else {
1125 $self->{current_token}->{data} # comment
1126 .= '-' . chr ($self->{next_input_character});
1127 $self->{state} = 'comment';
1128 !!!next-input-character;
1129 redo A;
1130 }
1131 } elsif ($self->{state} eq 'comment') {
1132 if ($self->{next_input_character} == 0x002D) { # -
1133 $self->{state} = 'comment end dash';
1134 !!!next-input-character;
1135 redo A;
1136 } elsif ($self->{next_input_character} == -1) {
1137 !!!parse-error (type => 'unclosed comment');
1138 $self->{state} = 'data';
1139 ## reconsume
1140
1141 !!!emit ($self->{current_token}); # comment
1142
1143 redo A;
1144 } else {
1145 $self->{current_token}->{data} .= chr ($self->{next_input_character}); # comment
1146 ## Stay in the state
1147 !!!next-input-character;
1148 redo A;
1149 }
1150 } elsif ($self->{state} eq 'comment end dash') {
1151 if ($self->{next_input_character} == 0x002D) { # -
1152 $self->{state} = 'comment end';
1153 !!!next-input-character;
1154 redo A;
1155 } elsif ($self->{next_input_character} == -1) {
1156 !!!parse-error (type => 'unclosed comment');
1157 $self->{state} = 'data';
1158 ## reconsume
1159
1160 !!!emit ($self->{current_token}); # comment
1161
1162 redo A;
1163 } else {
1164 $self->{current_token}->{data} .= '-' . chr ($self->{next_input_character}); # comment
1165 $self->{state} = 'comment';
1166 !!!next-input-character;
1167 redo A;
1168 }
1169 } elsif ($self->{state} eq 'comment end') {
1170 if ($self->{next_input_character} == 0x003E) { # >
1171 $self->{state} = 'data';
1172 !!!next-input-character;
1173
1174 !!!emit ($self->{current_token}); # comment
1175
1176 redo A;
1177 } elsif ($self->{next_input_character} == 0x002D) { # -
1178 !!!parse-error (type => 'dash in comment');
1179 $self->{current_token}->{data} .= '-'; # comment
1180 ## Stay in the state
1181 !!!next-input-character;
1182 redo A;
1183 } elsif ($self->{next_input_character} == -1) {
1184 !!!parse-error (type => 'unclosed comment');
1185 $self->{state} = 'data';
1186 ## reconsume
1187
1188 !!!emit ($self->{current_token}); # comment
1189
1190 redo A;
1191 } else {
1192 !!!parse-error (type => 'dash in comment');
1193 $self->{current_token}->{data} .= '--' . chr ($self->{next_input_character}); # comment
1194 $self->{state} = 'comment';
1195 !!!next-input-character;
1196 redo A;
1197 }
1198 } elsif ($self->{state} eq 'DOCTYPE') {
1199 if ($self->{next_input_character} == 0x0009 or # HT
1200 $self->{next_input_character} == 0x000A or # LF
1201 $self->{next_input_character} == 0x000B or # VT
1202 $self->{next_input_character} == 0x000C or # FF
1203 $self->{next_input_character} == 0x0020) { # SP
1204 $self->{state} = 'before DOCTYPE name';
1205 !!!next-input-character;
1206 redo A;
1207 } else {
1208 !!!parse-error (type => 'no space before DOCTYPE name');
1209 $self->{state} = 'before DOCTYPE name';
1210 ## reconsume
1211 redo A;
1212 }
1213 } elsif ($self->{state} eq 'before DOCTYPE name') {
1214 if ($self->{next_input_character} == 0x0009 or # HT
1215 $self->{next_input_character} == 0x000A or # LF
1216 $self->{next_input_character} == 0x000B or # VT
1217 $self->{next_input_character} == 0x000C or # FF
1218 $self->{next_input_character} == 0x0020) { # SP
1219 ## Stay in the state
1220 !!!next-input-character;
1221 redo A;
1222 } elsif ($self->{next_input_character} == 0x003E) { # >
1223 !!!parse-error (type => 'no DOCTYPE name');
1224 $self->{state} = 'data';
1225 !!!next-input-character;
1226
1227 !!!emit ({type => DOCTYPE_TOKEN}); # incorrect
1228
1229 redo A;
1230 } elsif ($self->{next_input_character} == -1) {
1231 !!!parse-error (type => 'no DOCTYPE name');
1232 $self->{state} = 'data';
1233 ## reconsume
1234
1235 !!!emit ({type => DOCTYPE_TOKEN}); # incorrect
1236
1237 redo A;
1238 } else {
1239 $self->{current_token}
1240 = {type => DOCTYPE_TOKEN,
1241 name => chr ($self->{next_input_character}),
1242 correct => 1};
1243 ## ISSUE: "Set the token's name name to the" in the spec
1244 $self->{state} = 'DOCTYPE name';
1245 !!!next-input-character;
1246 redo A;
1247 }
1248 } elsif ($self->{state} eq 'DOCTYPE name') {
1249 ## ISSUE: Redundant "First," in the spec.
1250 if ($self->{next_input_character} == 0x0009 or # HT
1251 $self->{next_input_character} == 0x000A or # LF
1252 $self->{next_input_character} == 0x000B or # VT
1253 $self->{next_input_character} == 0x000C or # FF
1254 $self->{next_input_character} == 0x0020) { # SP
1255 $self->{state} = 'after DOCTYPE name';
1256 !!!next-input-character;
1257 redo A;
1258 } elsif ($self->{next_input_character} == 0x003E) { # >
1259 $self->{state} = 'data';
1260 !!!next-input-character;
1261
1262 !!!emit ($self->{current_token}); # DOCTYPE
1263
1264 redo A;
1265 } elsif ($self->{next_input_character} == -1) {
1266 !!!parse-error (type => 'unclosed DOCTYPE');
1267 $self->{state} = 'data';
1268 ## reconsume
1269
1270 delete $self->{current_token}->{correct};
1271 !!!emit ($self->{current_token}); # DOCTYPE
1272
1273 redo A;
1274 } else {
1275 $self->{current_token}->{name}
1276 .= chr ($self->{next_input_character}); # DOCTYPE
1277 ## Stay in the state
1278 !!!next-input-character;
1279 redo A;
1280 }
1281 } elsif ($self->{state} eq 'after DOCTYPE name') {
1282 if ($self->{next_input_character} == 0x0009 or # HT
1283 $self->{next_input_character} == 0x000A or # LF
1284 $self->{next_input_character} == 0x000B or # VT
1285 $self->{next_input_character} == 0x000C or # FF
1286 $self->{next_input_character} == 0x0020) { # SP
1287 ## Stay in the state
1288 !!!next-input-character;
1289 redo A;
1290 } elsif ($self->{next_input_character} == 0x003E) { # >
1291 $self->{state} = 'data';
1292 !!!next-input-character;
1293
1294 !!!emit ($self->{current_token}); # DOCTYPE
1295
1296 redo A;
1297 } elsif ($self->{next_input_character} == -1) {
1298 !!!parse-error (type => 'unclosed DOCTYPE');
1299 $self->{state} = 'data';
1300 ## reconsume
1301
1302 delete $self->{current_token}->{correct};
1303 !!!emit ($self->{current_token}); # DOCTYPE
1304
1305 redo A;
1306 } elsif ($self->{next_input_character} == 0x0050 or # P
1307 $self->{next_input_character} == 0x0070) { # p
1308 !!!next-input-character;
1309 if ($self->{next_input_character} == 0x0055 or # U
1310 $self->{next_input_character} == 0x0075) { # u
1311 !!!next-input-character;
1312 if ($self->{next_input_character} == 0x0042 or # B
1313 $self->{next_input_character} == 0x0062) { # b
1314 !!!next-input-character;
1315 if ($self->{next_input_character} == 0x004C or # L
1316 $self->{next_input_character} == 0x006C) { # l
1317 !!!next-input-character;
1318 if ($self->{next_input_character} == 0x0049 or # I
1319 $self->{next_input_character} == 0x0069) { # i
1320 !!!next-input-character;
1321 if ($self->{next_input_character} == 0x0043 or # C
1322 $self->{next_input_character} == 0x0063) { # c
1323 $self->{state} = 'before DOCTYPE public identifier';
1324 !!!next-input-character;
1325 redo A;
1326 }
1327 }
1328 }
1329 }
1330 }
1331
1332 #
1333 } elsif ($self->{next_input_character} == 0x0053 or # S
1334 $self->{next_input_character} == 0x0073) { # s
1335 !!!next-input-character;
1336 if ($self->{next_input_character} == 0x0059 or # Y
1337 $self->{next_input_character} == 0x0079) { # y
1338 !!!next-input-character;
1339 if ($self->{next_input_character} == 0x0053 or # S
1340 $self->{next_input_character} == 0x0073) { # s
1341 !!!next-input-character;
1342 if ($self->{next_input_character} == 0x0054 or # T
1343 $self->{next_input_character} == 0x0074) { # t
1344 !!!next-input-character;
1345 if ($self->{next_input_character} == 0x0045 or # E
1346 $self->{next_input_character} == 0x0065) { # e
1347 !!!next-input-character;
1348 if ($self->{next_input_character} == 0x004D or # M
1349 $self->{next_input_character} == 0x006D) { # m
1350 $self->{state} = 'before DOCTYPE system identifier';
1351 !!!next-input-character;
1352 redo A;
1353 }
1354 }
1355 }
1356 }
1357 }
1358
1359 #
1360 } else {
1361 !!!next-input-character;
1362 #
1363 }
1364
1365 !!!parse-error (type => 'string after DOCTYPE name');
1366 $self->{state} = 'bogus DOCTYPE';
1367 # next-input-character is already done
1368 redo A;
1369 } elsif ($self->{state} eq 'before DOCTYPE public identifier') {
1370 if ({
1371 0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,
1372 #0x000D => 1, # HT, LF, VT, FF, SP, CR
1373 }->{$self->{next_input_character}}) {
1374 ## Stay in the state
1375 !!!next-input-character;
1376 redo A;
1377 } elsif ($self->{next_input_character} eq 0x0022) { # "
1378 $self->{current_token}->{public_identifier} = ''; # DOCTYPE
1379 $self->{state} = 'DOCTYPE public identifier (double-quoted)';
1380 !!!next-input-character;
1381 redo A;
1382 } elsif ($self->{next_input_character} eq 0x0027) { # '
1383 $self->{current_token}->{public_identifier} = ''; # DOCTYPE
1384 $self->{state} = 'DOCTYPE public identifier (single-quoted)';
1385 !!!next-input-character;
1386 redo A;
1387 } elsif ($self->{next_input_character} eq 0x003E) { # >
1388 !!!parse-error (type => 'no PUBLIC literal');
1389
1390 $self->{state} = 'data';
1391 !!!next-input-character;
1392
1393 delete $self->{current_token}->{correct};
1394 !!!emit ($self->{current_token}); # DOCTYPE
1395
1396 redo A;
1397 } elsif ($self->{next_input_character} == -1) {
1398 !!!parse-error (type => 'unclosed DOCTYPE');
1399
1400 $self->{state} = 'data';
1401 ## reconsume
1402
1403 delete $self->{current_token}->{correct};
1404 !!!emit ($self->{current_token}); # DOCTYPE
1405
1406 redo A;
1407 } else {
1408 !!!parse-error (type => 'string after PUBLIC');
1409 $self->{state} = 'bogus DOCTYPE';
1410 !!!next-input-character;
1411 redo A;
1412 }
1413 } elsif ($self->{state} eq 'DOCTYPE public identifier (double-quoted)') {
1414 if ($self->{next_input_character} == 0x0022) { # "
1415 $self->{state} = 'after DOCTYPE public identifier';
1416 !!!next-input-character;
1417 redo A;
1418 } elsif ($self->{next_input_character} == -1) {
1419 !!!parse-error (type => 'unclosed PUBLIC literal');
1420
1421 $self->{state} = 'data';
1422 ## reconsume
1423
1424 delete $self->{current_token}->{correct};
1425 !!!emit ($self->{current_token}); # DOCTYPE
1426
1427 redo A;
1428 } else {
1429 $self->{current_token}->{public_identifier} # DOCTYPE
1430 .= chr $self->{next_input_character};
1431 ## Stay in the state
1432 !!!next-input-character;
1433 redo A;
1434 }
1435 } elsif ($self->{state} eq 'DOCTYPE public identifier (single-quoted)') {
1436 if ($self->{next_input_character} == 0x0027) { # '
1437 $self->{state} = 'after DOCTYPE public identifier';
1438 !!!next-input-character;
1439 redo A;
1440 } elsif ($self->{next_input_character} == -1) {
1441 !!!parse-error (type => 'unclosed PUBLIC literal');
1442
1443 $self->{state} = 'data';
1444 ## reconsume
1445
1446 delete $self->{current_token}->{correct};
1447 !!!emit ($self->{current_token}); # DOCTYPE
1448
1449 redo A;
1450 } else {
1451 $self->{current_token}->{public_identifier} # DOCTYPE
1452 .= chr $self->{next_input_character};
1453 ## Stay in the state
1454 !!!next-input-character;
1455 redo A;
1456 }
1457 } elsif ($self->{state} eq 'after DOCTYPE public identifier') {
1458 if ({
1459 0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,
1460 #0x000D => 1, # HT, LF, VT, FF, SP, CR
1461 }->{$self->{next_input_character}}) {
1462 ## Stay in the state
1463 !!!next-input-character;
1464 redo A;
1465 } elsif ($self->{next_input_character} == 0x0022) { # "
1466 $self->{current_token}->{system_identifier} = ''; # DOCTYPE
1467 $self->{state} = 'DOCTYPE system identifier (double-quoted)';
1468 !!!next-input-character;
1469 redo A;
1470 } elsif ($self->{next_input_character} == 0x0027) { # '
1471 $self->{current_token}->{system_identifier} = ''; # DOCTYPE
1472 $self->{state} = 'DOCTYPE system identifier (single-quoted)';
1473 !!!next-input-character;
1474 redo A;
1475 } elsif ($self->{next_input_character} == 0x003E) { # >
1476 $self->{state} = 'data';
1477 !!!next-input-character;
1478
1479 !!!emit ($self->{current_token}); # DOCTYPE
1480
1481 redo A;
1482 } elsif ($self->{next_input_character} == -1) {
1483 !!!parse-error (type => 'unclosed DOCTYPE');
1484
1485 $self->{state} = 'data';
1486 ## reconsume
1487
1488 delete $self->{current_token}->{correct};
1489 !!!emit ($self->{current_token}); # DOCTYPE
1490
1491 redo A;
1492 } else {
1493 !!!parse-error (type => 'string after PUBLIC literal');
1494 $self->{state} = 'bogus DOCTYPE';
1495 !!!next-input-character;
1496 redo A;
1497 }
1498 } elsif ($self->{state} eq 'before DOCTYPE system identifier') {
1499 if ({
1500 0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,
1501 #0x000D => 1, # HT, LF, VT, FF, SP, CR
1502 }->{$self->{next_input_character}}) {
1503 ## Stay in the state
1504 !!!next-input-character;
1505 redo A;
1506 } elsif ($self->{next_input_character} == 0x0022) { # "
1507 $self->{current_token}->{system_identifier} = ''; # DOCTYPE
1508 $self->{state} = 'DOCTYPE system identifier (double-quoted)';
1509 !!!next-input-character;
1510 redo A;
1511 } elsif ($self->{next_input_character} == 0x0027) { # '
1512 $self->{current_token}->{system_identifier} = ''; # DOCTYPE
1513 $self->{state} = 'DOCTYPE system identifier (single-quoted)';
1514 !!!next-input-character;
1515 redo A;
1516 } elsif ($self->{next_input_character} == 0x003E) { # >
1517 !!!parse-error (type => 'no SYSTEM literal');
1518 $self->{state} = 'data';
1519 !!!next-input-character;
1520
1521 delete $self->{current_token}->{correct};
1522 !!!emit ($self->{current_token}); # DOCTYPE
1523
1524 redo A;
1525 } elsif ($self->{next_input_character} == -1) {
1526 !!!parse-error (type => 'unclosed DOCTYPE');
1527
1528 $self->{state} = 'data';
1529 ## reconsume
1530
1531 delete $self->{current_token}->{correct};
1532 !!!emit ($self->{current_token}); # DOCTYPE
1533
1534 redo A;
1535 } else {
1536 !!!parse-error (type => 'string after SYSTEM');
1537 $self->{state} = 'bogus DOCTYPE';
1538 !!!next-input-character;
1539 redo A;
1540 }
1541 } elsif ($self->{state} eq 'DOCTYPE system identifier (double-quoted)') {
1542 if ($self->{next_input_character} == 0x0022) { # "
1543 $self->{state} = 'after DOCTYPE system identifier';
1544 !!!next-input-character;
1545 redo A;
1546 } elsif ($self->{next_input_character} == -1) {
1547 !!!parse-error (type => 'unclosed SYSTEM literal');
1548
1549 $self->{state} = 'data';
1550 ## reconsume
1551
1552 delete $self->{current_token}->{correct};
1553 !!!emit ($self->{current_token}); # DOCTYPE
1554
1555 redo A;
1556 } else {
1557 $self->{current_token}->{system_identifier} # DOCTYPE
1558 .= chr $self->{next_input_character};
1559 ## Stay in the state
1560 !!!next-input-character;
1561 redo A;
1562 }
1563 } elsif ($self->{state} eq 'DOCTYPE system identifier (single-quoted)') {
1564 if ($self->{next_input_character} == 0x0027) { # '
1565 $self->{state} = 'after DOCTYPE system identifier';
1566 !!!next-input-character;
1567 redo A;
1568 } elsif ($self->{next_input_character} == -1) {
1569 !!!parse-error (type => 'unclosed SYSTEM literal');
1570
1571 $self->{state} = 'data';
1572 ## reconsume
1573
1574 delete $self->{current_token}->{correct};
1575 !!!emit ($self->{current_token}); # DOCTYPE
1576
1577 redo A;
1578 } else {
1579 $self->{current_token}->{system_identifier} # DOCTYPE
1580 .= chr $self->{next_input_character};
1581 ## Stay in the state
1582 !!!next-input-character;
1583 redo A;
1584 }
1585 } elsif ($self->{state} eq 'after DOCTYPE system identifier') {
1586 if ({
1587 0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,
1588 #0x000D => 1, # HT, LF, VT, FF, SP, CR
1589 }->{$self->{next_input_character}}) {
1590 ## Stay in the state
1591 !!!next-input-character;
1592 redo A;
1593 } elsif ($self->{next_input_character} == 0x003E) { # >
1594 $self->{state} = 'data';
1595 !!!next-input-character;
1596
1597 !!!emit ($self->{current_token}); # DOCTYPE
1598
1599 redo A;
1600 } elsif ($self->{next_input_character} == -1) {
1601 !!!parse-error (type => 'unclosed DOCTYPE');
1602
1603 $self->{state} = 'data';
1604 ## reconsume
1605
1606 delete $self->{current_token}->{correct};
1607 !!!emit ($self->{current_token}); # DOCTYPE
1608
1609 redo A;
1610 } else {
1611 !!!parse-error (type => 'string after SYSTEM literal');
1612 $self->{state} = 'bogus DOCTYPE';
1613 !!!next-input-character;
1614 redo A;
1615 }
1616 } elsif ($self->{state} eq 'bogus DOCTYPE') {
1617 if ($self->{next_input_character} == 0x003E) { # >
1618 $self->{state} = 'data';
1619 !!!next-input-character;
1620
1621 delete $self->{current_token}->{correct};
1622 !!!emit ($self->{current_token}); # DOCTYPE
1623
1624 redo A;
1625 } elsif ($self->{next_input_character} == -1) {
1626 !!!parse-error (type => 'unclosed DOCTYPE');
1627 $self->{state} = 'data';
1628 ## reconsume
1629
1630 delete $self->{current_token}->{correct};
1631 !!!emit ($self->{current_token}); # DOCTYPE
1632
1633 redo A;
1634 } else {
1635 ## Stay in the state
1636 !!!next-input-character;
1637 redo A;
1638 }
1639 } else {
1640 die "$0: $self->{state}: Unknown state";
1641 }
1642 } # A
1643
1644 die "$0: _get_next_token: unexpected case";
1645 } # _get_next_token
1646
1647 sub _tokenize_attempt_to_consume_an_entity ($$) {
1648 my ($self, $in_attr) = @_;
1649
1650 if ({
1651 0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, # HT, LF, VT, FF,
1652 0x0020 => 1, 0x003C => 1, 0x0026 => 1, -1 => 1, # SP, <, & # 0x000D # CR
1653 }->{$self->{next_input_character}}) {
1654 ## Don't consume
1655 ## No error
1656 return undef;
1657 } elsif ($self->{next_input_character} == 0x0023) { # #
1658 !!!next-input-character;
1659 if ($self->{next_input_character} == 0x0078 or # x
1660 $self->{next_input_character} == 0x0058) { # X
1661 my $code;
1662 X: {
1663 my $x_char = $self->{next_input_character};
1664 !!!next-input-character;
1665 if (0x0030 <= $self->{next_input_character} and
1666 $self->{next_input_character} <= 0x0039) { # 0..9
1667 $code ||= 0;
1668 $code *= 0x10;
1669 $code += $self->{next_input_character} - 0x0030;
1670 redo X;
1671 } elsif (0x0061 <= $self->{next_input_character} and
1672 $self->{next_input_character} <= 0x0066) { # a..f
1673 $code ||= 0;
1674 $code *= 0x10;
1675 $code += $self->{next_input_character} - 0x0060 + 9;
1676 redo X;
1677 } elsif (0x0041 <= $self->{next_input_character} and
1678 $self->{next_input_character} <= 0x0046) { # A..F
1679 $code ||= 0;
1680 $code *= 0x10;
1681 $code += $self->{next_input_character} - 0x0040 + 9;
1682 redo X;
1683 } elsif (not defined $code) { # no hexadecimal digit
1684 !!!parse-error (type => 'bare hcro');
1685 !!!back-next-input-character ($x_char, $self->{next_input_character});
1686 $self->{next_input_character} = 0x0023; # #
1687 return undef;
1688 } elsif ($self->{next_input_character} == 0x003B) { # ;
1689 !!!next-input-character;
1690 } else {
1691 !!!parse-error (type => 'no refc');
1692 }
1693
1694 if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) {
1695 !!!parse-error (type => sprintf 'invalid character reference:U+%04X', $code);
1696 $code = 0xFFFD;
1697 } elsif ($code > 0x10FFFF) {
1698 !!!parse-error (type => sprintf 'invalid character reference:U-%08X', $code);
1699 $code = 0xFFFD;
1700 } elsif ($code == 0x000D) {
1701 !!!parse-error (type => 'CR character reference');
1702 $code = 0x000A;
1703 } elsif (0x80 <= $code and $code <= 0x9F) {
1704 !!!parse-error (type => sprintf 'C1 character reference:U+%04X', $code);
1705 $code = $c1_entity_char->{$code};
1706 }
1707
1708 return {type => CHARACTER_TOKEN, data => chr $code};
1709 } # X
1710 } elsif (0x0030 <= $self->{next_input_character} and
1711 $self->{next_input_character} <= 0x0039) { # 0..9
1712 my $code = $self->{next_input_character} - 0x0030;
1713 !!!next-input-character;
1714
1715 while (0x0030 <= $self->{next_input_character} and
1716 $self->{next_input_character} <= 0x0039) { # 0..9
1717 $code *= 10;
1718 $code += $self->{next_input_character} - 0x0030;
1719
1720 !!!next-input-character;
1721 }
1722
1723 if ($self->{next_input_character} == 0x003B) { # ;
1724 !!!next-input-character;
1725 } else {
1726 !!!parse-error (type => 'no refc');
1727 }
1728
1729 if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) {
1730 !!!parse-error (type => sprintf 'invalid character reference:U+%04X', $code);
1731 $code = 0xFFFD;
1732 } elsif ($code > 0x10FFFF) {
1733 !!!parse-error (type => sprintf 'invalid character reference:U-%08X', $code);
1734 $code = 0xFFFD;
1735 } elsif ($code == 0x000D) {
1736 !!!parse-error (type => 'CR character reference');
1737 $code = 0x000A;
1738 } elsif (0x80 <= $code and $code <= 0x9F) {
1739 !!!parse-error (type => sprintf 'C1 character reference:U+%04X', $code);
1740 $code = $c1_entity_char->{$code};
1741 }
1742
1743 return {type => CHARACTER_TOKEN, data => chr $code};
1744 } else {
1745 !!!parse-error (type => 'bare nero');
1746 !!!back-next-input-character ($self->{next_input_character});
1747 $self->{next_input_character} = 0x0023; # #
1748 return undef;
1749 }
1750 } elsif ((0x0041 <= $self->{next_input_character} and
1751 $self->{next_input_character} <= 0x005A) or
1752 (0x0061 <= $self->{next_input_character} and
1753 $self->{next_input_character} <= 0x007A)) {
1754 my $entity_name = chr $self->{next_input_character};
1755 !!!next-input-character;
1756
1757 my $value = $entity_name;
1758 my $match = 0;
1759 require Whatpm::_NamedEntityList;
1760 our $EntityChar;
1761
1762 while (length $entity_name < 10 and
1763 ## NOTE: Some number greater than the maximum length of entity name
1764 ((0x0041 <= $self->{next_input_character} and # a
1765 $self->{next_input_character} <= 0x005A) or # x
1766 (0x0061 <= $self->{next_input_character} and # a
1767 $self->{next_input_character} <= 0x007A) or # z
1768 (0x0030 <= $self->{next_input_character} and # 0
1769 $self->{next_input_character} <= 0x0039) or # 9
1770 $self->{next_input_character} == 0x003B)) { # ;
1771 $entity_name .= chr $self->{next_input_character};
1772 if (defined $EntityChar->{$entity_name}) {
1773 if ($self->{next_input_character} == 0x003B) { # ;
1774 $value = $EntityChar->{$entity_name};
1775 $match = 1;
1776 !!!next-input-character;
1777 last;
1778 } else {
1779 $value = $EntityChar->{$entity_name};
1780 $match = -1;
1781 !!!next-input-character;
1782 }
1783 } else {
1784 $value .= chr $self->{next_input_character};
1785 $match *= 2;
1786 !!!next-input-character;
1787 }
1788 }
1789
1790 if ($match > 0) {
1791 return {type => CHARACTER_TOKEN, data => $value};
1792 } elsif ($match < 0) {
1793 !!!parse-error (type => 'no refc');
1794 if ($in_attr and $match < -1) {
1795 return {type => CHARACTER_TOKEN, data => '&'.$entity_name};
1796 } else {
1797 return {type => CHARACTER_TOKEN, data => $value};
1798 }
1799 } else {
1800 !!!parse-error (type => 'bare ero');
1801 ## NOTE: No characters are consumed in the spec.
1802 return {type => CHARACTER_TOKEN, data => '&'.$value};
1803 }
1804 } else {
1805 ## no characters are consumed
1806 !!!parse-error (type => 'bare ero');
1807 return undef;
1808 }
1809 } # _tokenize_attempt_to_consume_an_entity
1810
1811 sub _initialize_tree_constructor ($) {
1812 my $self = shift;
1813 ## NOTE: $self->{document} MUST be specified before this method is called
1814 $self->{document}->strict_error_checking (0);
1815 ## TODO: Turn mutation events off # MUST
1816 ## TODO: Turn loose Document option (manakai extension) on
1817 $self->{document}->manakai_is_html (1); # MUST
1818 } # _initialize_tree_constructor
1819
1820 sub _terminate_tree_constructor ($) {
1821 my $self = shift;
1822 $self->{document}->strict_error_checking (1);
1823 ## TODO: Turn mutation events on
1824 } # _terminate_tree_constructor
1825
1826 ## ISSUE: Should append_child (for example) in script executed in tree construction stage fire mutation events?
1827
1828 { # tree construction stage
1829 my $token;
1830
1831 sub _construct_tree ($) {
1832 my ($self) = @_;
1833
1834 ## When an interactive UA render the $self->{document} available
1835 ## to the user, or when it begin accepting user input, are
1836 ## not defined.
1837
1838 ## Append a character: collect it and all subsequent consecutive
1839 ## characters and insert one Text node whose data is concatenation
1840 ## of all those characters. # MUST
1841
1842 !!!next-token;
1843
1844 $self->{insertion_mode} = BEFORE_HEAD_IM;
1845 undef $self->{form_element};
1846 undef $self->{head_element};
1847 $self->{open_elements} = [];
1848 undef $self->{inner_html_node};
1849
1850 $self->_tree_construction_initial; # MUST
1851 $self->_tree_construction_root_element;
1852 $self->_tree_construction_main;
1853 } # _construct_tree
1854
1855 sub _tree_construction_initial ($) {
1856 my $self = shift;
1857 INITIAL: {
1858 if ($token->{type} == DOCTYPE_TOKEN) {
1859 ## NOTE: Conformance checkers MAY, instead of reporting "not HTML5"
1860 ## error, switch to a conformance checking mode for another
1861 ## language.
1862 my $doctype_name = $token->{name};
1863 $doctype_name = '' unless defined $doctype_name;
1864 $doctype_name =~ tr/a-z/A-Z/;
1865 if (not defined $token->{name} or # <!DOCTYPE>
1866 defined $token->{public_identifier} or
1867 defined $token->{system_identifier}) {
1868 !!!parse-error (type => 'not HTML5');
1869 } elsif ($doctype_name ne 'HTML') {
1870 ## ISSUE: ASCII case-insensitive? (in fact it does not matter)
1871 !!!parse-error (type => 'not HTML5');
1872 }
1873
1874 my $doctype = $self->{document}->create_document_type_definition
1875 ($token->{name}); ## ISSUE: If name is missing (e.g. <!DOCTYPE>)?
1876 $doctype->public_id ($token->{public_identifier})
1877 if defined $token->{public_identifier};
1878 $doctype->system_id ($token->{system_identifier})
1879 if defined $token->{system_identifier};
1880 ## NOTE: Other DocumentType attributes are null or empty lists.
1881 ## ISSUE: internalSubset = null??
1882 $self->{document}->append_child ($doctype);
1883
1884 if (not $token->{correct} or $doctype_name ne 'HTML') {
1885 $self->{document}->manakai_compat_mode ('quirks');
1886 } elsif (defined $token->{public_identifier}) {
1887 my $pubid = $token->{public_identifier};
1888 $pubid =~ tr/a-z/A-z/;
1889 if ({
1890 "+//SILMARIL//DTD HTML PRO V0R11 19970101//EN" => 1,
1891 "-//ADVASOFT LTD//DTD HTML 3.0 ASWEDIT + EXTENSIONS//EN" => 1,
1892 "-//AS//DTD HTML 3.0 ASWEDIT + EXTENSIONS//EN" => 1,
1893 "-//IETF//DTD HTML 2.0 LEVEL 1//EN" => 1,
1894 "-//IETF//DTD HTML 2.0 LEVEL 2//EN" => 1,
1895 "-//IETF//DTD HTML 2.0 STRICT LEVEL 1//EN" => 1,
1896 "-//IETF//DTD HTML 2.0 STRICT LEVEL 2//EN" => 1,
1897 "-//IETF//DTD HTML 2.0 STRICT//EN" => 1,
1898 "-//IETF//DTD HTML 2.0//EN" => 1,
1899 "-//IETF//DTD HTML 2.1E//EN" => 1,
1900 "-//IETF//DTD HTML 3.0//EN" => 1,
1901 "-//IETF//DTD HTML 3.0//EN//" => 1,
1902 "-//IETF//DTD HTML 3.2 FINAL//EN" => 1,
1903 "-//IETF//DTD HTML 3.2//EN" => 1,
1904 "-//IETF//DTD HTML 3//EN" => 1,
1905 "-//IETF//DTD HTML LEVEL 0//EN" => 1,
1906 "-//IETF//DTD HTML LEVEL 0//EN//2.0" => 1,
1907 "-//IETF//DTD HTML LEVEL 1//EN" => 1,
1908 "-//IETF//DTD HTML LEVEL 1//EN//2.0" => 1,
1909 "-//IETF//DTD HTML LEVEL 2//EN" => 1,
1910 "-//IETF//DTD HTML LEVEL 2//EN//2.0" => 1,
1911 "-//IETF//DTD HTML LEVEL 3//EN" => 1,
1912 "-//IETF//DTD HTML LEVEL 3//EN//3.0" => 1,
1913 "-//IETF//DTD HTML STRICT LEVEL 0//EN" => 1,
1914 "-//IETF//DTD HTML STRICT LEVEL 0//EN//2.0" => 1,
1915 "-//IETF//DTD HTML STRICT LEVEL 1//EN" => 1,
1916 "-//IETF//DTD HTML STRICT LEVEL 1//EN//2.0" => 1,
1917 "-//IETF//DTD HTML STRICT LEVEL 2//EN" => 1,
1918 "-//IETF//DTD HTML STRICT LEVEL 2//EN//2.0" => 1,
1919 "-//IETF//DTD HTML STRICT LEVEL 3//EN" => 1,
1920 "-//IETF//DTD HTML STRICT LEVEL 3//EN//3.0" => 1,
1921 "-//IETF//DTD HTML STRICT//EN" => 1,
1922 "-//IETF//DTD HTML STRICT//EN//2.0" => 1,
1923 "-//IETF//DTD HTML STRICT//EN//3.0" => 1,
1924 "-//IETF//DTD HTML//EN" => 1,
1925 "-//IETF//DTD HTML//EN//2.0" => 1,
1926 "-//IETF//DTD HTML//EN//3.0" => 1,
1927 "-//METRIUS//DTD METRIUS PRESENTATIONAL//EN" => 1,
1928 "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 HTML STRICT//EN" => 1,
1929 "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 HTML//EN" => 1,
1930 "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 TABLES//EN" => 1,
1931 "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 HTML STRICT//EN" => 1,
1932 "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 HTML//EN" => 1,
1933 "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 TABLES//EN" => 1,
1934 "-//NETSCAPE COMM. CORP.//DTD HTML//EN" => 1,
1935 "-//NETSCAPE COMM. CORP.//DTD STRICT HTML//EN" => 1,
1936 "-//O'REILLY AND ASSOCIATES//DTD HTML 2.0//EN" => 1,
1937 "-//O'REILLY AND ASSOCIATES//DTD HTML EXTENDED 1.0//EN" => 1,
1938 "-//SPYGLASS//DTD HTML 2.0 EXTENDED//EN" => 1,
1939 "-//SQ//DTD HTML 2.0 HOTMETAL + EXTENSIONS//EN" => 1,
1940 "-//SUN MICROSYSTEMS CORP.//DTD HOTJAVA HTML//EN" => 1,
1941 "-//SUN MICROSYSTEMS CORP.//DTD HOTJAVA STRICT HTML//EN" => 1,
1942 "-//W3C//DTD HTML 3 1995-03-24//EN" => 1,
1943 "-//W3C//DTD HTML 3.2 DRAFT//EN" => 1,
1944 "-//W3C//DTD HTML 3.2 FINAL//EN" => 1,
1945 "-//W3C//DTD HTML 3.2//EN" => 1,
1946 "-//W3C//DTD HTML 3.2S DRAFT//EN" => 1,
1947 "-//W3C//DTD HTML 4.0 FRAMESET//EN" => 1,
1948 "-//W3C//DTD HTML 4.0 TRANSITIONAL//EN" => 1,
1949 "-//W3C//DTD HTML EXPERIMETNAL 19960712//EN" => 1,
1950 "-//W3C//DTD HTML EXPERIMENTAL 970421//EN" => 1,
1951 "-//W3C//DTD W3 HTML//EN" => 1,
1952 "-//W3O//DTD W3 HTML 3.0//EN" => 1,
1953 "-//W3O//DTD W3 HTML 3.0//EN//" => 1,
1954 "-//W3O//DTD W3 HTML STRICT 3.0//EN//" => 1,
1955 "-//WEBTECHS//DTD MOZILLA HTML 2.0//EN" => 1,
1956 "-//WEBTECHS//DTD MOZILLA HTML//EN" => 1,
1957 "-/W3C/DTD HTML 4.0 TRANSITIONAL/EN" => 1,
1958 "HTML" => 1,
1959 }->{$pubid}) {
1960 $self->{document}->manakai_compat_mode ('quirks');
1961 } elsif ($pubid eq "-//W3C//DTD HTML 4.01 FRAMESET//EN" or
1962 $pubid eq "-//W3C//DTD HTML 4.01 TRANSITIONAL//EN") {
1963 if (defined $token->{system_identifier}) {
1964 $self->{document}->manakai_compat_mode ('quirks');
1965 } else {
1966 $self->{document}->manakai_compat_mode ('limited quirks');
1967 }
1968 } elsif ($pubid eq "-//W3C//DTD XHTML 1.0 Frameset//EN" or
1969 $pubid eq "-//W3C//DTD XHTML 1.0 Transitional//EN") {
1970 $self->{document}->manakai_compat_mode ('limited quirks');
1971 }
1972 }
1973 if (defined $token->{system_identifier}) {
1974 my $sysid = $token->{system_identifier};
1975 $sysid =~ tr/A-Z/a-z/;
1976 if ($sysid eq "http://www.ibm.com/data/dtd/v11/ibmxhtml1-transitional.dtd") {
1977 $self->{document}->manakai_compat_mode ('quirks');
1978 }
1979 }
1980
1981 ## Go to the root element phase.
1982 !!!next-token;
1983 return;
1984 } elsif ({
1985 START_TAG_TOKEN, 1,
1986 END_TAG_TOKEN, 1,
1987 END_OF_FILE_TOKEN, 1,
1988 }->{$token->{type}}) {
1989 !!!parse-error (type => 'no DOCTYPE');
1990 $self->{document}->manakai_compat_mode ('quirks');
1991 ## Go to the root element phase
1992 ## reprocess
1993 return;
1994 } elsif ($token->{type} == CHARACTER_TOKEN) {
1995 if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D
1996 ## Ignore the token
1997
1998 unless (length $token->{data}) {
1999 ## Stay in the phase
2000 !!!next-token;
2001 redo INITIAL;
2002 }
2003 }
2004
2005 !!!parse-error (type => 'no DOCTYPE');
2006 $self->{document}->manakai_compat_mode ('quirks');
2007 ## Go to the root element phase
2008 ## reprocess
2009 return;
2010 } elsif ($token->{type} == COMMENT_TOKEN) {
2011 my $comment = $self->{document}->create_comment ($token->{data});
2012 $self->{document}->append_child ($comment);
2013
2014 ## Stay in the phase.
2015 !!!next-token;
2016 redo INITIAL;
2017 } else {
2018 die "$0: $token->{type}: Unknown token type";
2019 }
2020 } # INITIAL
2021 } # _tree_construction_initial
2022
2023 sub _tree_construction_root_element ($) {
2024 my $self = shift;
2025
2026 B: {
2027 if ($token->{type} == DOCTYPE_TOKEN) {
2028 !!!parse-error (type => 'in html:#DOCTYPE');
2029 ## Ignore the token
2030 ## Stay in the phase
2031 !!!next-token;
2032 redo B;
2033 } elsif ($token->{type} == COMMENT_TOKEN) {
2034 my $comment = $self->{document}->create_comment ($token->{data});
2035 $self->{document}->append_child ($comment);
2036 ## Stay in the phase
2037 !!!next-token;
2038 redo B;
2039 } elsif ($token->{type} == CHARACTER_TOKEN) {
2040 if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D
2041 ## Ignore the token.
2042
2043 unless (length $token->{data}) {
2044 ## Stay in the phase
2045 !!!next-token;
2046 redo B;
2047 }
2048 }
2049 #
2050 } elsif ({
2051 START_TAG_TOKEN, 1,
2052 END_TAG_TOKEN, 1,
2053 END_OF_FILE_TOKEN, 1,
2054 }->{$token->{type}}) {
2055 ## ISSUE: There is an issue in the spec
2056 #
2057 } else {
2058 die "$0: $token->{type}: Unknown token type";
2059 }
2060 my $root_element; !!!create-element ($root_element, 'html');
2061 $self->{document}->append_child ($root_element);
2062 push @{$self->{open_elements}}, [$root_element, 'html'];
2063 ## reprocess
2064 #redo B;
2065 return; ## Go to the main phase.
2066 } # B
2067 } # _tree_construction_root_element
2068
2069 sub _reset_insertion_mode ($) {
2070 my $self = shift;
2071
2072 ## Step 1
2073 my $last;
2074
2075 ## Step 2
2076 my $i = -1;
2077 my $node = $self->{open_elements}->[$i];
2078
2079 ## Step 3
2080 S3: {
2081 ## ISSUE: Oops! "If node is the first node in the stack of open
2082 ## elements, then set last to true. If the context element of the
2083 ## HTML fragment parsing algorithm is neither a td element nor a
2084 ## th element, then set node to the context element. (fragment case)":
2085 ## The second "if" is in the scope of the first "if"!?
2086 if ($self->{open_elements}->[0]->[0] eq $node->[0]) {
2087 $last = 1;
2088 if (defined $self->{inner_html_node}) {
2089 if ($self->{inner_html_node}->[1] eq 'td' or
2090 $self->{inner_html_node}->[1] eq 'th') {
2091 #
2092 } else {
2093 $node = $self->{inner_html_node};
2094 }
2095 }
2096 }
2097
2098 ## Step 4..13
2099 my $new_mode = {
2100 select => IN_SELECT_IM,
2101 td => IN_CELL_IM,
2102 th => IN_CELL_IM,
2103 tr => IN_ROW_IM,
2104 tbody => IN_TABLE_BODY_IM,
2105 thead => IN_TABLE_BODY_IM,
2106 tfoot => IN_TABLE_BODY_IM,
2107 caption => IN_CAPTION_IM,
2108 colgroup => IN_COLUMN_GROUP_IM,
2109 table => IN_TABLE_IM,
2110 head => IN_BODY_IM, # not in head!
2111 body => IN_BODY_IM,
2112 frameset => IN_FRAMESET_IM,
2113 }->{$node->[1]};
2114 $self->{insertion_mode} = $new_mode and return if defined $new_mode;
2115
2116 ## Step 14
2117 if ($node->[1] eq 'html') {
2118 unless (defined $self->{head_element}) {
2119 $self->{insertion_mode} = BEFORE_HEAD_IM;
2120 } else {
2121 $self->{insertion_mode} = AFTER_HEAD_IM;
2122 }
2123 return;
2124 }
2125
2126 ## Step 15
2127 $self->{insertion_mode} = IN_BODY_IM and return if $last;
2128
2129 ## Step 16
2130 $i--;
2131 $node = $self->{open_elements}->[$i];
2132
2133 ## Step 17
2134 redo S3;
2135 } # S3
2136 } # _reset_insertion_mode
2137
2138 sub _tree_construction_main ($) {
2139 my $self = shift;
2140
2141 my $active_formatting_elements = [];
2142
2143 my $reconstruct_active_formatting_elements = sub { # MUST
2144 my $insert = shift;
2145
2146 ## Step 1
2147 return unless @$active_formatting_elements;
2148
2149 ## Step 3
2150 my $i = -1;
2151 my $entry = $active_formatting_elements->[$i];
2152
2153 ## Step 2
2154 return if $entry->[0] eq '#marker';
2155 for (@{$self->{open_elements}}) {
2156 if ($entry->[0] eq $_->[0]) {
2157 return;
2158 }
2159 }
2160
2161 S4: {
2162 ## Step 4
2163 last S4 if $active_formatting_elements->[0]->[0] eq $entry->[0];
2164
2165 ## Step 5
2166 $i--;
2167 $entry = $active_formatting_elements->[$i];
2168
2169 ## Step 6
2170 if ($entry->[0] eq '#marker') {
2171 #
2172 } else {
2173 my $in_open_elements;
2174 OE: for (@{$self->{open_elements}}) {
2175 if ($entry->[0] eq $_->[0]) {
2176 $in_open_elements = 1;
2177 last OE;
2178 }
2179 }
2180 if ($in_open_elements) {
2181 #
2182 } else {
2183 redo S4;
2184 }
2185 }
2186
2187 ## Step 7
2188 $i++;
2189 $entry = $active_formatting_elements->[$i];
2190 } # S4
2191
2192 S7: {
2193 ## Step 8
2194 my $clone = [$entry->[0]->clone_node (0), $entry->[1]];
2195
2196 ## Step 9
2197 $insert->($clone->[0]);
2198 push @{$self->{open_elements}}, $clone;
2199
2200 ## Step 10
2201 $active_formatting_elements->[$i] = $self->{open_elements}->[-1];
2202
2203 ## Step 11
2204 unless ($clone->[0] eq $active_formatting_elements->[-1]->[0]) {
2205 ## Step 7'
2206 $i++;
2207 $entry = $active_formatting_elements->[$i];
2208
2209 redo S7;
2210 }
2211 } # S7
2212 }; # $reconstruct_active_formatting_elements
2213
2214 my $clear_up_to_marker = sub {
2215 for (reverse 0..$#$active_formatting_elements) {
2216 if ($active_formatting_elements->[$_]->[0] eq '#marker') {
2217 splice @$active_formatting_elements, $_;
2218 return;
2219 }
2220 }
2221 }; # $clear_up_to_marker
2222
2223 my $parse_rcdata = sub ($$) {
2224 my ($content_model_flag, $insert) = @_;
2225
2226 ## Step 1
2227 my $start_tag_name = $token->{tag_name};
2228 my $el;
2229 !!!create-element ($el, $start_tag_name, $token->{attributes});
2230
2231 ## Step 2
2232 $insert->($el); # /context node/->append_child ($el)
2233
2234 ## Step 3
2235 $self->{content_model} = $content_model_flag; # CDATA or RCDATA
2236 delete $self->{escape}; # MUST
2237
2238 ## Step 4
2239 my $text = '';
2240 !!!next-token;
2241 while ($token->{type} == CHARACTER_TOKEN) { # or until stop tokenizing
2242 $text .= $token->{data};
2243 !!!next-token;
2244 }
2245
2246 ## Step 5
2247 if (length $text) {
2248 my $text = $self->{document}->create_text_node ($text);
2249 $el->append_child ($text);
2250 }
2251
2252 ## Step 6
2253 $self->{content_model} = PCDATA_CONTENT_MODEL;
2254
2255 ## Step 7
2256 if ($token->{type} == END_TAG_TOKEN and $token->{tag_name} eq $start_tag_name) {
2257 ## Ignore the token
2258 } elsif ($content_model_flag == CDATA_CONTENT_MODEL) {
2259 !!!parse-error (type => 'in CDATA:#'.$token->{type});
2260 } elsif ($content_model_flag == RCDATA_CONTENT_MODEL) {
2261 !!!parse-error (type => 'in RCDATA:#'.$token->{type});
2262 } else {
2263 die "$0: $content_model_flag in parse_rcdata";
2264 }
2265 !!!next-token;
2266 }; # $parse_rcdata
2267
2268 my $script_start_tag = sub ($) {
2269 my $insert = $_[0];
2270 my $script_el;
2271 !!!create-element ($script_el, 'script', $token->{attributes});
2272 ## TODO: mark as "parser-inserted"
2273
2274 $self->{content_model} = CDATA_CONTENT_MODEL;
2275 delete $self->{escape}; # MUST
2276
2277 my $text = '';
2278 !!!next-token;
2279 while ($token->{type} == CHARACTER_TOKEN) {
2280 $text .= $token->{data};
2281 !!!next-token;
2282 } # stop if non-character token or tokenizer stops tokenising
2283 if (length $text) {
2284 $script_el->manakai_append_text ($text);
2285 }
2286
2287 $self->{content_model} = PCDATA_CONTENT_MODEL;
2288
2289 if ($token->{type} == END_TAG_TOKEN and
2290 $token->{tag_name} eq 'script') {
2291 ## Ignore the token
2292 } else {
2293 !!!parse-error (type => 'in CDATA:#'.$token->{type});
2294 ## ISSUE: And ignore?
2295 ## TODO: mark as "already executed"
2296 }
2297
2298 if (defined $self->{inner_html_node}) {
2299 ## TODO: mark as "already executed"
2300 } else {
2301 ## TODO: $old_insertion_point = current insertion point
2302 ## TODO: insertion point = just before the next input character
2303
2304 $insert->($script_el);
2305
2306 ## TODO: insertion point = $old_insertion_point (might be "undefined")
2307
2308 ## TODO: if there is a script that will execute as soon as the parser resume, then...
2309 }
2310
2311 !!!next-token;
2312 }; # $script_start_tag
2313
2314 my $formatting_end_tag = sub {
2315 my $tag_name = shift;
2316
2317 FET: {
2318 ## Step 1
2319 my $formatting_element;
2320 my $formatting_element_i_in_active;
2321 AFE: for (reverse 0..$#$active_formatting_elements) {
2322 if ($active_formatting_elements->[$_]->[1] eq $tag_name) {
2323 $formatting_element = $active_formatting_elements->[$_];
2324 $formatting_element_i_in_active = $_;
2325 last AFE;
2326 } elsif ($active_formatting_elements->[$_]->[0] eq '#marker') {
2327 last AFE;
2328 }
2329 } # AFE
2330 unless (defined $formatting_element) {
2331 !!!parse-error (type => 'unmatched end tag:'.$tag_name);
2332 ## Ignore the token
2333 !!!next-token;
2334 return;
2335 }
2336 ## has an element in scope
2337 my $in_scope = 1;
2338 my $formatting_element_i_in_open;
2339 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
2340 my $node = $self->{open_elements}->[$_];
2341 if ($node->[0] eq $formatting_element->[0]) {
2342 if ($in_scope) {
2343 $formatting_element_i_in_open = $_;
2344 last INSCOPE;
2345 } else { # in open elements but not in scope
2346 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
2347 ## Ignore the token
2348 !!!next-token;
2349 return;
2350 }
2351 } elsif ({
2352 table => 1, caption => 1, td => 1, th => 1,
2353 button => 1, marquee => 1, object => 1, html => 1,
2354 }->{$node->[1]}) {
2355 $in_scope = 0;
2356 }
2357 } # INSCOPE
2358 unless (defined $formatting_element_i_in_open) {
2359 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
2360 pop @$active_formatting_elements; # $formatting_element
2361 !!!next-token; ## TODO: ok?
2362 return;
2363 }
2364 if (not $self->{open_elements}->[-1]->[0] eq $formatting_element->[0]) {
2365 !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);
2366 }
2367
2368 ## Step 2
2369 my $furthest_block;
2370 my $furthest_block_i_in_open;
2371 OE: for (reverse 0..$#{$self->{open_elements}}) {
2372 my $node = $self->{open_elements}->[$_];
2373 if (not $formatting_category->{$node->[1]} and
2374 #not $phrasing_category->{$node->[1]} and
2375 ($special_category->{$node->[1]} or
2376 $scoping_category->{$node->[1]})) {
2377 $furthest_block = $node;
2378 $furthest_block_i_in_open = $_;
2379 } elsif ($node->[0] eq $formatting_element->[0]) {
2380 last OE;
2381 }
2382 } # OE
2383
2384 ## Step 3
2385 unless (defined $furthest_block) { # MUST
2386 splice @{$self->{open_elements}}, $formatting_element_i_in_open;
2387 splice @$active_formatting_elements, $formatting_element_i_in_active, 1;
2388 !!!next-token;
2389 return;
2390 }
2391
2392 ## Step 4
2393 my $common_ancestor_node = $self->{open_elements}->[$formatting_element_i_in_open - 1];
2394
2395 ## Step 5
2396 my $furthest_block_parent = $furthest_block->[0]->parent_node;
2397 if (defined $furthest_block_parent) {
2398 $furthest_block_parent->remove_child ($furthest_block->[0]);
2399 }
2400
2401 ## Step 6
2402 my $bookmark_prev_el
2403 = $active_formatting_elements->[$formatting_element_i_in_active - 1]
2404 ->[0];
2405
2406 ## Step 7
2407 my $node = $furthest_block;
2408 my $node_i_in_open = $furthest_block_i_in_open;
2409 my $last_node = $furthest_block;
2410 S7: {
2411 ## Step 1
2412 $node_i_in_open--;
2413 $node = $self->{open_elements}->[$node_i_in_open];
2414
2415 ## Step 2
2416 my $node_i_in_active;
2417 S7S2: {
2418 for (reverse 0..$#$active_formatting_elements) {
2419 if ($active_formatting_elements->[$_]->[0] eq $node->[0]) {
2420 $node_i_in_active = $_;
2421 last S7S2;
2422 }
2423 }
2424 splice @{$self->{open_elements}}, $node_i_in_open, 1;
2425 redo S7;
2426 } # S7S2
2427
2428 ## Step 3
2429 last S7 if $node->[0] eq $formatting_element->[0];
2430
2431 ## Step 4
2432 if ($last_node->[0] eq $furthest_block->[0]) {
2433 $bookmark_prev_el = $node->[0];
2434 }
2435
2436 ## Step 5
2437 if ($node->[0]->has_child_nodes ()) {
2438 my $clone = [$node->[0]->clone_node (0), $node->[1]];
2439 $active_formatting_elements->[$node_i_in_active] = $clone;
2440 $self->{open_elements}->[$node_i_in_open] = $clone;
2441 $node = $clone;
2442 }
2443
2444 ## Step 6
2445 $node->[0]->append_child ($last_node->[0]);
2446
2447 ## Step 7
2448 $last_node = $node;
2449
2450 ## Step 8
2451 redo S7;
2452 } # S7
2453
2454 ## Step 8
2455 $common_ancestor_node->[0]->append_child ($last_node->[0]);
2456
2457 ## Step 9
2458 my $clone = [$formatting_element->[0]->clone_node (0),
2459 $formatting_element->[1]];
2460
2461 ## Step 10
2462 my @cn = @{$furthest_block->[0]->child_nodes};
2463 $clone->[0]->append_child ($_) for @cn;
2464
2465 ## Step 11
2466 $furthest_block->[0]->append_child ($clone->[0]);
2467
2468 ## Step 12
2469 my $i;
2470 AFE: for (reverse 0..$#$active_formatting_elements) {
2471 if ($active_formatting_elements->[$_]->[0] eq $formatting_element->[0]) {
2472 splice @$active_formatting_elements, $_, 1;
2473 $i-- and last AFE if defined $i;
2474 } elsif ($active_formatting_elements->[$_]->[0] eq $bookmark_prev_el) {
2475 $i = $_;
2476 }
2477 } # AFE
2478 splice @$active_formatting_elements, $i + 1, 0, $clone;
2479
2480 ## Step 13
2481 undef $i;
2482 OE: for (reverse 0..$#{$self->{open_elements}}) {
2483 if ($self->{open_elements}->[$_]->[0] eq $formatting_element->[0]) {
2484 splice @{$self->{open_elements}}, $_, 1;
2485 $i-- and last OE if defined $i;
2486 } elsif ($self->{open_elements}->[$_]->[0] eq $furthest_block->[0]) {
2487 $i = $_;
2488 }
2489 } # OE
2490 splice @{$self->{open_elements}}, $i + 1, 1, $clone;
2491
2492 ## Step 14
2493 redo FET;
2494 } # FET
2495 }; # $formatting_end_tag
2496
2497 my $insert_to_current = sub {
2498 $self->{open_elements}->[-1]->[0]->append_child ($_[0]);
2499 }; # $insert_to_current
2500
2501 my $insert_to_foster = sub {
2502 my $child = shift;
2503 if ({
2504 table => 1, tbody => 1, tfoot => 1,
2505 thead => 1, tr => 1,
2506 }->{$self->{open_elements}->[-1]->[1]}) {
2507 # MUST
2508 my $foster_parent_element;
2509 my $next_sibling;
2510 OE: for (reverse 0..$#{$self->{open_elements}}) {
2511 if ($self->{open_elements}->[$_]->[1] eq 'table') {
2512 my $parent = $self->{open_elements}->[$_]->[0]->parent_node;
2513 if (defined $parent and $parent->node_type == 1) {
2514 $foster_parent_element = $parent;
2515 $next_sibling = $self->{open_elements}->[$_]->[0];
2516 } else {
2517 $foster_parent_element
2518 = $self->{open_elements}->[$_ - 1]->[0];
2519 }
2520 last OE;
2521 }
2522 } # OE
2523 $foster_parent_element = $self->{open_elements}->[0]->[0]
2524 unless defined $foster_parent_element;
2525 $foster_parent_element->insert_before
2526 ($child, $next_sibling);
2527 } else {
2528 $self->{open_elements}->[-1]->[0]->append_child ($child);
2529 }
2530 }; # $insert_to_foster
2531
2532 my $insert;
2533
2534 B: {
2535 if ($token->{type} == DOCTYPE_TOKEN) {
2536 !!!parse-error (type => 'DOCTYPE in the middle');
2537 ## Ignore the token
2538 ## Stay in the phase
2539 !!!next-token;
2540 redo B;
2541 } elsif ($token->{type} == END_OF_FILE_TOKEN) {
2542 if ($self->{insertion_mode} == AFTER_HTML_BODY_IM or
2543 $self->{insertion_mode} == AFTER_HTML_FRAMESET_IM) {
2544 #
2545 } else {
2546 ## Generate implied end tags
2547 if ({
2548 dd => 1, dt => 1, li => 1, p => 1, td => 1, th => 1, tr => 1,
2549 tbody => 1, tfoot=> 1, thead => 1,
2550 }->{$self->{open_elements}->[-1]->[1]}) {
2551 !!!back-token;
2552 $token = {type => END_TAG_TOKEN, tag_name => $self->{open_elements}->[-1]->[1]};
2553 redo B;
2554 }
2555
2556 if (@{$self->{open_elements}} > 2 or
2557 (@{$self->{open_elements}} == 2 and $self->{open_elements}->[1]->[1] ne 'body')) {
2558 !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);
2559 } elsif (defined $self->{inner_html_node} and
2560 @{$self->{open_elements}} > 1 and
2561 $self->{open_elements}->[1]->[1] ne 'body') {
2562 !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);
2563 }
2564
2565 ## ISSUE: There is an issue in the spec.
2566 }
2567
2568 ## Stop parsing
2569 last B;
2570 } elsif ($token->{type} == START_TAG_TOKEN and
2571 $token->{tag_name} eq 'html') {
2572 if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
2573 ## Turn into the main phase
2574 !!!parse-error (type => 'after html:html');
2575 $self->{insertion_mode} = AFTER_BODY_IM;
2576 } elsif ($self->{insertion_mode} == AFTER_HTML_FRAMESET_IM) {
2577 ## Turn into the main phase
2578 !!!parse-error (type => 'after html:html');
2579 $self->{insertion_mode} = AFTER_FRAMESET_IM;
2580 }
2581
2582 ## ISSUE: "aa<html>" is not a parse error.
2583 ## ISSUE: "<html>" in fragment is not a parse error.
2584 unless ($token->{first_start_tag}) {
2585 !!!parse-error (type => 'not first start tag');
2586 }
2587 my $top_el = $self->{open_elements}->[0]->[0];
2588 for my $attr_name (keys %{$token->{attributes}}) {
2589 unless ($top_el->has_attribute_ns (undef, $attr_name)) {
2590 $top_el->set_attribute_ns
2591 (undef, [undef, $attr_name],
2592 $token->{attributes}->{$attr_name}->{value});
2593 }
2594 }
2595 !!!next-token;
2596 redo B;
2597 } elsif ($token->{type} == COMMENT_TOKEN) {
2598 my $comment = $self->{document}->create_comment ($token->{data});
2599 if ($self->{insertion_mode} == AFTER_HTML_BODY_IM or
2600 $self->{insertion_mode} == AFTER_HTML_FRAMESET_IM) {
2601 $self->{document}->append_child ($comment);
2602 } elsif ($self->{insertion_mode} == AFTER_BODY_IM) {
2603 $self->{open_elements}->[0]->[0]->append_child ($comment);
2604 } else {
2605 $self->{open_elements}->[-1]->[0]->append_child ($comment);
2606 }
2607 !!!next-token;
2608 redo B;
2609 } elsif ($self->{insertion_mode} == IN_HEAD_IM or
2610 $self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM or
2611 $self->{insertion_mode} == AFTER_HEAD_IM or
2612 $self->{insertion_mode} == BEFORE_HEAD_IM) {
2613 if ($token->{type} == CHARACTER_TOKEN) {
2614 if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {
2615 $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
2616 unless (length $token->{data}) {
2617 !!!next-token;
2618 redo B;
2619 }
2620 }
2621
2622 if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
2623 ## As if <head>
2624 !!!create-element ($self->{head_element}, 'head');
2625 $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
2626 push @{$self->{open_elements}}, [$self->{head_element}, 'head'];
2627
2628 ## Reprocess in the "in head" insertion mode...
2629 pop @{$self->{open_elements}};
2630
2631 ## Reprocess in the "after head" insertion mode...
2632 } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
2633 ## As if </noscript>
2634 pop @{$self->{open_elements}};
2635 !!!parse-error (type => 'in noscript:#character');
2636
2637 ## Reprocess in the "in head" insertion mode...
2638 ## As if </head>
2639 pop @{$self->{open_elements}};
2640
2641 ## Reprocess in the "after head" insertion mode...
2642 } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
2643 pop @{$self->{open_elements}};
2644
2645 ## Reprocess in the "after head" insertion mode...
2646 }
2647
2648 ## "after head" insertion mode
2649 ## As if <body>
2650 !!!insert-element ('body');
2651 $self->{insertion_mode} = IN_BODY_IM;
2652 ## reprocess
2653 redo B;
2654 } elsif ($token->{type} == START_TAG_TOKEN) {
2655 if ($token->{tag_name} eq 'head') {
2656 if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
2657 !!!create-element ($self->{head_element}, $token->{tag_name}, $token->{attributes});
2658 $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
2659 push @{$self->{open_elements}}, [$self->{head_element}, $token->{tag_name}];
2660 $self->{insertion_mode} = IN_HEAD_IM;
2661 !!!next-token;
2662 redo B;
2663 } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
2664 #
2665 } else {
2666 !!!parse-error (type => 'in head:head'); # or in head noscript
2667 ## Ignore the token
2668 !!!next-token;
2669 redo B;
2670 }
2671 } elsif ($self->{insertion_mode} == BEFORE_HEAD_IM) {
2672 ## As if <head>
2673 !!!create-element ($self->{head_element}, 'head');
2674 $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
2675 push @{$self->{open_elements}}, [$self->{head_element}, 'head'];
2676
2677 $self->{insertion_mode} = IN_HEAD_IM;
2678 ## Reprocess in the "in head" insertion mode...
2679 }
2680
2681 if ($token->{tag_name} eq 'base') {
2682 if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
2683 ## As if </noscript>
2684 pop @{$self->{open_elements}};
2685 !!!parse-error (type => 'in noscript:base');
2686
2687 $self->{insertion_mode} = IN_HEAD_IM;
2688 ## Reprocess in the "in head" insertion mode...
2689 }
2690
2691 ## NOTE: There is a "as if in head" code clone.
2692 if ($self->{insertion_mode} == AFTER_HEAD_IM) {
2693 !!!parse-error (type => 'after head:'.$token->{tag_name});
2694 push @{$self->{open_elements}}, [$self->{head_element}, 'head'];
2695 }
2696 !!!insert-element ($token->{tag_name}, $token->{attributes});
2697 pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.
2698 pop @{$self->{open_elements}}
2699 if $self->{insertion_mode} == AFTER_HEAD_IM;
2700 !!!next-token;
2701 redo B;
2702 } elsif ($token->{tag_name} eq 'link') {
2703 ## NOTE: There is a "as if in head" code clone.
2704 if ($self->{insertion_mode} == AFTER_HEAD_IM) {
2705 !!!parse-error (type => 'after head:'.$token->{tag_name});
2706 push @{$self->{open_elements}}, [$self->{head_element}, 'head'];
2707 }
2708 !!!insert-element ($token->{tag_name}, $token->{attributes});
2709 pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.
2710 pop @{$self->{open_elements}}
2711 if $self->{insertion_mode} == AFTER_HEAD_IM;
2712 !!!next-token;
2713 redo B;
2714 } elsif ($token->{tag_name} eq 'meta') {
2715 ## NOTE: There is a "as if in head" code clone.
2716 if ($self->{insertion_mode} == AFTER_HEAD_IM) {
2717 !!!parse-error (type => 'after head:'.$token->{tag_name});
2718 push @{$self->{open_elements}}, [$self->{head_element}, 'head'];
2719 }
2720 !!!insert-element ($token->{tag_name}, $token->{attributes});
2721 pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.
2722
2723 unless ($self->{confident}) {
2724 my $charset;
2725 if ($token->{attributes}->{charset}) { ## TODO: And if supported
2726 $charset = $token->{attributes}->{charset}->{value};
2727 }
2728 if ($token->{attributes}->{'http-equiv'}) {
2729 ## ISSUE: Algorithm name in the spec was incorrect so that not linked to the definition.
2730 if ($token->{attributes}->{'http-equiv'}->{value}
2731 =~ /\A[^;]*;[\x09-\x0D\x20]*charset[\x09-\x0D\x20]*=
2732 [\x09-\x0D\x20]*(?>"([^"]*)"|'([^']*)'|
2733 ([^"'\x09-\x0D\x20][^\x09-\x0D\x20]*))/x) {
2734 $charset = defined $1 ? $1 : defined $2 ? $2 : $3;
2735 } ## TODO: And if supported
2736 }
2737 ## TODO: Change the encoding
2738 }
2739
2740 ## TODO: Extracting |charset| from |meta|.
2741 pop @{$self->{open_elements}}
2742 if $self->{insertion_mode} == AFTER_HEAD_IM;
2743 !!!next-token;
2744 redo B;
2745 } elsif ($token->{tag_name} eq 'title') {
2746 if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
2747 ## As if </noscript>
2748 pop @{$self->{open_elements}};
2749 !!!parse-error (type => 'in noscript:title');
2750
2751 $self->{insertion_mode} = IN_HEAD_IM;
2752 ## Reprocess in the "in head" insertion mode...
2753 } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
2754 !!!parse-error (type => 'after head:'.$token->{tag_name});
2755 push @{$self->{open_elements}}, [$self->{head_element}, 'head'];
2756 }
2757
2758 ## NOTE: There is a "as if in head" code clone.
2759 my $parent = defined $self->{head_element} ? $self->{head_element}
2760 : $self->{open_elements}->[-1]->[0];
2761 $parse_rcdata->(RCDATA_CONTENT_MODEL,
2762 sub { $parent->append_child ($_[0]) });
2763 pop @{$self->{open_elements}}
2764 if $self->{insertion_mode} == AFTER_HEAD_IM;
2765 redo B;
2766 } elsif ($token->{tag_name} eq 'style') {
2767 ## NOTE: Or (scripting is enabled and tag_name eq 'noscript' and
2768 ## insertion mode IN_HEAD_IM)
2769 ## NOTE: There is a "as if in head" code clone.
2770 if ($self->{insertion_mode} == AFTER_HEAD_IM) {
2771 !!!parse-error (type => 'after head:'.$token->{tag_name});
2772 push @{$self->{open_elements}}, [$self->{head_element}, 'head'];
2773 }
2774 $parse_rcdata->(CDATA_CONTENT_MODEL, $insert_to_current);
2775 pop @{$self->{open_elements}}
2776 if $self->{insertion_mode} == AFTER_HEAD_IM;
2777 redo B;
2778 } elsif ($token->{tag_name} eq 'noscript') {
2779 if ($self->{insertion_mode} == IN_HEAD_IM) {
2780 ## NOTE: and scripting is disalbed
2781 !!!insert-element ($token->{tag_name}, $token->{attributes});
2782 $self->{insertion_mode} = IN_HEAD_NOSCRIPT_IM;
2783 !!!next-token;
2784 redo B;
2785 } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
2786 !!!parse-error (type => 'in noscript:noscript');
2787 ## Ignore the token
2788 !!!next-token;
2789 redo B;
2790 } else {
2791 #
2792 }
2793 } elsif ($token->{tag_name} eq 'script') {
2794 if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
2795 ## As if </noscript>
2796 pop @{$self->{open_elements}};
2797 !!!parse-error (type => 'in noscript:script');
2798
2799 $self->{insertion_mode} = IN_HEAD_IM;
2800 ## Reprocess in the "in head" insertion mode...
2801 } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
2802 !!!parse-error (type => 'after head:'.$token->{tag_name});
2803 push @{$self->{open_elements}}, [$self->{head_element}, 'head'];
2804 }
2805
2806 ## NOTE: There is a "as if in head" code clone.
2807 $script_start_tag->($insert_to_current);
2808 pop @{$self->{open_elements}}
2809 if $self->{insertion_mode} == AFTER_HEAD_IM;
2810 redo B;
2811 } elsif ($token->{tag_name} eq 'body' or
2812 $token->{tag_name} eq 'frameset') {
2813 if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
2814 ## As if </noscript>
2815 pop @{$self->{open_elements}};
2816 !!!parse-error (type => 'in noscript:'.$token->{tag_name});
2817
2818 ## Reprocess in the "in head" insertion mode...
2819 ## As if </head>
2820 pop @{$self->{open_elements}};
2821
2822 ## Reprocess in the "after head" insertion mode...
2823 } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
2824 pop @{$self->{open_elements}};
2825
2826 ## Reprocess in the "after head" insertion mode...
2827 }
2828
2829 ## "after head" insertion mode
2830 !!!insert-element ($token->{tag_name}, $token->{attributes});
2831 if ($token->{tag_name} eq 'body') {
2832 $self->{insertion_mode} = IN_BODY_IM;
2833 } elsif ($token->{tag_name} eq 'frameset') {
2834 $self->{insertion_mode} = IN_FRAMESET_IM;
2835 } else {
2836 die "$0: tag name: $self->{tag_name}";
2837 }
2838 !!!next-token;
2839 redo B;
2840 } else {
2841 #
2842 }
2843
2844 if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
2845 ## As if </noscript>
2846 pop @{$self->{open_elements}};
2847 !!!parse-error (type => 'in noscript:/'.$token->{tag_name});
2848
2849 ## Reprocess in the "in head" insertion mode...
2850 ## As if </head>
2851 pop @{$self->{open_elements}};
2852
2853 ## Reprocess in the "after head" insertion mode...
2854 } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
2855 ## As if </head>
2856 pop @{$self->{open_elements}};
2857
2858 ## Reprocess in the "after head" insertion mode...
2859 }
2860
2861 ## "after head" insertion mode
2862 ## As if <body>
2863 !!!insert-element ('body');
2864 $self->{insertion_mode} = IN_BODY_IM;
2865 ## reprocess
2866 redo B;
2867 } elsif ($token->{type} == END_TAG_TOKEN) {
2868 if ($token->{tag_name} eq 'head') {
2869 if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
2870 ## As if <head>
2871 !!!create-element ($self->{head_element}, 'head');
2872 $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
2873 push @{$self->{open_elements}}, [$self->{head_element}, 'head'];
2874
2875 ## Reprocess in the "in head" insertion mode...
2876 pop @{$self->{open_elements}};
2877 $self->{insertion_mode} = AFTER_HEAD_IM;
2878 !!!next-token;
2879 redo B;
2880 } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
2881 ## As if </noscript>
2882 pop @{$self->{open_elements}};
2883 !!!parse-error (type => 'in noscript:script');
2884
2885 ## Reprocess in the "in head" insertion mode...
2886 pop @{$self->{open_elements}};
2887 $self->{insertion_mode} = AFTER_HEAD_IM;
2888 !!!next-token;
2889 redo B;
2890 } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
2891 pop @{$self->{open_elements}};
2892 $self->{insertion_mode} = AFTER_HEAD_IM;
2893 !!!next-token;
2894 redo B;
2895 } else {
2896 #
2897 }
2898 } elsif ($token->{tag_name} eq 'noscript') {
2899 if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
2900 pop @{$self->{open_elements}};
2901 $self->{insertion_mode} = IN_HEAD_IM;
2902 !!!next-token;
2903 redo B;
2904 } elsif ($self->{insertion_mode} == BEFORE_HEAD_IM) {
2905 !!!parse-error (type => 'unmatched end tag:noscript');
2906 ## Ignore the token ## ISSUE: An issue in the spec.
2907 !!!next-token;
2908 redo B;
2909 } else {
2910 #
2911 }
2912 } elsif ({
2913 body => 1, html => 1,
2914 }->{$token->{tag_name}}) {
2915 if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
2916 ## As if <head>
2917 !!!create-element ($self->{head_element}, 'head');
2918 $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
2919 push @{$self->{open_elements}}, [$self->{head_element}, 'head'];
2920
2921 $self->{insertion_mode} = IN_HEAD_IM;
2922 ## Reprocess in the "in head" insertion mode...
2923 } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
2924 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
2925 ## Ignore the token
2926 !!!next-token;
2927 redo B;
2928 }
2929
2930 #
2931 } elsif ({
2932 p => 1, br => 1,
2933 }->{$token->{tag_name}}) {
2934 if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
2935 ## As if <head>
2936 !!!create-element ($self->{head_element}, 'head');
2937 $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
2938 push @{$self->{open_elements}}, [$self->{head_element}, 'head'];
2939
2940 $self->{insertion_mode} = IN_HEAD_IM;
2941 ## Reprocess in the "in head" insertion mode...
2942 }
2943
2944 #
2945 } else {
2946 if ($self->{insertion_mode} == AFTER_HEAD_IM) {
2947 #
2948 } else {
2949 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
2950 ## Ignore the token
2951 !!!next-token;
2952 redo B;
2953 }
2954 }
2955
2956 if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
2957 ## As if </noscript>
2958 pop @{$self->{open_elements}};
2959 !!!parse-error (type => 'in noscript:/'.$token->{tag_name});
2960
2961 ## Reprocess in the "in head" insertion mode...
2962 ## As if </head>
2963 pop @{$self->{open_elements}};
2964
2965 ## Reprocess in the "after head" insertion mode...
2966 } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
2967 ## As if </head>
2968 pop @{$self->{open_elements}};
2969
2970 ## Reprocess in the "after head" insertion mode...
2971 } elsif ($self->{insertion_mode} == BEFORE_HEAD_IM) {
2972 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
2973 ## Ignore the token ## ISSUE: An issue in the spec.
2974 !!!next-token;
2975 redo B;
2976 }
2977
2978 ## "after head" insertion mode
2979 ## As if <body>
2980 !!!insert-element ('body');
2981 $self->{insertion_mode} = IN_BODY_IM;
2982 ## reprocess
2983 redo B;
2984 } else {
2985 die "$0: $token->{type}: Unknown token type";
2986 }
2987
2988 ## ISSUE: An issue in the spec.
2989 } elsif ($self->{insertion_mode} == IN_BODY_IM or
2990 $self->{insertion_mode} == IN_CELL_IM or
2991 $self->{insertion_mode} == IN_CAPTION_IM) {
2992 if ($token->{type} == CHARACTER_TOKEN) {
2993 ## NOTE: There is a code clone of "character in body".
2994 $reconstruct_active_formatting_elements->($insert_to_current);
2995
2996 $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});
2997
2998 !!!next-token;
2999 redo B;
3000 } elsif ($token->{type} == START_TAG_TOKEN) {
3001 if ({
3002 caption => 1, col => 1, colgroup => 1, tbody => 1,
3003 td => 1, tfoot => 1, th => 1, thead => 1, tr => 1,
3004 }->{$token->{tag_name}}) {
3005 if ($self->{insertion_mode} == IN_CELL_IM) {
3006 ## have an element in table scope
3007 my $tn;
3008 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3009 my $node = $self->{open_elements}->[$_];
3010 if ($node->[1] eq 'td' or $node->[1] eq 'th') {
3011 $tn = $node->[1];
3012 last INSCOPE;
3013 } elsif ({
3014 table => 1, html => 1,
3015 }->{$node->[1]}) {
3016 last INSCOPE;
3017 }
3018 } # INSCOPE
3019 unless (defined $tn) {
3020 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
3021 ## Ignore the token
3022 !!!next-token;
3023 redo B;
3024 }
3025
3026 ## Close the cell
3027 !!!back-token; # <?>
3028 $token = {type => END_TAG_TOKEN, tag_name => $tn};
3029 redo B;
3030 } elsif ($self->{insertion_mode} == IN_CAPTION_IM) {
3031 !!!parse-error (type => 'not closed:caption');
3032
3033 ## As if </caption>
3034 ## have a table element in table scope
3035 my $i;
3036 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3037 my $node = $self->{open_elements}->[$_];
3038 if ($node->[1] eq 'caption') {
3039 $i = $_;
3040 last INSCOPE;
3041 } elsif ({
3042 table => 1, html => 1,
3043 }->{$node->[1]}) {
3044 last INSCOPE;
3045 }
3046 } # INSCOPE
3047 unless (defined $i) {
3048 !!!parse-error (type => 'unmatched end tag:caption');
3049 ## Ignore the token
3050 !!!next-token;
3051 redo B;
3052 }
3053
3054 ## generate implied end tags
3055 if ({
3056 dd => 1, dt => 1, li => 1, p => 1,
3057 td => 1, th => 1, tr => 1,
3058 tbody => 1, tfoot=> 1, thead => 1,
3059 }->{$self->{open_elements}->[-1]->[1]}) {
3060 !!!back-token; # <?>
3061 $token = {type => END_TAG_TOKEN, tag_name => 'caption'};
3062 !!!back-token;
3063 $token = {type => END_TAG_TOKEN,
3064 tag_name => $self->{open_elements}->[-1]->[1]}; # MUST
3065 redo B;
3066 }
3067
3068 if ($self->{open_elements}->[-1]->[1] ne 'caption') {
3069 !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);
3070 }
3071
3072 splice @{$self->{open_elements}}, $i;
3073
3074 $clear_up_to_marker->();
3075
3076 $self->{insertion_mode} = IN_TABLE_IM;
3077
3078 ## reprocess
3079 redo B;
3080 } else {
3081 #
3082 }
3083 } else {
3084 #
3085 }
3086 } elsif ($token->{type} == END_TAG_TOKEN) {
3087 if ($token->{tag_name} eq 'td' or $token->{tag_name} eq 'th') {
3088 if ($self->{insertion_mode} == IN_CELL_IM) {
3089 ## have an element in table scope
3090 my $i;
3091 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3092 my $node = $self->{open_elements}->[$_];
3093 if ($node->[1] eq $token->{tag_name}) {
3094 $i = $_;
3095 last INSCOPE;
3096 } elsif ({
3097 table => 1, html => 1,
3098 }->{$node->[1]}) {
3099 last INSCOPE;
3100 }
3101 } # INSCOPE
3102 unless (defined $i) {
3103 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
3104 ## Ignore the token
3105 !!!next-token;
3106 redo B;
3107 }
3108
3109 ## generate implied end tags
3110 if ({
3111 dd => 1, dt => 1, li => 1, p => 1,
3112 td => ($token->{tag_name} eq 'th'),
3113 th => ($token->{tag_name} eq 'td'),
3114 tr => 1,
3115 tbody => 1, tfoot=> 1, thead => 1,
3116 }->{$self->{open_elements}->[-1]->[1]}) {
3117 !!!back-token;
3118 $token = {type => END_TAG_TOKEN,
3119 tag_name => $self->{open_elements}->[-1]->[1]}; # MUST
3120 redo B;
3121 }
3122
3123 if ($self->{open_elements}->[-1]->[1] ne $token->{tag_name}) {
3124 !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);
3125 }
3126
3127 splice @{$self->{open_elements}}, $i;
3128
3129 $clear_up_to_marker->();
3130
3131 $self->{insertion_mode} = IN_ROW_IM;
3132
3133 !!!next-token;
3134 redo B;
3135 } elsif ($self->{insertion_mode} == IN_CAPTION_IM) {
3136 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
3137 ## Ignore the token
3138 !!!next-token;
3139 redo B;
3140 } else {
3141 #
3142 }
3143 } elsif ($token->{tag_name} eq 'caption') {
3144 if ($self->{insertion_mode} == IN_CAPTION_IM) {
3145 ## have a table element in table scope
3146 my $i;
3147 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3148 my $node = $self->{open_elements}->[$_];
3149 if ($node->[1] eq $token->{tag_name}) {
3150 $i = $_;
3151 last INSCOPE;
3152 } elsif ({
3153 table => 1, html => 1,
3154 }->{$node->[1]}) {
3155 last INSCOPE;
3156 }
3157 } # INSCOPE
3158 unless (defined $i) {
3159 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
3160 ## Ignore the token
3161 !!!next-token;
3162 redo B;
3163 }
3164
3165 ## generate implied end tags
3166 if ({
3167 dd => 1, dt => 1, li => 1, p => 1,
3168 td => 1, th => 1, tr => 1,
3169 tbody => 1, tfoot=> 1, thead => 1,
3170 }->{$self->{open_elements}->[-1]->[1]}) {
3171 !!!back-token;
3172 $token = {type => END_TAG_TOKEN,
3173 tag_name => $self->{open_elements}->[-1]->[1]}; # MUST
3174 redo B;
3175 }
3176
3177 if ($self->{open_elements}->[-1]->[1] ne 'caption') {
3178 !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);
3179 }
3180
3181 splice @{$self->{open_elements}}, $i;
3182
3183 $clear_up_to_marker->();
3184
3185 $self->{insertion_mode} = IN_TABLE_IM;
3186
3187 !!!next-token;
3188 redo B;
3189 } elsif ($self->{insertion_mode} == IN_CELL_IM) {
3190 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
3191 ## Ignore the token
3192 !!!next-token;
3193 redo B;
3194 } else {
3195 #
3196 }
3197 } elsif ({
3198 table => 1, tbody => 1, tfoot => 1,
3199 thead => 1, tr => 1,
3200 }->{$token->{tag_name}} and
3201 $self->{insertion_mode} == IN_CELL_IM) {
3202 ## have an element in table scope
3203 my $i;
3204 my $tn;
3205 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3206 my $node = $self->{open_elements}->[$_];
3207 if ($node->[1] eq $token->{tag_name}) {
3208 $i = $_;
3209 last INSCOPE;
3210 } elsif ($node->[1] eq 'td' or $node->[1] eq 'th') {
3211 $tn = $node->[1];
3212 ## NOTE: There is exactly one |td| or |th| element
3213 ## in scope in the stack of open elements by definition.
3214 } elsif ({
3215 table => 1, html => 1,
3216 }->{$node->[1]}) {
3217 last INSCOPE;
3218 }
3219 } # INSCOPE
3220 unless (defined $i) {
3221 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
3222 ## Ignore the token
3223 !!!next-token;
3224 redo B;
3225 }
3226
3227 ## Close the cell
3228 !!!back-token; # </?>
3229 $token = {type => END_TAG_TOKEN, tag_name => $tn};
3230 redo B;
3231 } elsif ($token->{tag_name} eq 'table' and
3232 $self->{insertion_mode} == IN_CAPTION_IM) {
3233 !!!parse-error (type => 'not closed:caption');
3234
3235 ## As if </caption>
3236 ## have a table element in table scope
3237 my $i;
3238 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3239 my $node = $self->{open_elements}->[$_];
3240 if ($node->[1] eq 'caption') {
3241 $i = $_;
3242 last INSCOPE;
3243 } elsif ({
3244 table => 1, html => 1,
3245 }->{$node->[1]}) {
3246 last INSCOPE;
3247 }
3248 } # INSCOPE
3249 unless (defined $i) {
3250 !!!parse-error (type => 'unmatched end tag:caption');
3251 ## Ignore the token
3252 !!!next-token;
3253 redo B;
3254 }
3255
3256 ## generate implied end tags
3257 if ({
3258 dd => 1, dt => 1, li => 1, p => 1,
3259 td => 1, th => 1, tr => 1,
3260 tbody => 1, tfoot=> 1, thead => 1,
3261 }->{$self->{open_elements}->[-1]->[1]}) {
3262 !!!back-token; # </table>
3263 $token = {type => END_TAG_TOKEN, tag_name => 'caption'};
3264 !!!back-token;
3265 $token = {type => END_TAG_TOKEN,
3266 tag_name => $self->{open_elements}->[-1]->[1]}; # MUST
3267 redo B;
3268 }
3269
3270 if ($self->{open_elements}->[-1]->[1] ne 'caption') {
3271 !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);
3272 }
3273
3274 splice @{$self->{open_elements}}, $i;
3275
3276 $clear_up_to_marker->();
3277
3278 $self->{insertion_mode} = IN_TABLE_IM;
3279
3280 ## reprocess
3281 redo B;
3282 } elsif ({
3283 body => 1, col => 1, colgroup => 1, html => 1,
3284 }->{$token->{tag_name}}) {
3285 if ($self->{insertion_mode} == IN_CELL_IM or
3286 $self->{insertion_mode} == IN_CAPTION_IM) {
3287 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
3288 ## Ignore the token
3289 !!!next-token;
3290 redo B;
3291 } else {
3292 #
3293 }
3294 } elsif ({
3295 tbody => 1, tfoot => 1,
3296 thead => 1, tr => 1,
3297 }->{$token->{tag_name}} and
3298 $self->{insertion_mode} == IN_CAPTION_IM) {
3299 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
3300 ## Ignore the token
3301 !!!next-token;
3302 redo B;
3303 } else {
3304 #
3305 }
3306 } else {
3307 die "$0: $token->{type}: Unknown token type";
3308 }
3309
3310 $insert = $insert_to_current;
3311 #
3312 } elsif ($self->{insertion_mode} == IN_ROW_IM or
3313 $self->{insertion_mode} == IN_TABLE_BODY_IM or
3314 $self->{insertion_mode} == IN_TABLE_IM) {
3315 if ($token->{type} == CHARACTER_TOKEN) {
3316 ## NOTE: There are "character in table" code clones.
3317 if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {
3318 $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
3319
3320 unless (length $token->{data}) {
3321 !!!next-token;
3322 redo B;
3323 }
3324 }
3325
3326 !!!parse-error (type => 'in table:#character');
3327
3328 ## As if in body, but insert into foster parent element
3329 ## ISSUE: Spec says that "whenever a node would be inserted
3330 ## into the current node" while characters might not be
3331 ## result in a new Text node.
3332 $reconstruct_active_formatting_elements->($insert_to_foster);
3333
3334 if ({
3335 table => 1, tbody => 1, tfoot => 1,
3336 thead => 1, tr => 1,
3337 }->{$self->{open_elements}->[-1]->[1]}) {
3338 # MUST
3339 my $foster_parent_element;
3340 my $next_sibling;
3341 my $prev_sibling;
3342 OE: for (reverse 0..$#{$self->{open_elements}}) {
3343 if ($self->{open_elements}->[$_]->[1] eq 'table') {
3344 my $parent = $self->{open_elements}->[$_]->[0]->parent_node;
3345 if (defined $parent and $parent->node_type == 1) {
3346 $foster_parent_element = $parent;
3347 $next_sibling = $self->{open_elements}->[$_]->[0];
3348 $prev_sibling = $next_sibling->previous_sibling;
3349 } else {
3350 $foster_parent_element = $self->{open_elements}->[$_ - 1]->[0];
3351 $prev_sibling = $foster_parent_element->last_child;
3352 }
3353 last OE;
3354 }
3355 } # OE
3356 $foster_parent_element = $self->{open_elements}->[0]->[0] and
3357 $prev_sibling = $foster_parent_element->last_child
3358 unless defined $foster_parent_element;
3359 if (defined $prev_sibling and
3360 $prev_sibling->node_type == 3) {
3361 $prev_sibling->manakai_append_text ($token->{data});
3362 } else {
3363 $foster_parent_element->insert_before
3364 ($self->{document}->create_text_node ($token->{data}),
3365 $next_sibling);
3366 }
3367 } else {
3368 $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});
3369 }
3370
3371 !!!next-token;
3372 redo B;
3373 } elsif ($token->{type} == START_TAG_TOKEN) {
3374 if ({
3375 tr => ($self->{insertion_mode} != IN_ROW_IM),
3376 th => 1, td => 1,
3377 }->{$token->{tag_name}}) {
3378 if ($self->{insertion_mode} == IN_TABLE_IM) {
3379 ## Clear back to table context
3380 while ($self->{open_elements}->[-1]->[1] ne 'table' and
3381 $self->{open_elements}->[-1]->[1] ne 'html') {
3382 !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);
3383 pop @{$self->{open_elements}};
3384 }
3385
3386 !!!insert-element ('tbody');
3387 $self->{insertion_mode} = IN_TABLE_BODY_IM;
3388 ## reprocess in the "in table body" insertion mode...
3389 }
3390
3391 if ($self->{insertion_mode} == IN_TABLE_BODY_IM) {
3392 unless ($token->{tag_name} eq 'tr') {
3393 !!!parse-error (type => 'missing start tag:tr');
3394 }
3395
3396 ## Clear back to table body context
3397 while (not {
3398 tbody => 1, tfoot => 1, thead => 1, html => 1,
3399 }->{$self->{open_elements}->[-1]->[1]}) {
3400 !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);
3401 pop @{$self->{open_elements}};
3402 }
3403
3404 $self->{insertion_mode} = IN_ROW_IM;
3405 if ($token->{tag_name} eq 'tr') {
3406 !!!insert-element ($token->{tag_name}, $token->{attributes});
3407 !!!next-token;
3408 redo B;
3409 } else {
3410 !!!insert-element ('tr');
3411 ## reprocess in the "in row" insertion mode
3412 }
3413 }
3414
3415 ## Clear back to table row context
3416 while (not {
3417 tr => 1, html => 1,
3418 }->{$self->{open_elements}->[-1]->[1]}) {
3419 !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);
3420 pop @{$self->{open_elements}};
3421 }
3422
3423 !!!insert-element ($token->{tag_name}, $token->{attributes});
3424 $self->{insertion_mode} = IN_CELL_IM;
3425
3426 push @$active_formatting_elements, ['#marker', ''];
3427
3428 !!!next-token;
3429 redo B;
3430 } elsif ({
3431 caption => 1, col => 1, colgroup => 1,
3432 tbody => 1, tfoot => 1, thead => 1,
3433 tr => 1, # $self->{insertion_mode} == IN_ROW_IM
3434 }->{$token->{tag_name}}) {
3435 if ($self->{insertion_mode} == IN_ROW_IM) {
3436 ## As if </tr>
3437 ## have an element in table scope
3438 my $i;
3439 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3440 my $node = $self->{open_elements}->[$_];
3441 if ($node->[1] eq 'tr') {
3442 $i = $_;
3443 last INSCOPE;
3444 } elsif ({
3445 table => 1, html => 1,
3446 }->{$node->[1]}) {
3447 last INSCOPE;
3448 }
3449 } # INSCOPE
3450 unless (defined $i) {
3451 !!!parse-error (type => 'unmacthed end tag:'.$token->{tag_name});
3452 ## Ignore the token
3453 !!!next-token;
3454 redo B;
3455 }
3456
3457 ## Clear back to table row context
3458 while (not {
3459 tr => 1, html => 1,
3460 }->{$self->{open_elements}->[-1]->[1]}) {
3461 !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);
3462 pop @{$self->{open_elements}};
3463 }
3464
3465 pop @{$self->{open_elements}}; # tr
3466 $self->{insertion_mode} = IN_TABLE_BODY_IM;
3467 if ($token->{tag_name} eq 'tr') {
3468 ## reprocess
3469 redo B;
3470 } else {
3471 ## reprocess in the "in table body" insertion mode...
3472 }
3473 }
3474
3475 if ($self->{insertion_mode} == IN_TABLE_BODY_IM) {
3476 ## have an element in table scope
3477 my $i;
3478 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3479 my $node = $self->{open_elements}->[$_];
3480 if ({
3481 tbody => 1, thead => 1, tfoot => 1,
3482 }->{$node->[1]}) {
3483 $i = $_;
3484 last INSCOPE;
3485 } elsif ({
3486 table => 1, html => 1,
3487 }->{$node->[1]}) {
3488 last INSCOPE;
3489 }
3490 } # INSCOPE
3491 unless (defined $i) {
3492 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
3493 ## Ignore the token
3494 !!!next-token;
3495 redo B;
3496 }
3497
3498 ## Clear back to table body context
3499 while (not {
3500 tbody => 1, tfoot => 1, thead => 1, html => 1,
3501 }->{$self->{open_elements}->[-1]->[1]}) {
3502 !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);
3503 pop @{$self->{open_elements}};
3504 }
3505
3506 ## As if <{current node}>
3507 ## have an element in table scope
3508 ## true by definition
3509
3510 ## Clear back to table body context
3511 ## nop by definition
3512
3513 pop @{$self->{open_elements}};
3514 $self->{insertion_mode} = IN_TABLE_IM;
3515 ## reprocess in "in table" insertion mode...
3516 }
3517
3518 if ($token->{tag_name} eq 'col') {
3519 ## Clear back to table context
3520 while ($self->{open_elements}->[-1]->[1] ne 'table' and
3521 $self->{open_elements}->[-1]->[1] ne 'html') {
3522 !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);
3523 pop @{$self->{open_elements}};
3524 }
3525
3526 !!!insert-element ('colgroup');
3527 $self->{insertion_mode} = IN_COLUMN_GROUP_IM;
3528 ## reprocess
3529 redo B;
3530 } elsif ({
3531 caption => 1,
3532 colgroup => 1,
3533 tbody => 1, tfoot => 1, thead => 1,
3534 }->{$token->{tag_name}}) {
3535 ## Clear back to table context
3536 while ($self->{open_elements}->[-1]->[1] ne 'table' and
3537 $self->{open_elements}->[-1]->[1] ne 'html') {
3538 !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);
3539 pop @{$self->{open_elements}};
3540 }
3541
3542 push @$active_formatting_elements, ['#marker', '']
3543 if $token->{tag_name} eq 'caption';
3544
3545 !!!insert-element ($token->{tag_name}, $token->{attributes});
3546 $self->{insertion_mode} = {
3547 caption => IN_CAPTION_IM,
3548 colgroup => IN_COLUMN_GROUP_IM,
3549 tbody => IN_TABLE_BODY_IM,
3550 tfoot => IN_TABLE_BODY_IM,
3551 thead => IN_TABLE_BODY_IM,
3552 }->{$token->{tag_name}};
3553 !!!next-token;
3554 redo B;
3555 } else {
3556 die "$0: in table: <>: $token->{tag_name}";
3557 }
3558 } elsif ($token->{tag_name} eq 'table') {
3559 ## NOTE: There are code clones for this "table in table"
3560 !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);
3561
3562 ## As if </table>
3563 ## have a table element in table scope
3564 my $i;
3565 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3566 my $node = $self->{open_elements}->[$_];
3567 if ($node->[1] eq 'table') {
3568 $i = $_;
3569 last INSCOPE;
3570 } elsif ({
3571 table => 1, html => 1,
3572 }->{$node->[1]}) {
3573 last INSCOPE;
3574 }
3575 } # INSCOPE
3576 unless (defined $i) {
3577 !!!parse-error (type => 'unmatched end tag:table');
3578 ## Ignore tokens </table><table>
3579 !!!next-token;
3580 redo B;
3581 }
3582
3583 ## generate implied end tags
3584 if ({
3585 dd => 1, dt => 1, li => 1, p => 1,
3586 td => 1, th => 1, tr => 1,
3587 tbody => 1, tfoot=> 1, thead => 1,
3588 }->{$self->{open_elements}->[-1]->[1]}) {
3589 !!!back-token; # <table>
3590 $token = {type => END_TAG_TOKEN, tag_name => 'table'};
3591 !!!back-token;
3592 $token = {type => END_TAG_TOKEN,
3593 tag_name => $self->{open_elements}->[-1]->[1]}; # MUST
3594 redo B;
3595 }
3596
3597 if ($self->{open_elements}->[-1]->[1] ne 'table') {
3598 !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);
3599 }
3600
3601 splice @{$self->{open_elements}}, $i;
3602
3603 $self->_reset_insertion_mode;
3604
3605 ## reprocess
3606 redo B;
3607 } else {
3608 #
3609 }
3610 } elsif ($token->{type} == END_TAG_TOKEN) {
3611 if ($token->{tag_name} eq 'tr' and
3612 $self->{insertion_mode} == IN_ROW_IM) {
3613 ## have an element in table scope
3614 my $i;
3615 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3616 my $node = $self->{open_elements}->[$_];
3617 if ($node->[1] eq $token->{tag_name}) {
3618 $i = $_;
3619 last INSCOPE;
3620 } elsif ({
3621 table => 1, html => 1,
3622 }->{$node->[1]}) {
3623 last INSCOPE;
3624 }
3625 } # INSCOPE
3626 unless (defined $i) {
3627 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
3628 ## Ignore the token
3629 !!!next-token;
3630 redo B;
3631 }
3632
3633 ## Clear back to table row context
3634 while (not {
3635 tr => 1, html => 1,
3636 }->{$self->{open_elements}->[-1]->[1]}) {
3637 !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);
3638 pop @{$self->{open_elements}};
3639 }
3640
3641 pop @{$self->{open_elements}}; # tr
3642 $self->{insertion_mode} = IN_TABLE_BODY_IM;
3643 !!!next-token;
3644 redo B;
3645 } elsif ($token->{tag_name} eq 'table') {
3646 if ($self->{insertion_mode} == IN_ROW_IM) {
3647 ## As if </tr>
3648 ## have an element in table scope
3649 my $i;
3650 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3651 my $node = $self->{open_elements}->[$_];
3652 if ($node->[1] eq 'tr') {
3653 $i = $_;
3654 last INSCOPE;
3655 } elsif ({
3656 table => 1, html => 1,
3657 }->{$node->[1]}) {
3658 last INSCOPE;
3659 }
3660 } # INSCOPE
3661 unless (defined $i) {
3662 !!!parse-error (type => 'unmatched end tag:'.$token->{type});
3663 ## Ignore the token
3664 !!!next-token;
3665 redo B;
3666 }
3667
3668 ## Clear back to table row context
3669 while (not {
3670 tr => 1, html => 1,
3671 }->{$self->{open_elements}->[-1]->[1]}) {
3672 !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);
3673 pop @{$self->{open_elements}};
3674 }
3675
3676 pop @{$self->{open_elements}}; # tr
3677 $self->{insertion_mode} = IN_TABLE_BODY_IM;
3678 ## reprocess in the "in table body" insertion mode...
3679 }
3680
3681 if ($self->{insertion_mode} == IN_TABLE_BODY_IM) {
3682 ## have an element in table scope
3683 my $i;
3684 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3685 my $node = $self->{open_elements}->[$_];
3686 if ({
3687 tbody => 1, thead => 1, tfoot => 1,
3688 }->{$node->[1]}) {
3689 $i = $_;
3690 last INSCOPE;
3691 } elsif ({
3692 table => 1, html => 1,
3693 }->{$node->[1]}) {
3694 last INSCOPE;
3695 }
3696 } # INSCOPE
3697 unless (defined $i) {
3698 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
3699 ## Ignore the token
3700 !!!next-token;
3701 redo B;
3702 }
3703
3704 ## Clear back to table body context
3705 while (not {
3706 tbody => 1, tfoot => 1, thead => 1, html => 1,
3707 }->{$self->{open_elements}->[-1]->[1]}) {
3708 !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);
3709 pop @{$self->{open_elements}};
3710 }
3711
3712 ## As if <{current node}>
3713 ## have an element in table scope
3714 ## true by definition
3715
3716 ## Clear back to table body context
3717 ## nop by definition
3718
3719 pop @{$self->{open_elements}};
3720 $self->{insertion_mode} = IN_TABLE_IM;
3721 ## reprocess in the "in table" insertion mode...
3722 }
3723
3724 ## have a table element in table scope
3725 my $i;
3726 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3727 my $node = $self->{open_elements}->[$_];
3728 if ($node->[1] eq $token->{tag_name}) {
3729 $i = $_;
3730 last INSCOPE;
3731 } elsif ({
3732 table => 1, html => 1,
3733 }->{$node->[1]}) {
3734 last INSCOPE;
3735 }
3736 } # INSCOPE
3737 unless (defined $i) {
3738 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
3739 ## Ignore the token
3740 !!!next-token;
3741 redo B;
3742 }
3743
3744 ## generate implied end tags
3745 if ({
3746 dd => 1, dt => 1, li => 1, p => 1,
3747 td => 1, th => 1, tr => 1,
3748 tbody => 1, tfoot=> 1, thead => 1,
3749 }->{$self->{open_elements}->[-1]->[1]}) {
3750 !!!back-token;
3751 $token = {type => END_TAG_TOKEN,
3752 tag_name => $self->{open_elements}->[-1]->[1]}; # MUST
3753 redo B;
3754 }
3755
3756 if ($self->{open_elements}->[-1]->[1] ne 'table') {
3757 !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);
3758 }
3759
3760 splice @{$self->{open_elements}}, $i;
3761
3762 $self->_reset_insertion_mode;
3763
3764 !!!next-token;
3765 redo B;
3766 } elsif ({
3767 tbody => 1, tfoot => 1, thead => 1,
3768 }->{$token->{tag_name}} and
3769 ($self->{insertion_mode} == IN_ROW_IM or
3770 $self->{insertion_mode} == IN_TABLE_BODY_IM)) {
3771 if ($self->{insertion_mode} == IN_ROW_IM) {
3772 ## have an element in table scope
3773 my $i;
3774 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3775 my $node = $self->{open_elements}->[$_];
3776 if ($node->[1] eq $token->{tag_name}) {
3777 $i = $_;
3778 last INSCOPE;
3779 } elsif ({
3780 table => 1, html => 1,
3781 }->{$node->[1]}) {
3782 last INSCOPE;
3783 }
3784 } # INSCOPE
3785 unless (defined $i) {
3786 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
3787 ## Ignore the token
3788 !!!next-token;
3789 redo B;
3790 }
3791
3792 ## As if </tr>
3793 ## have an element in table scope
3794 my $i;
3795 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3796 my $node = $self->{open_elements}->[$_];
3797 if ($node->[1] eq 'tr') {
3798 $i = $_;
3799 last INSCOPE;
3800 } elsif ({
3801 table => 1, html => 1,
3802 }->{$node->[1]}) {
3803 last INSCOPE;
3804 }
3805 } # INSCOPE
3806 unless (defined $i) {
3807 !!!parse-error (type => 'unmatched end tag:tr');
3808 ## Ignore the token
3809 !!!next-token;
3810 redo B;
3811 }
3812
3813 ## Clear back to table row context
3814 while (not {
3815 tr => 1, html => 1,
3816 }->{$self->{open_elements}->[-1]->[1]}) {
3817 !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);
3818 pop @{$self->{open_elements}};
3819 }
3820
3821 pop @{$self->{open_elements}}; # tr
3822 $self->{insertion_mode} = IN_TABLE_BODY_IM;
3823 ## reprocess in the "in table body" insertion mode...
3824 }
3825
3826 ## have an element in table scope
3827 my $i;
3828 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3829 my $node = $self->{open_elements}->[$_];
3830 if ($node->[1] eq $token->{tag_name}) {
3831 $i = $_;
3832 last INSCOPE;
3833 } elsif ({
3834 table => 1, html => 1,
3835 }->{$node->[1]}) {
3836 last INSCOPE;
3837 }
3838 } # INSCOPE
3839 unless (defined $i) {
3840 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
3841 ## Ignore the token
3842 !!!next-token;
3843 redo B;
3844 }
3845
3846 ## Clear back to table body context
3847 while (not {
3848 tbody => 1, tfoot => 1, thead => 1, html => 1,
3849 }->{$self->{open_elements}->[-1]->[1]}) {
3850 !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);
3851 pop @{$self->{open_elements}};
3852 }
3853
3854 pop @{$self->{open_elements}};
3855 $self->{insertion_mode} = IN_TABLE_IM;
3856 !!!next-token;
3857 redo B;
3858 } elsif ({
3859 body => 1, caption => 1, col => 1, colgroup => 1,
3860 html => 1, td => 1, th => 1,
3861 tr => 1, # $self->{insertion_mode} == IN_ROW_IM
3862 tbody => 1, tfoot => 1, thead => 1, # $self->{insertion_mode} == IN_TABLE_IM
3863 }->{$token->{tag_name}}) {
3864 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
3865 ## Ignore the token
3866 !!!next-token;
3867 redo B;
3868 } else {
3869 #
3870 }
3871 } else {
3872 die "$0: $token->{type}: Unknown token type";
3873 }
3874
3875 !!!parse-error (type => 'in table:'.$token->{tag_name});
3876
3877 $insert = $insert_to_foster;
3878 #
3879 } elsif ($self->{insertion_mode} == IN_COLUMN_GROUP_IM) {
3880 if ($token->{type} == CHARACTER_TOKEN) {
3881 if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {
3882 $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
3883 unless (length $token->{data}) {
3884 !!!next-token;
3885 redo B;
3886 }
3887 }
3888
3889 #
3890 } elsif ($token->{type} == START_TAG_TOKEN) {
3891 if ($token->{tag_name} eq 'col') {
3892 !!!insert-element ($token->{tag_name}, $token->{attributes});
3893 pop @{$self->{open_elements}};
3894 !!!next-token;
3895 redo B;
3896 } else {
3897 #
3898 }
3899 } elsif ($token->{type} == END_TAG_TOKEN) {
3900 if ($token->{tag_name} eq 'colgroup') {
3901 if ($self->{open_elements}->[-1]->[1] eq 'html') {
3902 !!!parse-error (type => 'unmatched end tag:colgroup');
3903 ## Ignore the token
3904 !!!next-token;
3905 redo B;
3906 } else {
3907 pop @{$self->{open_elements}}; # colgroup
3908 $self->{insertion_mode} = IN_TABLE_IM;
3909 !!!next-token;
3910 redo B;
3911 }
3912 } elsif ($token->{tag_name} eq 'col') {
3913 !!!parse-error (type => 'unmatched end tag:col');
3914 ## Ignore the token
3915 !!!next-token;
3916 redo B;
3917 } else {
3918 #
3919 }
3920 } else {
3921 #
3922 }
3923
3924 ## As if </colgroup>
3925 if ($self->{open_elements}->[-1]->[1] eq 'html') {
3926 !!!parse-error (type => 'unmatched end tag:colgroup');
3927 ## Ignore the token
3928 !!!next-token;
3929 redo B;
3930 } else {
3931 pop @{$self->{open_elements}}; # colgroup
3932 $self->{insertion_mode} = IN_TABLE_IM;
3933 ## reprocess
3934 redo B;
3935 }
3936 } elsif ($self->{insertion_mode} == IN_SELECT_IM) {
3937 if ($token->{type} == CHARACTER_TOKEN) {
3938 $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});
3939 !!!next-token;
3940 redo B;
3941 } elsif ($token->{type} == START_TAG_TOKEN) {
3942 if ($token->{tag_name} eq 'option') {
3943 if ($self->{open_elements}->[-1]->[1] eq 'option') {
3944 ## As if </option>
3945 pop @{$self->{open_elements}};
3946 }
3947
3948 !!!insert-element ($token->{tag_name}, $token->{attributes});
3949 !!!next-token;
3950 redo B;
3951 } elsif ($token->{tag_name} eq 'optgroup') {
3952 if ($self->{open_elements}->[-1]->[1] eq 'option') {
3953 ## As if </option>
3954 pop @{$self->{open_elements}};
3955 }
3956
3957 if ($self->{open_elements}->[-1]->[1] eq 'optgroup') {
3958 ## As if </optgroup>
3959 pop @{$self->{open_elements}};
3960 }
3961
3962 !!!insert-element ($token->{tag_name}, $token->{attributes});
3963 !!!next-token;
3964 redo B;
3965 } elsif ($token->{tag_name} eq 'select') {
3966 !!!parse-error (type => 'not closed:select');
3967 ## As if </select> instead
3968 ## have an element in table scope
3969 my $i;
3970 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3971 my $node = $self->{open_elements}->[$_];
3972 if ($node->[1] eq $token->{tag_name}) {
3973 $i = $_;
3974 last INSCOPE;
3975 } elsif ({
3976 table => 1, html => 1,
3977 }->{$node->[1]}) {
3978 last INSCOPE;
3979 }
3980 } # INSCOPE
3981 unless (defined $i) {
3982 !!!parse-error (type => 'unmatched end tag:select');
3983 ## Ignore the token
3984 !!!next-token;
3985 redo B;
3986 }
3987
3988 splice @{$self->{open_elements}}, $i;
3989
3990 $self->_reset_insertion_mode;
3991
3992 !!!next-token;
3993 redo B;
3994 } else {
3995 #
3996 }
3997 } elsif ($token->{type} == END_TAG_TOKEN) {
3998 if ($token->{tag_name} eq 'optgroup') {
3999 if ($self->{open_elements}->[-1]->[1] eq 'option' and
4000 $self->{open_elements}->[-2]->[1] eq 'optgroup') {
4001 ## As if </option>
4002 splice @{$self->{open_elements}}, -2;
4003 } elsif ($self->{open_elements}->[-1]->[1] eq 'optgroup') {
4004 pop @{$self->{open_elements}};
4005 } else {
4006 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
4007 ## Ignore the token
4008 }
4009 !!!next-token;
4010 redo B;
4011 } elsif ($token->{tag_name} eq 'option') {
4012 if ($self->{open_elements}->[-1]->[1] eq 'option') {
4013 pop @{$self->{open_elements}};
4014 } else {
4015 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
4016 ## Ignore the token
4017 }
4018 !!!next-token;
4019 redo B;
4020 } elsif ($token->{tag_name} eq 'select') {
4021 ## have an element in table scope
4022 my $i;
4023 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
4024 my $node = $self->{open_elements}->[$_];
4025 if ($node->[1] eq $token->{tag_name}) {
4026 $i = $_;
4027 last INSCOPE;
4028 } elsif ({
4029 table => 1, html => 1,
4030 }->{$node->[1]}) {
4031 last INSCOPE;
4032 }
4033 } # INSCOPE
4034 unless (defined $i) {
4035 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
4036 ## Ignore the token
4037 !!!next-token;
4038 redo B;
4039 }
4040
4041 splice @{$self->{open_elements}}, $i;
4042
4043 $self->_reset_insertion_mode;
4044
4045 !!!next-token;
4046 redo B;
4047 } elsif ({
4048 caption => 1, table => 1, tbody => 1,
4049 tfoot => 1, thead => 1, tr => 1, td => 1, th => 1,
4050 }->{$token->{tag_name}}) {
4051 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
4052
4053 ## have an element in table scope
4054 my $i;
4055 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
4056 my $node = $self->{open_elements}->[$_];
4057 if ($node->[1] eq $token->{tag_name}) {
4058 $i = $_;
4059 last INSCOPE;
4060 } elsif ({
4061 table => 1, html => 1,
4062 }->{$node->[1]}) {
4063 last INSCOPE;
4064 }
4065 } # INSCOPE
4066 unless (defined $i) {
4067 ## Ignore the token
4068 !!!next-token;
4069 redo B;
4070 }
4071
4072 ## As if </select>
4073 ## have an element in table scope
4074 undef $i;
4075 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
4076 my $node = $self->{open_elements}->[$_];
4077 if ($node->[1] eq 'select') {
4078 $i = $_;
4079 last INSCOPE;
4080 } elsif ({
4081 table => 1, html => 1,
4082 }->{$node->[1]}) {
4083 last INSCOPE;
4084 }
4085 } # INSCOPE
4086 unless (defined $i) {
4087 !!!parse-error (type => 'unmatched end tag:select');
4088 ## Ignore the </select> token
4089 !!!next-token; ## TODO: ok?
4090 redo B;
4091 }
4092
4093 splice @{$self->{open_elements}}, $i;
4094
4095 $self->_reset_insertion_mode;
4096
4097 ## reprocess
4098 redo B;
4099 } else {
4100 #
4101 }
4102 } else {
4103 #
4104 }
4105
4106 !!!parse-error (type => 'in select:'.$token->{tag_name});
4107 ## Ignore the token
4108 !!!next-token;
4109 redo B;
4110 } elsif ($self->{insertion_mode} == AFTER_BODY_IM or
4111 $self->{insertion_mode} == AFTER_HTML_BODY_IM) {
4112 if ($token->{type} == CHARACTER_TOKEN) {
4113 if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {
4114 my $data = $1;
4115 ## As if in body
4116 $reconstruct_active_formatting_elements->($insert_to_current);
4117
4118 $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
4119
4120 unless (length $token->{data}) {
4121 !!!next-token;
4122 redo B;
4123 }
4124 }
4125
4126 if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
4127 !!!parse-error (type => 'after html:#character');
4128
4129 ## Reprocess in the "main" phase, "after body" insertion mode...
4130 }
4131
4132 ## "after body" insertion mode
4133 !!!parse-error (type => 'after body:#character');
4134
4135 $self->{insertion_mode} = IN_BODY_IM;
4136 ## reprocess
4137 redo B;
4138 } elsif ($token->{type} == START_TAG_TOKEN) {
4139 if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
4140 !!!parse-error (type => 'after html:'.$token->{tag_name});
4141
4142 ## Reprocess in the "main" phase, "after body" insertion mode...
4143 }
4144
4145 ## "after body" insertion mode
4146 !!!parse-error (type => 'after body:'.$token->{tag_name});
4147
4148 $self->{insertion_mode} = IN_BODY_IM;
4149 ## reprocess
4150 redo B;
4151 } elsif ($token->{type} == END_TAG_TOKEN) {
4152 if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
4153 !!!parse-error (type => 'after html:/'.$token->{tag_name});
4154
4155 $self->{insertion_mode} = AFTER_BODY_IM;
4156 ## Reprocess in the "main" phase, "after body" insertion mode...
4157 }
4158
4159 ## "after body" insertion mode
4160 if ($token->{tag_name} eq 'html') {
4161 if (defined $self->{inner_html_node}) {
4162 !!!parse-error (type => 'unmatched end tag:html');
4163 ## Ignore the token
4164 !!!next-token;
4165 redo B;
4166 } else {
4167 $self->{insertion_mode} = AFTER_HTML_BODY_IM;
4168 !!!next-token;
4169 redo B;
4170 }
4171 } else {
4172 !!!parse-error (type => 'after body:/'.$token->{tag_name});
4173
4174 $self->{insertion_mode} = IN_BODY_IM;
4175 ## reprocess
4176 redo B;
4177 }
4178 } else {
4179 die "$0: $token->{type}: Unknown token type";
4180 }
4181 } elsif ($self->{insertion_mode} == IN_FRAMESET_IM or
4182 $self->{insertion_mode} == AFTER_FRAMESET_IM or
4183 $self->{insertion_mode} == AFTER_HTML_FRAMESET_IM) {
4184 if ($token->{type} == CHARACTER_TOKEN) {
4185 if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {
4186 $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
4187
4188 unless (length $token->{data}) {
4189 !!!next-token;
4190 redo B;
4191 }
4192 }
4193
4194 if ($token->{data} =~ s/^[^\x09\x0A\x0B\x0C\x20]+//) {
4195 if ($self->{insertion_mode} == IN_FRAMESET_IM) {
4196 !!!parse-error (type => 'in frameset:#character');
4197 } elsif ($self->{insertion_mode} == AFTER_FRAMESET_IM) {
4198 !!!parse-error (type => 'after frameset:#character');
4199 } else { # "after html frameset"
4200 !!!parse-error (type => 'after html:#character');
4201
4202 $self->{insertion_mode} = AFTER_FRAMESET_IM;
4203 ## Reprocess in the "main" phase, "after frameset"...
4204 !!!parse-error (type => 'after frameset:#character');
4205 }
4206
4207 ## Ignore the token.
4208 if (length $token->{data}) {
4209 ## reprocess the rest of characters
4210 } else {
4211 !!!next-token;
4212 }
4213 redo B;
4214 }
4215
4216 die qq[$0: Character "$token->{data}"];
4217 } elsif ($token->{type} == START_TAG_TOKEN) {
4218 if ($self->{insertion_mode} == AFTER_HTML_FRAMESET_IM) {
4219 !!!parse-error (type => 'after html:'.$token->{tag_name});
4220
4221 $self->{insertion_mode} = AFTER_FRAMESET_IM;
4222 ## Process in the "main" phase, "after frameset" insertion mode...
4223 }
4224
4225 if ($token->{tag_name} eq 'frameset' and
4226 $self->{insertion_mode} == IN_FRAMESET_IM) {
4227 !!!insert-element ($token->{tag_name}, $token->{attributes});
4228 !!!next-token;
4229 redo B;
4230 } elsif ($token->{tag_name} eq 'frame' and
4231 $self->{insertion_mode} == IN_FRAMESET_IM) {
4232 !!!insert-element ($token->{tag_name}, $token->{attributes});
4233 pop @{$self->{open_elements}};
4234 !!!next-token;
4235 redo B;
4236 } elsif ($token->{tag_name} eq 'noframes') {
4237 ## NOTE: As if in body.
4238 $parse_rcdata->(CDATA_CONTENT_MODEL, $insert_to_current);
4239 redo B;
4240 } else {
4241 if ($self->{insertion_mode} == IN_FRAMESET_IM) {
4242 !!!parse-error (type => 'in frameset:'.$token->{tag_name});
4243 } else {
4244 !!!parse-error (type => 'after frameset:'.$token->{tag_name});
4245 }
4246 ## Ignore the token
4247 !!!next-token;
4248 redo B;
4249 }
4250 } elsif ($token->{type} == END_TAG_TOKEN) {
4251 if ($self->{insertion_mode} == AFTER_HTML_FRAMESET_IM) {
4252 !!!parse-error (type => 'after html:/'.$token->{tag_name});
4253
4254 $self->{insertion_mode} = AFTER_FRAMESET_IM;
4255 ## Process in the "main" phase, "after frameset" insertion mode...
4256 }
4257
4258 if ($token->{tag_name} eq 'frameset' and
4259 $self->{insertion_mode} == IN_FRAMESET_IM) {
4260 if ($self->{open_elements}->[-1]->[1] eq 'html' and
4261 @{$self->{open_elements}} == 1) {
4262 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
4263 ## Ignore the token
4264 !!!next-token;
4265 } else {
4266 pop @{$self->{open_elements}};
4267 !!!next-token;
4268 }
4269
4270 if (not defined $self->{inner_html_node} and
4271 $self->{open_elements}->[-1]->[1] ne 'frameset') {
4272 $self->{insertion_mode} = AFTER_FRAMESET_IM;
4273 }
4274 redo B;
4275 } elsif ($token->{tag_name} eq 'html' and
4276 $self->{insertion_mode} == AFTER_FRAMESET_IM) {
4277 $self->{insertion_mode} = AFTER_HTML_FRAMESET_IM;
4278 !!!next-token;
4279 redo B;
4280 } else {
4281 if ($self->{insertion_mode} == IN_FRAMESET_IM) {
4282 !!!parse-error (type => 'in frameset:/'.$token->{tag_name});
4283 } else {
4284 !!!parse-error (type => 'after frameset:/'.$token->{tag_name});
4285 }
4286 ## Ignore the token
4287 !!!next-token;
4288 redo B;
4289 }
4290 } else {
4291 die "$0: $token->{type}: Unknown token type";
4292 }
4293
4294 ## ISSUE: An issue in spec here
4295 } else {
4296 die "$0: $self->{insertion_mode}: Unknown insertion mode";
4297 }
4298
4299 ## "in body" insertion mode
4300 if ($token->{type} == START_TAG_TOKEN) {
4301 if ($token->{tag_name} eq 'script') {
4302 ## NOTE: This is an "as if in head" code clone
4303 $script_start_tag->($insert);
4304 redo B;
4305 } elsif ($token->{tag_name} eq 'style') {
4306 ## NOTE: This is an "as if in head" code clone
4307 $parse_rcdata->(CDATA_CONTENT_MODEL, $insert);
4308 redo B;
4309 } elsif ({
4310 base => 1, link => 1,
4311 }->{$token->{tag_name}}) {
4312 ## NOTE: This is an "as if in head" code clone, only "-t" differs
4313 !!!insert-element-t ($token->{tag_name}, $token->{attributes});
4314 pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.
4315 !!!next-token;
4316 redo B;
4317 } elsif ($token->{tag_name} eq 'meta') {
4318 ## NOTE: This is an "as if in head" code clone, only "-t" differs
4319 !!!insert-element-t ($token->{tag_name}, $token->{attributes});
4320 pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.
4321
4322 unless ($self->{confident}) {
4323 my $charset;
4324 if ($token->{attributes}->{charset}) { ## TODO: And if supported
4325 $charset = $token->{attributes}->{charset}->{value};
4326 }
4327 if ($token->{attributes}->{'http-equiv'}) {
4328 ## ISSUE: Algorithm name in the spec was incorrect so that not linked to the definition.
4329 if ($token->{attributes}->{'http-equiv'}->{value}
4330 =~ /\A[^;]*;[\x09-\x0D\x20]*charset[\x09-\x0D\x20]*=
4331 [\x09-\x0D\x20]*(?>"([^"]*)"|'([^']*)'|
4332 ([^"'\x09-\x0D\x20][^\x09-\x0D\x20]*))/x) {
4333 $charset = defined $1 ? $1 : defined $2 ? $2 : $3;
4334 } ## TODO: And if supported
4335 }
4336 ## TODO: Change the encoding
4337 }
4338
4339 !!!next-token;
4340 redo B;
4341 } elsif ($token->{tag_name} eq 'title') {
4342 !!!parse-error (type => 'in body:title');
4343 ## NOTE: This is an "as if in head" code clone
4344 $parse_rcdata->(RCDATA_CONTENT_MODEL, sub {
4345 if (defined $self->{head_element}) {
4346 $self->{head_element}->append_child ($_[0]);
4347 } else {
4348 $insert->($_[0]);
4349 }
4350 });
4351 redo B;
4352 } elsif ($token->{tag_name} eq 'body') {
4353 !!!parse-error (type => 'in body:body');
4354
4355 if (@{$self->{open_elements}} == 1 or
4356 $self->{open_elements}->[1]->[1] ne 'body') {
4357 ## Ignore the token
4358 } else {
4359 my $body_el = $self->{open_elements}->[1]->[0];
4360 for my $attr_name (keys %{$token->{attributes}}) {
4361 unless ($body_el->has_attribute_ns (undef, $attr_name)) {
4362 $body_el->set_attribute_ns
4363 (undef, [undef, $attr_name],
4364 $token->{attributes}->{$attr_name}->{value});
4365 }
4366 }
4367 }
4368 !!!next-token;
4369 redo B;
4370 } elsif ({
4371 address => 1, blockquote => 1, center => 1, dir => 1,
4372 div => 1, dl => 1, fieldset => 1, listing => 1,
4373 menu => 1, ol => 1, p => 1, ul => 1,
4374 pre => 1,
4375 }->{$token->{tag_name}}) {
4376 ## has a p element in scope
4377 INSCOPE: for (reverse @{$self->{open_elements}}) {
4378 if ($_->[1] eq 'p') {
4379 !!!back-token;
4380 $token = {type => END_TAG_TOKEN, tag_name => 'p'};
4381 redo B;
4382 } elsif ({
4383 table => 1, caption => 1, td => 1, th => 1,
4384 button => 1, marquee => 1, object => 1, html => 1,
4385 }->{$_->[1]}) {
4386 last INSCOPE;
4387 }
4388 } # INSCOPE
4389
4390 !!!insert-element-t ($token->{tag_name}, $token->{attributes});
4391 if ($token->{tag_name} eq 'pre') {
4392 !!!next-token;
4393 if ($token->{type} == CHARACTER_TOKEN) {
4394 $token->{data} =~ s/^\x0A//;
4395 unless (length $token->{data}) {
4396 !!!next-token;
4397 }
4398 }
4399 } else {
4400 !!!next-token;
4401 }
4402 redo B;
4403 } elsif ($token->{tag_name} eq 'form') {
4404 if (defined $self->{form_element}) {
4405 !!!parse-error (type => 'in form:form');
4406 ## Ignore the token
4407 !!!next-token;
4408 redo B;
4409 } else {
4410 ## has a p element in scope
4411 INSCOPE: for (reverse @{$self->{open_elements}}) {
4412 if ($_->[1] eq 'p') {
4413 !!!back-token;
4414 $token = {type => END_TAG_TOKEN, tag_name => 'p'};
4415 redo B;
4416 } elsif ({
4417 table => 1, caption => 1, td => 1, th => 1,
4418 button => 1, marquee => 1, object => 1, html => 1,
4419 }->{$_->[1]}) {
4420 last INSCOPE;
4421 }
4422 } # INSCOPE
4423
4424 !!!insert-element-t ($token->{tag_name}, $token->{attributes});
4425 $self->{form_element} = $self->{open_elements}->[-1]->[0];
4426 !!!next-token;
4427 redo B;
4428 }
4429 } elsif ($token->{tag_name} eq 'li') {
4430 ## has a p element in scope
4431 INSCOPE: for (reverse @{$self->{open_elements}}) {
4432 if ($_->[1] eq 'p') {
4433 !!!back-token;
4434 $token = {type => END_TAG_TOKEN, tag_name => 'p'};
4435 redo B;
4436 } elsif ({
4437 table => 1, caption => 1, td => 1, th => 1,
4438 button => 1, marquee => 1, object => 1, html => 1,
4439 }->{$_->[1]}) {
4440 last INSCOPE;
4441 }
4442 } # INSCOPE
4443
4444 ## Step 1
4445 my $i = -1;
4446 my $node = $self->{open_elements}->[$i];
4447 LI: {
4448 ## Step 2
4449 if ($node->[1] eq 'li') {
4450 if ($i != -1) {
4451 !!!parse-error (type => 'end tag missing:'.
4452 $self->{open_elements}->[-1]->[1]);
4453 }
4454 splice @{$self->{open_elements}}, $i;
4455 last LI;
4456 }
4457
4458 ## Step 3
4459 if (not $formatting_category->{$node->[1]} and
4460 #not $phrasing_category->{$node->[1]} and
4461 ($special_category->{$node->[1]} or
4462 $scoping_category->{$node->[1]}) and
4463 $node->[1] ne 'address' and $node->[1] ne 'div') {
4464 last LI;
4465 }
4466
4467 ## Step 4
4468 $i--;
4469 $node = $self->{open_elements}->[$i];
4470 redo LI;
4471 } # LI
4472
4473 !!!insert-element-t ($token->{tag_name}, $token->{attributes});
4474 !!!next-token;
4475 redo B;
4476 } elsif ($token->{tag_name} eq 'dd' or $token->{tag_name} eq 'dt') {
4477 ## has a p element in scope
4478 INSCOPE: for (reverse @{$self->{open_elements}}) {
4479 if ($_->[1] eq 'p') {
4480 !!!back-token;
4481 $token = {type => END_TAG_TOKEN, tag_name => 'p'};
4482 redo B;
4483 } elsif ({
4484 table => 1, caption => 1, td => 1, th => 1,
4485 button => 1, marquee => 1, object => 1, html => 1,
4486 }->{$_->[1]}) {
4487 last INSCOPE;
4488 }
4489 } # INSCOPE
4490
4491 ## Step 1
4492 my $i = -1;
4493 my $node = $self->{open_elements}->[$i];
4494 LI: {
4495 ## Step 2
4496 if ($node->[1] eq 'dt' or $node->[1] eq 'dd') {
4497 if ($i != -1) {
4498 !!!parse-error (type => 'end tag missing:'.
4499 $self->{open_elements}->[-1]->[1]);
4500 }
4501 splice @{$self->{open_elements}}, $i;
4502 last LI;
4503 }
4504
4505 ## Step 3
4506 if (not $formatting_category->{$node->[1]} and
4507 #not $phrasing_category->{$node->[1]} and
4508 ($special_category->{$node->[1]} or
4509 $scoping_category->{$node->[1]}) and
4510 $node->[1] ne 'address' and $node->[1] ne 'div') {
4511 last LI;
4512 }
4513
4514 ## Step 4
4515 $i--;
4516 $node = $self->{open_elements}->[$i];
4517 redo LI;
4518 } # LI
4519
4520 !!!insert-element-t ($token->{tag_name}, $token->{attributes});
4521 !!!next-token;
4522 redo B;
4523 } elsif ($token->{tag_name} eq 'plaintext') {
4524 ## has a p element in scope
4525 INSCOPE: for (reverse @{$self->{open_elements}}) {
4526 if ($_->[1] eq 'p') {
4527 !!!back-token;
4528 $token = {type => END_TAG_TOKEN, tag_name => 'p'};
4529 redo B;
4530 } elsif ({
4531 table => 1, caption => 1, td => 1, th => 1,
4532 button => 1, marquee => 1, object => 1, html => 1,
4533 }->{$_->[1]}) {
4534 last INSCOPE;
4535 }
4536 } # INSCOPE
4537
4538 !!!insert-element-t ($token->{tag_name}, $token->{attributes});
4539
4540 $self->{content_model} = PLAINTEXT_CONTENT_MODEL;
4541
4542 !!!next-token;
4543 redo B;
4544 } elsif ({
4545 h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,
4546 }->{$token->{tag_name}}) {
4547 ## has a p element in scope
4548 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
4549 my $node = $self->{open_elements}->[$_];
4550 if ($node->[1] eq 'p') {
4551 !!!back-token;
4552 $token = {type => END_TAG_TOKEN, tag_name => 'p'};
4553 redo B;
4554 } elsif ({
4555 table => 1, caption => 1, td => 1, th => 1,
4556 button => 1, marquee => 1, object => 1, html => 1,
4557 }->{$node->[1]}) {
4558 last INSCOPE;
4559 }
4560 } # INSCOPE
4561
4562 ## NOTE: See <http://html5.org/tools/web-apps-tracker?from=925&to=926>
4563 ## has an element in scope
4564 #my $i;
4565 #INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
4566 # my $node = $self->{open_elements}->[$_];
4567 # if ({
4568 # h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,
4569 # }->{$node->[1]}) {
4570 # $i = $_;
4571 # last INSCOPE;
4572 # } elsif ({
4573 # table => 1, caption => 1, td => 1, th => 1,
4574 # button => 1, marquee => 1, object => 1, html => 1,
4575 # }->{$node->[1]}) {
4576 # last INSCOPE;
4577 # }
4578 #} # INSCOPE
4579 #
4580 #if (defined $i) {
4581 # !!! parse-error (type => 'in hn:hn');
4582 # splice @{$self->{open_elements}}, $i;
4583 #}
4584
4585 !!!insert-element-t ($token->{tag_name}, $token->{attributes});
4586
4587 !!!next-token;
4588 redo B;
4589 } elsif ($token->{tag_name} eq 'a') {
4590 AFE: for my $i (reverse 0..$#$active_formatting_elements) {
4591 my $node = $active_formatting_elements->[$i];
4592 if ($node->[1] eq 'a') {
4593 !!!parse-error (type => 'in a:a');
4594
4595 !!!back-token;
4596 $token = {type => END_TAG_TOKEN, tag_name => 'a'};
4597 $formatting_end_tag->($token->{tag_name});
4598
4599 AFE2: for (reverse 0..$#$active_formatting_elements) {
4600 if ($active_formatting_elements->[$_]->[0] eq $node->[0]) {
4601 splice @$active_formatting_elements, $_, 1;
4602 last AFE2;
4603 }
4604 } # AFE2
4605 OE: for (reverse 0..$#{$self->{open_elements}}) {
4606 if ($self->{open_elements}->[$_]->[0] eq $node->[0]) {
4607 splice @{$self->{open_elements}}, $_, 1;
4608 last OE;
4609 }
4610 } # OE
4611 last AFE;
4612 } elsif ($node->[0] eq '#marker') {
4613 last AFE;
4614 }
4615 } # AFE
4616
4617 $reconstruct_active_formatting_elements->($insert_to_current);
4618
4619 !!!insert-element-t ($token->{tag_name}, $token->{attributes});
4620 push @$active_formatting_elements, $self->{open_elements}->[-1];
4621
4622 !!!next-token;
4623 redo B;
4624 } elsif ({
4625 b => 1, big => 1, em => 1, font => 1, i => 1,
4626 s => 1, small => 1, strile => 1,
4627 strong => 1, tt => 1, u => 1,
4628 }->{$token->{tag_name}}) {
4629 $reconstruct_active_formatting_elements->($insert_to_current);
4630
4631 !!!insert-element-t ($token->{tag_name}, $token->{attributes});
4632 push @$active_formatting_elements, $self->{open_elements}->[-1];
4633
4634 !!!next-token;
4635 redo B;
4636 } elsif ($token->{tag_name} eq 'nobr') {
4637 $reconstruct_active_formatting_elements->($insert_to_current);
4638
4639 ## has a |nobr| element in scope
4640 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
4641 my $node = $self->{open_elements}->[$_];
4642 if ($node->[1] eq 'nobr') {
4643 !!!parse-error (type => 'not closed:nobr');
4644 !!!back-token;
4645 $token = {type => END_TAG_TOKEN, tag_name => 'nobr'};
4646 redo B;
4647 } elsif ({
4648 table => 1, caption => 1, td => 1, th => 1,
4649 button => 1, marquee => 1, object => 1, html => 1,
4650 }->{$node->[1]}) {
4651 last INSCOPE;
4652 }
4653 } # INSCOPE
4654
4655 !!!insert-element-t ($token->{tag_name}, $token->{attributes});
4656 push @$active_formatting_elements, $self->{open_elements}->[-1];
4657
4658 !!!next-token;
4659 redo B;
4660 } elsif ($token->{tag_name} eq 'button') {
4661 ## has a button element in scope
4662 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
4663 my $node = $self->{open_elements}->[$_];
4664 if ($node->[1] eq 'button') {
4665 !!!parse-error (type => 'in button:button');
4666 !!!back-token;
4667 $token = {type => END_TAG_TOKEN, tag_name => 'button'};
4668 redo B;
4669 } elsif ({
4670 table => 1, caption => 1, td => 1, th => 1,
4671 button => 1, marquee => 1, object => 1, html => 1,
4672 }->{$node->[1]}) {
4673 last INSCOPE;
4674 }
4675 } # INSCOPE
4676
4677 $reconstruct_active_formatting_elements->($insert_to_current);
4678
4679 !!!insert-element-t ($token->{tag_name}, $token->{attributes});
4680 push @$active_formatting_elements, ['#marker', ''];
4681
4682 !!!next-token;
4683 redo B;
4684 } elsif ($token->{tag_name} eq 'marquee' or
4685 $token->{tag_name} eq 'object') {
4686 $reconstruct_active_formatting_elements->($insert_to_current);
4687
4688 !!!insert-element-t ($token->{tag_name}, $token->{attributes});
4689 push @$active_formatting_elements, ['#marker', ''];
4690
4691 !!!next-token;
4692 redo B;
4693 } elsif ($token->{tag_name} eq 'xmp') {
4694 $reconstruct_active_formatting_elements->($insert_to_current);
4695 $parse_rcdata->(CDATA_CONTENT_MODEL, $insert);
4696 redo B;
4697 } elsif ($token->{tag_name} eq 'table') {
4698 ## has a p element in scope
4699 INSCOPE: for (reverse @{$self->{open_elements}}) {
4700 if ($_->[1] eq 'p') {
4701 !!!back-token;
4702 $token = {type => END_TAG_TOKEN, tag_name => 'p'};
4703 redo B;
4704 } elsif ({
4705 table => 1, caption => 1, td => 1, th => 1,
4706 button => 1, marquee => 1, object => 1, html => 1,
4707 }->{$_->[1]}) {
4708 last INSCOPE;
4709 }
4710 } # INSCOPE
4711
4712 !!!insert-element-t ($token->{tag_name}, $token->{attributes});
4713
4714 $self->{insertion_mode} = IN_TABLE_IM;
4715
4716 !!!next-token;
4717 redo B;
4718 } elsif ({
4719 area => 1, basefont => 1, bgsound => 1, br => 1,
4720 embed => 1, img => 1, param => 1, spacer => 1, wbr => 1,
4721 image => 1,
4722 }->{$token->{tag_name}}) {
4723 if ($token->{tag_name} eq 'image') {
4724 !!!parse-error (type => 'image');
4725 $token->{tag_name} = 'img';
4726 }
4727
4728 ## NOTE: There is an "as if <br>" code clone.
4729 $reconstruct_active_formatting_elements->($insert_to_current);
4730
4731 !!!insert-element-t ($token->{tag_name}, $token->{attributes});
4732 pop @{$self->{open_elements}};
4733
4734 !!!next-token;
4735 redo B;
4736 } elsif ($token->{tag_name} eq 'hr') {
4737 ## has a p element in scope
4738 INSCOPE: for (reverse @{$self->{open_elements}}) {
4739 if ($_->[1] eq 'p') {
4740 !!!back-token;
4741 $token = {type => END_TAG_TOKEN, tag_name => 'p'};
4742 redo B;
4743 } elsif ({
4744 table => 1, caption => 1, td => 1, th => 1,
4745 button => 1, marquee => 1, object => 1, html => 1,
4746 }->{$_->[1]}) {
4747 last INSCOPE;
4748 }
4749 } # INSCOPE
4750
4751 !!!insert-element-t ($token->{tag_name}, $token->{attributes});
4752 pop @{$self->{open_elements}};
4753
4754 !!!next-token;
4755 redo B;
4756 } elsif ($token->{tag_name} eq 'input') {
4757 $reconstruct_active_formatting_elements->($insert_to_current);
4758
4759 !!!insert-element-t ($token->{tag_name}, $token->{attributes});
4760 ## TODO: associate with $self->{form_element} if defined
4761 pop @{$self->{open_elements}};
4762
4763 !!!next-token;
4764 redo B;
4765 } elsif ($token->{tag_name} eq 'isindex') {
4766 !!!parse-error (type => 'isindex');
4767
4768 if (defined $self->{form_element}) {
4769 ## Ignore the token
4770 !!!next-token;
4771 redo B;
4772 } else {
4773 my $at = $token->{attributes};
4774 my $form_attrs;
4775 $form_attrs->{action} = $at->{action} if $at->{action};
4776 my $prompt_attr = $at->{prompt};
4777 $at->{name} = {name => 'name', value => 'isindex'};
4778 delete $at->{action};
4779 delete $at->{prompt};
4780 my @tokens = (
4781 {type => START_TAG_TOKEN, tag_name => 'form',
4782 attributes => $form_attrs},
4783 {type => START_TAG_TOKEN, tag_name => 'hr'},
4784 {type => START_TAG_TOKEN, tag_name => 'p'},
4785 {type => START_TAG_TOKEN, tag_name => 'label'},
4786 );
4787 if ($prompt_attr) {
4788 push @tokens, {type => CHARACTER_TOKEN, data => $prompt_attr->{value}};
4789 } else {
4790 push @tokens, {type => CHARACTER_TOKEN,
4791 data => 'This is a searchable index. Insert your search keywords here: '}; # SHOULD
4792 ## TODO: make this configurable
4793 }
4794 push @tokens,
4795 {type => START_TAG_TOKEN, tag_name => 'input', attributes => $at},
4796 #{type => CHARACTER_TOKEN, data => ''}, # SHOULD
4797 {type => END_TAG_TOKEN, tag_name => 'label'},
4798 {type => END_TAG_TOKEN, tag_name => 'p'},
4799 {type => START_TAG_TOKEN, tag_name => 'hr'},
4800 {type => END_TAG_TOKEN, tag_name => 'form'};
4801 $token = shift @tokens;
4802 !!!back-token (@tokens);
4803 redo B;
4804 }
4805 } elsif ($token->{tag_name} eq 'textarea') {
4806 my $tag_name = $token->{tag_name};
4807 my $el;
4808 !!!create-element ($el, $token->{tag_name}, $token->{attributes});
4809
4810 ## TODO: $self->{form_element} if defined
4811 $self->{content_model} = RCDATA_CONTENT_MODEL;
4812 delete $self->{escape}; # MUST
4813
4814 $insert->($el);
4815
4816 my $text = '';
4817 !!!next-token;
4818 if ($token->{type} == CHARACTER_TOKEN) {
4819 $token->{data} =~ s/^\x0A//;
4820 unless (length $token->{data}) {
4821 !!!next-token;
4822 }
4823 }
4824 while ($token->{type} == CHARACTER_TOKEN) {
4825 $text .= $token->{data};
4826 !!!next-token;
4827 }
4828 if (length $text) {
4829 $el->manakai_append_text ($text);
4830 }
4831
4832 $self->{content_model} = PCDATA_CONTENT_MODEL;
4833
4834 if ($token->{type} == END_TAG_TOKEN and
4835 $token->{tag_name} eq $tag_name) {
4836 ## Ignore the token
4837 } else {
4838 !!!parse-error (type => 'in RCDATA:#'.$token->{type});
4839 }
4840 !!!next-token;
4841 redo B;
4842 } elsif ({
4843 iframe => 1,
4844 noembed => 1,
4845 noframes => 1,
4846 noscript => 0, ## TODO: 1 if scripting is enabled
4847 }->{$token->{tag_name}}) {
4848 ## NOTE: There are two "as if in body" code clones.
4849 $parse_rcdata->(CDATA_CONTENT_MODEL, $insert);
4850 redo B;
4851 } elsif ($token->{tag_name} eq 'select') {
4852 $reconstruct_active_formatting_elements->($insert_to_current);
4853
4854 !!!insert-element-t ($token->{tag_name}, $token->{attributes});
4855
4856 $self->{insertion_mode} = IN_SELECT_IM;
4857 !!!next-token;
4858 redo B;
4859 } elsif ({
4860 caption => 1, col => 1, colgroup => 1, frame => 1,
4861 frameset => 1, head => 1, option => 1, optgroup => 1,
4862 tbody => 1, td => 1, tfoot => 1, th => 1,
4863 thead => 1, tr => 1,
4864 }->{$token->{tag_name}}) {
4865 !!!parse-error (type => 'in body:'.$token->{tag_name});
4866 ## Ignore the token
4867 !!!next-token;
4868 redo B;
4869
4870 ## ISSUE: An issue on HTML5 new elements in the spec.
4871 } else {
4872 $reconstruct_active_formatting_elements->($insert_to_current);
4873
4874 !!!insert-element-t ($token->{tag_name}, $token->{attributes});
4875
4876 !!!next-token;
4877 redo B;
4878 }
4879 } elsif ($token->{type} == END_TAG_TOKEN) {
4880 if ($token->{tag_name} eq 'body') {
4881 if (@{$self->{open_elements}} > 1 and
4882 $self->{open_elements}->[1]->[1] eq 'body') {
4883 for (@{$self->{open_elements}}) {
4884 unless ({
4885 dd => 1, dt => 1, li => 1, p => 1, td => 1,
4886 th => 1, tr => 1, body => 1, html => 1,
4887 tbody => 1, tfoot => 1, thead => 1,
4888 }->{$_->[1]}) {
4889 !!!parse-error (type => 'not closed:'.$_->[1]);
4890 }
4891 }
4892
4893 $self->{insertion_mode} = AFTER_BODY_IM;
4894 !!!next-token;
4895 redo B;
4896 } else {
4897 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
4898 ## Ignore the token
4899 !!!next-token;
4900 redo B;
4901 }
4902 } elsif ($token->{tag_name} eq 'html') {
4903 if (@{$self->{open_elements}} > 1 and $self->{open_elements}->[1]->[1] eq 'body') {
4904 ## ISSUE: There is an issue in the spec.
4905 if ($self->{open_elements}->[-1]->[1] ne 'body') {
4906 !!!parse-error (type => 'not closed:'.$self->{open_elements}->[1]->[1]);
4907 }
4908 $self->{insertion_mode} = AFTER_BODY_IM;
4909 ## reprocess
4910 redo B;
4911 } else {
4912 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
4913 ## Ignore the token
4914 !!!next-token;
4915 redo B;
4916 }
4917 } elsif ({
4918 address => 1, blockquote => 1, center => 1, dir => 1,
4919 div => 1, dl => 1, fieldset => 1, listing => 1,
4920 menu => 1, ol => 1, pre => 1, ul => 1,
4921 p => 1,
4922 dd => 1, dt => 1, li => 1,
4923 button => 1, marquee => 1, object => 1,
4924 }->{$token->{tag_name}}) {
4925 ## has an element in scope
4926 my $i;
4927 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
4928 my $node = $self->{open_elements}->[$_];
4929 if ($node->[1] eq $token->{tag_name}) {
4930 ## generate implied end tags
4931 if ({
4932 dd => ($token->{tag_name} ne 'dd'),
4933 dt => ($token->{tag_name} ne 'dt'),
4934 li => ($token->{tag_name} ne 'li'),
4935 p => ($token->{tag_name} ne 'p'),
4936 td => 1, th => 1, tr => 1,
4937 tbody => 1, tfoot=> 1, thead => 1,
4938 }->{$self->{open_elements}->[-1]->[1]}) {
4939 !!!back-token;
4940 $token = {type => END_TAG_TOKEN,
4941 tag_name => $self->{open_elements}->[-1]->[1]}; # MUST
4942 redo B;
4943 }
4944 $i = $_;
4945 last INSCOPE unless $token->{tag_name} eq 'p';
4946 } elsif ({
4947 table => 1, caption => 1, td => 1, th => 1,
4948 button => 1, marquee => 1, object => 1, html => 1,
4949 }->{$node->[1]}) {
4950 last INSCOPE;
4951 }
4952 } # INSCOPE
4953
4954 if ($self->{open_elements}->[-1]->[1] ne $token->{tag_name}) {
4955 if (defined $i) {
4956 !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);
4957 } else {
4958 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
4959 }
4960 }
4961
4962 if (defined $i) {
4963 splice @{$self->{open_elements}}, $i;
4964 } elsif ($token->{tag_name} eq 'p') {
4965 ## As if <p>, then reprocess the current token
4966 my $el;
4967 !!!create-element ($el, 'p');
4968 $insert->($el);
4969 }
4970 $clear_up_to_marker->()
4971 if {
4972 button => 1, marquee => 1, object => 1,
4973 }->{$token->{tag_name}};
4974 !!!next-token;
4975 redo B;
4976 } elsif ($token->{tag_name} eq 'form') {
4977 ## has an element in scope
4978 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
4979 my $node = $self->{open_elements}->[$_];
4980 if ($node->[1] eq $token->{tag_name}) {
4981 ## generate implied end tags
4982 if ({
4983 dd => 1, dt => 1, li => 1, p => 1,
4984 td => 1, th => 1, tr => 1,
4985 tbody => 1, tfoot=> 1, thead => 1,
4986 }->{$self->{open_elements}->[-1]->[1]}) {
4987 !!!back-token;
4988 $token = {type => END_TAG_TOKEN,
4989 tag_name => $self->{open_elements}->[-1]->[1]}; # MUST
4990 redo B;
4991 }
4992 last INSCOPE;
4993 } elsif ({
4994 table => 1, caption => 1, td => 1, th => 1,
4995 button => 1, marquee => 1, object => 1, html => 1,
4996 }->{$node->[1]}) {
4997 last INSCOPE;
4998 }
4999 } # INSCOPE
5000
5001 if ($self->{open_elements}->[-1]->[1] eq $token->{tag_name}) {
5002 pop @{$self->{open_elements}};
5003 } else {
5004 !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);
5005 }
5006
5007 undef $self->{form_element};
5008 !!!next-token;
5009 redo B;
5010 } elsif ({
5011 h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,
5012 }->{$token->{tag_name}}) {
5013 ## has an element in scope
5014 my $i;
5015 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5016 my $node = $self->{open_elements}->[$_];
5017 if ({
5018 h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,
5019 }->{$node->[1]}) {
5020 ## generate implied end tags
5021 if ({
5022 dd => 1, dt => 1, li => 1, p => 1,
5023 td => 1, th => 1, tr => 1,
5024 tbody => 1, tfoot=> 1, thead => 1,
5025 }->{$self->{open_elements}->[-1]->[1]}) {
5026 !!!back-token;
5027 $token = {type => END_TAG_TOKEN,
5028 tag_name => $self->{open_elements}->[-1]->[1]}; # MUST
5029 redo B;
5030 }
5031 $i = $_;
5032 last INSCOPE;
5033 } elsif ({
5034 table => 1, caption => 1, td => 1, th => 1,
5035 button => 1, marquee => 1, object => 1, html => 1,
5036 }->{$node->[1]}) {
5037 last INSCOPE;
5038 }
5039 } # INSCOPE
5040
5041 if ($self->{open_elements}->[-1]->[1] ne $token->{tag_name}) {
5042 !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);
5043 }
5044
5045 splice @{$self->{open_elements}}, $i if defined $i;
5046 !!!next-token;
5047 redo B;
5048 } elsif ({
5049 a => 1,
5050 b => 1, big => 1, em => 1, font => 1, i => 1,
5051 nobr => 1, s => 1, small => 1, strile => 1,
5052 strong => 1, tt => 1, u => 1,
5053 }->{$token->{tag_name}}) {
5054 $formatting_end_tag->($token->{tag_name});
5055 redo B;
5056 } elsif ($token->{tag_name} eq 'br') {
5057 !!!parse-error (type => 'unmatched end tag:br');
5058
5059 ## As if <br>
5060 $reconstruct_active_formatting_elements->($insert_to_current);
5061
5062 my $el;
5063 !!!create-element ($el, 'br');
5064 $insert->($el);
5065
5066 ## Ignore the token.
5067 !!!next-token;
5068 redo B;
5069 } elsif ({
5070 caption => 1, col => 1, colgroup => 1, frame => 1,
5071 frameset => 1, head => 1, option => 1, optgroup => 1,
5072 tbody => 1, td => 1, tfoot => 1, th => 1,
5073 thead => 1, tr => 1,
5074 area => 1, basefont => 1, bgsound => 1,
5075 embed => 1, hr => 1, iframe => 1, image => 1,
5076 img => 1, input => 1, isindex => 1, noembed => 1,
5077 noframes => 1, param => 1, select => 1, spacer => 1,
5078 table => 1, textarea => 1, wbr => 1,
5079 noscript => 0, ## TODO: if scripting is enabled
5080 }->{$token->{tag_name}}) {
5081 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
5082 ## Ignore the token
5083 !!!next-token;
5084 redo B;
5085
5086 ## ISSUE: Issue on HTML5 new elements in spec
5087
5088 } else {
5089 ## Step 1
5090 my $node_i = -1;
5091 my $node = $self->{open_elements}->[$node_i];
5092
5093 ## Step 2
5094 S2: {
5095 if ($node->[1] eq $token->{tag_name}) {
5096 ## Step 1
5097 ## generate implied end tags
5098 if ({
5099 dd => 1, dt => 1, li => 1, p => 1,
5100 td => 1, th => 1, tr => 1,
5101 tbody => 1, tfoot => 1, thead => 1,
5102 }->{$self->{open_elements}->[-1]->[1]}) {
5103 !!!back-token;
5104 $token = {type => END_TAG_TOKEN,
5105 tag_name => $self->{open_elements}->[-1]->[1]}; # MUST
5106 redo B;
5107 }
5108
5109 ## Step 2
5110 if ($token->{tag_name} ne $self->{open_elements}->[-1]->[1]) {
5111 !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);
5112 }
5113
5114 ## Step 3
5115 splice @{$self->{open_elements}}, $node_i;
5116
5117 !!!next-token;
5118 last S2;
5119 } else {
5120 ## Step 3
5121 if (not $formatting_category->{$node->[1]} and
5122 #not $phrasing_category->{$node->[1]} and
5123 ($special_category->{$node->[1]} or
5124 $scoping_category->{$node->[1]})) {
5125 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
5126 ## Ignore the token
5127 !!!next-token;
5128 last S2;
5129 }
5130 }
5131
5132 ## Step 4
5133 $node_i--;
5134 $node = $self->{open_elements}->[$node_i];
5135
5136 ## Step 5;
5137 redo S2;
5138 } # S2
5139 redo B;
5140 }
5141 }
5142 redo B;
5143 } # B
5144
5145 ## NOTE: The "trailing end" phase in HTML5 is split into
5146 ## two insertion modes: "after html body" and "after html frameset".
5147 ## NOTE: States in the main stage is preserved while
5148 ## the parser stays in the trailing end phase. # MUST
5149
5150 ## Stop parsing # MUST
5151
5152 ## TODO: script stuffs
5153 } # _tree_construct_main
5154
5155 sub set_inner_html ($$$) {
5156 my $class = shift;
5157 my $node = shift;
5158 my $s = \$_[0];
5159 my $onerror = $_[1];
5160
5161 my $nt = $node->node_type;
5162 if ($nt == 9) {
5163 # MUST
5164
5165 ## Step 1 # MUST
5166 ## TODO: If the document has an active parser, ...
5167 ## ISSUE: There is an issue in the spec.
5168
5169 ## Step 2 # MUST
5170 my @cn = @{$node->child_nodes};
5171 for (@cn) {
5172 $node->remove_child ($_);
5173 }
5174
5175 ## Step 3, 4, 5 # MUST
5176 $class->parse_string ($$s => $node, $onerror);
5177 } elsif ($nt == 1) {
5178 ## TODO: If non-html element
5179
5180 ## NOTE: Most of this code is copied from |parse_string|
5181
5182 ## Step 1 # MUST
5183 my $this_doc = $node->owner_document;
5184 my $doc = $this_doc->implementation->create_document;
5185 $doc->manakai_is_html (1);
5186 my $p = $class->new;
5187 $p->{document} = $doc;
5188
5189 ## Step 9 # MUST
5190 my $i = 0;
5191 my $line = 1;
5192 my $column = 0;
5193 $p->{set_next_input_character} = sub {
5194 my $self = shift;
5195
5196 pop @{$self->{prev_input_character}};
5197 unshift @{$self->{prev_input_character}}, $self->{next_input_character};
5198
5199 $self->{next_input_character} = -1 and return if $i >= length $$s;
5200 $self->{next_input_character} = ord substr $$s, $i++, 1;
5201 $column++;
5202
5203 if ($self->{next_input_character} == 0x000A) { # LF
5204 $line++;
5205 $column = 0;
5206 } elsif ($self->{next_input_character} == 0x000D) { # CR
5207 $i++ if substr ($$s, $i, 1) eq "\x0A";
5208 $self->{next_input_character} = 0x000A; # LF # MUST
5209 $line++;
5210 $column = 0;
5211 } elsif ($self->{next_input_character} > 0x10FFFF) {
5212 $self->{next_input_character} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
5213 } elsif ($self->{next_input_character} == 0x0000) { # NULL
5214 !!!parse-error (type => 'NULL');
5215 $self->{next_input_character} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
5216 }
5217 };
5218 $p->{prev_input_character} = [-1, -1, -1];
5219 $p->{next_input_character} = -1;
5220
5221 my $ponerror = $onerror || sub {
5222 my (%opt) = @_;
5223 warn "Parse error ($opt{type}) at line $opt{line} column $opt{column}\n";
5224 };
5225 $p->{parse_error} = sub {
5226 $ponerror->(@_, line => $line, column => $column);
5227 };
5228
5229 $p->_initialize_tokenizer;
5230 $p->_initialize_tree_constructor;
5231
5232 ## Step 2
5233 my $node_ln = $node->local_name;
5234 $p->{content_model} = {
5235 title => RCDATA_CONTENT_MODEL,
5236 textarea => RCDATA_CONTENT_MODEL,
5237 style => CDATA_CONTENT_MODEL,
5238 script => CDATA_CONTENT_MODEL,
5239 xmp => CDATA_CONTENT_MODEL,
5240 iframe => CDATA_CONTENT_MODEL,
5241 noembed => CDATA_CONTENT_MODEL,
5242 noframes => CDATA_CONTENT_MODEL,
5243 noscript => CDATA_CONTENT_MODEL,
5244 plaintext => PLAINTEXT_CONTENT_MODEL,
5245 }->{$node_ln};
5246 $p->{content_model} = PCDATA_CONTENT_MODEL
5247 unless defined $p->{content_model};
5248 ## ISSUE: What is "the name of the element"? local name?
5249
5250 $p->{inner_html_node} = [$node, $node_ln];
5251
5252 ## Step 4
5253 my $root = $doc->create_element_ns
5254 ('http://www.w3.org/1999/xhtml', [undef, 'html']);
5255
5256 ## Step 5 # MUST
5257 $doc->append_child ($root);
5258
5259 ## Step 6 # MUST
5260 push @{$p->{open_elements}}, [$root, 'html'];
5261
5262 undef $p->{head_element};
5263
5264 ## Step 7 # MUST
5265 $p->_reset_insertion_mode;
5266
5267 ## Step 8 # MUST
5268 my $anode = $node;
5269 AN: while (defined $anode) {
5270 if ($anode->node_type == 1) {
5271 my $nsuri = $anode->namespace_uri;
5272 if (defined $nsuri and $nsuri eq 'http://www.w3.org/1999/xhtml') {
5273 if ($anode->local_name eq 'form') { ## TODO: case?
5274 $p->{form_element} = $anode;
5275 last AN;
5276 }
5277 }
5278 }
5279 $anode = $anode->parent_node;
5280 } # AN
5281
5282 ## Step 3 # MUST
5283 ## Step 10 # MUST
5284 {
5285 my $self = $p;
5286 !!!next-token;
5287 }
5288 $p->_tree_construction_main;
5289
5290 ## Step 11 # MUST
5291 my @cn = @{$node->child_nodes};
5292 for (@cn) {
5293 $node->remove_child ($_);
5294 }
5295 ## ISSUE: mutation events? read-only?
5296
5297 ## Step 12 # MUST
5298 @cn = @{$root->child_nodes};
5299 for (@cn) {
5300 $this_doc->adopt_node ($_);
5301 $node->append_child ($_);
5302 }
5303 ## ISSUE: mutation events?
5304
5305 $p->_terminate_tree_constructor;
5306 } else {
5307 die "$0: |set_inner_html| is not defined for node of type $nt";
5308 }
5309 } # set_inner_html
5310
5311 } # tree construction stage
5312
5313 sub get_inner_html ($$$) {
5314 my (undef, $node, $on_error) = @_;
5315
5316 ## Step 1
5317 my $s = '';
5318
5319 my $in_cdata;
5320 my $parent = $node;
5321 while (defined $parent) {
5322 if ($parent->node_type == 1 and
5323 $parent->namespace_uri eq 'http://www.w3.org/1999/xhtml' and
5324 {
5325 style => 1, script => 1, xmp => 1, iframe => 1,
5326 noembed => 1, noframes => 1, noscript => 1,
5327 }->{$parent->local_name}) { ## TODO: case thingy
5328 $in_cdata = 1;
5329 }
5330 $parent = $parent->parent_node;
5331 }
5332
5333 ## Step 2
5334 my @node = @{$node->child_nodes};
5335 C: while (@node) {
5336 my $child = shift @node;
5337 unless (ref $child) {
5338 if ($child eq 'cdata-out') {
5339 $in_cdata = 0;
5340 } else {
5341 $s .= $child; # end tag
5342 }
5343 next C;
5344 }
5345
5346 my $nt = $child->node_type;
5347 if ($nt == 1) { # Element
5348 my $tag_name = $child->tag_name; ## TODO: manakai_tag_name
5349 $s .= '<' . $tag_name;
5350 ## NOTE: Non-HTML case:
5351 ## <http://permalink.gmane.org/gmane.org.w3c.whatwg.discuss/11191>
5352
5353 my @attrs = @{$child->attributes}; # sort order MUST be stable
5354 for my $attr (@attrs) { # order is implementation dependent
5355 my $attr_name = $attr->name; ## TODO: manakai_name
5356 $s .= ' ' . $attr_name . '="';
5357 my $attr_value = $attr->value;
5358 ## escape
5359 $attr_value =~ s/&/&amp;/g;
5360 $attr_value =~ s/</&lt;/g;
5361 $attr_value =~ s/>/&gt;/g;
5362 $attr_value =~ s/"/&quot;/g;
5363 $s .= $attr_value . '"';
5364 }
5365 $s .= '>';
5366
5367 next C if {
5368 area => 1, base => 1, basefont => 1, bgsound => 1,
5369 br => 1, col => 1, embed => 1, frame => 1, hr => 1,
5370 img => 1, input => 1, link => 1, meta => 1, param => 1,
5371 spacer => 1, wbr => 1,
5372 }->{$tag_name};
5373
5374 $s .= "\x0A" if $tag_name eq 'pre' or $tag_name eq 'textarea';
5375
5376 if (not $in_cdata and {
5377 style => 1, script => 1, xmp => 1, iframe => 1,
5378 noembed => 1, noframes => 1, noscript => 1,
5379 plaintext => 1,
5380 }->{$tag_name}) {
5381 unshift @node, 'cdata-out';
5382 $in_cdata = 1;
5383 }
5384
5385 unshift @node, @{$child->child_nodes}, '</' . $tag_name . '>';
5386 } elsif ($nt == 3 or $nt == 4) {
5387 if ($in_cdata) {
5388 $s .= $child->data;
5389 } else {
5390 my $value = $child->data;
5391 $value =~ s/&/&amp;/g;
5392 $value =~ s/</&lt;/g;
5393 $value =~ s/>/&gt;/g;
5394 $value =~ s/"/&quot;/g;
5395 $s .= $value;
5396 }
5397 } elsif ($nt == 8) {
5398 $s .= '<!--' . $child->data . '-->';
5399 } elsif ($nt == 10) {
5400 $s .= '<!DOCTYPE ' . $child->name . '>';
5401 } elsif ($nt == 5) { # entrefs
5402 push @node, @{$child->child_nodes};
5403 } else {
5404 $on_error->($child) if defined $on_error;
5405 }
5406 ## ISSUE: This code does not support PIs.
5407 } # C
5408
5409 ## Step 3
5410 return \$s;
5411 } # get_inner_html
5412
5413 1;
5414 # $Date: 2007/08/11 06:37:12 $

[email protected]
ViewVC Help
Powered by ViewVC 1.1.24