| 7 |
## doc.write (''); |
## doc.write (''); |
| 8 |
## alert (doc.compatMode); |
## alert (doc.compatMode); |
| 9 |
|
|
| 10 |
|
## ISSUE: HTML5 revision 967 says that the encoding layer MUST NOT |
| 11 |
|
## strip BOM and the HTML layer MUST ignore it. Whether we can do it |
| 12 |
|
## is not yet clear. |
| 13 |
|
## "{U+FEFF}..." in UTF-16BE/UTF-16LE is three or four characters? |
| 14 |
|
## "{U+FEFF}..." in GB18030? |
| 15 |
|
|
| 16 |
my $permitted_slash_tag_name = { |
my $permitted_slash_tag_name = { |
| 17 |
base => 1, |
base => 1, |
| 18 |
link => 1, |
link => 1, |
| 333 |
if ($self->{content_model_flag} eq 'RCDATA' or |
if ($self->{content_model_flag} eq 'RCDATA' or |
| 334 |
$self->{content_model_flag} eq 'CDATA') { |
$self->{content_model_flag} eq 'CDATA') { |
| 335 |
if (defined $self->{last_emitted_start_tag_name}) { |
if (defined $self->{last_emitted_start_tag_name}) { |
| 336 |
|
## NOTE: <http://krijnhoetmer.nl/irc-logs/whatwg/20070626#l-564> |
| 337 |
my @next_char; |
my @next_char; |
| 338 |
TAGNAME: for (my $i = 0; $i < length $self->{last_emitted_start_tag_name}; $i++) { |
TAGNAME: for (my $i = 0; $i < length $self->{last_emitted_start_tag_name}; $i++) { |
| 339 |
push @next_char, $self->{next_input_character}; |
push @next_char, $self->{next_input_character}; |
| 705 |
# |
# |
| 706 |
} else { |
} else { |
| 707 |
!!!parse-error (type => 'nestc'); |
!!!parse-error (type => 'nestc'); |
| 708 |
|
## TODO: Different error type for <aa / bb> than <aa/> |
| 709 |
} |
} |
| 710 |
$self->{state} = 'before attribute name'; |
$self->{state} = 'before attribute name'; |
| 711 |
# next-input-character is already done |
# next-input-character is already done |
| 1026 |
} |
} |
| 1027 |
} |
} |
| 1028 |
|
|
| 1029 |
!!!parse-error (type => 'bogus comment open'); |
!!!parse-error (type => 'bogus comment'); |
| 1030 |
$self->{next_input_character} = shift @next_char; |
$self->{next_input_character} = shift @next_char; |
| 1031 |
!!!back-next-input-character (@next_char); |
!!!back-next-input-character (@next_char); |
| 1032 |
$self->{state} = 'bogus comment'; |
$self->{state} = 'bogus comment'; |
| 1085 |
redo A; |
redo A; |
| 1086 |
} else { |
} else { |
| 1087 |
$self->{current_token}->{data} # comment |
$self->{current_token}->{data} # comment |
| 1088 |
.= chr ($self->{next_input_character}); |
.= '-' . chr ($self->{next_input_character}); |
| 1089 |
$self->{state} = 'comment'; |
$self->{state} = 'comment'; |
| 1090 |
!!!next-input-character; |
!!!next-input-character; |
| 1091 |
redo A; |
redo A; |
| 1495 |
|
|
| 1496 |
redo A; |
redo A; |
| 1497 |
} else { |
} else { |
| 1498 |
!!!parse-error (type => 'string after PUBLIC literal'); |
!!!parse-error (type => 'string after SYSTEM'); |
| 1499 |
$self->{state} = 'bogus DOCTYPE'; |
$self->{state} = 'bogus DOCTYPE'; |
| 1500 |
!!!next-input-character; |
!!!next-input-character; |
| 1501 |
redo A; |
redo A; |
| 1663 |
!!!parse-error (type => 'CR character reference'); |
!!!parse-error (type => 'CR character reference'); |
| 1664 |
$code = 0x000A; |
$code = 0x000A; |
| 1665 |
} elsif (0x80 <= $code and $code <= 0x9F) { |
} elsif (0x80 <= $code and $code <= 0x9F) { |
| 1666 |
!!!parse-error (type => sprintf 'c1 entity:U+%04X', $code); |
!!!parse-error (type => sprintf 'C1 character reference:U+%04X', $code); |
| 1667 |
$code = $c1_entity_char->{$code}; |
$code = $c1_entity_char->{$code}; |
| 1668 |
} |
} |
| 1669 |
|
|
| 1698 |
!!!parse-error (type => 'CR character reference'); |
!!!parse-error (type => 'CR character reference'); |
| 1699 |
$code = 0x000A; |
$code = 0x000A; |
| 1700 |
} elsif (0x80 <= $code and $code <= 0x9F) { |
} elsif (0x80 <= $code and $code <= 0x9F) { |
| 1701 |
!!!parse-error (type => sprintf 'c1 entity:U+%04X', $code); |
!!!parse-error (type => sprintf 'C1 character reference:U+%04X', $code); |
| 1702 |
$code = $c1_entity_char->{$code}; |
$code = $c1_entity_char->{$code}; |
| 1703 |
} |
} |
| 1704 |
|
|
| 1752 |
if ($match > 0) { |
if ($match > 0) { |
| 1753 |
return {type => 'character', data => $value}; |
return {type => 'character', data => $value}; |
| 1754 |
} elsif ($match < 0) { |
} elsif ($match < 0) { |
| 1755 |
!!!parse-error (type => 'refc'); |
!!!parse-error (type => 'no refc'); |
| 1756 |
return {type => 'character', data => $value}; |
return {type => 'character', data => $value}; |
| 1757 |
} else { |
} else { |
| 1758 |
!!!parse-error (type => 'bare ero'); |
!!!parse-error (type => 'bare ero'); |
| 2018 |
my $root_element; !!!create-element ($root_element, 'html'); |
my $root_element; !!!create-element ($root_element, 'html'); |
| 2019 |
$self->{document}->append_child ($root_element); |
$self->{document}->append_child ($root_element); |
| 2020 |
push @{$self->{open_elements}}, [$root_element, 'html']; |
push @{$self->{open_elements}}, [$root_element, 'html']; |
|
#$phase = 'main'; |
|
| 2021 |
## reprocess |
## reprocess |
| 2022 |
#redo B; |
#redo B; |
| 2023 |
return; |
return; ## Go to the main phase. |
| 2024 |
} # B |
} # B |
| 2025 |
} # _tree_construction_root_element |
} # _tree_construction_root_element |
| 2026 |
|
|
| 2096 |
sub _tree_construction_main ($) { |
sub _tree_construction_main ($) { |
| 2097 |
my $self = shift; |
my $self = shift; |
| 2098 |
|
|
| 2099 |
my $phase = 'main'; |
my $previous_insertion_mode; |
| 2100 |
|
|
| 2101 |
my $active_formatting_elements = []; |
my $active_formatting_elements = []; |
| 2102 |
|
|
| 2497 |
$parse_rcdata->('CDATA', $insert); |
$parse_rcdata->('CDATA', $insert); |
| 2498 |
return; |
return; |
| 2499 |
} elsif ({ |
} elsif ({ |
| 2500 |
base => 1, link => 1, meta => 1, |
base => 1, link => 1, |
| 2501 |
}->{$token->{tag_name}}) { |
}->{$token->{tag_name}}) { |
| 2502 |
## NOTE: This is an "as if in head" code clone, only "-t" differs |
## NOTE: This is an "as if in head" code clone, only "-t" differs |
| 2503 |
!!!insert-element-t ($token->{tag_name}, $token->{attributes}); |
!!!insert-element-t ($token->{tag_name}, $token->{attributes}); |
| 2504 |
pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec. |
pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec. |
| 2505 |
!!!next-token; |
!!!next-token; |
| 2506 |
## TODO: Extracting |charset| from |meta|. |
return; |
| 2507 |
|
} elsif ($token->{tag_name} eq 'meta') { |
| 2508 |
|
## NOTE: This is an "as if in head" code clone, only "-t" differs |
| 2509 |
|
!!!insert-element-t ($token->{tag_name}, $token->{attributes}); |
| 2510 |
|
pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec. |
| 2511 |
|
|
| 2512 |
|
unless ($self->{confident}) { |
| 2513 |
|
my $charset; |
| 2514 |
|
if ($token->{attributes}->{charset}) { ## TODO: And if supported |
| 2515 |
|
$charset = $token->{attributes}->{charset}->{value}; |
| 2516 |
|
} |
| 2517 |
|
if ($token->{attributes}->{'http-equiv'}) { |
| 2518 |
|
## ISSUE: Algorithm name in the spec was incorrect so that not linked to the definition. |
| 2519 |
|
if ($token->{attributes}->{'http-equiv'}->{value} |
| 2520 |
|
=~ /\A[^;]*;[\x09-\x0D\x20]*charset[\x09-\x0D\x20]*= |
| 2521 |
|
[\x09-\x0D\x20]*(?>"([^"]*)"|'([^']*)'| |
| 2522 |
|
([^"'\x09-\x0D\x20][^\x09-\x0D\x20]*))/x) { |
| 2523 |
|
$charset = defined $1 ? $1 : defined $2 ? $2 : $3; |
| 2524 |
|
} ## TODO: And if supported |
| 2525 |
|
} |
| 2526 |
|
## TODO: Change the encoding |
| 2527 |
|
} |
| 2528 |
|
|
| 2529 |
|
!!!next-token; |
| 2530 |
return; |
return; |
| 2531 |
} elsif ($token->{tag_name} eq 'title') { |
} elsif ($token->{tag_name} eq 'title') { |
| 2532 |
!!!parse-error (type => 'in body:title'); |
!!!parse-error (type => 'in body:title'); |
| 2533 |
## NOTE: This is an "as if in head" code clone |
## NOTE: This is an "as if in head" code clone |
| 2534 |
$parse_rcdata->('RCDATA', $insert); |
$parse_rcdata->('RCDATA', sub { |
| 2535 |
|
if (defined $self->{head_element}) { |
| 2536 |
|
$self->{head_element}->append_child ($_[0]); |
| 2537 |
|
} else { |
| 2538 |
|
$insert->($_[0]); |
| 2539 |
|
} |
| 2540 |
|
}); |
| 2541 |
return; |
return; |
| 2542 |
} elsif ($token->{tag_name} eq 'body') { |
} elsif ($token->{tag_name} eq 'body') { |
| 2543 |
!!!parse-error (type => 'in body:body'); |
!!!parse-error (type => 'in body:body'); |
| 2830 |
INSCOPE: for (reverse 0..$#{$self->{open_elements}}) { |
INSCOPE: for (reverse 0..$#{$self->{open_elements}}) { |
| 2831 |
my $node = $self->{open_elements}->[$_]; |
my $node = $self->{open_elements}->[$_]; |
| 2832 |
if ($node->[1] eq 'nobr') { |
if ($node->[1] eq 'nobr') { |
| 2833 |
|
!!!parse-error (type => 'not closed:nobr'); |
| 2834 |
!!!back-token; |
!!!back-token; |
| 2835 |
$token = {type => 'end tag', tag_name => 'nobr'}; |
$token = {type => 'end tag', tag_name => 'nobr'}; |
| 2836 |
return; |
return; |
| 2914 |
!!!parse-error (type => 'image'); |
!!!parse-error (type => 'image'); |
| 2915 |
$token->{tag_name} = 'img'; |
$token->{tag_name} = 'img'; |
| 2916 |
} |
} |
| 2917 |
|
|
| 2918 |
|
## NOTE: There is an "as if <br>" code clone. |
| 2919 |
$reconstruct_active_formatting_elements->($insert_to_current); |
$reconstruct_active_formatting_elements->($insert_to_current); |
| 2920 |
|
|
| 2921 |
!!!insert-element-t ($token->{tag_name}, $token->{attributes}); |
!!!insert-element-t ($token->{tag_name}, $token->{attributes}); |
| 3073 |
unless ({ |
unless ({ |
| 3074 |
dd => 1, dt => 1, li => 1, p => 1, td => 1, |
dd => 1, dt => 1, li => 1, p => 1, td => 1, |
| 3075 |
th => 1, tr => 1, body => 1, html => 1, |
th => 1, tr => 1, body => 1, html => 1, |
| 3076 |
|
tbody => 1, tfoot => 1, thead => 1, |
| 3077 |
}->{$_->[1]}) { |
}->{$_->[1]}) { |
| 3078 |
!!!parse-error (type => 'not closed:'.$_->[1]); |
!!!parse-error (type => 'not closed:'.$_->[1]); |
| 3079 |
} |
} |
| 3123 |
li => ($token->{tag_name} ne 'li'), |
li => ($token->{tag_name} ne 'li'), |
| 3124 |
p => ($token->{tag_name} ne 'p'), |
p => ($token->{tag_name} ne 'p'), |
| 3125 |
td => 1, th => 1, tr => 1, |
td => 1, th => 1, tr => 1, |
| 3126 |
|
tbody => 1, tfoot=> 1, thead => 1, |
| 3127 |
}->{$self->{open_elements}->[-1]->[1]}) { |
}->{$self->{open_elements}->[-1]->[1]}) { |
| 3128 |
!!!back-token; |
!!!back-token; |
| 3129 |
$token = {type => 'end tag', |
$token = {type => 'end tag', |
| 3141 |
} # INSCOPE |
} # INSCOPE |
| 3142 |
|
|
| 3143 |
if ($self->{open_elements}->[-1]->[1] ne $token->{tag_name}) { |
if ($self->{open_elements}->[-1]->[1] ne $token->{tag_name}) { |
| 3144 |
!!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]); |
if (defined $i) { |
| 3145 |
|
!!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]); |
| 3146 |
|
} else { |
| 3147 |
|
!!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}); |
| 3148 |
|
} |
| 3149 |
} |
} |
| 3150 |
|
|
| 3151 |
splice @{$self->{open_elements}}, $i if defined $i; |
if (defined $i) { |
| 3152 |
|
splice @{$self->{open_elements}}, $i; |
| 3153 |
|
} elsif ($token->{tag_name} eq 'p') { |
| 3154 |
|
## As if <p>, then reprocess the current token |
| 3155 |
|
my $el; |
| 3156 |
|
!!!create-element ($el, 'p'); |
| 3157 |
|
$insert->($el); |
| 3158 |
|
} |
| 3159 |
$clear_up_to_marker->() |
$clear_up_to_marker->() |
| 3160 |
if { |
if { |
| 3161 |
button => 1, marquee => 1, object => 1, |
button => 1, marquee => 1, object => 1, |
| 3171 |
if ({ |
if ({ |
| 3172 |
dd => 1, dt => 1, li => 1, p => 1, |
dd => 1, dt => 1, li => 1, p => 1, |
| 3173 |
td => 1, th => 1, tr => 1, |
td => 1, th => 1, tr => 1, |
| 3174 |
|
tbody => 1, tfoot=> 1, thead => 1, |
| 3175 |
}->{$self->{open_elements}->[-1]->[1]}) { |
}->{$self->{open_elements}->[-1]->[1]}) { |
| 3176 |
!!!back-token; |
!!!back-token; |
| 3177 |
$token = {type => 'end tag', |
$token = {type => 'end tag', |
| 3210 |
if ({ |
if ({ |
| 3211 |
dd => 1, dt => 1, li => 1, p => 1, |
dd => 1, dt => 1, li => 1, p => 1, |
| 3212 |
td => 1, th => 1, tr => 1, |
td => 1, th => 1, tr => 1, |
| 3213 |
|
tbody => 1, tfoot=> 1, thead => 1, |
| 3214 |
}->{$self->{open_elements}->[-1]->[1]}) { |
}->{$self->{open_elements}->[-1]->[1]}) { |
| 3215 |
!!!back-token; |
!!!back-token; |
| 3216 |
$token = {type => 'end tag', |
$token = {type => 'end tag', |
| 3241 |
strong => 1, tt => 1, u => 1, |
strong => 1, tt => 1, u => 1, |
| 3242 |
}->{$token->{tag_name}}) { |
}->{$token->{tag_name}}) { |
| 3243 |
$formatting_end_tag->($token->{tag_name}); |
$formatting_end_tag->($token->{tag_name}); |
| 3244 |
## TODO: <http://html5.org/tools/web-apps-tracker?from=883&to=884> |
return; |
| 3245 |
|
} elsif ($token->{tag_name} eq 'br') { |
| 3246 |
|
!!!parse-error (type => 'unmatched end tag:br'); |
| 3247 |
|
|
| 3248 |
|
## As if <br> |
| 3249 |
|
$reconstruct_active_formatting_elements->($insert_to_current); |
| 3250 |
|
|
| 3251 |
|
my $el; |
| 3252 |
|
!!!create-element ($el, 'br'); |
| 3253 |
|
$insert->($el); |
| 3254 |
|
|
| 3255 |
|
## Ignore the token. |
| 3256 |
|
!!!next-token; |
| 3257 |
return; |
return; |
| 3258 |
} elsif ({ |
} elsif ({ |
| 3259 |
caption => 1, col => 1, colgroup => 1, frame => 1, |
caption => 1, col => 1, colgroup => 1, frame => 1, |
| 3260 |
frameset => 1, head => 1, option => 1, optgroup => 1, |
frameset => 1, head => 1, option => 1, optgroup => 1, |
| 3261 |
tbody => 1, td => 1, tfoot => 1, th => 1, |
tbody => 1, td => 1, tfoot => 1, th => 1, |
| 3262 |
thead => 1, tr => 1, |
thead => 1, tr => 1, |
| 3263 |
area => 1, basefont => 1, bgsound => 1, br => 1, |
area => 1, basefont => 1, bgsound => 1, |
| 3264 |
embed => 1, hr => 1, iframe => 1, image => 1, |
embed => 1, hr => 1, iframe => 1, image => 1, |
| 3265 |
img => 1, input => 1, isindex => 1, noembed => 1, |
img => 1, input => 1, isindex => 1, noembed => 1, |
| 3266 |
noframes => 1, param => 1, select => 1, spacer => 1, |
noframes => 1, param => 1, select => 1, spacer => 1, |
| 3287 |
if ({ |
if ({ |
| 3288 |
dd => 1, dt => 1, li => 1, p => 1, |
dd => 1, dt => 1, li => 1, p => 1, |
| 3289 |
td => 1, th => 1, tr => 1, |
td => 1, th => 1, tr => 1, |
| 3290 |
|
tbody => 1, tfoot=> 1, thead => 1, |
| 3291 |
}->{$self->{open_elements}->[-1]->[1]}) { |
}->{$self->{open_elements}->[-1]->[1]}) { |
| 3292 |
!!!back-token; |
!!!back-token; |
| 3293 |
$token = {type => 'end tag', |
$token = {type => 'end tag', |
| 3331 |
}; # $in_body |
}; # $in_body |
| 3332 |
|
|
| 3333 |
B: { |
B: { |
| 3334 |
if ($phase eq 'main') { |
if ($token->{type} eq 'DOCTYPE') { |
| 3335 |
if ($token->{type} eq 'DOCTYPE') { |
!!!parse-error (type => 'DOCTYPE in the middle'); |
| 3336 |
!!!parse-error (type => 'in html:#DOCTYPE'); |
## Ignore the token |
| 3337 |
## Ignore the token |
## Stay in the phase |
| 3338 |
## Stay in the phase |
!!!next-token; |
| 3339 |
!!!next-token; |
redo B; |
| 3340 |
redo B; |
} elsif ($token->{type} eq 'end-of-file') { |
| 3341 |
} elsif ($token->{type} eq 'start tag' and |
if ($token->{insertion_mode} ne 'trailing end') { |
|
$token->{tag_name} eq 'html') { |
|
|
## ISSUE: "aa<html>" is not a parse error. |
|
|
## ISSUE: "<html>" in fragment is not a parse error. |
|
|
unless ($token->{first_start_tag}) { |
|
|
!!!parse-error (type => 'not first start tag'); |
|
|
} |
|
|
my $top_el = $self->{open_elements}->[0]->[0]; |
|
|
for my $attr_name (keys %{$token->{attributes}}) { |
|
|
unless ($top_el->has_attribute_ns (undef, $attr_name)) { |
|
|
$top_el->set_attribute_ns |
|
|
(undef, [undef, $attr_name], |
|
|
$token->{attributes}->{$attr_name}->{value}); |
|
|
} |
|
|
} |
|
|
!!!next-token; |
|
|
redo B; |
|
|
} elsif ($token->{type} eq 'end-of-file') { |
|
| 3342 |
## Generate implied end tags |
## Generate implied end tags |
| 3343 |
if ({ |
if ({ |
| 3344 |
dd => 1, dt => 1, li => 1, p => 1, td => 1, th => 1, tr => 1, |
dd => 1, dt => 1, li => 1, p => 1, td => 1, th => 1, tr => 1, |
| 3345 |
|
tbody => 1, tfoot=> 1, thead => 1, |
| 3346 |
}->{$self->{open_elements}->[-1]->[1]}) { |
}->{$self->{open_elements}->[-1]->[1]}) { |
| 3347 |
!!!back-token; |
!!!back-token; |
| 3348 |
$token = {type => 'end tag', tag_name => $self->{open_elements}->[-1]->[1]}; |
$token = {type => 'end tag', tag_name => $self->{open_elements}->[-1]->[1]}; |
| 3358 |
!!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]); |
!!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]); |
| 3359 |
} |
} |
| 3360 |
|
|
|
## Stop parsing |
|
|
last B; |
|
|
|
|
| 3361 |
## ISSUE: There is an issue in the spec. |
## ISSUE: There is an issue in the spec. |
| 3362 |
|
} |
| 3363 |
|
|
| 3364 |
|
## Stop parsing |
| 3365 |
|
last B; |
| 3366 |
|
} elsif ($token->{type} eq 'start tag' and |
| 3367 |
|
$token->{tag_name} eq 'html') { |
| 3368 |
|
if ($self->{insertion_mode} eq 'trailing end') { |
| 3369 |
|
## Turn into the main phase |
| 3370 |
|
!!!parse-error (type => 'after html:html'); |
| 3371 |
|
$self->{insertion_mode} = $previous_insertion_mode; |
| 3372 |
|
} |
| 3373 |
|
|
| 3374 |
|
## ISSUE: "aa<html>" is not a parse error. |
| 3375 |
|
## ISSUE: "<html>" in fragment is not a parse error. |
| 3376 |
|
unless ($token->{first_start_tag}) { |
| 3377 |
|
!!!parse-error (type => 'not first start tag'); |
| 3378 |
|
} |
| 3379 |
|
my $top_el = $self->{open_elements}->[0]->[0]; |
| 3380 |
|
for my $attr_name (keys %{$token->{attributes}}) { |
| 3381 |
|
unless ($top_el->has_attribute_ns (undef, $attr_name)) { |
| 3382 |
|
$top_el->set_attribute_ns |
| 3383 |
|
(undef, [undef, $attr_name], |
| 3384 |
|
$token->{attributes}->{$attr_name}->{value}); |
| 3385 |
|
} |
| 3386 |
|
} |
| 3387 |
|
!!!next-token; |
| 3388 |
|
redo B; |
| 3389 |
|
} elsif ($token->{type} eq 'comment') { |
| 3390 |
|
my $comment = $self->{document}->create_comment ($token->{data}); |
| 3391 |
|
if ($self->{insertion_mode} eq 'trailing end') { |
| 3392 |
|
$self->{document}->append_child ($comment); |
| 3393 |
|
} elsif ($self->{insertion_mode} eq 'after body') { |
| 3394 |
|
$self->{open_elements}->[0]->[0]->append_child ($comment); |
| 3395 |
} else { |
} else { |
| 3396 |
if ($self->{insertion_mode} eq 'before head') { |
$self->{open_elements}->[-1]->[0]->append_child ($comment); |
| 3397 |
|
} |
| 3398 |
|
!!!next-token; |
| 3399 |
|
redo B; |
| 3400 |
|
} elsif ($self->{insertion_mode} eq 'before head') { |
| 3401 |
if ($token->{type} eq 'character') { |
if ($token->{type} eq 'character') { |
| 3402 |
if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { |
if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { |
| 3403 |
$self->{open_elements}->[-1]->[0]->manakai_append_text ($1); |
$self->{open_elements}->[-1]->[0]->manakai_append_text ($1); |
| 3413 |
$self->{insertion_mode} = 'in head'; |
$self->{insertion_mode} = 'in head'; |
| 3414 |
## reprocess |
## reprocess |
| 3415 |
redo B; |
redo B; |
|
} elsif ($token->{type} eq 'comment') { |
|
|
my $comment = $self->{document}->create_comment ($token->{data}); |
|
|
$self->{open_elements}->[-1]->[0]->append_child ($comment); |
|
|
!!!next-token; |
|
|
redo B; |
|
| 3416 |
} elsif ($token->{type} eq 'start tag') { |
} elsif ($token->{type} eq 'start tag') { |
| 3417 |
my $attr = $token->{tag_name} eq 'head' ? $token->{attributes} : {}; |
my $attr = $token->{tag_name} eq 'head' ? $token->{attributes} : {}; |
| 3418 |
!!!create-element ($self->{head_element}, 'head', $attr); |
!!!create-element ($self->{head_element}, 'head', $attr); |
| 3431 |
} |
} |
| 3432 |
redo B; |
redo B; |
| 3433 |
} elsif ($token->{type} eq 'end tag') { |
} elsif ($token->{type} eq 'end tag') { |
| 3434 |
if ({head => 1, body => 1, html => 1}->{$token->{tag_name}}) { |
if ({ |
| 3435 |
|
head => 1, body => 1, html => 1, |
| 3436 |
|
p => 1, br => 1, |
| 3437 |
|
}->{$token->{tag_name}}) { |
| 3438 |
## As if <head> |
## As if <head> |
| 3439 |
!!!create-element ($self->{head_element}, 'head'); |
!!!create-element ($self->{head_element}, 'head'); |
| 3440 |
$self->{open_elements}->[-1]->[0]->append_child ($self->{head_element}); |
$self->{open_elements}->[-1]->[0]->append_child ($self->{head_element}); |
| 3464 |
} |
} |
| 3465 |
|
|
| 3466 |
# |
# |
|
} elsif ($token->{type} eq 'comment') { |
|
|
my $comment = $self->{document}->create_comment ($token->{data}); |
|
|
$self->{open_elements}->[-1]->[0]->append_child ($comment); |
|
|
!!!next-token; |
|
|
redo B; |
|
| 3467 |
} elsif ($token->{type} eq 'start tag') { |
} elsif ($token->{type} eq 'start tag') { |
| 3468 |
if ({base => ($self->{insertion_mode} eq 'in head' or |
if ({base => ($self->{insertion_mode} eq 'in head' or |
| 3469 |
$self->{insertion_mode} eq 'after head'), |
$self->{insertion_mode} eq 'after head'), |
| 3470 |
link => 1, meta => 1}->{$token->{tag_name}}) { |
link => 1}->{$token->{tag_name}}) { |
| 3471 |
## NOTE: There is a "as if in head" code clone. |
## NOTE: There is a "as if in head" code clone. |
| 3472 |
if ($self->{insertion_mode} eq 'after head') { |
if ($self->{insertion_mode} eq 'after head') { |
| 3473 |
!!!parse-error (type => 'after head:'.$token->{tag_name}); |
!!!parse-error (type => 'after head:'.$token->{tag_name}); |
| 3475 |
} |
} |
| 3476 |
!!!insert-element ($token->{tag_name}, $token->{attributes}); |
!!!insert-element ($token->{tag_name}, $token->{attributes}); |
| 3477 |
pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec. |
pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec. |
| 3478 |
|
pop @{$self->{open_elements}} |
| 3479 |
|
if $self->{insertion_mode} eq 'after head'; |
| 3480 |
|
!!!next-token; |
| 3481 |
|
redo B; |
| 3482 |
|
} elsif ($token->{tag_name} eq 'meta') { |
| 3483 |
|
## NOTE: There is a "as if in head" code clone. |
| 3484 |
|
if ($self->{insertion_mode} eq 'after head') { |
| 3485 |
|
!!!parse-error (type => 'after head:'.$token->{tag_name}); |
| 3486 |
|
push @{$self->{open_elements}}, [$self->{head_element}, 'head']; |
| 3487 |
|
} |
| 3488 |
|
!!!insert-element ($token->{tag_name}, $token->{attributes}); |
| 3489 |
|
pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec. |
| 3490 |
|
|
| 3491 |
|
unless ($self->{confident}) { |
| 3492 |
|
my $charset; |
| 3493 |
|
if ($token->{attributes}->{charset}) { ## TODO: And if supported |
| 3494 |
|
$charset = $token->{attributes}->{charset}->{value}; |
| 3495 |
|
} |
| 3496 |
|
if ($token->{attributes}->{'http-equiv'}) { |
| 3497 |
|
## ISSUE: Algorithm name in the spec was incorrect so that not linked to the definition. |
| 3498 |
|
if ($token->{attributes}->{'http-equiv'}->{value} |
| 3499 |
|
=~ /\A[^;]*;[\x09-\x0D\x20]*charset[\x09-\x0D\x20]*= |
| 3500 |
|
[\x09-\x0D\x20]*(?>"([^"]*)"|'([^']*)'| |
| 3501 |
|
([^"'\x09-\x0D\x20][^\x09-\x0D\x20]*))/x) { |
| 3502 |
|
$charset = defined $1 ? $1 : defined $2 ? $2 : $3; |
| 3503 |
|
} ## TODO: And if supported |
| 3504 |
|
} |
| 3505 |
|
## TODO: Change the encoding |
| 3506 |
|
} |
| 3507 |
|
|
| 3508 |
## TODO: Extracting |charset| from |meta|. |
## TODO: Extracting |charset| from |meta|. |
| 3509 |
pop @{$self->{open_elements}} |
pop @{$self->{open_elements}} |
| 3510 |
if $self->{insertion_mode} eq 'after head'; |
if $self->{insertion_mode} eq 'after head'; |
| 3517 |
!!!parse-error (type => 'after head:'.$token->{tag_name}); |
!!!parse-error (type => 'after head:'.$token->{tag_name}); |
| 3518 |
push @{$self->{open_elements}}, [$self->{head_element}, 'head']; |
push @{$self->{open_elements}}, [$self->{head_element}, 'head']; |
| 3519 |
} |
} |
| 3520 |
$parse_rcdata->('RCDATA', $insert_to_current); |
my $parent = defined $self->{head_element} ? $self->{head_element} |
| 3521 |
|
: $self->{open_elements}->[-1]->[0]; |
| 3522 |
|
$parse_rcdata->('RCDATA', sub { $parent->append_child ($_[0]) }); |
| 3523 |
pop @{$self->{open_elements}} |
pop @{$self->{open_elements}} |
| 3524 |
if $self->{insertion_mode} eq 'after head'; |
if $self->{insertion_mode} eq 'after head'; |
| 3525 |
redo B; |
redo B; |
| 3543 |
!!!next-token; |
!!!next-token; |
| 3544 |
redo B; |
redo B; |
| 3545 |
} elsif ($self->{insertion_mode} eq 'in head noscript') { |
} elsif ($self->{insertion_mode} eq 'in head noscript') { |
| 3546 |
!!!parse-error (type => 'noscript in noscript'); |
!!!parse-error (type => 'in noscript:noscript'); |
| 3547 |
## Ignore the token |
## Ignore the token |
| 3548 |
redo B; |
redo B; |
| 3549 |
} else { |
} else { |
| 3595 |
!!!next-token; |
!!!next-token; |
| 3596 |
redo B; |
redo B; |
| 3597 |
} elsif ($self->{insertion_mode} eq 'in head' and |
} elsif ($self->{insertion_mode} eq 'in head' and |
| 3598 |
($token->{tag_name} eq 'body' or |
{ |
| 3599 |
$token->{tag_name} eq 'html')) { |
body => 1, html => 1, |
| 3600 |
|
p => 1, br => 1, |
| 3601 |
|
}->{$token->{tag_name}}) { |
| 3602 |
|
# |
| 3603 |
|
} elsif ($self->{insertion_mode} eq 'in head noscript' and |
| 3604 |
|
{ |
| 3605 |
|
p => 1, br => 1, |
| 3606 |
|
}->{$token->{tag_name}}) { |
| 3607 |
# |
# |
| 3608 |
} elsif ($self->{insertion_mode} ne 'after head') { |
} elsif ($self->{insertion_mode} ne 'after head') { |
| 3609 |
!!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}); |
!!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}); |
| 3642 |
|
|
| 3643 |
!!!next-token; |
!!!next-token; |
| 3644 |
redo B; |
redo B; |
|
} elsif ($token->{type} eq 'comment') { |
|
|
## NOTE: There is a code clone of "comment in body". |
|
|
my $comment = $self->{document}->create_comment ($token->{data}); |
|
|
$self->{open_elements}->[-1]->[0]->append_child ($comment); |
|
|
!!!next-token; |
|
|
redo B; |
|
| 3645 |
} else { |
} else { |
| 3646 |
$in_body->($insert_to_current); |
$in_body->($insert_to_current); |
| 3647 |
redo B; |
redo B; |
| 3705 |
|
|
| 3706 |
!!!next-token; |
!!!next-token; |
| 3707 |
redo B; |
redo B; |
|
} elsif ($token->{type} eq 'comment') { |
|
|
my $comment = $self->{document}->create_comment ($token->{data}); |
|
|
$self->{open_elements}->[-1]->[0]->append_child ($comment); |
|
|
!!!next-token; |
|
|
redo B; |
|
| 3708 |
} elsif ($token->{type} eq 'start tag') { |
} elsif ($token->{type} eq 'start tag') { |
| 3709 |
if ({ |
if ({ |
| 3710 |
caption => 1, |
caption => 1, |
| 3776 |
if ({ |
if ({ |
| 3777 |
dd => 1, dt => 1, li => 1, p => 1, |
dd => 1, dt => 1, li => 1, p => 1, |
| 3778 |
td => 1, th => 1, tr => 1, |
td => 1, th => 1, tr => 1, |
| 3779 |
|
tbody => 1, tfoot=> 1, thead => 1, |
| 3780 |
}->{$self->{open_elements}->[-1]->[1]}) { |
}->{$self->{open_elements}->[-1]->[1]}) { |
| 3781 |
!!!back-token; # <table> |
!!!back-token; # <table> |
| 3782 |
$token = {type => 'end tag', tag_name => 'table'}; |
$token = {type => 'end tag', tag_name => 'table'}; |
| 3825 |
if ({ |
if ({ |
| 3826 |
dd => 1, dt => 1, li => 1, p => 1, |
dd => 1, dt => 1, li => 1, p => 1, |
| 3827 |
td => 1, th => 1, tr => 1, |
td => 1, th => 1, tr => 1, |
| 3828 |
|
tbody => 1, tfoot=> 1, thead => 1, |
| 3829 |
}->{$self->{open_elements}->[-1]->[1]}) { |
}->{$self->{open_elements}->[-1]->[1]}) { |
| 3830 |
!!!back-token; |
!!!back-token; |
| 3831 |
$token = {type => 'end tag', |
$token = {type => 'end tag', |
| 3871 |
|
|
| 3872 |
!!!next-token; |
!!!next-token; |
| 3873 |
redo B; |
redo B; |
|
} elsif ($token->{type} eq 'comment') { |
|
|
## NOTE: This is a code clone of "comment in body". |
|
|
my $comment = $self->{document}->create_comment ($token->{data}); |
|
|
$self->{open_elements}->[-1]->[0]->append_child ($comment); |
|
|
!!!next-token; |
|
|
redo B; |
|
| 3874 |
} elsif ($token->{type} eq 'start tag') { |
} elsif ($token->{type} eq 'start tag') { |
| 3875 |
if ({ |
if ({ |
| 3876 |
caption => 1, col => 1, colgroup => 1, tbody => 1, |
caption => 1, col => 1, colgroup => 1, tbody => 1, |
| 3903 |
if ({ |
if ({ |
| 3904 |
dd => 1, dt => 1, li => 1, p => 1, |
dd => 1, dt => 1, li => 1, p => 1, |
| 3905 |
td => 1, th => 1, tr => 1, |
td => 1, th => 1, tr => 1, |
| 3906 |
|
tbody => 1, tfoot=> 1, thead => 1, |
| 3907 |
}->{$self->{open_elements}->[-1]->[1]}) { |
}->{$self->{open_elements}->[-1]->[1]}) { |
| 3908 |
!!!back-token; # <?> |
!!!back-token; # <?> |
| 3909 |
$token = {type => 'end tag', tag_name => 'caption'}; |
$token = {type => 'end tag', tag_name => 'caption'}; |
| 3954 |
if ({ |
if ({ |
| 3955 |
dd => 1, dt => 1, li => 1, p => 1, |
dd => 1, dt => 1, li => 1, p => 1, |
| 3956 |
td => 1, th => 1, tr => 1, |
td => 1, th => 1, tr => 1, |
| 3957 |
|
tbody => 1, tfoot=> 1, thead => 1, |
| 3958 |
}->{$self->{open_elements}->[-1]->[1]}) { |
}->{$self->{open_elements}->[-1]->[1]}) { |
| 3959 |
!!!back-token; |
!!!back-token; |
| 3960 |
$token = {type => 'end tag', |
$token = {type => 'end tag', |
| 4002 |
if ({ |
if ({ |
| 4003 |
dd => 1, dt => 1, li => 1, p => 1, |
dd => 1, dt => 1, li => 1, p => 1, |
| 4004 |
td => 1, th => 1, tr => 1, |
td => 1, th => 1, tr => 1, |
| 4005 |
|
tbody => 1, tfoot=> 1, thead => 1, |
| 4006 |
}->{$self->{open_elements}->[-1]->[1]}) { |
}->{$self->{open_elements}->[-1]->[1]}) { |
| 4007 |
!!!back-token; # </table> |
!!!back-token; # </table> |
| 4008 |
$token = {type => 'end tag', tag_name => 'caption'}; |
$token = {type => 'end tag', tag_name => 'caption'}; |
| 4052 |
} |
} |
| 4053 |
|
|
| 4054 |
# |
# |
|
} elsif ($token->{type} eq 'comment') { |
|
|
my $comment = $self->{document}->create_comment ($token->{data}); |
|
|
$self->{open_elements}->[-1]->[0]->append_child ($comment); |
|
|
!!!next-token; |
|
|
redo B; |
|
| 4055 |
} elsif ($token->{type} eq 'start tag') { |
} elsif ($token->{type} eq 'start tag') { |
| 4056 |
if ($token->{tag_name} eq 'col') { |
if ($token->{tag_name} eq 'col') { |
| 4057 |
!!!insert-element ($token->{tag_name}, $token->{attributes}); |
!!!insert-element ($token->{tag_name}, $token->{attributes}); |
| 4157 |
|
|
| 4158 |
!!!next-token; |
!!!next-token; |
| 4159 |
redo B; |
redo B; |
|
} elsif ($token->{type} eq 'comment') { |
|
|
## Copied from 'in table' |
|
|
my $comment = $self->{document}->create_comment ($token->{data}); |
|
|
$self->{open_elements}->[-1]->[0]->append_child ($comment); |
|
|
!!!next-token; |
|
|
redo B; |
|
| 4160 |
} elsif ($token->{type} eq 'start tag') { |
} elsif ($token->{type} eq 'start tag') { |
| 4161 |
if ({ |
if ({ |
| 4162 |
tr => 1, |
tr => 1, |
| 4257 |
if ({ |
if ({ |
| 4258 |
dd => 1, dt => 1, li => 1, p => 1, |
dd => 1, dt => 1, li => 1, p => 1, |
| 4259 |
td => 1, th => 1, tr => 1, |
td => 1, th => 1, tr => 1, |
| 4260 |
|
tbody => 1, tfoot=> 1, thead => 1, |
| 4261 |
}->{$self->{open_elements}->[-1]->[1]}) { |
}->{$self->{open_elements}->[-1]->[1]}) { |
| 4262 |
!!!back-token; # <table> |
!!!back-token; # <table> |
| 4263 |
$token = {type => 'end tag', tag_name => 'table'}; |
$token = {type => 'end tag', tag_name => 'table'}; |
| 4436 |
|
|
| 4437 |
!!!next-token; |
!!!next-token; |
| 4438 |
redo B; |
redo B; |
|
} elsif ($token->{type} eq 'comment') { |
|
|
## Copied from 'in table' |
|
|
my $comment = $self->{document}->create_comment ($token->{data}); |
|
|
$self->{open_elements}->[-1]->[0]->append_child ($comment); |
|
|
!!!next-token; |
|
|
redo B; |
|
| 4439 |
} elsif ($token->{type} eq 'start tag') { |
} elsif ($token->{type} eq 'start tag') { |
| 4440 |
if ($token->{tag_name} eq 'th' or |
if ($token->{tag_name} eq 'th' or |
| 4441 |
$token->{tag_name} eq 'td') { |
$token->{tag_name} eq 'td') { |
| 4520 |
if ({ |
if ({ |
| 4521 |
dd => 1, dt => 1, li => 1, p => 1, |
dd => 1, dt => 1, li => 1, p => 1, |
| 4522 |
td => 1, th => 1, tr => 1, |
td => 1, th => 1, tr => 1, |
| 4523 |
|
tbody => 1, tfoot=> 1, thead => 1, |
| 4524 |
}->{$self->{open_elements}->[-1]->[1]}) { |
}->{$self->{open_elements}->[-1]->[1]}) { |
| 4525 |
!!!back-token; # <table> |
!!!back-token; # <table> |
| 4526 |
$token = {type => 'end tag', tag_name => 'table'}; |
$token = {type => 'end tag', tag_name => 'table'}; |
| 4695 |
|
|
| 4696 |
!!!next-token; |
!!!next-token; |
| 4697 |
redo B; |
redo B; |
|
} elsif ($token->{type} eq 'comment') { |
|
|
## NOTE: This is a code clone of "comment in body". |
|
|
my $comment = $self->{document}->create_comment ($token->{data}); |
|
|
$self->{open_elements}->[-1]->[0]->append_child ($comment); |
|
|
!!!next-token; |
|
|
redo B; |
|
| 4698 |
} elsif ($token->{type} eq 'start tag') { |
} elsif ($token->{type} eq 'start tag') { |
| 4699 |
if ({ |
if ({ |
| 4700 |
caption => 1, col => 1, colgroup => 1, |
caption => 1, col => 1, colgroup => 1, |
| 4756 |
td => ($token->{tag_name} eq 'th'), |
td => ($token->{tag_name} eq 'th'), |
| 4757 |
th => ($token->{tag_name} eq 'td'), |
th => ($token->{tag_name} eq 'td'), |
| 4758 |
tr => 1, |
tr => 1, |
| 4759 |
|
tbody => 1, tfoot=> 1, thead => 1, |
| 4760 |
}->{$self->{open_elements}->[-1]->[1]}) { |
}->{$self->{open_elements}->[-1]->[1]}) { |
| 4761 |
!!!back-token; |
!!!back-token; |
| 4762 |
$token = {type => 'end tag', |
$token = {type => 'end tag', |
| 4831 |
$self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data}); |
$self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data}); |
| 4832 |
!!!next-token; |
!!!next-token; |
| 4833 |
redo B; |
redo B; |
|
} elsif ($token->{type} eq 'comment') { |
|
|
my $comment = $self->{document}->create_comment ($token->{data}); |
|
|
$self->{open_elements}->[-1]->[0]->append_child ($comment); |
|
|
!!!next-token; |
|
|
redo B; |
|
| 4834 |
} elsif ($token->{type} eq 'start tag') { |
} elsif ($token->{type} eq 'start tag') { |
| 4835 |
if ($token->{tag_name} eq 'option') { |
if ($token->{tag_name} eq 'option') { |
| 4836 |
if ($self->{open_elements}->[-1]->[1] eq 'option') { |
if ($self->{open_elements}->[-1]->[1] eq 'option') { |
| 5003 |
} elsif ($self->{insertion_mode} eq 'after body') { |
} elsif ($self->{insertion_mode} eq 'after body') { |
| 5004 |
if ($token->{type} eq 'character') { |
if ($token->{type} eq 'character') { |
| 5005 |
if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { |
if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { |
| 5006 |
|
my $data = $1; |
| 5007 |
## As if in body |
## As if in body |
| 5008 |
$reconstruct_active_formatting_elements->($insert_to_current); |
$reconstruct_active_formatting_elements->($insert_to_current); |
| 5009 |
|
|
| 5010 |
$self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data}); |
$self->{open_elements}->[-1]->[0]->manakai_append_text ($1); |
| 5011 |
|
|
| 5012 |
unless (length $token->{data}) { |
unless (length $token->{data}) { |
| 5013 |
!!!next-token; |
!!!next-token; |
| 5016 |
} |
} |
| 5017 |
|
|
| 5018 |
# |
# |
| 5019 |
!!!parse-error (type => 'after body:#'.$token->{type}); |
!!!parse-error (type => 'after body:#character'); |
|
} elsif ($token->{type} eq 'comment') { |
|
|
my $comment = $self->{document}->create_comment ($token->{data}); |
|
|
$self->{open_elements}->[0]->[0]->append_child ($comment); |
|
|
!!!next-token; |
|
|
redo B; |
|
| 5020 |
} elsif ($token->{type} eq 'start tag') { |
} elsif ($token->{type} eq 'start tag') { |
| 5021 |
!!!parse-error (type => 'after body:'.$token->{tag_name}); |
!!!parse-error (type => 'after body:'.$token->{tag_name}); |
| 5022 |
# |
# |
| 5028 |
!!!next-token; |
!!!next-token; |
| 5029 |
redo B; |
redo B; |
| 5030 |
} else { |
} else { |
| 5031 |
$phase = 'trailing end'; |
$previous_insertion_mode = $self->{insertion_mode}; |
| 5032 |
|
$self->{insertion_mode} = 'trailing end'; |
| 5033 |
!!!next-token; |
!!!next-token; |
| 5034 |
redo B; |
redo B; |
| 5035 |
} |
} |
| 5037 |
!!!parse-error (type => 'after body:/'.$token->{tag_name}); |
!!!parse-error (type => 'after body:/'.$token->{tag_name}); |
| 5038 |
} |
} |
| 5039 |
} else { |
} else { |
| 5040 |
!!!parse-error (type => 'after body:#'.$token->{type}); |
die "$0: $token->{type}: Unknown token type"; |
| 5041 |
} |
} |
| 5042 |
|
|
| 5043 |
$self->{insertion_mode} = 'in body'; |
$self->{insertion_mode} = 'in body'; |
| 5044 |
## reprocess |
## reprocess |
| 5045 |
redo B; |
redo B; |
| 5046 |
} elsif ($self->{insertion_mode} eq 'in frameset') { |
} elsif ($self->{insertion_mode} eq 'in frameset') { |
| 5047 |
if ($token->{type} eq 'character') { |
if ($token->{type} eq 'character') { |
| 5048 |
if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { |
if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { |
| 5049 |
$self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data}); |
$self->{open_elements}->[-1]->[0]->manakai_append_text ($1); |
|
|
|
|
unless (length $token->{data}) { |
|
|
!!!next-token; |
|
|
redo B; |
|
|
} |
|
|
} |
|
| 5050 |
|
|
| 5051 |
# |
unless (length $token->{data}) { |
|
} elsif ($token->{type} eq 'comment') { |
|
|
my $comment = $self->{document}->create_comment ($token->{data}); |
|
|
$self->{open_elements}->[-1]->[0]->append_child ($comment); |
|
| 5052 |
!!!next-token; |
!!!next-token; |
| 5053 |
redo B; |
redo B; |
|
} elsif ($token->{type} eq 'start tag') { |
|
|
if ($token->{tag_name} eq 'frameset') { |
|
|
!!!insert-element ($token->{tag_name}, $token->{attributes}); |
|
|
!!!next-token; |
|
|
redo B; |
|
|
} elsif ($token->{tag_name} eq 'frame') { |
|
|
!!!insert-element ($token->{tag_name}, $token->{attributes}); |
|
|
pop @{$self->{open_elements}}; |
|
|
!!!next-token; |
|
|
redo B; |
|
|
} elsif ($token->{tag_name} eq 'noframes') { |
|
|
$in_body->($insert_to_current); |
|
|
redo B; |
|
|
} else { |
|
|
# |
|
|
} |
|
|
} elsif ($token->{type} eq 'end tag') { |
|
|
if ($token->{tag_name} eq 'frameset') { |
|
|
if ($self->{open_elements}->[-1]->[1] eq 'html' and |
|
|
@{$self->{open_elements}} == 1) { |
|
|
!!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}); |
|
|
## Ignore the token |
|
|
!!!next-token; |
|
|
} else { |
|
|
pop @{$self->{open_elements}}; |
|
|
!!!next-token; |
|
|
} |
|
|
|
|
|
## if not inner_html and |
|
|
if ($self->{open_elements}->[-1]->[1] ne 'frameset') { |
|
|
$self->{insertion_mode} = 'after frameset'; |
|
|
} |
|
|
redo B; |
|
|
} else { |
|
|
# |
|
|
} |
|
|
} else { |
|
|
# |
|
| 5054 |
} |
} |
| 5055 |
|
} |
| 5056 |
if (defined $token->{tag_name}) { |
|
| 5057 |
!!!parse-error (type => 'in frameset:'.$token->{tag_name}); |
!!!parse-error (type => 'in frameset:#character'); |
| 5058 |
|
## Ignore the token |
| 5059 |
|
!!!next-token; |
| 5060 |
|
redo B; |
| 5061 |
|
} elsif ($token->{type} eq 'start tag') { |
| 5062 |
|
if ($token->{tag_name} eq 'frameset') { |
| 5063 |
|
!!!insert-element ($token->{tag_name}, $token->{attributes}); |
| 5064 |
|
!!!next-token; |
| 5065 |
|
redo B; |
| 5066 |
|
} elsif ($token->{tag_name} eq 'frame') { |
| 5067 |
|
!!!insert-element ($token->{tag_name}, $token->{attributes}); |
| 5068 |
|
pop @{$self->{open_elements}}; |
| 5069 |
|
!!!next-token; |
| 5070 |
|
redo B; |
| 5071 |
|
} elsif ($token->{tag_name} eq 'noframes') { |
| 5072 |
|
$in_body->($insert_to_current); |
| 5073 |
|
redo B; |
| 5074 |
|
} else { |
| 5075 |
|
!!!parse-error (type => 'in frameset:'.$token->{tag_name}); |
| 5076 |
|
## Ignore the token |
| 5077 |
|
!!!next-token; |
| 5078 |
|
redo B; |
| 5079 |
|
} |
| 5080 |
|
} elsif ($token->{type} eq 'end tag') { |
| 5081 |
|
if ($token->{tag_name} eq 'frameset') { |
| 5082 |
|
if ($self->{open_elements}->[-1]->[1] eq 'html' and |
| 5083 |
|
@{$self->{open_elements}} == 1) { |
| 5084 |
|
!!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}); |
| 5085 |
|
## Ignore the token |
| 5086 |
|
!!!next-token; |
| 5087 |
} else { |
} else { |
| 5088 |
!!!parse-error (type => 'in frameset:#'.$token->{type}); |
pop @{$self->{open_elements}}; |
| 5089 |
|
!!!next-token; |
| 5090 |
|
} |
| 5091 |
|
|
| 5092 |
|
if (not defined $self->{inner_html_node} and |
| 5093 |
|
$self->{open_elements}->[-1]->[1] ne 'frameset') { |
| 5094 |
|
$self->{insertion_mode} = 'after frameset'; |
| 5095 |
} |
} |
| 5096 |
|
redo B; |
| 5097 |
|
} else { |
| 5098 |
|
!!!parse-error (type => 'in frameset:/'.$token->{tag_name}); |
| 5099 |
## Ignore the token |
## Ignore the token |
| 5100 |
!!!next-token; |
!!!next-token; |
| 5101 |
redo B; |
redo B; |
| 5102 |
} elsif ($self->{insertion_mode} eq 'after frameset') { |
} |
| 5103 |
if ($token->{type} eq 'character') { |
} else { |
| 5104 |
|
die "$0: $token->{type}: Unknown token type"; |
| 5105 |
|
} |
| 5106 |
|
} elsif ($self->{insertion_mode} eq 'after frameset') { |
| 5107 |
|
if ($token->{type} eq 'character') { |
| 5108 |
if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { |
if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { |
| 5109 |
$self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data}); |
$self->{open_elements}->[-1]->[0]->manakai_append_text ($1); |
| 5110 |
|
|
| 5111 |
unless (length $token->{data}) { |
unless (length $token->{data}) { |
| 5112 |
!!!next-token; |
!!!next-token; |
| 5114 |
} |
} |
| 5115 |
} |
} |
| 5116 |
|
|
| 5117 |
# |
if ($token->{data} =~ s/^[^\x09\x0A\x0B\x0C\x20]+//) { |
| 5118 |
} elsif ($token->{type} eq 'comment') { |
!!!parse-error (type => 'after frameset:#character'); |
| 5119 |
my $comment = $self->{document}->create_comment ($token->{data}); |
|
| 5120 |
$self->{open_elements}->[-1]->[0]->append_child ($comment); |
## Ignore the token. |
| 5121 |
!!!next-token; |
if (length $token->{data}) { |
| 5122 |
redo B; |
## reprocess the rest of characters |
| 5123 |
} elsif ($token->{type} eq 'start tag') { |
} else { |
| 5124 |
if ($token->{tag_name} eq 'noframes') { |
!!!next-token; |
| 5125 |
$in_body->($insert_to_current); |
} |
|
redo B; |
|
|
} else { |
|
|
# |
|
|
} |
|
|
} elsif ($token->{type} eq 'end tag') { |
|
|
if ($token->{tag_name} eq 'html') { |
|
|
$phase = 'trailing end'; |
|
|
!!!next-token; |
|
| 5126 |
redo B; |
redo B; |
|
} else { |
|
|
# |
|
| 5127 |
} |
} |
| 5128 |
} else { |
|
| 5129 |
# |
die qq[$0: Character "$token->{data}"]; |
| 5130 |
} |
} elsif ($token->{type} eq 'start tag') { |
| 5131 |
|
if ($token->{tag_name} eq 'noframes') { |
| 5132 |
if (defined $token->{tag_name}) { |
$in_body->($insert_to_current); |
| 5133 |
!!!parse-error (type => 'after frameset:'.$token->{tag_name}); |
redo B; |
| 5134 |
} else { |
} else { |
| 5135 |
!!!parse-error (type => 'after frameset:#'.$token->{type}); |
!!!parse-error (type => 'after frameset:'.$token->{tag_name}); |
|
} |
|
| 5136 |
## Ignore the token |
## Ignore the token |
| 5137 |
!!!next-token; |
!!!next-token; |
| 5138 |
redo B; |
redo B; |
| 5139 |
|
} |
| 5140 |
## ISSUE: An issue in spec there |
} elsif ($token->{type} eq 'end tag') { |
| 5141 |
|
if ($token->{tag_name} eq 'html') { |
| 5142 |
|
$previous_insertion_mode = $self->{insertion_mode}; |
| 5143 |
|
$self->{insertion_mode} = 'trailing end'; |
| 5144 |
|
!!!next-token; |
| 5145 |
|
redo B; |
| 5146 |
} else { |
} else { |
| 5147 |
die "$0: $self->{insertion_mode}: Unknown insertion mode"; |
!!!parse-error (type => 'after frameset:/'.$token->{tag_name}); |
| 5148 |
|
## Ignore the token |
| 5149 |
|
!!!next-token; |
| 5150 |
|
redo B; |
| 5151 |
} |
} |
| 5152 |
|
} else { |
| 5153 |
|
die "$0: $token->{type}: Unknown token type"; |
| 5154 |
} |
} |
| 5155 |
} elsif ($phase eq 'trailing end') { |
|
| 5156 |
|
## ISSUE: An issue in spec here |
| 5157 |
|
} elsif ($self->{insertion_mode} eq 'trailing end') { |
| 5158 |
## states in the main stage is preserved yet # MUST |
## states in the main stage is preserved yet # MUST |
| 5159 |
|
|
| 5160 |
if ($token->{type} eq 'DOCTYPE') { |
if ($token->{type} eq 'character') { |
|
!!!parse-error (type => 'after html:#DOCTYPE'); |
|
|
## Ignore the token |
|
|
!!!next-token; |
|
|
redo B; |
|
|
} elsif ($token->{type} eq 'comment') { |
|
|
my $comment = $self->{document}->create_comment ($token->{data}); |
|
|
$self->{document}->append_child ($comment); |
|
|
!!!next-token; |
|
|
redo B; |
|
|
} elsif ($token->{type} eq 'character') { |
|
| 5161 |
if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { |
if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { |
| 5162 |
my $data = $1; |
my $data = $1; |
| 5163 |
## As if in the main phase. |
## As if in the main phase. |
| 5164 |
## NOTE: The insertion mode in the main phase |
## NOTE: The insertion mode in the main phase |
| 5165 |
## just before the phase has been changed to the trailing |
## just before the phase has been changed to the trailing |
| 5166 |
## end phase is either "after body" or "after frameset". |
## end phase is either "after body" or "after frameset". |
| 5167 |
$reconstruct_active_formatting_elements->($insert_to_current) |
$reconstruct_active_formatting_elements->($insert_to_current); |
|
if $phase eq 'main'; |
|
| 5168 |
|
|
| 5169 |
$self->{open_elements}->[-1]->[0]->manakai_append_text ($data); |
$self->{open_elements}->[-1]->[0]->manakai_append_text ($data); |
| 5170 |
|
|
| 5175 |
} |
} |
| 5176 |
|
|
| 5177 |
!!!parse-error (type => 'after html:#character'); |
!!!parse-error (type => 'after html:#character'); |
| 5178 |
$phase = 'main'; |
$self->{insertion_mode} = $previous_insertion_mode; |
| 5179 |
## reprocess |
## reprocess |
| 5180 |
redo B; |
redo B; |
| 5181 |
} elsif ($token->{type} eq 'start tag' or |
} elsif ($token->{type} eq 'start tag') { |
|
$token->{type} eq 'end tag') { |
|
| 5182 |
!!!parse-error (type => 'after html:'.$token->{tag_name}); |
!!!parse-error (type => 'after html:'.$token->{tag_name}); |
| 5183 |
$phase = 'main'; |
$self->{insertion_mode} = $previous_insertion_mode; |
| 5184 |
|
## reprocess |
| 5185 |
|
redo B; |
| 5186 |
|
} elsif ($token->{type} eq 'end tag') { |
| 5187 |
|
!!!parse-error (type => 'after html:/'.$token->{tag_name}); |
| 5188 |
|
$self->{insertion_mode} = $previous_insertion_mode; |
| 5189 |
## reprocess |
## reprocess |
| 5190 |
redo B; |
redo B; |
|
} elsif ($token->{type} eq 'end-of-file') { |
|
|
## Stop parsing |
|
|
last B; |
|
| 5191 |
} else { |
} else { |
| 5192 |
die "$0: $token->{type}: Unknown token"; |
die "$0: $token->{type}: Unknown token"; |
| 5193 |
} |
} |
| 5194 |
|
} else { |
| 5195 |
|
die "$0: $self->{insertion_mode}: Unknown insertion mode"; |
| 5196 |
} |
} |
| 5197 |
} # B |
} # B |
| 5198 |
|
|