| 2018 |
my $root_element; !!!create-element ($root_element, 'html'); |
my $root_element; !!!create-element ($root_element, 'html'); |
| 2019 |
$self->{document}->append_child ($root_element); |
$self->{document}->append_child ($root_element); |
| 2020 |
push @{$self->{open_elements}}, [$root_element, 'html']; |
push @{$self->{open_elements}}, [$root_element, 'html']; |
|
#$phase = 'main'; |
|
| 2021 |
## reprocess |
## reprocess |
| 2022 |
#redo B; |
#redo B; |
| 2023 |
return; |
return; ## Go to the main phase. |
| 2024 |
} # B |
} # B |
| 2025 |
} # _tree_construction_root_element |
} # _tree_construction_root_element |
| 2026 |
|
|
| 2096 |
sub _tree_construction_main ($) { |
sub _tree_construction_main ($) { |
| 2097 |
my $self = shift; |
my $self = shift; |
| 2098 |
|
|
| 2099 |
my $phase = 'main'; |
my $previous_insertion_mode; |
| 2100 |
|
|
| 2101 |
my $active_formatting_elements = []; |
my $active_formatting_elements = []; |
| 2102 |
|
|
| 2497 |
$parse_rcdata->('CDATA', $insert); |
$parse_rcdata->('CDATA', $insert); |
| 2498 |
return; |
return; |
| 2499 |
} elsif ({ |
} elsif ({ |
| 2500 |
base => 1, link => 1, meta => 1, |
base => 1, link => 1, |
| 2501 |
}->{$token->{tag_name}}) { |
}->{$token->{tag_name}}) { |
| 2502 |
## NOTE: This is an "as if in head" code clone, only "-t" differs |
## NOTE: This is an "as if in head" code clone, only "-t" differs |
| 2503 |
!!!insert-element-t ($token->{tag_name}, $token->{attributes}); |
!!!insert-element-t ($token->{tag_name}, $token->{attributes}); |
| 2504 |
pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec. |
pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec. |
| 2505 |
!!!next-token; |
!!!next-token; |
| 2506 |
## TODO: Extracting |charset| from |meta|. |
return; |
| 2507 |
|
} elsif ($token->{tag_name} eq 'meta') { |
| 2508 |
|
## NOTE: This is an "as if in head" code clone, only "-t" differs |
| 2509 |
|
!!!insert-element-t ($token->{tag_name}, $token->{attributes}); |
| 2510 |
|
pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec. |
| 2511 |
|
|
| 2512 |
|
unless ($self->{confident}) { |
| 2513 |
|
my $charset; |
| 2514 |
|
if ($token->{attributes}->{charset}) { ## TODO: And if supported |
| 2515 |
|
$charset = $token->{attributes}->{charset}->{value}; |
| 2516 |
|
} |
| 2517 |
|
if ($token->{attributes}->{'http-equiv'}) { |
| 2518 |
|
## ISSUE: Algorithm name in the spec was incorrect so that not linked to the definition. |
| 2519 |
|
if ($token->{attributes}->{'http-equiv'}->{value} |
| 2520 |
|
=~ /\A[^;]*;[\x09-\x0D\x20]*charset[\x09-\x0D\x20]*= |
| 2521 |
|
[\x09-\x0D\x20]*(?>"([^"]*)"|'([^']*)'| |
| 2522 |
|
([^"'\x09-\x0D\x20][^\x09-\x0D\x20]*))/x) { |
| 2523 |
|
$charset = defined $1 ? $1 : defined $2 ? $2 : $3; |
| 2524 |
|
} ## TODO: And if supported |
| 2525 |
|
} |
| 2526 |
|
## TODO: Change the encoding |
| 2527 |
|
} |
| 2528 |
|
|
| 2529 |
|
!!!next-token; |
| 2530 |
return; |
return; |
| 2531 |
} elsif ($token->{tag_name} eq 'title') { |
} elsif ($token->{tag_name} eq 'title') { |
| 2532 |
!!!parse-error (type => 'in body:title'); |
!!!parse-error (type => 'in body:title'); |
| 3331 |
}; # $in_body |
}; # $in_body |
| 3332 |
|
|
| 3333 |
B: { |
B: { |
| 3334 |
if ($phase eq 'main') { |
if ($token->{type} eq 'DOCTYPE') { |
| 3335 |
if ($token->{type} eq 'DOCTYPE') { |
!!!parse-error (type => 'DOCTYPE in the middle'); |
| 3336 |
!!!parse-error (type => 'in html:#DOCTYPE'); |
## Ignore the token |
| 3337 |
## Ignore the token |
## Stay in the phase |
| 3338 |
## Stay in the phase |
!!!next-token; |
| 3339 |
!!!next-token; |
redo B; |
| 3340 |
redo B; |
} elsif ($token->{type} eq 'end-of-file') { |
| 3341 |
} elsif ($token->{type} eq 'start tag' and |
if ($token->{insertion_mode} ne 'trailing end') { |
|
$token->{tag_name} eq 'html') { |
|
|
## ISSUE: "aa<html>" is not a parse error. |
|
|
## ISSUE: "<html>" in fragment is not a parse error. |
|
|
unless ($token->{first_start_tag}) { |
|
|
!!!parse-error (type => 'not first start tag'); |
|
|
} |
|
|
my $top_el = $self->{open_elements}->[0]->[0]; |
|
|
for my $attr_name (keys %{$token->{attributes}}) { |
|
|
unless ($top_el->has_attribute_ns (undef, $attr_name)) { |
|
|
$top_el->set_attribute_ns |
|
|
(undef, [undef, $attr_name], |
|
|
$token->{attributes}->{$attr_name}->{value}); |
|
|
} |
|
|
} |
|
|
!!!next-token; |
|
|
redo B; |
|
|
} elsif ($token->{type} eq 'end-of-file') { |
|
| 3342 |
## Generate implied end tags |
## Generate implied end tags |
| 3343 |
if ({ |
if ({ |
| 3344 |
dd => 1, dt => 1, li => 1, p => 1, td => 1, th => 1, tr => 1, |
dd => 1, dt => 1, li => 1, p => 1, td => 1, th => 1, tr => 1, |
| 3358 |
!!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]); |
!!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]); |
| 3359 |
} |
} |
| 3360 |
|
|
|
## Stop parsing |
|
|
last B; |
|
|
|
|
| 3361 |
## ISSUE: There is an issue in the spec. |
## ISSUE: There is an issue in the spec. |
| 3362 |
|
} |
| 3363 |
|
|
| 3364 |
|
## Stop parsing |
| 3365 |
|
last B; |
| 3366 |
|
} elsif ($token->{type} eq 'start tag' and |
| 3367 |
|
$token->{tag_name} eq 'html') { |
| 3368 |
|
if ($self->{insertion_mode} eq 'trailing end') { |
| 3369 |
|
## Turn into the main phase |
| 3370 |
|
!!!parse-error (type => 'after html:html'); |
| 3371 |
|
$self->{insertion_mode} = $previous_insertion_mode; |
| 3372 |
|
} |
| 3373 |
|
|
| 3374 |
|
## ISSUE: "aa<html>" is not a parse error. |
| 3375 |
|
## ISSUE: "<html>" in fragment is not a parse error. |
| 3376 |
|
unless ($token->{first_start_tag}) { |
| 3377 |
|
!!!parse-error (type => 'not first start tag'); |
| 3378 |
|
} |
| 3379 |
|
my $top_el = $self->{open_elements}->[0]->[0]; |
| 3380 |
|
for my $attr_name (keys %{$token->{attributes}}) { |
| 3381 |
|
unless ($top_el->has_attribute_ns (undef, $attr_name)) { |
| 3382 |
|
$top_el->set_attribute_ns |
| 3383 |
|
(undef, [undef, $attr_name], |
| 3384 |
|
$token->{attributes}->{$attr_name}->{value}); |
| 3385 |
|
} |
| 3386 |
|
} |
| 3387 |
|
!!!next-token; |
| 3388 |
|
redo B; |
| 3389 |
|
} elsif ($token->{type} eq 'comment') { |
| 3390 |
|
my $comment = $self->{document}->create_comment ($token->{data}); |
| 3391 |
|
if ($self->{insertion_mode} eq 'trailing end') { |
| 3392 |
|
$self->{document}->append_child ($comment); |
| 3393 |
|
} elsif ($self->{insertion_mode} eq 'after body') { |
| 3394 |
|
$self->{open_elements}->[0]->[0]->append_child ($comment); |
| 3395 |
} else { |
} else { |
| 3396 |
if ($self->{insertion_mode} eq 'before head') { |
$self->{open_elements}->[-1]->[0]->append_child ($comment); |
| 3397 |
|
} |
| 3398 |
|
!!!next-token; |
| 3399 |
|
redo B; |
| 3400 |
|
} elsif ($self->{insertion_mode} eq 'before head') { |
| 3401 |
if ($token->{type} eq 'character') { |
if ($token->{type} eq 'character') { |
| 3402 |
if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { |
if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { |
| 3403 |
$self->{open_elements}->[-1]->[0]->manakai_append_text ($1); |
$self->{open_elements}->[-1]->[0]->manakai_append_text ($1); |
| 3413 |
$self->{insertion_mode} = 'in head'; |
$self->{insertion_mode} = 'in head'; |
| 3414 |
## reprocess |
## reprocess |
| 3415 |
redo B; |
redo B; |
|
} elsif ($token->{type} eq 'comment') { |
|
|
my $comment = $self->{document}->create_comment ($token->{data}); |
|
|
$self->{open_elements}->[-1]->[0]->append_child ($comment); |
|
|
!!!next-token; |
|
|
redo B; |
|
| 3416 |
} elsif ($token->{type} eq 'start tag') { |
} elsif ($token->{type} eq 'start tag') { |
| 3417 |
my $attr = $token->{tag_name} eq 'head' ? $token->{attributes} : {}; |
my $attr = $token->{tag_name} eq 'head' ? $token->{attributes} : {}; |
| 3418 |
!!!create-element ($self->{head_element}, 'head', $attr); |
!!!create-element ($self->{head_element}, 'head', $attr); |
| 3464 |
} |
} |
| 3465 |
|
|
| 3466 |
# |
# |
|
} elsif ($token->{type} eq 'comment') { |
|
|
my $comment = $self->{document}->create_comment ($token->{data}); |
|
|
$self->{open_elements}->[-1]->[0]->append_child ($comment); |
|
|
!!!next-token; |
|
|
redo B; |
|
| 3467 |
} elsif ($token->{type} eq 'start tag') { |
} elsif ($token->{type} eq 'start tag') { |
| 3468 |
if ({base => ($self->{insertion_mode} eq 'in head' or |
if ({base => ($self->{insertion_mode} eq 'in head' or |
| 3469 |
$self->{insertion_mode} eq 'after head'), |
$self->{insertion_mode} eq 'after head'), |
| 3470 |
link => 1, meta => 1}->{$token->{tag_name}}) { |
link => 1}->{$token->{tag_name}}) { |
| 3471 |
|
## NOTE: There is a "as if in head" code clone. |
| 3472 |
|
if ($self->{insertion_mode} eq 'after head') { |
| 3473 |
|
!!!parse-error (type => 'after head:'.$token->{tag_name}); |
| 3474 |
|
push @{$self->{open_elements}}, [$self->{head_element}, 'head']; |
| 3475 |
|
} |
| 3476 |
|
!!!insert-element ($token->{tag_name}, $token->{attributes}); |
| 3477 |
|
pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec. |
| 3478 |
|
pop @{$self->{open_elements}} |
| 3479 |
|
if $self->{insertion_mode} eq 'after head'; |
| 3480 |
|
!!!next-token; |
| 3481 |
|
redo B; |
| 3482 |
|
} elsif ($token->{tag_name} eq 'meta') { |
| 3483 |
## NOTE: There is a "as if in head" code clone. |
## NOTE: There is a "as if in head" code clone. |
| 3484 |
if ($self->{insertion_mode} eq 'after head') { |
if ($self->{insertion_mode} eq 'after head') { |
| 3485 |
!!!parse-error (type => 'after head:'.$token->{tag_name}); |
!!!parse-error (type => 'after head:'.$token->{tag_name}); |
| 3487 |
} |
} |
| 3488 |
!!!insert-element ($token->{tag_name}, $token->{attributes}); |
!!!insert-element ($token->{tag_name}, $token->{attributes}); |
| 3489 |
pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec. |
pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec. |
| 3490 |
|
|
| 3491 |
|
unless ($self->{confident}) { |
| 3492 |
|
my $charset; |
| 3493 |
|
if ($token->{attributes}->{charset}) { ## TODO: And if supported |
| 3494 |
|
$charset = $token->{attributes}->{charset}->{value}; |
| 3495 |
|
} |
| 3496 |
|
if ($token->{attributes}->{'http-equiv'}) { |
| 3497 |
|
## ISSUE: Algorithm name in the spec was incorrect so that not linked to the definition. |
| 3498 |
|
if ($token->{attributes}->{'http-equiv'}->{value} |
| 3499 |
|
=~ /\A[^;]*;[\x09-\x0D\x20]*charset[\x09-\x0D\x20]*= |
| 3500 |
|
[\x09-\x0D\x20]*(?>"([^"]*)"|'([^']*)'| |
| 3501 |
|
([^"'\x09-\x0D\x20][^\x09-\x0D\x20]*))/x) { |
| 3502 |
|
$charset = defined $1 ? $1 : defined $2 ? $2 : $3; |
| 3503 |
|
} ## TODO: And if supported |
| 3504 |
|
} |
| 3505 |
|
## TODO: Change the encoding |
| 3506 |
|
} |
| 3507 |
|
|
| 3508 |
## TODO: Extracting |charset| from |meta|. |
## TODO: Extracting |charset| from |meta|. |
| 3509 |
pop @{$self->{open_elements}} |
pop @{$self->{open_elements}} |
| 3510 |
if $self->{insertion_mode} eq 'after head'; |
if $self->{insertion_mode} eq 'after head'; |
| 3642 |
|
|
| 3643 |
!!!next-token; |
!!!next-token; |
| 3644 |
redo B; |
redo B; |
|
} elsif ($token->{type} eq 'comment') { |
|
|
## NOTE: There is a code clone of "comment in body". |
|
|
my $comment = $self->{document}->create_comment ($token->{data}); |
|
|
$self->{open_elements}->[-1]->[0]->append_child ($comment); |
|
|
!!!next-token; |
|
|
redo B; |
|
| 3645 |
} else { |
} else { |
| 3646 |
$in_body->($insert_to_current); |
$in_body->($insert_to_current); |
| 3647 |
redo B; |
redo B; |
| 3705 |
|
|
| 3706 |
!!!next-token; |
!!!next-token; |
| 3707 |
redo B; |
redo B; |
|
} elsif ($token->{type} eq 'comment') { |
|
|
my $comment = $self->{document}->create_comment ($token->{data}); |
|
|
$self->{open_elements}->[-1]->[0]->append_child ($comment); |
|
|
!!!next-token; |
|
|
redo B; |
|
| 3708 |
} elsif ($token->{type} eq 'start tag') { |
} elsif ($token->{type} eq 'start tag') { |
| 3709 |
if ({ |
if ({ |
| 3710 |
caption => 1, |
caption => 1, |
| 3871 |
|
|
| 3872 |
!!!next-token; |
!!!next-token; |
| 3873 |
redo B; |
redo B; |
|
} elsif ($token->{type} eq 'comment') { |
|
|
## NOTE: This is a code clone of "comment in body". |
|
|
my $comment = $self->{document}->create_comment ($token->{data}); |
|
|
$self->{open_elements}->[-1]->[0]->append_child ($comment); |
|
|
!!!next-token; |
|
|
redo B; |
|
| 3874 |
} elsif ($token->{type} eq 'start tag') { |
} elsif ($token->{type} eq 'start tag') { |
| 3875 |
if ({ |
if ({ |
| 3876 |
caption => 1, col => 1, colgroup => 1, tbody => 1, |
caption => 1, col => 1, colgroup => 1, tbody => 1, |
| 4052 |
} |
} |
| 4053 |
|
|
| 4054 |
# |
# |
|
} elsif ($token->{type} eq 'comment') { |
|
|
my $comment = $self->{document}->create_comment ($token->{data}); |
|
|
$self->{open_elements}->[-1]->[0]->append_child ($comment); |
|
|
!!!next-token; |
|
|
redo B; |
|
| 4055 |
} elsif ($token->{type} eq 'start tag') { |
} elsif ($token->{type} eq 'start tag') { |
| 4056 |
if ($token->{tag_name} eq 'col') { |
if ($token->{tag_name} eq 'col') { |
| 4057 |
!!!insert-element ($token->{tag_name}, $token->{attributes}); |
!!!insert-element ($token->{tag_name}, $token->{attributes}); |
| 4157 |
|
|
| 4158 |
!!!next-token; |
!!!next-token; |
| 4159 |
redo B; |
redo B; |
|
} elsif ($token->{type} eq 'comment') { |
|
|
## Copied from 'in table' |
|
|
my $comment = $self->{document}->create_comment ($token->{data}); |
|
|
$self->{open_elements}->[-1]->[0]->append_child ($comment); |
|
|
!!!next-token; |
|
|
redo B; |
|
| 4160 |
} elsif ($token->{type} eq 'start tag') { |
} elsif ($token->{type} eq 'start tag') { |
| 4161 |
if ({ |
if ({ |
| 4162 |
tr => 1, |
tr => 1, |
| 4436 |
|
|
| 4437 |
!!!next-token; |
!!!next-token; |
| 4438 |
redo B; |
redo B; |
|
} elsif ($token->{type} eq 'comment') { |
|
|
## Copied from 'in table' |
|
|
my $comment = $self->{document}->create_comment ($token->{data}); |
|
|
$self->{open_elements}->[-1]->[0]->append_child ($comment); |
|
|
!!!next-token; |
|
|
redo B; |
|
| 4439 |
} elsif ($token->{type} eq 'start tag') { |
} elsif ($token->{type} eq 'start tag') { |
| 4440 |
if ($token->{tag_name} eq 'th' or |
if ($token->{tag_name} eq 'th' or |
| 4441 |
$token->{tag_name} eq 'td') { |
$token->{tag_name} eq 'td') { |
| 4695 |
|
|
| 4696 |
!!!next-token; |
!!!next-token; |
| 4697 |
redo B; |
redo B; |
|
} elsif ($token->{type} eq 'comment') { |
|
|
## NOTE: This is a code clone of "comment in body". |
|
|
my $comment = $self->{document}->create_comment ($token->{data}); |
|
|
$self->{open_elements}->[-1]->[0]->append_child ($comment); |
|
|
!!!next-token; |
|
|
redo B; |
|
| 4698 |
} elsif ($token->{type} eq 'start tag') { |
} elsif ($token->{type} eq 'start tag') { |
| 4699 |
if ({ |
if ({ |
| 4700 |
caption => 1, col => 1, colgroup => 1, |
caption => 1, col => 1, colgroup => 1, |
| 4831 |
$self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data}); |
$self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data}); |
| 4832 |
!!!next-token; |
!!!next-token; |
| 4833 |
redo B; |
redo B; |
|
} elsif ($token->{type} eq 'comment') { |
|
|
my $comment = $self->{document}->create_comment ($token->{data}); |
|
|
$self->{open_elements}->[-1]->[0]->append_child ($comment); |
|
|
!!!next-token; |
|
|
redo B; |
|
| 4834 |
} elsif ($token->{type} eq 'start tag') { |
} elsif ($token->{type} eq 'start tag') { |
| 4835 |
if ($token->{tag_name} eq 'option') { |
if ($token->{tag_name} eq 'option') { |
| 4836 |
if ($self->{open_elements}->[-1]->[1] eq 'option') { |
if ($self->{open_elements}->[-1]->[1] eq 'option') { |
| 5003 |
} elsif ($self->{insertion_mode} eq 'after body') { |
} elsif ($self->{insertion_mode} eq 'after body') { |
| 5004 |
if ($token->{type} eq 'character') { |
if ($token->{type} eq 'character') { |
| 5005 |
if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { |
if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { |
| 5006 |
|
my $data = $1; |
| 5007 |
## As if in body |
## As if in body |
| 5008 |
$reconstruct_active_formatting_elements->($insert_to_current); |
$reconstruct_active_formatting_elements->($insert_to_current); |
| 5009 |
|
|
| 5010 |
$self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data}); |
$self->{open_elements}->[-1]->[0]->manakai_append_text ($1); |
| 5011 |
|
|
| 5012 |
unless (length $token->{data}) { |
unless (length $token->{data}) { |
| 5013 |
!!!next-token; |
!!!next-token; |
| 5016 |
} |
} |
| 5017 |
|
|
| 5018 |
# |
# |
| 5019 |
!!!parse-error (type => 'after body:#'.$token->{type}); |
!!!parse-error (type => 'after body:#character'); |
|
} elsif ($token->{type} eq 'comment') { |
|
|
my $comment = $self->{document}->create_comment ($token->{data}); |
|
|
$self->{open_elements}->[0]->[0]->append_child ($comment); |
|
|
!!!next-token; |
|
|
redo B; |
|
| 5020 |
} elsif ($token->{type} eq 'start tag') { |
} elsif ($token->{type} eq 'start tag') { |
| 5021 |
!!!parse-error (type => 'after body:'.$token->{tag_name}); |
!!!parse-error (type => 'after body:'.$token->{tag_name}); |
| 5022 |
# |
# |
| 5028 |
!!!next-token; |
!!!next-token; |
| 5029 |
redo B; |
redo B; |
| 5030 |
} else { |
} else { |
| 5031 |
$phase = 'trailing end'; |
$previous_insertion_mode = $self->{insertion_mode}; |
| 5032 |
|
$self->{insertion_mode} = 'trailing end'; |
| 5033 |
!!!next-token; |
!!!next-token; |
| 5034 |
redo B; |
redo B; |
| 5035 |
} |
} |
| 5037 |
!!!parse-error (type => 'after body:/'.$token->{tag_name}); |
!!!parse-error (type => 'after body:/'.$token->{tag_name}); |
| 5038 |
} |
} |
| 5039 |
} else { |
} else { |
| 5040 |
!!!parse-error (type => 'after body:#'.$token->{type}); |
die "$0: $token->{type}: Unknown token type"; |
| 5041 |
} |
} |
| 5042 |
|
|
| 5043 |
$self->{insertion_mode} = 'in body'; |
$self->{insertion_mode} = 'in body'; |
| 5044 |
## reprocess |
## reprocess |
| 5045 |
redo B; |
redo B; |
| 5046 |
} elsif ($self->{insertion_mode} eq 'in frameset') { |
} elsif ($self->{insertion_mode} eq 'in frameset') { |
| 5047 |
if ($token->{type} eq 'character') { |
if ($token->{type} eq 'character') { |
| 5048 |
if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { |
if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { |
| 5049 |
$self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data}); |
$self->{open_elements}->[-1]->[0]->manakai_append_text ($1); |
|
|
|
|
unless (length $token->{data}) { |
|
|
!!!next-token; |
|
|
redo B; |
|
|
} |
|
|
} |
|
| 5050 |
|
|
| 5051 |
# |
unless (length $token->{data}) { |
|
} elsif ($token->{type} eq 'comment') { |
|
|
my $comment = $self->{document}->create_comment ($token->{data}); |
|
|
$self->{open_elements}->[-1]->[0]->append_child ($comment); |
|
| 5052 |
!!!next-token; |
!!!next-token; |
| 5053 |
redo B; |
redo B; |
|
} elsif ($token->{type} eq 'start tag') { |
|
|
if ($token->{tag_name} eq 'frameset') { |
|
|
!!!insert-element ($token->{tag_name}, $token->{attributes}); |
|
|
!!!next-token; |
|
|
redo B; |
|
|
} elsif ($token->{tag_name} eq 'frame') { |
|
|
!!!insert-element ($token->{tag_name}, $token->{attributes}); |
|
|
pop @{$self->{open_elements}}; |
|
|
!!!next-token; |
|
|
redo B; |
|
|
} elsif ($token->{tag_name} eq 'noframes') { |
|
|
$in_body->($insert_to_current); |
|
|
redo B; |
|
|
} else { |
|
|
# |
|
|
} |
|
|
} elsif ($token->{type} eq 'end tag') { |
|
|
if ($token->{tag_name} eq 'frameset') { |
|
|
if ($self->{open_elements}->[-1]->[1] eq 'html' and |
|
|
@{$self->{open_elements}} == 1) { |
|
|
!!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}); |
|
|
## Ignore the token |
|
|
!!!next-token; |
|
|
} else { |
|
|
pop @{$self->{open_elements}}; |
|
|
!!!next-token; |
|
|
} |
|
|
|
|
|
## if not inner_html and |
|
|
if ($self->{open_elements}->[-1]->[1] ne 'frameset') { |
|
|
$self->{insertion_mode} = 'after frameset'; |
|
|
} |
|
|
redo B; |
|
|
} else { |
|
|
# |
|
|
} |
|
|
} else { |
|
|
# |
|
| 5054 |
} |
} |
| 5055 |
|
} |
| 5056 |
if (defined $token->{tag_name}) { |
|
| 5057 |
!!!parse-error (type => 'in frameset:'.($token->{type} eq 'end tag' ? '/' : '').$token->{tag_name}); |
!!!parse-error (type => 'in frameset:#character'); |
| 5058 |
|
## Ignore the token |
| 5059 |
|
!!!next-token; |
| 5060 |
|
redo B; |
| 5061 |
|
} elsif ($token->{type} eq 'start tag') { |
| 5062 |
|
if ($token->{tag_name} eq 'frameset') { |
| 5063 |
|
!!!insert-element ($token->{tag_name}, $token->{attributes}); |
| 5064 |
|
!!!next-token; |
| 5065 |
|
redo B; |
| 5066 |
|
} elsif ($token->{tag_name} eq 'frame') { |
| 5067 |
|
!!!insert-element ($token->{tag_name}, $token->{attributes}); |
| 5068 |
|
pop @{$self->{open_elements}}; |
| 5069 |
|
!!!next-token; |
| 5070 |
|
redo B; |
| 5071 |
|
} elsif ($token->{tag_name} eq 'noframes') { |
| 5072 |
|
$in_body->($insert_to_current); |
| 5073 |
|
redo B; |
| 5074 |
|
} else { |
| 5075 |
|
!!!parse-error (type => 'in frameset:'.$token->{tag_name}); |
| 5076 |
|
## Ignore the token |
| 5077 |
|
!!!next-token; |
| 5078 |
|
redo B; |
| 5079 |
|
} |
| 5080 |
|
} elsif ($token->{type} eq 'end tag') { |
| 5081 |
|
if ($token->{tag_name} eq 'frameset') { |
| 5082 |
|
if ($self->{open_elements}->[-1]->[1] eq 'html' and |
| 5083 |
|
@{$self->{open_elements}} == 1) { |
| 5084 |
|
!!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}); |
| 5085 |
|
## Ignore the token |
| 5086 |
|
!!!next-token; |
| 5087 |
} else { |
} else { |
| 5088 |
!!!parse-error (type => 'in frameset:#'.$token->{type}); |
pop @{$self->{open_elements}}; |
| 5089 |
|
!!!next-token; |
| 5090 |
|
} |
| 5091 |
|
|
| 5092 |
|
if (not defined $self->{inner_html_node} and |
| 5093 |
|
$self->{open_elements}->[-1]->[1] ne 'frameset') { |
| 5094 |
|
$self->{insertion_mode} = 'after frameset'; |
| 5095 |
} |
} |
| 5096 |
|
redo B; |
| 5097 |
|
} else { |
| 5098 |
|
!!!parse-error (type => 'in frameset:/'.$token->{tag_name}); |
| 5099 |
## Ignore the token |
## Ignore the token |
| 5100 |
!!!next-token; |
!!!next-token; |
| 5101 |
redo B; |
redo B; |
| 5102 |
} elsif ($self->{insertion_mode} eq 'after frameset') { |
} |
| 5103 |
if ($token->{type} eq 'character') { |
} else { |
| 5104 |
|
die "$0: $token->{type}: Unknown token type"; |
| 5105 |
|
} |
| 5106 |
|
} elsif ($self->{insertion_mode} eq 'after frameset') { |
| 5107 |
|
if ($token->{type} eq 'character') { |
| 5108 |
if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { |
if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { |
| 5109 |
$self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data}); |
$self->{open_elements}->[-1]->[0]->manakai_append_text ($1); |
| 5110 |
|
|
| 5111 |
unless (length $token->{data}) { |
unless (length $token->{data}) { |
| 5112 |
!!!next-token; |
!!!next-token; |
| 5114 |
} |
} |
| 5115 |
} |
} |
| 5116 |
|
|
| 5117 |
# |
if ($token->{data} =~ s/^[^\x09\x0A\x0B\x0C\x20]+//) { |
| 5118 |
} elsif ($token->{type} eq 'comment') { |
!!!parse-error (type => 'after frameset:#character'); |
| 5119 |
my $comment = $self->{document}->create_comment ($token->{data}); |
|
| 5120 |
$self->{open_elements}->[-1]->[0]->append_child ($comment); |
## Ignore the token. |
| 5121 |
!!!next-token; |
if (length $token->{data}) { |
| 5122 |
redo B; |
## reprocess the rest of characters |
| 5123 |
} elsif ($token->{type} eq 'start tag') { |
} else { |
| 5124 |
if ($token->{tag_name} eq 'noframes') { |
!!!next-token; |
| 5125 |
$in_body->($insert_to_current); |
} |
|
redo B; |
|
|
} else { |
|
|
# |
|
|
} |
|
|
} elsif ($token->{type} eq 'end tag') { |
|
|
if ($token->{tag_name} eq 'html') { |
|
|
$phase = 'trailing end'; |
|
|
!!!next-token; |
|
| 5126 |
redo B; |
redo B; |
|
} else { |
|
|
# |
|
| 5127 |
} |
} |
| 5128 |
} else { |
|
| 5129 |
# |
die qq[$0: Character "$token->{data}"]; |
| 5130 |
} |
} elsif ($token->{type} eq 'start tag') { |
| 5131 |
|
if ($token->{tag_name} eq 'noframes') { |
| 5132 |
if (defined $token->{tag_name}) { |
$in_body->($insert_to_current); |
| 5133 |
!!!parse-error (type => 'after frameset:'.($token->{tag_name} eq 'end tag' ? '/' : '').$token->{tag_name}); |
redo B; |
| 5134 |
} else { |
} else { |
| 5135 |
!!!parse-error (type => 'after frameset:#'.$token->{type}); |
!!!parse-error (type => 'after frameset:'.$token->{tag_name}); |
|
} |
|
| 5136 |
## Ignore the token |
## Ignore the token |
| 5137 |
!!!next-token; |
!!!next-token; |
| 5138 |
redo B; |
redo B; |
| 5139 |
|
} |
| 5140 |
## ISSUE: An issue in spec there |
} elsif ($token->{type} eq 'end tag') { |
| 5141 |
|
if ($token->{tag_name} eq 'html') { |
| 5142 |
|
$previous_insertion_mode = $self->{insertion_mode}; |
| 5143 |
|
$self->{insertion_mode} = 'trailing end'; |
| 5144 |
|
!!!next-token; |
| 5145 |
|
redo B; |
| 5146 |
} else { |
} else { |
| 5147 |
die "$0: $self->{insertion_mode}: Unknown insertion mode"; |
!!!parse-error (type => 'after frameset:/'.$token->{tag_name}); |
| 5148 |
|
## Ignore the token |
| 5149 |
|
!!!next-token; |
| 5150 |
|
redo B; |
| 5151 |
} |
} |
| 5152 |
|
} else { |
| 5153 |
|
die "$0: $token->{type}: Unknown token type"; |
| 5154 |
} |
} |
| 5155 |
} elsif ($phase eq 'trailing end') { |
|
| 5156 |
|
## ISSUE: An issue in spec here |
| 5157 |
|
} elsif ($self->{insertion_mode} eq 'trailing end') { |
| 5158 |
## states in the main stage is preserved yet # MUST |
## states in the main stage is preserved yet # MUST |
| 5159 |
|
|
| 5160 |
if ($token->{type} eq 'DOCTYPE') { |
if ($token->{type} eq 'character') { |
|
!!!parse-error (type => 'after html:#DOCTYPE'); |
|
|
## Ignore the token |
|
|
!!!next-token; |
|
|
redo B; |
|
|
} elsif ($token->{type} eq 'comment') { |
|
|
my $comment = $self->{document}->create_comment ($token->{data}); |
|
|
$self->{document}->append_child ($comment); |
|
|
!!!next-token; |
|
|
redo B; |
|
|
} elsif ($token->{type} eq 'character') { |
|
| 5161 |
if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { |
if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { |
| 5162 |
my $data = $1; |
my $data = $1; |
| 5163 |
## As if in the main phase. |
## As if in the main phase. |
| 5164 |
## NOTE: The insertion mode in the main phase |
## NOTE: The insertion mode in the main phase |
| 5165 |
## just before the phase has been changed to the trailing |
## just before the phase has been changed to the trailing |
| 5166 |
## end phase is either "after body" or "after frameset". |
## end phase is either "after body" or "after frameset". |
| 5167 |
$reconstruct_active_formatting_elements->($insert_to_current) |
$reconstruct_active_formatting_elements->($insert_to_current); |
|
if $phase eq 'main'; |
|
| 5168 |
|
|
| 5169 |
$self->{open_elements}->[-1]->[0]->manakai_append_text ($data); |
$self->{open_elements}->[-1]->[0]->manakai_append_text ($data); |
| 5170 |
|
|
| 5175 |
} |
} |
| 5176 |
|
|
| 5177 |
!!!parse-error (type => 'after html:#character'); |
!!!parse-error (type => 'after html:#character'); |
| 5178 |
$phase = 'main'; |
$self->{insertion_mode} = $previous_insertion_mode; |
| 5179 |
## reprocess |
## reprocess |
| 5180 |
redo B; |
redo B; |
| 5181 |
} elsif ($token->{type} eq 'start tag' or |
} elsif ($token->{type} eq 'start tag') { |
| 5182 |
$token->{type} eq 'end tag') { |
!!!parse-error (type => 'after html:'.$token->{tag_name}); |
| 5183 |
!!!parse-error (type => 'after html:'.($token->{type} eq 'end tag' ? '/' : '').$token->{tag_name}); |
$self->{insertion_mode} = $previous_insertion_mode; |
| 5184 |
$phase = 'main'; |
## reprocess |
| 5185 |
|
redo B; |
| 5186 |
|
} elsif ($token->{type} eq 'end tag') { |
| 5187 |
|
!!!parse-error (type => 'after html:/'.$token->{tag_name}); |
| 5188 |
|
$self->{insertion_mode} = $previous_insertion_mode; |
| 5189 |
## reprocess |
## reprocess |
| 5190 |
redo B; |
redo B; |
|
} elsif ($token->{type} eq 'end-of-file') { |
|
|
## Stop parsing |
|
|
last B; |
|
| 5191 |
} else { |
} else { |
| 5192 |
die "$0: $token->{type}: Unknown token"; |
die "$0: $token->{type}: Unknown token"; |
| 5193 |
} |
} |
| 5194 |
|
} else { |
| 5195 |
|
die "$0: $self->{insertion_mode}: Unknown insertion mode"; |
| 5196 |
} |
} |
| 5197 |
} # B |
} # B |
| 5198 |
|
|