| 372 |
my ($char_stream, $e_status); |
my ($char_stream, $e_status); |
| 373 |
|
|
| 374 |
SNIFFING: { |
SNIFFING: { |
| 375 |
|
## NOTE: By setting |allow_fallback| option true when the |
| 376 |
|
## |get_decode_handle| method is invoked, we ignore what the HTML5 |
| 377 |
|
## spec requires, i.e. unsupported encoding should be ignored. |
| 378 |
|
## TODO: We should not do this unless the parser is invoked |
| 379 |
|
## in the conformance checking mode, in which this behavior |
| 380 |
|
## would be useful. |
| 381 |
|
|
| 382 |
## Step 1 |
## Step 1 |
| 383 |
if (defined $charset_name) { |
if (defined $charset_name) { |
| 481 |
$self->{confident} = 0; |
$self->{confident} = 0; |
| 482 |
} # SNIFFING |
} # SNIFFING |
| 483 |
|
|
|
$self->{input_encoding} = $charset->get_iana_name; |
|
| 484 |
if ($e_status & Message::Charset::Info::FALLBACK_ENCODING_IMPL ()) { |
if ($e_status & Message::Charset::Info::FALLBACK_ENCODING_IMPL ()) { |
| 485 |
|
$self->{input_encoding} = $charset->get_iana_name; ## TODO: Should we set actual charset decoder's encoding name? |
| 486 |
!!!parse-error (type => 'chardecode:fallback', |
!!!parse-error (type => 'chardecode:fallback', |
| 487 |
text => $self->{input_encoding}, |
#text => $self->{input_encoding}, |
| 488 |
level => $self->{level}->{uncertain}, |
level => $self->{level}->{uncertain}, |
| 489 |
line => 1, column => 1, |
line => 1, column => 1, |
| 490 |
layer => 'encode'); |
layer => 'encode'); |
| 491 |
} elsif (not ($e_status & |
} elsif (not ($e_status & |
| 492 |
Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL())) { |
Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL())) { |
| 493 |
|
$self->{input_encoding} = $charset->get_iana_name; |
| 494 |
!!!parse-error (type => 'chardecode:no error', |
!!!parse-error (type => 'chardecode:no error', |
| 495 |
text => $self->{input_encoding}, |
text => $self->{input_encoding}, |
| 496 |
level => $self->{level}->{uncertain}, |
level => $self->{level}->{uncertain}, |
| 497 |
line => 1, column => 1, |
line => 1, column => 1, |
| 498 |
layer => 'encode'); |
layer => 'encode'); |
| 499 |
|
} else { |
| 500 |
|
$self->{input_encoding} = $charset->get_iana_name; |
| 501 |
} |
} |
| 502 |
|
|
| 503 |
$self->{change_encoding} = sub { |
$self->{change_encoding} = sub { |
| 569 |
} catch Whatpm::HTML::RestartParser with { |
} catch Whatpm::HTML::RestartParser with { |
| 570 |
## NOTE: Invoked after {change_encoding}. |
## NOTE: Invoked after {change_encoding}. |
| 571 |
|
|
|
$self->{input_encoding} = $charset->get_iana_name; |
|
| 572 |
if ($e_status & Message::Charset::Info::FALLBACK_ENCODING_IMPL ()) { |
if ($e_status & Message::Charset::Info::FALLBACK_ENCODING_IMPL ()) { |
| 573 |
|
$self->{input_encoding} = $charset->get_iana_name; ## TODO: Should we set actual charset decoder's encoding name? |
| 574 |
!!!parse-error (type => 'chardecode:fallback', |
!!!parse-error (type => 'chardecode:fallback', |
|
text => $self->{input_encoding}, |
|
| 575 |
level => $self->{level}->{uncertain}, |
level => $self->{level}->{uncertain}, |
| 576 |
|
#text => $self->{input_encoding}, |
| 577 |
line => 1, column => 1, |
line => 1, column => 1, |
| 578 |
layer => 'encode'); |
layer => 'encode'); |
| 579 |
} elsif (not ($e_status & |
} elsif (not ($e_status & |
| 580 |
Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL())) { |
Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL())) { |
| 581 |
|
$self->{input_encoding} = $charset->get_iana_name; |
| 582 |
!!!parse-error (type => 'chardecode:no error', |
!!!parse-error (type => 'chardecode:no error', |
| 583 |
text => $self->{input_encoding}, |
text => $self->{input_encoding}, |
| 584 |
level => $self->{level}->{uncertain}, |
level => $self->{level}->{uncertain}, |
| 585 |
line => 1, column => 1, |
line => 1, column => 1, |
| 586 |
layer => 'encode'); |
layer => 'encode'); |
| 587 |
|
} else { |
| 588 |
|
$self->{input_encoding} = $charset->get_iana_name; |
| 589 |
} |
} |
| 590 |
$self->{confident} = 1; |
$self->{confident} = 1; |
| 591 |
$char_stream->onerror ($char_onerror); |
$char_stream->onerror ($char_onerror); |
| 720 |
my $class = shift; |
my $class = shift; |
| 721 |
my $self = bless { |
my $self = bless { |
| 722 |
level => {must => 'm', |
level => {must => 'm', |
| 723 |
|
should => 's', |
| 724 |
warn => 'w', |
warn => 'w', |
| 725 |
info => 'i', |
info => 'i', |
| 726 |
uncertain => 'u'}, |
uncertain => 'u'}, |
| 3167 |
## language. |
## language. |
| 3168 |
my $doctype_name = $token->{name}; |
my $doctype_name = $token->{name}; |
| 3169 |
$doctype_name = '' unless defined $doctype_name; |
$doctype_name = '' unless defined $doctype_name; |
| 3170 |
$doctype_name =~ tr/a-z/A-Z/; |
$doctype_name =~ tr/a-z/A-Z/; # ASCII case-insensitive |
| 3171 |
if (not defined $token->{name} or # <!DOCTYPE> |
if (not defined $token->{name} or # <!DOCTYPE> |
|
defined $token->{public_identifier} or |
|
| 3172 |
defined $token->{system_identifier}) { |
defined $token->{system_identifier}) { |
| 3173 |
!!!cp ('t1'); |
!!!cp ('t1'); |
| 3174 |
!!!parse-error (type => 'not HTML5', token => $token); |
!!!parse-error (type => 'not HTML5', token => $token); |
| 3175 |
} elsif ($doctype_name ne 'HTML') { |
} elsif ($doctype_name ne 'HTML') { |
| 3176 |
!!!cp ('t2'); |
!!!cp ('t2'); |
|
## ISSUE: ASCII case-insensitive? (in fact it does not matter) |
|
| 3177 |
!!!parse-error (type => 'not HTML5', token => $token); |
!!!parse-error (type => 'not HTML5', token => $token); |
| 3178 |
|
} elsif (defined $token->{public_identifier}) { |
| 3179 |
|
if ($token->{public_identifier} eq 'XSLT-compat') { |
| 3180 |
|
!!!cp ('t1.2'); |
| 3181 |
|
!!!parse-error (type => 'XSLT-compat', token => $token, |
| 3182 |
|
level => $self->{level}->{should}); |
| 3183 |
|
} else { |
| 3184 |
|
!!!parse-error (type => 'not HTML5', token => $token); |
| 3185 |
|
} |
| 3186 |
} else { |
} else { |
| 3187 |
!!!cp ('t3'); |
!!!cp ('t3'); |
| 3188 |
|
# |
| 3189 |
} |
} |
| 3190 |
|
|
| 3191 |
my $doctype = $self->{document}->create_document_type_definition |
my $doctype = $self->{document}->create_document_type_definition |
| 6321 |
} elsif ($self->{insertion_mode} == AFTER_FRAMESET_IM) { |
} elsif ($self->{insertion_mode} == AFTER_FRAMESET_IM) { |
| 6322 |
!!!cp ('t312'); |
!!!cp ('t312'); |
| 6323 |
!!!parse-error (type => 'after frameset:#text', token => $token); |
!!!parse-error (type => 'after frameset:#text', token => $token); |
| 6324 |
} else { # "after html frameset" |
} else { # "after after frameset" |
| 6325 |
!!!cp ('t313'); |
!!!cp ('t313'); |
| 6326 |
!!!parse-error (type => 'after html:#text', token => $token); |
!!!parse-error (type => 'after html:#text', token => $token); |
|
|
|
|
$self->{insertion_mode} = AFTER_FRAMESET_IM; |
|
|
## Reprocess in the "after frameset" insertion mode. |
|
|
!!!parse-error (type => 'after frameset:#text', token => $token); |
|
| 6327 |
} |
} |
| 6328 |
|
|
| 6329 |
## Ignore the token. |
## Ignore the token. |
| 6339 |
|
|
| 6340 |
die qq[$0: Character "$token->{data}"]; |
die qq[$0: Character "$token->{data}"]; |
| 6341 |
} elsif ($token->{type} == START_TAG_TOKEN) { |
} elsif ($token->{type} == START_TAG_TOKEN) { |
|
if ($self->{insertion_mode} == AFTER_HTML_FRAMESET_IM) { |
|
|
!!!cp ('t316'); |
|
|
!!!parse-error (type => 'after html', |
|
|
text => $token->{tag_name}, token => $token); |
|
|
|
|
|
$self->{insertion_mode} = AFTER_FRAMESET_IM; |
|
|
## Process in the "after frameset" insertion mode. |
|
|
} else { |
|
|
!!!cp ('t317'); |
|
|
} |
|
|
|
|
| 6342 |
if ($token->{tag_name} eq 'frameset' and |
if ($token->{tag_name} eq 'frameset' and |
| 6343 |
$self->{insertion_mode} == IN_FRAMESET_IM) { |
$self->{insertion_mode} == IN_FRAMESET_IM) { |
| 6344 |
!!!cp ('t318'); |
!!!cp ('t318'); |
| 6359 |
## NOTE: As if in head. |
## NOTE: As if in head. |
| 6360 |
$parse_rcdata->(CDATA_CONTENT_MODEL); |
$parse_rcdata->(CDATA_CONTENT_MODEL); |
| 6361 |
next B; |
next B; |
| 6362 |
|
|
| 6363 |
|
## NOTE: |<!DOCTYPE HTML><frameset></frameset></html><noframes></noframes>| |
| 6364 |
|
## has no parse error. |
| 6365 |
} else { |
} else { |
| 6366 |
if ($self->{insertion_mode} == IN_FRAMESET_IM) { |
if ($self->{insertion_mode} == IN_FRAMESET_IM) { |
| 6367 |
!!!cp ('t321'); |
!!!cp ('t321'); |
| 6368 |
!!!parse-error (type => 'in frameset', |
!!!parse-error (type => 'in frameset', |
| 6369 |
text => $token->{tag_name}, token => $token); |
text => $token->{tag_name}, token => $token); |
| 6370 |
} else { |
} elsif ($self->{insertion_mode} == AFTER_FRAMESET_IM) { |
| 6371 |
!!!cp ('t322'); |
!!!cp ('t322'); |
| 6372 |
!!!parse-error (type => 'after frameset', |
!!!parse-error (type => 'after frameset', |
| 6373 |
text => $token->{tag_name}, token => $token); |
text => $token->{tag_name}, token => $token); |
| 6374 |
|
} else { # "after after frameset" |
| 6375 |
|
!!!cp ('t322.2'); |
| 6376 |
|
!!!parse-error (type => 'after after frameset', |
| 6377 |
|
text => $token->{tag_name}, token => $token); |
| 6378 |
} |
} |
| 6379 |
## Ignore the token |
## Ignore the token |
| 6380 |
!!!nack ('t322.1'); |
!!!nack ('t322.1'); |
| 6382 |
next B; |
next B; |
| 6383 |
} |
} |
| 6384 |
} elsif ($token->{type} == END_TAG_TOKEN) { |
} elsif ($token->{type} == END_TAG_TOKEN) { |
|
if ($self->{insertion_mode} == AFTER_HTML_FRAMESET_IM) { |
|
|
!!!cp ('t323'); |
|
|
!!!parse-error (type => 'after html:/', |
|
|
text => $token->{tag_name}, token => $token); |
|
|
|
|
|
$self->{insertion_mode} = AFTER_FRAMESET_IM; |
|
|
## Process in the "after frameset" insertion mode. |
|
|
} else { |
|
|
!!!cp ('t324'); |
|
|
} |
|
|
|
|
| 6385 |
if ($token->{tag_name} eq 'frameset' and |
if ($token->{tag_name} eq 'frameset' and |
| 6386 |
$self->{insertion_mode} == IN_FRAMESET_IM) { |
$self->{insertion_mode} == IN_FRAMESET_IM) { |
| 6387 |
if ($self->{open_elements}->[-1]->[1] & HTML_EL and |
if ($self->{open_elements}->[-1]->[1] & HTML_EL and |
| 6416 |
!!!cp ('t330'); |
!!!cp ('t330'); |
| 6417 |
!!!parse-error (type => 'in frameset:/', |
!!!parse-error (type => 'in frameset:/', |
| 6418 |
text => $token->{tag_name}, token => $token); |
text => $token->{tag_name}, token => $token); |
| 6419 |
} else { |
} elsif ($self->{insertion_mode} == AFTER_FRAMESET_IM) { |
| 6420 |
!!!cp ('t331'); |
!!!cp ('t330.1'); |
| 6421 |
!!!parse-error (type => 'after frameset:/', |
!!!parse-error (type => 'after frameset:/', |
| 6422 |
text => $token->{tag_name}, token => $token); |
text => $token->{tag_name}, token => $token); |
| 6423 |
|
} else { # "after after html" |
| 6424 |
|
!!!cp ('t331'); |
| 6425 |
|
!!!parse-error (type => 'after after frameset:/', |
| 6426 |
|
text => $token->{tag_name}, token => $token); |
| 6427 |
} |
} |
| 6428 |
## Ignore the token |
## Ignore the token |
| 6429 |
!!!next-token; |
!!!next-token; |
| 7146 |
!!!cp ('t413'); |
!!!cp ('t413'); |
| 7147 |
!!!parse-error (type => 'unmatched end tag', |
!!!parse-error (type => 'unmatched end tag', |
| 7148 |
text => $token->{tag_name}, token => $token); |
text => $token->{tag_name}, token => $token); |
| 7149 |
|
## NOTE: Ignore the token. |
| 7150 |
} else { |
} else { |
| 7151 |
## Step 1. generate implied end tags |
## Step 1. generate implied end tags |
| 7152 |
while ({ |
while ({ |
| 7206 |
!!!cp ('t421'); |
!!!cp ('t421'); |
| 7207 |
!!!parse-error (type => 'unmatched end tag', |
!!!parse-error (type => 'unmatched end tag', |
| 7208 |
text => $token->{tag_name}, token => $token); |
text => $token->{tag_name}, token => $token); |
| 7209 |
|
## NOTE: Ignore the token. |
| 7210 |
} else { |
} else { |
| 7211 |
## Step 1. generate implied end tags |
## Step 1. generate implied end tags |
| 7212 |
while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) { |
while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) { |
| 7253 |
!!!cp ('t425.1'); |
!!!cp ('t425.1'); |
| 7254 |
!!!parse-error (type => 'unmatched end tag', |
!!!parse-error (type => 'unmatched end tag', |
| 7255 |
text => $token->{tag_name}, token => $token); |
text => $token->{tag_name}, token => $token); |
| 7256 |
|
## NOTE: Ignore the token. |
| 7257 |
} else { |
} else { |
| 7258 |
## Step 1. generate implied end tags |
## Step 1. generate implied end tags |
| 7259 |
while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) { |
while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) { |