| 278 |
zeta => "\x{03B6}", |
zeta => "\x{03B6}", |
| 279 |
zwj => "\x{200D}", |
zwj => "\x{200D}", |
| 280 |
zwnj => "\x{200C}", |
zwnj => "\x{200C}", |
| 281 |
}; |
}; # $entity_char |
| 282 |
|
|
| 283 |
|
my $c1_entity_char = { |
| 284 |
|
0x80 => 0x20AC, |
| 285 |
|
0x81 => 0xFFFD, |
| 286 |
|
0x82 => 0x201A, |
| 287 |
|
0x83 => 0x0192, |
| 288 |
|
0x84 => 0x201E, |
| 289 |
|
0x85 => 0x2026, |
| 290 |
|
0x86 => 0x2020, |
| 291 |
|
0x87 => 0x2021, |
| 292 |
|
0x88 => 0x02C6, |
| 293 |
|
0x89 => 0x2030, |
| 294 |
|
0x8A => 0x0160, |
| 295 |
|
0x8B => 0x2039, |
| 296 |
|
0x8C => 0x0152, |
| 297 |
|
0x8D => 0xFFFD, |
| 298 |
|
0x8E => 0x017D, |
| 299 |
|
0x8F => 0xFFFD, |
| 300 |
|
0x90 => 0xFFFD, |
| 301 |
|
0x91 => 0x2018, |
| 302 |
|
0x92 => 0x2019, |
| 303 |
|
0x93 => 0x201C, |
| 304 |
|
0x94 => 0x201D, |
| 305 |
|
0x95 => 0x2022, |
| 306 |
|
0x96 => 0x2013, |
| 307 |
|
0x97 => 0x2014, |
| 308 |
|
0x98 => 0x02DC, |
| 309 |
|
0x99 => 0x2122, |
| 310 |
|
0x9A => 0x0161, |
| 311 |
|
0x9B => 0x203A, |
| 312 |
|
0x9C => 0x0153, |
| 313 |
|
0x9D => 0xFFFD, |
| 314 |
|
0x9E => 0x017E, |
| 315 |
|
0x9F => 0x0178, |
| 316 |
|
}; # $c1_entity_char |
| 317 |
|
|
| 318 |
my $special_category = { |
my $special_category = { |
| 319 |
address => 1, area => 1, base => 1, basefont => 1, bgsound => 1, |
address => 1, area => 1, base => 1, basefont => 1, bgsound => 1, |
| 353 |
$self->{next_input_character} = ord substr $$s, $i++, 1; |
$self->{next_input_character} = ord substr $$s, $i++, 1; |
| 354 |
$column++; |
$column++; |
| 355 |
|
|
| 356 |
if ($self->{next_input_character} == 0x000D) { # CR |
if ($self->{next_input_character} == 0x000A) { # LF |
| 357 |
|
$line++; |
| 358 |
|
$column = 0; |
| 359 |
|
} elsif ($self->{next_input_character} == 0x000D) { # CR |
| 360 |
if ($i >= length $$s) { |
if ($i >= length $$s) { |
| 361 |
# |
# |
| 362 |
} else { |
} else { |
| 369 |
} |
} |
| 370 |
$self->{next_input_character} = 0x000A; # LF # MUST |
$self->{next_input_character} = 0x000A; # LF # MUST |
| 371 |
$line++; |
$line++; |
| 372 |
$column = -1; |
$column = 0; |
| 373 |
} elsif ($self->{next_input_character} > 0x10FFFF) { |
} elsif ($self->{next_input_character} > 0x10FFFF) { |
| 374 |
$self->{next_input_character} = 0xFFFD; # REPLACEMENT CHARACTER # MUST |
$self->{next_input_character} = 0xFFFD; # REPLACEMENT CHARACTER # MUST |
| 375 |
} elsif ($self->{next_input_character} == 0x0000) { # NULL |
} elsif ($self->{next_input_character} == 0x0000) { # NULL |
| 376 |
|
!!!parse-error (type => 'NULL'); |
| 377 |
|
## TODO: test |
| 378 |
$self->{next_input_character} = 0xFFFD; # REPLACEMENT CHARACTER # MUST |
$self->{next_input_character} = 0xFFFD; # REPLACEMENT CHARACTER # MUST |
| 379 |
} |
} |
| 380 |
}; |
}; |
| 445 |
## has completed loading. If one has, then it MUST be executed |
## has completed loading. If one has, then it MUST be executed |
| 446 |
## and removed from the list. |
## and removed from the list. |
| 447 |
|
|
| 448 |
|
## ISSUE: <http://html5.org/tools/web-apps-tracker?from=874&to=876> |
| 449 |
|
|
| 450 |
sub _get_next_token ($) { |
sub _get_next_token ($) { |
| 451 |
my $self = shift; |
my $self = shift; |
| 452 |
if (@{$self->{token}}) { |
if (@{$self->{token}}) { |
| 1353 |
redo A; |
redo A; |
| 1354 |
} elsif (0x0061 <= $self->{next_input_character} and |
} elsif (0x0061 <= $self->{next_input_character} and |
| 1355 |
$self->{next_input_character} <= 0x007A) { # a..z |
$self->{next_input_character} <= 0x007A) { # a..z |
| 1356 |
|
## ISSUE: "Set the token's name name to the" in the spec |
| 1357 |
$self->{current_token} = {type => 'DOCTYPE', |
$self->{current_token} = {type => 'DOCTYPE', |
| 1358 |
name => chr ($self->{next_input_character} - 0x0020), |
name => chr ($self->{next_input_character} - 0x0020), |
| 1359 |
error => 1}; |
error => 1}; |
| 1380 |
$self->{current_token} = {type => 'DOCTYPE', |
$self->{current_token} = {type => 'DOCTYPE', |
| 1381 |
name => chr ($self->{next_input_character}), |
name => chr ($self->{next_input_character}), |
| 1382 |
error => 1}; |
error => 1}; |
| 1383 |
|
## ISSUE: "Set the token's name name to the" in the spec |
| 1384 |
$self->{state} = 'DOCTYPE name'; |
$self->{state} = 'DOCTYPE name'; |
| 1385 |
!!!next-input-character; |
!!!next-input-character; |
| 1386 |
redo A; |
redo A; |
| 1498 |
|
|
| 1499 |
if ($self->{next_input_character} == 0x0023) { # # |
if ($self->{next_input_character} == 0x0023) { # # |
| 1500 |
!!!next-input-character; |
!!!next-input-character; |
|
my $num; |
|
| 1501 |
if ($self->{next_input_character} == 0x0078 or # x |
if ($self->{next_input_character} == 0x0078 or # x |
| 1502 |
$self->{next_input_character} == 0x0058) { # X |
$self->{next_input_character} == 0x0058) { # X |
| 1503 |
|
my $num; |
| 1504 |
X: { |
X: { |
| 1505 |
my $x_char = $self->{next_input_character}; |
my $x_char = $self->{next_input_character}; |
| 1506 |
!!!next-input-character; |
!!!next-input-character; |
| 1536 |
} |
} |
| 1537 |
|
|
| 1538 |
## TODO: check the definition for |a valid Unicode character|. |
## TODO: check the definition for |a valid Unicode character|. |
| 1539 |
|
## <http://lists.whatwg.org/pipermail/whatwg-whatwg.org/2006-December/thread.html#8189> |
| 1540 |
if ($num > 1114111 or $num == 0) { |
if ($num > 1114111 or $num == 0) { |
| 1541 |
$num = 0xFFFD; # REPLACEMENT CHARACTER |
$num = 0xFFFD; # REPLACEMENT CHARACTER |
| 1542 |
## ISSUE: Why this is not an error? |
## ISSUE: Why this is not an error? |
| 1543 |
|
} elsif (0x80 <= $num and $num <= 0x9F) { |
| 1544 |
|
!!!parse-error (type => sprintf 'c1 entity:U+%04X', $num); |
| 1545 |
|
$num = $c1_entity_char->{$num}; |
| 1546 |
} |
} |
| 1547 |
|
|
| 1548 |
return {type => 'character', data => chr $num}; |
return {type => 'character', data => chr $num}; |
| 1570 |
if ($code > 1114111 or $code == 0) { |
if ($code > 1114111 or $code == 0) { |
| 1571 |
$code = 0xFFFD; # REPLACEMENT CHARACTER |
$code = 0xFFFD; # REPLACEMENT CHARACTER |
| 1572 |
## ISSUE: Why this is not an error? |
## ISSUE: Why this is not an error? |
| 1573 |
|
} elsif (0x80 <= $code and $code <= 0x9F) { |
| 1574 |
|
!!!parse-error (type => sprintf 'c1 entity:U+%04X', $code); |
| 1575 |
|
$code = $c1_entity_char->{$code}; |
| 1576 |
} |
} |
| 1577 |
|
|
| 1578 |
return {type => 'character', data => chr $code}; |
return {type => 'character', data => chr $code}; |
| 1921 |
}; # $clear_up_to_marker |
}; # $clear_up_to_marker |
| 1922 |
|
|
| 1923 |
my $style_start_tag = sub { |
my $style_start_tag = sub { |
| 1924 |
my $style_el; !!!create-element ($style_el, 'style'); |
my $style_el; !!!create-element ($style_el, 'style', $token->{attributes}); |
| 1925 |
## $self->{insertion_mode} eq 'in head' and ... (always true) |
## $self->{insertion_mode} eq 'in head' and ... (always true) |
| 1926 |
(($self->{insertion_mode} eq 'in head' and defined $self->{head_element}) |
(($self->{insertion_mode} eq 'in head' and defined $self->{head_element}) |
| 1927 |
? $self->{head_element} : $self->{open_elements}->[-1]->[0]) |
? $self->{head_element} : $self->{open_elements}->[-1]->[0]) |
| 2026 |
$formatting_element_i_in_open = $_; |
$formatting_element_i_in_open = $_; |
| 2027 |
last INSCOPE; |
last INSCOPE; |
| 2028 |
} else { # in open elements but not in scope |
} else { # in open elements but not in scope |
| 2029 |
!!!parse-error; |
!!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}); |
| 2030 |
## Ignore the token |
## Ignore the token |
| 2031 |
!!!next-token; |
!!!next-token; |
| 2032 |
return; |
return; |
| 2039 |
} |
} |
| 2040 |
} # INSCOPE |
} # INSCOPE |
| 2041 |
unless (defined $formatting_element_i_in_open) { |
unless (defined $formatting_element_i_in_open) { |
| 2042 |
!!!parse-error; |
!!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}); |
| 2043 |
pop @$active_formatting_elements; # $formatting_element |
pop @$active_formatting_elements; # $formatting_element |
| 2044 |
!!!next-token; ## TODO: ok? |
!!!next-token; ## TODO: ok? |
| 2045 |
return; |
return; |
| 2046 |
} |
} |
| 2047 |
if (not $self->{open_elements}->[-1]->[0] eq $formatting_element->[0]) { |
if (not $self->{open_elements}->[-1]->[0] eq $formatting_element->[0]) { |
| 2048 |
!!!parse-error; |
!!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]); |
| 2049 |
} |
} |
| 2050 |
|
|
| 2051 |
## Step 2 |
## Step 2 |
| 2321 |
if (defined $self->{form_element}) { |
if (defined $self->{form_element}) { |
| 2322 |
!!!parse-error (type => 'in form:form'); |
!!!parse-error (type => 'in form:form'); |
| 2323 |
## Ignore the token |
## Ignore the token |
| 2324 |
|
!!!next-token; |
| 2325 |
|
return; |
| 2326 |
} else { |
} else { |
| 2327 |
## has a p element in scope |
## has a p element in scope |
| 2328 |
INSCOPE: for (reverse @{$self->{open_elements}}) { |
INSCOPE: for (reverse @{$self->{open_elements}}) { |
| 2364 |
LI: { |
LI: { |
| 2365 |
## Step 2 |
## Step 2 |
| 2366 |
if ($node->[1] eq 'li') { |
if ($node->[1] eq 'li') { |
| 2367 |
|
if ($i != -1) { |
| 2368 |
|
!!!parse-error (type => 'end tag missing:'. |
| 2369 |
|
$self->{open_elements}->[-1]->[1]); |
| 2370 |
|
## TODO: test |
| 2371 |
|
} |
| 2372 |
splice @{$self->{open_elements}}, $i; |
splice @{$self->{open_elements}}, $i; |
| 2373 |
last LI; |
last LI; |
| 2374 |
} |
} |
| 2412 |
LI: { |
LI: { |
| 2413 |
## Step 2 |
## Step 2 |
| 2414 |
if ($node->[1] eq 'dt' or $node->[1] eq 'dd') { |
if ($node->[1] eq 'dt' or $node->[1] eq 'dd') { |
| 2415 |
|
if ($i != -1) { |
| 2416 |
|
!!!parse-error (type => 'end tag missing:'. |
| 2417 |
|
$self->{open_elements}->[-1]->[1]); |
| 2418 |
|
## TODO: test |
| 2419 |
|
} |
| 2420 |
splice @{$self->{open_elements}}, $i; |
splice @{$self->{open_elements}}, $i; |
| 2421 |
last LI; |
last LI; |
| 2422 |
} |
} |
| 2691 |
} |
} |
| 2692 |
} elsif ({ |
} elsif ({ |
| 2693 |
textarea => 1, |
textarea => 1, |
| 2694 |
|
iframe => 1, |
| 2695 |
noembed => 1, |
noembed => 1, |
| 2696 |
noframes => 1, |
noframes => 1, |
| 2697 |
noscript => 0, ## TODO: 1 if scripting is enabled |
noscript => 0, ## TODO: 1 if scripting is enabled |
| 2710 |
$insert->($el); |
$insert->($el); |
| 2711 |
|
|
| 2712 |
my $text = ''; |
my $text = ''; |
| 2713 |
!!!next-token; |
if ($token->{tag_name} eq 'textarea') { |
| 2714 |
|
!!!next-token; |
| 2715 |
|
if ($token->{type} eq 'character') { |
| 2716 |
|
$token->{data} =~ s/^\x0A//; |
| 2717 |
|
unless (length $token->{data}) { |
| 2718 |
|
!!!next-token; |
| 2719 |
|
} |
| 2720 |
|
} |
| 2721 |
|
} else { |
| 2722 |
|
!!!next-token; |
| 2723 |
|
} |
| 2724 |
while ($token->{type} eq 'character') { |
while ($token->{type} eq 'character') { |
| 2725 |
$text .= $token->{data}; |
$text .= $token->{data}; |
| 2726 |
!!!next-token; |
!!!next-token; |
| 2736 |
## Ignore the token |
## Ignore the token |
| 2737 |
} else { |
} else { |
| 2738 |
if ($token->{tag_name} eq 'textarea') { |
if ($token->{tag_name} eq 'textarea') { |
|
!!!parse-error (type => 'in CDATA:#'.$token->{type}); |
|
|
} else { |
|
| 2739 |
!!!parse-error (type => 'in RCDATA:#'.$token->{type}); |
!!!parse-error (type => 'in RCDATA:#'.$token->{type}); |
| 2740 |
|
} else { |
| 2741 |
|
!!!parse-error (type => 'in CDATA:#'.$token->{type}); |
| 2742 |
} |
} |
| 2743 |
## ISSUE: And ignore? |
## ISSUE: And ignore? |
| 2744 |
} |
} |
| 2896 |
strong => 1, tt => 1, u => 1, |
strong => 1, tt => 1, u => 1, |
| 2897 |
}->{$token->{tag_name}}) { |
}->{$token->{tag_name}}) { |
| 2898 |
$formatting_end_tag->($token->{tag_name}); |
$formatting_end_tag->($token->{tag_name}); |
| 2899 |
|
## TODO: <http://html5.org/tools/web-apps-tracker?from=883&to=884> |
| 2900 |
return; |
return; |
| 2901 |
} elsif ({ |
} elsif ({ |
| 2902 |
caption => 1, col => 1, colgroup => 1, frame => 1, |
caption => 1, col => 1, colgroup => 1, frame => 1, |
| 2905 |
thead => 1, tr => 1, |
thead => 1, tr => 1, |
| 2906 |
area => 1, basefont => 1, bgsound => 1, br => 1, |
area => 1, basefont => 1, bgsound => 1, br => 1, |
| 2907 |
embed => 1, hr => 1, iframe => 1, image => 1, |
embed => 1, hr => 1, iframe => 1, image => 1, |
| 2908 |
img => 1, input => 1, isindex=> 1, noembed => 1, |
img => 1, input => 1, isindex => 1, noembed => 1, |
| 2909 |
noframes => 1, param => 1, select => 1, spacer => 1, |
noframes => 1, param => 1, select => 1, spacer => 1, |
| 2910 |
table => 1, textarea => 1, wbr => 1, |
table => 1, textarea => 1, wbr => 1, |
| 2911 |
noscript => 0, ## TODO: if scripting is enabled |
noscript => 0, ## TODO: if scripting is enabled |
| 4882 |
$self->{next_input_character} = -1 and return if $i >= length $$s; |
$self->{next_input_character} = -1 and return if $i >= length $$s; |
| 4883 |
$self->{next_input_character} = ord substr $$s, $i++, 1; |
$self->{next_input_character} = ord substr $$s, $i++, 1; |
| 4884 |
$column++; |
$column++; |
| 4885 |
|
|
| 4886 |
if ($self->{next_input_character} == 0x000D) { # CR |
if ($self->{next_input_character} == 0x000A) { # LF |
| 4887 |
|
$line++; |
| 4888 |
|
$column = 0; |
| 4889 |
|
} elsif ($self->{next_input_character} == 0x000D) { # CR |
| 4890 |
if ($i >= length $$s) { |
if ($i >= length $$s) { |
| 4891 |
# |
# |
| 4892 |
} else { |
} else { |
| 4899 |
} |
} |
| 4900 |
$self->{next_input_character} = 0x000A; # LF # MUST |
$self->{next_input_character} = 0x000A; # LF # MUST |
| 4901 |
$line++; |
$line++; |
| 4902 |
$column = -1; |
$column = 0; |
| 4903 |
} elsif ($self->{next_input_character} > 0x10FFFF) { |
} elsif ($self->{next_input_character} > 0x10FFFF) { |
| 4904 |
$self->{next_input_character} = 0xFFFD; # REPLACEMENT CHARACTER # MUST |
$self->{next_input_character} = 0xFFFD; # REPLACEMENT CHARACTER # MUST |
| 4905 |
} elsif ($self->{next_input_character} == 0x0000) { # NULL |
} elsif ($self->{next_input_character} == 0x0000) { # NULL |