| 323 |
|
|
| 324 |
## ISSUE: xmlns:xlink="non-xlink-ns" is not an error. |
## ISSUE: xmlns:xlink="non-xlink-ns" is not an error. |
| 325 |
|
|
| 326 |
my $c1_entity_char = { |
my $charref_map = { |
| 327 |
|
0x0D => 0x000A, |
| 328 |
0x80 => 0x20AC, |
0x80 => 0x20AC, |
| 329 |
0x81 => 0xFFFD, |
0x81 => 0xFFFD, |
| 330 |
0x82 => 0x201A, |
0x82 => 0x201A, |
| 357 |
0x9D => 0xFFFD, |
0x9D => 0xFFFD, |
| 358 |
0x9E => 0x017E, |
0x9E => 0x017E, |
| 359 |
0x9F => 0x0178, |
0x9F => 0x0178, |
| 360 |
}; # $c1_entity_char |
}; # $charref_map |
| 361 |
|
$charref_map->{$_} = 0xFFFD |
| 362 |
|
for 0x0000..0x0008, 0x000B, 0x000E..0x001F, 0x007F, |
| 363 |
|
0xD800..0xDFFF, 0xFDD0..0xFDDF, ## ISSUE: 0xFDEF |
| 364 |
|
0xFFFE, 0xFFFF, 0x1FFFE, 0x1FFFF, 0x2FFFE, 0x2FFFF, 0x3FFFE, 0x3FFFF, |
| 365 |
|
0x4FFFE, 0x4FFFF, 0x5FFFE, 0x5FFFF, 0x6FFFE, 0x6FFFF, 0x7FFFE, |
| 366 |
|
0x7FFFF, 0x8FFFE, 0x8FFFF, 0x9FFFE, 0x9FFFF, 0xAFFFE, 0xAFFFF, |
| 367 |
|
0xBFFFE, 0xBFFFF, 0xCFFFE, 0xCFFFF, 0xDFFFE, 0xDFFFF, 0xEFFFE, |
| 368 |
|
0xEFFFF, 0xFFFFE, 0xFFFFF, 0x10FFFE, 0x10FFFF; |
| 369 |
|
|
| 370 |
sub parse_byte_string ($$$$;$) { |
sub parse_byte_string ($$$$;$) { |
| 371 |
my $self = shift; |
my $self = shift; |
| 410 |
## TODO: Is this ok? Transfer protocol's parameter should be |
## TODO: Is this ok? Transfer protocol's parameter should be |
| 411 |
## interpreted in its semantics? |
## interpreted in its semantics? |
| 412 |
|
|
|
## ISSUE: Unsupported encoding is not ignored according to the spec. |
|
| 413 |
($char_stream, $e_status) = $charset->get_decode_handle |
($char_stream, $e_status) = $charset->get_decode_handle |
| 414 |
($byte_stream, allow_error_reporting => 1, |
($byte_stream, allow_error_reporting => 1, |
| 415 |
allow_fallback => 1); |
allow_fallback => 1); |
| 417 |
$self->{confident} = 1; |
$self->{confident} = 1; |
| 418 |
last SNIFFING; |
last SNIFFING; |
| 419 |
} else { |
} else { |
| 420 |
## TODO: unsupported error |
!!!parse-error (type => 'charset:not supported', |
| 421 |
|
layer => 'encode', |
| 422 |
|
line => 1, column => 1, |
| 423 |
|
value => $charset_name, |
| 424 |
|
level => $self->{level}->{uncertain}); |
| 425 |
} |
} |
| 426 |
} |
} |
| 427 |
|
|
| 3183 |
my $code = $self->{s_kwd}; |
my $code = $self->{s_kwd}; |
| 3184 |
my $l = $self->{line_prev}; |
my $l = $self->{line_prev}; |
| 3185 |
my $c = $self->{column_prev}; |
my $c = $self->{column_prev}; |
| 3186 |
if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) { |
if ($charref_map->{$code}) { |
| 3187 |
!!!cp (1015); |
!!!cp (1015); |
| 3188 |
!!!parse-error (type => 'invalid character reference', |
!!!parse-error (type => 'invalid character reference', |
| 3189 |
text => (sprintf 'U+%04X', $code), |
text => (sprintf 'U+%04X', $code), |
| 3190 |
line => $l, column => $c); |
line => $l, column => $c); |
| 3191 |
$code = 0xFFFD; |
$code = $charref_map->{$code}; |
| 3192 |
} elsif ($code > 0x10FFFF) { |
} elsif ($code > 0x10FFFF) { |
| 3193 |
!!!cp (1016); |
!!!cp (1016); |
| 3194 |
!!!parse-error (type => 'invalid character reference', |
!!!parse-error (type => 'invalid character reference', |
| 3195 |
text => (sprintf 'U-%08X', $code), |
text => (sprintf 'U-%08X', $code), |
| 3196 |
line => $l, column => $c); |
line => $l, column => $c); |
| 3197 |
$code = 0xFFFD; |
$code = 0xFFFD; |
|
} elsif ($code == 0x000D) { |
|
|
!!!cp (1017); |
|
|
!!!parse-error (type => 'CR character reference', |
|
|
line => $l, column => $c); |
|
|
$code = 0x000A; |
|
|
} elsif (0x80 <= $code and $code <= 0x9F) { |
|
|
!!!cp (1018); |
|
|
!!!parse-error (type => 'C1 character reference', |
|
|
text => (sprintf 'U+%04X', $code), |
|
|
line => $l, column => $c); |
|
|
$code = $c1_entity_char->{$code}; |
|
| 3198 |
} |
} |
| 3199 |
|
|
| 3200 |
if ($self->{prev_state} == DATA_STATE) { |
if ($self->{prev_state} == DATA_STATE) { |
| 3291 |
my $code = $self->{s_kwd}; |
my $code = $self->{s_kwd}; |
| 3292 |
my $l = $self->{line_prev}; |
my $l = $self->{line_prev}; |
| 3293 |
my $c = $self->{column_prev}; |
my $c = $self->{column_prev}; |
| 3294 |
if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) { |
if ($charref_map->{$code}) { |
| 3295 |
!!!cp (1008); |
!!!cp (1008); |
| 3296 |
!!!parse-error (type => 'invalid character reference', |
!!!parse-error (type => 'invalid character reference', |
| 3297 |
text => (sprintf 'U+%04X', $code), |
text => (sprintf 'U+%04X', $code), |
| 3298 |
line => $l, column => $c); |
line => $l, column => $c); |
| 3299 |
$code = 0xFFFD; |
$code = $charref_map->{$code}; |
| 3300 |
} elsif ($code > 0x10FFFF) { |
} elsif ($code > 0x10FFFF) { |
| 3301 |
!!!cp (1009); |
!!!cp (1009); |
| 3302 |
!!!parse-error (type => 'invalid character reference', |
!!!parse-error (type => 'invalid character reference', |
| 3303 |
text => (sprintf 'U-%08X', $code), |
text => (sprintf 'U-%08X', $code), |
| 3304 |
line => $l, column => $c); |
line => $l, column => $c); |
| 3305 |
$code = 0xFFFD; |
$code = 0xFFFD; |
|
} elsif ($code == 0x000D) { |
|
|
!!!cp (1010); |
|
|
!!!parse-error (type => 'CR character reference', line => $l, column => $c); |
|
|
$code = 0x000A; |
|
|
} elsif (0x80 <= $code and $code <= 0x9F) { |
|
|
!!!cp (1011); |
|
|
!!!parse-error (type => 'C1 character reference', text => (sprintf 'U+%04X', $code), line => $l, column => $c); |
|
|
$code = $c1_entity_char->{$code}; |
|
| 3306 |
} |
} |
| 3307 |
|
|
| 3308 |
if ($self->{prev_state} == DATA_STATE) { |
if ($self->{prev_state} == DATA_STATE) { |
| 6809 |
} elsif ($token->{attributes}->{content}) { |
} elsif ($token->{attributes}->{content}) { |
| 6810 |
if ($token->{attributes}->{content}->{value} |
if ($token->{attributes}->{content}->{value} |
| 6811 |
=~ /[Cc][Hh][Aa][Rr][Ss][Ee][Tt] |
=~ /[Cc][Hh][Aa][Rr][Ss][Ee][Tt] |
| 6812 |
[\x09-\x0D\x20]*= |
[\x09\x0A\x0C\x0D\x20]*= |
| 6813 |
[\x09-\x0D\x20]*(?>"([^"]*)"|'([^']*)'| |
[\x09\x0A\x0C\x0D\x20]*(?>"([^"]*)"|'([^']*)'| |
| 6814 |
([^"'\x09-\x0D\x20][^\x09-\x0D\x20\x3B]*))/x) { |
([^"'\x09\x0A\x0C\x0D\x20][^\x09\x0A\x0C\x0D\x20\x3B]*)) |
| 6815 |
|
/x) { |
| 6816 |
!!!cp ('t336'); |
!!!cp ('t336'); |
| 6817 |
## NOTE: Whether the encoding is supported or not is handled |
## NOTE: Whether the encoding is supported or not is handled |
| 6818 |
## in the {change_encoding} callback. |
## in the {change_encoding} callback. |