| 603 |
|
|
| 604 |
if ($error) { |
if ($error) { |
| 605 |
$r = substr $self->{byte_buffer}, 0, 1, ''; |
$r = substr $self->{byte_buffer}, 0, 1, ''; |
| 606 |
$self->{onerror}->($self, 'illegal-octets-error', octets => \$r); |
my $fallback = $self->{fallback}->{$r}; |
| 607 |
|
if (defined $fallback) { |
| 608 |
|
## NOTE: This is an HTML5 parse error. Applied to Web ISO-8859-1 |
| 609 |
|
## and Web ISO-8859-11 encodings. |
| 610 |
|
$self->{onerror}->($self, 'fallback-char-error', octets => \$r, |
| 611 |
|
char => \$fallback, |
| 612 |
|
level => $self->{must_level}); |
| 613 |
|
return $fallback; |
| 614 |
|
} else { |
| 615 |
|
$self->{onerror}->($self, 'illegal-octets-error', octets => \$r); |
| 616 |
|
} |
| 617 |
} |
} |
| 618 |
|
|
| 619 |
return $r; |
return $r; |
| 873 |
if ($error) { |
if ($error) { |
| 874 |
$r = substr $self->{byte_buffer}, 0, 1, ''; |
$r = substr $self->{byte_buffer}, 0, 1, ''; |
| 875 |
my $etype = 'illegal-octets-error'; |
my $etype = 'illegal-octets-error'; |
| 876 |
if ($r =~ /^[\x81-\x9F\xE0-\xEF]/) { |
if ($r =~ /^[\x81-\x9F\xE0-\xFC]/) { |
| 877 |
if ($self->{byte_buffer} =~ s/(.)//s) { |
if ($self->{byte_buffer} =~ s/(.)//s) { |
| 878 |
$r .= $1; # not limited to \x40-\xFC - \x7F |
$r .= $1; # not limited to \x40-\xFC - \x7F |
| 879 |
$etype = 'unassigned-code-point-error'; |
$etype = 'unassigned-code-point-error'; |
| 880 |
} |
} |
| 881 |
} elsif ($r =~ /^[\x80\xA0\xF0-\xFF]/) { |
## NOTE: Range [\xF0-\xFC] is unassigned and may be used as a single-byte |
| 882 |
|
## character or as the first-byte of a double-byte character according |
| 883 |
|
## to JIS X 0208:1997 Appendix 1. However, the current practice is |
| 884 |
|
## use the range as the first-byte of double-byte characters. |
| 885 |
|
} elsif ($r =~ /^[\x80\xA0\xFD-\xFF]/) { |
| 886 |
$etype = 'unassigned-code-point-error'; |
$etype = 'unassigned-code-point-error'; |
| 887 |
} |
} |
| 888 |
$self->{onerror}->($self, $etype, octets => \$r); |
$self->{onerror}->($self, $etype, octets => \$r); |