| 3 |
our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r}; |
our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r}; |
| 4 |
use Error qw(:try); |
use Error qw(:try); |
| 5 |
|
|
| 6 |
|
## NOTE: This module don't check all HTML5 parse errors; character |
| 7 |
|
## encoding related parse errors are expected to be handled by relevant |
| 8 |
|
## modules. |
| 9 |
|
## Parse errors for control characters that are not allowed in HTML5 |
| 10 |
|
## documents, for surrogate code points, and for noncharacter code |
| 11 |
|
## points, as well as U+FFFD substitions for characters whose code points |
| 12 |
|
## is higher than U+10FFFF may be detected by combining the parser with |
| 13 |
|
## the checker implemented by Whatpm::Charset::UnicodeChecker (for its |
| 14 |
|
## usage example, see |t/HTML-tree.t| in the Whatpm package or the |
| 15 |
|
## WebHACC::Language::HTML module in the WebHACC package). |
| 16 |
|
|
| 17 |
## ISSUE: |
## ISSUE: |
| 18 |
## var doc = implementation.createDocument (null, null, null); |
## var doc = implementation.createDocument (null, null, null); |
| 19 |
## doc.write (''); |
## doc.write (''); |
| 507 |
line => 1, column => 1, |
line => 1, column => 1, |
| 508 |
layer => 'encode'); |
layer => 'encode'); |
| 509 |
} elsif (not ($e_status & |
} elsif (not ($e_status & |
| 510 |
Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL())) { |
Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL ())) { |
| 511 |
$self->{input_encoding} = $charset->get_iana_name; |
$self->{input_encoding} = $charset->get_iana_name; |
| 512 |
!!!parse-error (type => 'chardecode:no error', |
!!!parse-error (type => 'chardecode:no error', |
| 513 |
text => $self->{input_encoding}, |
text => $self->{input_encoding}, |
| 572 |
my $char_onerror = sub { |
my $char_onerror = sub { |
| 573 |
my (undef, $type, %opt) = @_; |
my (undef, $type, %opt) = @_; |
| 574 |
!!!parse-error (layer => 'encode', |
!!!parse-error (layer => 'encode', |
| 575 |
%opt, type => $type, |
line => $self->{line}, column => $self->{column} + 1, |
| 576 |
line => $self->{line}, column => $self->{column} + 1); |
%opt, type => $type); |
| 577 |
if ($opt{octets}) { |
if ($opt{octets}) { |
| 578 |
${$opt{octets}} = "\x{FFFD}"; # relacement character |
${$opt{octets}} = "\x{FFFD}"; # relacement character |
| 579 |
} |
} |
| 582 |
my $wrapped_char_stream = $get_wrapper->($char_stream); |
my $wrapped_char_stream = $get_wrapper->($char_stream); |
| 583 |
$wrapped_char_stream->onerror ($char_onerror); |
$wrapped_char_stream->onerror ($char_onerror); |
| 584 |
|
|
| 585 |
my @args = @_; shift @args; # $s |
my @args = ($_[1], $_[2]); # $doc, $onerror - $get_wrapper = undef; |
| 586 |
my $return; |
my $return; |
| 587 |
try { |
try { |
| 588 |
$return = $self->parse_char_stream ($wrapped_char_stream, @args); |
$return = $self->parse_char_stream ($wrapped_char_stream, @args); |
| 597 |
line => 1, column => 1, |
line => 1, column => 1, |
| 598 |
layer => 'encode'); |
layer => 'encode'); |
| 599 |
} elsif (not ($e_status & |
} elsif (not ($e_status & |
| 600 |
Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL())) { |
Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL ())) { |
| 601 |
$self->{input_encoding} = $charset->get_iana_name; |
$self->{input_encoding} = $charset->get_iana_name; |
| 602 |
!!!parse-error (type => 'chardecode:no error', |
!!!parse-error (type => 'chardecode:no error', |
| 603 |
text => $self->{input_encoding}, |
text => $self->{input_encoding}, |
| 629 |
sub parse_char_string ($$$;$$) { |
sub parse_char_string ($$$;$$) { |
| 630 |
#my ($self, $s, $doc, $onerror, $get_wrapper) = @_; |
#my ($self, $s, $doc, $onerror, $get_wrapper) = @_; |
| 631 |
my $self = shift; |
my $self = shift; |
|
require utf8; |
|
| 632 |
my $s = ref $_[0] ? $_[0] : \($_[0]); |
my $s = ref $_[0] ? $_[0] : \($_[0]); |
| 633 |
open my $input, '<' . (utf8::is_utf8 ($$s) ? ':utf8' : ''), $s; |
require Whatpm::Charset::DecodeHandle; |
| 634 |
if ($_[3]) { |
my $input = Whatpm::Charset::DecodeHandle::CharString->new ($s); |
|
$input = $_[3]->($input); |
|
|
} |
|
| 635 |
return $self->parse_char_stream ($input, @_[1..$#_]); |
return $self->parse_char_stream ($input, @_[1..$#_]); |
| 636 |
} # parse_char_string |
} # parse_char_string |
| 637 |
*parse_string = \&parse_char_string; ## NOTE: Alias for backward compatibility. |
*parse_string = \&parse_char_string; ## NOTE: Alias for backward compatibility. |
| 638 |
|
|
| 639 |
sub parse_char_stream ($$$;$) { |
sub parse_char_stream ($$$;$$) { |
| 640 |
my $self = ref $_[0] ? shift : shift->new; |
my $self = ref $_[0] ? shift : shift->new; |
| 641 |
my $input = $_[0]; |
my $input = $_[0]; |
| 642 |
$self->{document} = $_[1]; |
$self->{document} = $_[1]; |
| 647 |
$self->{confident} = 1 unless exists $self->{confident}; |
$self->{confident} = 1 unless exists $self->{confident}; |
| 648 |
$self->{document}->input_encoding ($self->{input_encoding}) |
$self->{document}->input_encoding ($self->{input_encoding}) |
| 649 |
if defined $self->{input_encoding}; |
if defined $self->{input_encoding}; |
| 650 |
|
## TODO: |{input_encoding}| is needless? |
| 651 |
|
|
|
my $i = 0; |
|
| 652 |
$self->{line_prev} = $self->{line} = 1; |
$self->{line_prev} = $self->{line} = 1; |
| 653 |
$self->{column_prev} = $self->{column} = 0; |
$self->{column_prev} = -1; |
| 654 |
|
$self->{column} = 0; |
| 655 |
$self->{set_next_char} = sub { |
$self->{set_next_char} = sub { |
| 656 |
my $self = shift; |
my $self = shift; |
| 657 |
|
|
| 658 |
pop @{$self->{prev_char}}; |
my $char = ''; |
|
unshift @{$self->{prev_char}}, $self->{next_char}; |
|
|
|
|
|
my $char; |
|
| 659 |
if (defined $self->{next_next_char}) { |
if (defined $self->{next_next_char}) { |
| 660 |
$char = $self->{next_next_char}; |
$char = $self->{next_next_char}; |
| 661 |
delete $self->{next_next_char}; |
delete $self->{next_next_char}; |
| 662 |
|
$self->{next_char} = ord $char; |
| 663 |
} else { |
} else { |
| 664 |
$char = $input->getc; |
$self->{char_buffer} = ''; |
| 665 |
|
$self->{char_buffer_pos} = 0; |
| 666 |
|
|
| 667 |
|
my $count = $input->manakai_read_until |
| 668 |
|
($self->{char_buffer}, qr/[^\x00\x0A\x0D]/, $self->{char_buffer_pos}); |
| 669 |
|
if ($count) { |
| 670 |
|
$self->{line_prev} = $self->{line}; |
| 671 |
|
$self->{column_prev} = $self->{column}; |
| 672 |
|
$self->{column}++; |
| 673 |
|
$self->{next_char} |
| 674 |
|
= ord substr ($self->{char_buffer}, $self->{char_buffer_pos}++, 1); |
| 675 |
|
return; |
| 676 |
|
} |
| 677 |
|
|
| 678 |
|
if ($input->read ($char, 1)) { |
| 679 |
|
$self->{next_char} = ord $char; |
| 680 |
|
} else { |
| 681 |
|
$self->{next_char} = -1; |
| 682 |
|
return; |
| 683 |
|
} |
| 684 |
} |
} |
|
$self->{next_char} = -1 and return unless defined $char; |
|
|
$self->{next_char} = ord $char; |
|
| 685 |
|
|
| 686 |
($self->{line_prev}, $self->{column_prev}) |
($self->{line_prev}, $self->{column_prev}) |
| 687 |
= ($self->{line}, $self->{column}); |
= ($self->{line}, $self->{column}); |
| 693 |
$self->{column} = 0; |
$self->{column} = 0; |
| 694 |
} elsif ($self->{next_char} == 0x000D) { # CR |
} elsif ($self->{next_char} == 0x000D) { # CR |
| 695 |
!!!cp ('j2'); |
!!!cp ('j2'); |
| 696 |
my $next = $input->getc; |
## TODO: support for abort/streaming |
| 697 |
if (defined $next and $next ne "\x0A") { |
my $next = ''; |
| 698 |
|
if ($input->read ($next, 1) and $next ne "\x0A") { |
| 699 |
$self->{next_next_char} = $next; |
$self->{next_next_char} = $next; |
| 700 |
} |
} |
| 701 |
$self->{next_char} = 0x000A; # LF # MUST |
$self->{next_char} = 0x000A; # LF # MUST |
| 702 |
$self->{line}++; |
$self->{line}++; |
| 703 |
$self->{column} = 0; |
$self->{column} = 0; |
|
} elsif ($self->{next_char} > 0x10FFFF) { |
|
|
!!!cp ('j3'); |
|
|
$self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST |
|
| 704 |
} elsif ($self->{next_char} == 0x0000) { # NULL |
} elsif ($self->{next_char} == 0x0000) { # NULL |
| 705 |
!!!cp ('j4'); |
!!!cp ('j4'); |
| 706 |
!!!parse-error (type => 'NULL'); |
!!!parse-error (type => 'NULL'); |
| 707 |
$self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST |
$self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST |
| 708 |
} elsif ($self->{next_char} <= 0x0008 or |
} |
| 709 |
(0x000E <= $self->{next_char} and $self->{next_char} <= 0x001F) or |
}; |
| 710 |
(0x007F <= $self->{next_char} and $self->{next_char} <= 0x009F) or |
|
| 711 |
(0xD800 <= $self->{next_char} and $self->{next_char} <= 0xDFFF) or |
$self->{read_until} = sub { |
| 712 |
(0xFDD0 <= $self->{next_char} and $self->{next_char} <= 0xFDDF) or |
#my ($scalar, $specials_range, $offset) = @_; |
| 713 |
{ |
return 0 if defined $self->{next_next_char}; |
| 714 |
0xFFFE => 1, 0xFFFF => 1, 0x1FFFE => 1, 0x1FFFF => 1, |
|
| 715 |
0x2FFFE => 1, 0x2FFFF => 1, 0x3FFFE => 1, 0x3FFFF => 1, |
my $pattern = qr/[^$_[1]\x00\x0A\x0D]/; |
| 716 |
0x4FFFE => 1, 0x4FFFF => 1, 0x5FFFE => 1, 0x5FFFF => 1, |
my $offset = $_[2] || 0; |
| 717 |
0x6FFFE => 1, 0x6FFFF => 1, 0x7FFFE => 1, 0x7FFFF => 1, |
|
| 718 |
0x8FFFE => 1, 0x8FFFF => 1, 0x9FFFE => 1, 0x9FFFF => 1, |
if ($self->{char_buffer_pos} < length $self->{char_buffer}) { |
| 719 |
0xAFFFE => 1, 0xAFFFF => 1, 0xBFFFE => 1, 0xBFFFF => 1, |
pos ($self->{char_buffer}) = $self->{char_buffer_pos}; |
| 720 |
0xCFFFE => 1, 0xCFFFF => 1, 0xDFFFE => 1, 0xDFFFF => 1, |
if ($self->{char_buffer} =~ /\G(?>$pattern)+/) { |
| 721 |
0xEFFFE => 1, 0xEFFFF => 1, 0xFFFFE => 1, 0xFFFFF => 1, |
substr ($_[0], $offset) |
| 722 |
0x10FFFE => 1, 0x10FFFF => 1, |
= substr ($self->{char_buffer}, $-[0], $+[0] - $-[0]); |
| 723 |
}->{$self->{next_char}}) { |
my $count = $+[0] - $-[0]; |
| 724 |
!!!cp ('j5'); |
if ($count) { |
| 725 |
if ($self->{next_char} < 0x10000) { |
$self->{column} += $count; |
| 726 |
!!!parse-error (type => 'control char', |
$self->{char_buffer_pos} += $count; |
| 727 |
text => (sprintf 'U+%04X', $self->{next_char})); |
$self->{line_prev} = $self->{line}; |
| 728 |
|
$self->{column_prev} = $self->{column} - 1; |
| 729 |
|
$self->{prev_char} = [-1, -1, -1]; |
| 730 |
|
$self->{next_char} = -1; |
| 731 |
|
} |
| 732 |
|
return $count; |
| 733 |
} else { |
} else { |
| 734 |
!!!parse-error (type => 'control char', |
return 0; |
|
text => (sprintf 'U-%08X', $self->{next_char})); |
|
| 735 |
} |
} |
| 736 |
|
} else { |
| 737 |
|
my $count = $input->manakai_read_until ($_[0], $pattern, $_[2]); |
| 738 |
|
if ($count) { |
| 739 |
|
$self->{column} += $count; |
| 740 |
|
$self->{line_prev} = $self->{line}; |
| 741 |
|
$self->{column_prev} = $self->{column} - 1; |
| 742 |
|
$self->{prev_char} = [-1, -1, -1]; |
| 743 |
|
$self->{next_char} = -1; |
| 744 |
|
} |
| 745 |
|
return $count; |
| 746 |
} |
} |
| 747 |
}; |
}; # $self->{read_until} |
|
$self->{prev_char} = [-1, -1, -1]; |
|
|
$self->{next_char} = -1; |
|
| 748 |
|
|
| 749 |
my $onerror = $_[2] || sub { |
my $onerror = $_[2] || sub { |
| 750 |
my (%opt) = @_; |
my (%opt) = @_; |
| 756 |
$onerror->(line => $self->{line}, column => $self->{column}, @_); |
$onerror->(line => $self->{line}, column => $self->{column}, @_); |
| 757 |
}; |
}; |
| 758 |
|
|
| 759 |
|
my $char_onerror = sub { |
| 760 |
|
my (undef, $type, %opt) = @_; |
| 761 |
|
!!!parse-error (layer => 'encode', |
| 762 |
|
line => $self->{line}, column => $self->{column} + 1, |
| 763 |
|
%opt, type => $type); |
| 764 |
|
}; # $char_onerror |
| 765 |
|
|
| 766 |
|
if ($_[3]) { |
| 767 |
|
$input = $_[3]->($input); |
| 768 |
|
$input->onerror ($char_onerror); |
| 769 |
|
} else { |
| 770 |
|
$input->onerror ($char_onerror) unless defined $input->onerror; |
| 771 |
|
} |
| 772 |
|
|
| 773 |
$self->_initialize_tokenizer; |
$self->_initialize_tokenizer; |
| 774 |
$self->_initialize_tree_constructor; |
$self->_initialize_tree_constructor; |
| 775 |
$self->_construct_tree; |
$self->_construct_tree; |
| 922 |
my $self = shift; |
my $self = shift; |
| 923 |
$self->{state} = DATA_STATE; # MUST |
$self->{state} = DATA_STATE; # MUST |
| 924 |
#$self->{state_keyword}; # initialized when used |
#$self->{state_keyword}; # initialized when used |
| 925 |
|
#$self->{entity__value}; # initialized when used |
| 926 |
|
#$self->{entity__match}; # initialized when used |
| 927 |
$self->{content_model} = PCDATA_CONTENT_MODEL; # be |
$self->{content_model} = PCDATA_CONTENT_MODEL; # be |
| 928 |
undef $self->{current_token}; |
undef $self->{current_token}; |
| 929 |
undef $self->{current_attribute}; |
undef $self->{current_attribute}; |
| 930 |
undef $self->{last_emitted_start_tag_name}; |
undef $self->{last_emitted_start_tag_name}; |
| 931 |
undef $self->{last_attribute_value_state}; |
#$self->{prev_state}; # initialized when used |
| 932 |
delete $self->{self_closing}; |
delete $self->{self_closing}; |
| 933 |
$self->{char} = []; |
$self->{char_buffer} = ''; |
| 934 |
# $self->{next_char} |
$self->{char_buffer_pos} = 0; |
| 935 |
|
$self->{prev_char} = [-1, -1, -1]; |
| 936 |
|
$self->{next_char} = -1; |
| 937 |
!!!next-input-character; |
!!!next-input-character; |
| 938 |
$self->{token} = []; |
$self->{token} = []; |
| 939 |
# $self->{escape} |
# $self->{escape} |
| 964 |
## has completed loading. If one has, then it MUST be executed |
## has completed loading. If one has, then it MUST be executed |
| 965 |
## and removed from the list. |
## and removed from the list. |
| 966 |
|
|
| 967 |
## NOTE: HTML5 "Writing HTML documents" section, applied to |
## TODO: Polytheistic slash SHOULD NOT be used. (Applied only to atheists.) |
| 968 |
## documents and not to user agents and conformance checkers, |
## (This requirement was dropped from HTML5 spec, unfortunately.) |
|
## contains some requirements that are not detected by the |
|
|
## parsing algorithm: |
|
|
## - Some requirements on character encoding declarations. ## TODO |
|
|
## - "Elements MUST NOT contain content that their content model disallows." |
|
|
## ... Some are parse error, some are not (will be reported by c.c.). |
|
|
## - Polytheistic slash SHOULD NOT be used. (Applied only to atheists.) ## TODO |
|
|
## - Text (in elements, attributes, and comments) SHOULD NOT contain |
|
|
## control characters other than space characters. ## TODO: (what is control character? C0, C1 and DEL? Unicode control character?) |
|
|
|
|
|
## TODO: HTML5 poses authors two SHOULD-level requirements that cannot |
|
|
## be detected by the HTML5 parsing algorithm: |
|
|
## - Text, |
|
| 969 |
|
|
| 970 |
sub _get_next_token ($) { |
sub _get_next_token ($) { |
| 971 |
my $self = shift; |
my $self = shift; |
| 993 |
## "entity data state". In this implementation, the tokenizer |
## "entity data state". In this implementation, the tokenizer |
| 994 |
## is switched to the |ENTITY_STATE|, which is an implementation |
## is switched to the |ENTITY_STATE|, which is an implementation |
| 995 |
## of the "consume a character reference" algorithm. |
## of the "consume a character reference" algorithm. |
|
$self->{entity_in_attr} = 0; |
|
| 996 |
$self->{entity_additional} = -1; |
$self->{entity_additional} = -1; |
| 997 |
|
$self->{prev_state} = DATA_STATE; |
| 998 |
$self->{state} = ENTITY_STATE; |
$self->{state} = ENTITY_STATE; |
| 999 |
!!!next-input-character; |
!!!next-input-character; |
| 1000 |
redo A; |
redo A; |
| 1059 |
data => chr $self->{next_char}, |
data => chr $self->{next_char}, |
| 1060 |
line => $self->{line}, column => $self->{column}, |
line => $self->{line}, column => $self->{column}, |
| 1061 |
}; |
}; |
| 1062 |
|
$self->{read_until}->($token->{data}, q[-!<>&], length $token->{data}); |
| 1063 |
|
|
| 1064 |
## Stay in the data state |
## Stay in the data state |
| 1065 |
!!!next-input-character; |
!!!next-input-character; |
| 1066 |
|
|
| 1740 |
redo A; |
redo A; |
| 1741 |
} elsif ($self->{next_char} == 0x0026) { # & |
} elsif ($self->{next_char} == 0x0026) { # & |
| 1742 |
!!!cp (96); |
!!!cp (96); |
|
$self->{last_attribute_value_state} = $self->{state}; |
|
| 1743 |
## NOTE: In the spec, the tokenizer is switched to the |
## NOTE: In the spec, the tokenizer is switched to the |
| 1744 |
## "entity in attribute value state". In this implementation, the |
## "entity in attribute value state". In this implementation, the |
| 1745 |
## tokenizer is switched to the |ENTITY_STATE|, which is an |
## tokenizer is switched to the |ENTITY_STATE|, which is an |
| 1746 |
## implementation of the "consume a character reference" algorithm. |
## implementation of the "consume a character reference" algorithm. |
| 1747 |
$self->{entity_in_attr} = 1; |
$self->{prev_state} = $self->{state}; |
| 1748 |
$self->{entity_additional} = 0x0022; # " |
$self->{entity_additional} = 0x0022; # " |
| 1749 |
$self->{state} = ENTITY_STATE; |
$self->{state} = ENTITY_STATE; |
| 1750 |
!!!next-input-character; |
!!!next-input-character; |
| 1775 |
} else { |
} else { |
| 1776 |
!!!cp (100); |
!!!cp (100); |
| 1777 |
$self->{current_attribute}->{value} .= chr ($self->{next_char}); |
$self->{current_attribute}->{value} .= chr ($self->{next_char}); |
| 1778 |
|
$self->{read_until}->($self->{current_attribute}->{value}, |
| 1779 |
|
q["&], |
| 1780 |
|
length $self->{current_attribute}->{value}); |
| 1781 |
|
|
| 1782 |
## Stay in the state |
## Stay in the state |
| 1783 |
!!!next-input-character; |
!!!next-input-character; |
| 1784 |
redo A; |
redo A; |
| 1791 |
redo A; |
redo A; |
| 1792 |
} elsif ($self->{next_char} == 0x0026) { # & |
} elsif ($self->{next_char} == 0x0026) { # & |
| 1793 |
!!!cp (102); |
!!!cp (102); |
|
$self->{last_attribute_value_state} = $self->{state}; |
|
| 1794 |
## NOTE: In the spec, the tokenizer is switched to the |
## NOTE: In the spec, the tokenizer is switched to the |
| 1795 |
## "entity in attribute value state". In this implementation, the |
## "entity in attribute value state". In this implementation, the |
| 1796 |
## tokenizer is switched to the |ENTITY_STATE|, which is an |
## tokenizer is switched to the |ENTITY_STATE|, which is an |
| 1797 |
## implementation of the "consume a character reference" algorithm. |
## implementation of the "consume a character reference" algorithm. |
|
$self->{entity_in_attr} = 1; |
|
| 1798 |
$self->{entity_additional} = 0x0027; # ' |
$self->{entity_additional} = 0x0027; # ' |
| 1799 |
|
$self->{prev_state} = $self->{state}; |
| 1800 |
$self->{state} = ENTITY_STATE; |
$self->{state} = ENTITY_STATE; |
| 1801 |
!!!next-input-character; |
!!!next-input-character; |
| 1802 |
redo A; |
redo A; |
| 1826 |
} else { |
} else { |
| 1827 |
!!!cp (106); |
!!!cp (106); |
| 1828 |
$self->{current_attribute}->{value} .= chr ($self->{next_char}); |
$self->{current_attribute}->{value} .= chr ($self->{next_char}); |
| 1829 |
|
$self->{read_until}->($self->{current_attribute}->{value}, |
| 1830 |
|
q['&], |
| 1831 |
|
length $self->{current_attribute}->{value}); |
| 1832 |
|
|
| 1833 |
## Stay in the state |
## Stay in the state |
| 1834 |
!!!next-input-character; |
!!!next-input-character; |
| 1835 |
redo A; |
redo A; |
| 1846 |
redo A; |
redo A; |
| 1847 |
} elsif ($self->{next_char} == 0x0026) { # & |
} elsif ($self->{next_char} == 0x0026) { # & |
| 1848 |
!!!cp (108); |
!!!cp (108); |
|
$self->{last_attribute_value_state} = $self->{state}; |
|
| 1849 |
## NOTE: In the spec, the tokenizer is switched to the |
## NOTE: In the spec, the tokenizer is switched to the |
| 1850 |
## "entity in attribute value state". In this implementation, the |
## "entity in attribute value state". In this implementation, the |
| 1851 |
## tokenizer is switched to the |ENTITY_STATE|, which is an |
## tokenizer is switched to the |ENTITY_STATE|, which is an |
| 1852 |
## implementation of the "consume a character reference" algorithm. |
## implementation of the "consume a character reference" algorithm. |
|
$self->{entity_in_attr} = 1; |
|
| 1853 |
$self->{entity_additional} = -1; |
$self->{entity_additional} = -1; |
| 1854 |
|
$self->{prev_state} = $self->{state}; |
| 1855 |
$self->{state} = ENTITY_STATE; |
$self->{state} = ENTITY_STATE; |
| 1856 |
!!!next-input-character; |
!!!next-input-character; |
| 1857 |
redo A; |
redo A; |
| 1912 |
!!!cp (116); |
!!!cp (116); |
| 1913 |
} |
} |
| 1914 |
$self->{current_attribute}->{value} .= chr ($self->{next_char}); |
$self->{current_attribute}->{value} .= chr ($self->{next_char}); |
| 1915 |
|
$self->{read_until}->($self->{current_attribute}->{value}, |
| 1916 |
|
q["'=& >], |
| 1917 |
|
length $self->{current_attribute}->{value}); |
| 1918 |
|
|
| 1919 |
## Stay in the state |
## Stay in the state |
| 1920 |
!!!next-input-character; |
!!!next-input-character; |
| 1921 |
redo A; |
redo A; |
| 2060 |
} else { |
} else { |
| 2061 |
!!!cp (126); |
!!!cp (126); |
| 2062 |
$self->{current_token}->{data} .= chr ($self->{next_char}); # comment |
$self->{current_token}->{data} .= chr ($self->{next_char}); # comment |
| 2063 |
|
$self->{read_until}->($self->{current_token}->{data}, |
| 2064 |
|
q[>], |
| 2065 |
|
length $self->{current_token}->{data}); |
| 2066 |
|
|
| 2067 |
## Stay in the state. |
## Stay in the state. |
| 2068 |
!!!next-input-character; |
!!!next-input-character; |
| 2069 |
redo A; |
redo A; |
| 2298 |
} else { |
} else { |
| 2299 |
!!!cp (147); |
!!!cp (147); |
| 2300 |
$self->{current_token}->{data} .= chr ($self->{next_char}); # comment |
$self->{current_token}->{data} .= chr ($self->{next_char}); # comment |
| 2301 |
|
$self->{read_until}->($self->{current_token}->{data}, |
| 2302 |
|
q[-], |
| 2303 |
|
length $self->{current_token}->{data}); |
| 2304 |
|
|
| 2305 |
## Stay in the state |
## Stay in the state |
| 2306 |
!!!next-input-character; |
!!!next-input-character; |
| 2307 |
redo A; |
redo A; |
| 2667 |
!!!cp (190); |
!!!cp (190); |
| 2668 |
$self->{current_token}->{public_identifier} # DOCTYPE |
$self->{current_token}->{public_identifier} # DOCTYPE |
| 2669 |
.= chr $self->{next_char}; |
.= chr $self->{next_char}; |
| 2670 |
|
$self->{read_until}->($self->{current_token}->{public_identifier}, |
| 2671 |
|
q[">], |
| 2672 |
|
length $self->{current_token}->{public_identifier}); |
| 2673 |
|
|
| 2674 |
## Stay in the state |
## Stay in the state |
| 2675 |
!!!next-input-character; |
!!!next-input-character; |
| 2676 |
redo A; |
redo A; |
| 2707 |
!!!cp (194); |
!!!cp (194); |
| 2708 |
$self->{current_token}->{public_identifier} # DOCTYPE |
$self->{current_token}->{public_identifier} # DOCTYPE |
| 2709 |
.= chr $self->{next_char}; |
.= chr $self->{next_char}; |
| 2710 |
|
$self->{read_until}->($self->{current_token}->{public_identifier}, |
| 2711 |
|
q['>], |
| 2712 |
|
length $self->{current_token}->{public_identifier}); |
| 2713 |
|
|
| 2714 |
## Stay in the state |
## Stay in the state |
| 2715 |
!!!next-input-character; |
!!!next-input-character; |
| 2716 |
redo A; |
redo A; |
| 2847 |
!!!cp (210); |
!!!cp (210); |
| 2848 |
$self->{current_token}->{system_identifier} # DOCTYPE |
$self->{current_token}->{system_identifier} # DOCTYPE |
| 2849 |
.= chr $self->{next_char}; |
.= chr $self->{next_char}; |
| 2850 |
|
$self->{read_until}->($self->{current_token}->{system_identifier}, |
| 2851 |
|
q[">], |
| 2852 |
|
length $self->{current_token}->{system_identifier}); |
| 2853 |
|
|
| 2854 |
## Stay in the state |
## Stay in the state |
| 2855 |
!!!next-input-character; |
!!!next-input-character; |
| 2856 |
redo A; |
redo A; |
| 2887 |
!!!cp (214); |
!!!cp (214); |
| 2888 |
$self->{current_token}->{system_identifier} # DOCTYPE |
$self->{current_token}->{system_identifier} # DOCTYPE |
| 2889 |
.= chr $self->{next_char}; |
.= chr $self->{next_char}; |
| 2890 |
|
$self->{read_until}->($self->{current_token}->{system_identifier}, |
| 2891 |
|
q['>], |
| 2892 |
|
length $self->{current_token}->{system_identifier}); |
| 2893 |
|
|
| 2894 |
## Stay in the state |
## Stay in the state |
| 2895 |
!!!next-input-character; |
!!!next-input-character; |
| 2896 |
redo A; |
redo A; |
| 2951 |
redo A; |
redo A; |
| 2952 |
} else { |
} else { |
| 2953 |
!!!cp (221); |
!!!cp (221); |
| 2954 |
|
my $s = ''; |
| 2955 |
|
$self->{read_until}->($s, q[>], 0); |
| 2956 |
|
|
| 2957 |
## Stay in the state |
## Stay in the state |
| 2958 |
!!!next-input-character; |
!!!next-input-character; |
| 2959 |
redo A; |
redo A; |
| 2982 |
} else { |
} else { |
| 2983 |
!!!cp (221.4); |
!!!cp (221.4); |
| 2984 |
$self->{current_token}->{data} .= chr $self->{next_char}; |
$self->{current_token}->{data} .= chr $self->{next_char}; |
| 2985 |
|
$self->{read_until}->($self->{current_token}->{data}, |
| 2986 |
|
q<]>, |
| 2987 |
|
length $self->{current_token}->{data}); |
| 2988 |
|
|
| 2989 |
## Stay in the state. |
## Stay in the state. |
| 2990 |
!!!next-input-character; |
!!!next-input-character; |
| 2991 |
redo A; |
redo A; |
| 3042 |
## Return nothing. |
## Return nothing. |
| 3043 |
# |
# |
| 3044 |
} elsif ($self->{next_char} == 0x0023) { # # |
} elsif ($self->{next_char} == 0x0023) { # # |
| 3045 |
|
!!!cp (999); |
| 3046 |
$self->{state} = ENTITY_HASH_STATE; |
$self->{state} = ENTITY_HASH_STATE; |
| 3047 |
$self->{state_keyword} = '#'; |
$self->{state_keyword} = '#'; |
| 3048 |
!!!next-input-character; |
!!!next-input-character; |
| 3051 |
$self->{next_char} <= 0x005A) or # A..Z |
$self->{next_char} <= 0x005A) or # A..Z |
| 3052 |
(0x0061 <= $self->{next_char} and |
(0x0061 <= $self->{next_char} and |
| 3053 |
$self->{next_char} <= 0x007A)) { # a..z |
$self->{next_char} <= 0x007A)) { # a..z |
| 3054 |
|
!!!cp (998); |
| 3055 |
require Whatpm::_NamedEntityList; |
require Whatpm::_NamedEntityList; |
| 3056 |
$self->{state} = ENTITY_NAME_STATE; |
$self->{state} = ENTITY_NAME_STATE; |
| 3057 |
$self->{state_keyword} = chr $self->{next_char}; |
$self->{state_keyword} = chr $self->{next_char}; |
| 3072 |
## appended to the parent element or the attribute value in later |
## appended to the parent element or the attribute value in later |
| 3073 |
## process of the tokenizer. |
## process of the tokenizer. |
| 3074 |
|
|
| 3075 |
if ($self->{entity_in_attr}) { |
if ($self->{prev_state} == DATA_STATE) { |
| 3076 |
$self->{current_attribute}->{value} .= '&'; |
!!!cp (997); |
| 3077 |
$self->{state} = $self->{last_attribute_value_state}; |
$self->{state} = $self->{prev_state}; |
|
## Reconsume. |
|
|
redo A; |
|
|
} else { |
|
|
$self->{state} = DATA_STATE; |
|
| 3078 |
## Reconsume. |
## Reconsume. |
| 3079 |
!!!emit ({type => CHARACTER_TOKEN, data => '&', |
!!!emit ({type => CHARACTER_TOKEN, data => '&', |
| 3080 |
line => $self->{line_prev}, |
line => $self->{line_prev}, |
| 3081 |
column => $self->{column_prev}, |
column => $self->{column_prev}, |
| 3082 |
}); |
}); |
| 3083 |
redo A; |
redo A; |
| 3084 |
|
} else { |
| 3085 |
|
!!!cp (996); |
| 3086 |
|
$self->{current_attribute}->{value} .= '&'; |
| 3087 |
|
$self->{state} = $self->{prev_state}; |
| 3088 |
|
## Reconsume. |
| 3089 |
|
redo A; |
| 3090 |
} |
} |
| 3091 |
} elsif ($self->{state} == ENTITY_HASH_STATE) { |
} elsif ($self->{state} == ENTITY_HASH_STATE) { |
| 3092 |
if ($self->{next_char} == 0x0078 or # x |
if ($self->{next_char} == 0x0078 or # x |
| 3093 |
$self->{next_char} == 0x0058) { # X |
$self->{next_char} == 0x0058) { # X |
| 3094 |
|
!!!cp (995); |
| 3095 |
$self->{state} = HEXREF_X_STATE; |
$self->{state} = HEXREF_X_STATE; |
| 3096 |
$self->{state_keyword} .= chr $self->{next_char}; |
$self->{state_keyword} .= chr $self->{next_char}; |
| 3097 |
!!!next-input-character; |
!!!next-input-character; |
| 3098 |
redo A; |
redo A; |
| 3099 |
} elsif (0x0030 <= $self->{next_char} and |
} elsif (0x0030 <= $self->{next_char} and |
| 3100 |
$self->{next_char} <= 0x0039) { # 0..9 |
$self->{next_char} <= 0x0039) { # 0..9 |
| 3101 |
|
!!!cp (994); |
| 3102 |
$self->{state} = NCR_NUM_STATE; |
$self->{state} = NCR_NUM_STATE; |
| 3103 |
$self->{state_keyword} = $self->{next_char} - 0x0030; |
$self->{state_keyword} = $self->{next_char} - 0x0030; |
| 3104 |
!!!next-input-character; |
!!!next-input-character; |
| 3105 |
redo A; |
redo A; |
| 3106 |
} else { |
} else { |
|
!!!cp (1019); |
|
| 3107 |
!!!parse-error (type => 'bare nero', |
!!!parse-error (type => 'bare nero', |
| 3108 |
line => $self->{line_prev}, |
line => $self->{line_prev}, |
| 3109 |
column => $self->{column_prev} - 1); |
column => $self->{column_prev} - 1); |
| 3112 |
## and then "&#" is appended to the parent element or the attribute |
## and then "&#" is appended to the parent element or the attribute |
| 3113 |
## value in the later processing. |
## value in the later processing. |
| 3114 |
|
|
| 3115 |
if ($self->{entity_in_attr}) { |
if ($self->{prev_state} == DATA_STATE) { |
| 3116 |
$self->{current_attribute}->{value} .= '&#'; |
!!!cp (1019); |
| 3117 |
$self->{state} = $self->{last_attribute_value_state}; |
$self->{state} = $self->{prev_state}; |
|
## Reconsume. |
|
|
redo A; |
|
|
} else { |
|
|
$self->{state} = DATA_STATE; |
|
| 3118 |
## Reconsume. |
## Reconsume. |
| 3119 |
!!!emit ({type => CHARACTER_TOKEN, |
!!!emit ({type => CHARACTER_TOKEN, |
| 3120 |
data => '&#', |
data => '&#', |
| 3122 |
column => $self->{column_prev} - 1, |
column => $self->{column_prev} - 1, |
| 3123 |
}); |
}); |
| 3124 |
redo A; |
redo A; |
| 3125 |
|
} else { |
| 3126 |
|
!!!cp (993); |
| 3127 |
|
$self->{current_attribute}->{value} .= '&#'; |
| 3128 |
|
$self->{state} = $self->{prev_state}; |
| 3129 |
|
## Reconsume. |
| 3130 |
|
redo A; |
| 3131 |
} |
} |
| 3132 |
} |
} |
| 3133 |
} elsif ($self->{state} == NCR_NUM_STATE) { |
} elsif ($self->{state} == NCR_NUM_STATE) { |
| 3179 |
$code = $c1_entity_char->{$code}; |
$code = $c1_entity_char->{$code}; |
| 3180 |
} |
} |
| 3181 |
|
|
| 3182 |
if ($self->{entity_in_attr}) { |
if ($self->{prev_state} == DATA_STATE) { |
| 3183 |
$self->{current_attribute}->{value} .= chr $code; |
!!!cp (992); |
| 3184 |
$self->{current_attribute}->{has_reference} = 1; |
$self->{state} = $self->{prev_state}; |
|
$self->{state} = $self->{last_attribute_value_state}; |
|
|
## Reconsume. |
|
|
redo A; |
|
|
} else { |
|
|
$self->{state} = DATA_STATE; |
|
| 3185 |
## Reconsume. |
## Reconsume. |
| 3186 |
!!!emit ({type => CHARACTER_TOKEN, data => chr $code, |
!!!emit ({type => CHARACTER_TOKEN, data => chr $code, |
|
has_reference => 1, |
|
| 3187 |
line => $l, column => $c, |
line => $l, column => $c, |
| 3188 |
}); |
}); |
| 3189 |
redo A; |
redo A; |
| 3190 |
|
} else { |
| 3191 |
|
!!!cp (991); |
| 3192 |
|
$self->{current_attribute}->{value} .= chr $code; |
| 3193 |
|
$self->{current_attribute}->{has_reference} = 1; |
| 3194 |
|
$self->{state} = $self->{prev_state}; |
| 3195 |
|
## Reconsume. |
| 3196 |
|
redo A; |
| 3197 |
} |
} |
| 3198 |
} elsif ($self->{state} == HEXREF_X_STATE) { |
} elsif ($self->{state} == HEXREF_X_STATE) { |
| 3199 |
if ((0x0030 <= $self->{next_char} and $self->{next_char} <= 0x0039) or |
if ((0x0030 <= $self->{next_char} and $self->{next_char} <= 0x0039) or |
| 3200 |
(0x0041 <= $self->{next_char} and $self->{next_char} <= 0x0046) or |
(0x0041 <= $self->{next_char} and $self->{next_char} <= 0x0046) or |
| 3201 |
(0x0061 <= $self->{next_char} and $self->{next_char} <= 0x0066)) { |
(0x0061 <= $self->{next_char} and $self->{next_char} <= 0x0066)) { |
| 3202 |
# 0..9, A..F, a..f |
# 0..9, A..F, a..f |
| 3203 |
|
!!!cp (990); |
| 3204 |
$self->{state} = HEXREF_HEX_STATE; |
$self->{state} = HEXREF_HEX_STATE; |
| 3205 |
$self->{state_keyword} = 0; |
$self->{state_keyword} = 0; |
| 3206 |
## Reconsume. |
## Reconsume. |
| 3207 |
redo A; |
redo A; |
| 3208 |
} else { |
} else { |
|
!!!cp (1005); |
|
| 3209 |
!!!parse-error (type => 'bare hcro', |
!!!parse-error (type => 'bare hcro', |
| 3210 |
line => $self->{line_prev}, |
line => $self->{line_prev}, |
| 3211 |
column => $self->{column_prev} - 2); |
column => $self->{column_prev} - 2); |
| 3214 |
## and then "&#" followed by "X" or "x" is appended to the parent |
## and then "&#" followed by "X" or "x" is appended to the parent |
| 3215 |
## element or the attribute value in the later processing. |
## element or the attribute value in the later processing. |
| 3216 |
|
|
| 3217 |
if ($self->{entity_in_attr}) { |
if ($self->{prev_state} == DATA_STATE) { |
| 3218 |
$self->{current_attribute}->{value} .= '&' . $self->{state_keyword}; |
!!!cp (1005); |
| 3219 |
$self->{state} = $self->{last_attribute_value_state}; |
$self->{state} = $self->{prev_state}; |
|
## Reconsume. |
|
|
redo A; |
|
|
} else { |
|
|
$self->{state} = DATA_STATE; |
|
| 3220 |
## Reconsume. |
## Reconsume. |
| 3221 |
!!!emit ({type => CHARACTER_TOKEN, |
!!!emit ({type => CHARACTER_TOKEN, |
| 3222 |
data => '&' . $self->{state_keyword}, |
data => '&' . $self->{state_keyword}, |
| 3224 |
column => $self->{column_prev} - length $self->{state_keyword}, |
column => $self->{column_prev} - length $self->{state_keyword}, |
| 3225 |
}); |
}); |
| 3226 |
redo A; |
redo A; |
| 3227 |
|
} else { |
| 3228 |
|
!!!cp (989); |
| 3229 |
|
$self->{current_attribute}->{value} .= '&' . $self->{state_keyword}; |
| 3230 |
|
$self->{state} = $self->{prev_state}; |
| 3231 |
|
## Reconsume. |
| 3232 |
|
redo A; |
| 3233 |
} |
} |
| 3234 |
} |
} |
| 3235 |
} elsif ($self->{state} == HEXREF_HEX_STATE) { |
} elsif ($self->{state} == HEXREF_HEX_STATE) { |
| 3295 |
$code = $c1_entity_char->{$code}; |
$code = $c1_entity_char->{$code}; |
| 3296 |
} |
} |
| 3297 |
|
|
| 3298 |
if ($self->{entity_in_attr}) { |
if ($self->{prev_state} == DATA_STATE) { |
| 3299 |
$self->{current_attribute}->{value} .= chr $code; |
!!!cp (988); |
| 3300 |
$self->{current_attribute}->{has_reference} = 1; |
$self->{state} = $self->{prev_state}; |
|
$self->{state} = $self->{last_attribute_value_state}; |
|
|
## Reconsume. |
|
|
redo A; |
|
|
} else { |
|
|
$self->{state} = DATA_STATE; |
|
| 3301 |
## Reconsume. |
## Reconsume. |
| 3302 |
!!!emit ({type => CHARACTER_TOKEN, data => chr $code, |
!!!emit ({type => CHARACTER_TOKEN, data => chr $code, |
|
has_reference => 1, |
|
| 3303 |
line => $l, column => $c, |
line => $l, column => $c, |
| 3304 |
}); |
}); |
| 3305 |
redo A; |
redo A; |
| 3306 |
|
} else { |
| 3307 |
|
!!!cp (987); |
| 3308 |
|
$self->{current_attribute}->{value} .= chr $code; |
| 3309 |
|
$self->{current_attribute}->{has_reference} = 1; |
| 3310 |
|
$self->{state} = $self->{prev_state}; |
| 3311 |
|
## Reconsume. |
| 3312 |
|
redo A; |
| 3313 |
} |
} |
| 3314 |
} elsif ($self->{state} == ENTITY_NAME_STATE) { |
} elsif ($self->{state} == ENTITY_NAME_STATE) { |
| 3315 |
if (length $self->{state_keyword} < 30 and |
if (length $self->{state_keyword} < 30 and |
| 3357 |
# |
# |
| 3358 |
} elsif ($self->{entity__match} < 0) { |
} elsif ($self->{entity__match} < 0) { |
| 3359 |
!!!parse-error (type => 'no refc'); |
!!!parse-error (type => 'no refc'); |
| 3360 |
if ($self->{entity_in_attr} and $self->{entity__match} < -1) { |
if ($self->{prev_state} != DATA_STATE and # in attribute |
| 3361 |
|
$self->{entity__match} < -1) { |
| 3362 |
!!!cp (1024); |
!!!cp (1024); |
| 3363 |
$data = '&' . $self->{state_keyword}; |
$data = '&' . $self->{state_keyword}; |
| 3364 |
# |
# |
| 3372 |
!!!cp (1026); |
!!!cp (1026); |
| 3373 |
!!!parse-error (type => 'bare ero', |
!!!parse-error (type => 'bare ero', |
| 3374 |
line => $self->{line_prev}, |
line => $self->{line_prev}, |
| 3375 |
column => $self->{column_prev}); |
column => $self->{column_prev} - length $self->{state_keyword}); |
| 3376 |
$data = '&' . $self->{state_keyword}; |
$data = '&' . $self->{state_keyword}; |
| 3377 |
# |
# |
| 3378 |
} |
} |
| 3387 |
## that would not be consumed are appended in the data state or in an |
## that would not be consumed are appended in the data state or in an |
| 3388 |
## appropriate attribute value state anyway. |
## appropriate attribute value state anyway. |
| 3389 |
|
|
| 3390 |
if ($self->{entity_in_attr}) { |
if ($self->{prev_state} == DATA_STATE) { |
| 3391 |
$self->{current_attribute}->{value} .= $data; |
!!!cp (986); |
| 3392 |
$self->{current_attribute}->{has_reference} = 1 if $has_ref; |
$self->{state} = $self->{prev_state}; |
|
$self->{state} = $self->{last_attribute_value_state}; |
|
|
## Reconsume. |
|
|
redo A; |
|
|
} else { |
|
|
$self->{state} = DATA_STATE; |
|
| 3393 |
## Reconsume. |
## Reconsume. |
| 3394 |
!!!emit ({type => CHARACTER_TOKEN, |
!!!emit ({type => CHARACTER_TOKEN, |
| 3395 |
data => $data, has_reference => $has_ref, |
data => $data, |
| 3396 |
line => $self->{line_prev}, |
line => $self->{line_prev}, |
| 3397 |
column => $self->{column_prev} + 1 - length $self->{state_keyword}, |
column => $self->{column_prev} + 1 - length $self->{state_keyword}, |
| 3398 |
}); |
}); |
| 3399 |
redo A; |
redo A; |
| 3400 |
|
} else { |
| 3401 |
|
!!!cp (985); |
| 3402 |
|
$self->{current_attribute}->{value} .= $data; |
| 3403 |
|
$self->{current_attribute}->{has_reference} = 1 if $has_ref; |
| 3404 |
|
$self->{state} = $self->{prev_state}; |
| 3405 |
|
## Reconsume. |
| 3406 |
|
redo A; |
| 3407 |
} |
} |
| 3408 |
} else { |
} else { |
| 3409 |
die "$0: $self->{state}: Unknown state"; |
die "$0: $self->{state}: Unknown state"; |
| 4510 |
unless ($self->{insertion_mode} == BEFORE_HEAD_IM) { |
unless ($self->{insertion_mode} == BEFORE_HEAD_IM) { |
| 4511 |
!!!cp ('t88.2'); |
!!!cp ('t88.2'); |
| 4512 |
$self->{open_elements}->[-1]->[0]->manakai_append_text ($1); |
$self->{open_elements}->[-1]->[0]->manakai_append_text ($1); |
| 4513 |
|
# |
| 4514 |
} else { |
} else { |
| 4515 |
!!!cp ('t88.1'); |
!!!cp ('t88.1'); |
| 4516 |
## Ignore the token. |
## Ignore the token. |
| 4517 |
!!!next-token; |
# |
|
next B; |
|
| 4518 |
} |
} |
| 4519 |
unless (length $token->{data}) { |
unless (length $token->{data}) { |
| 4520 |
!!!cp ('t88'); |
!!!cp ('t88'); |
| 4521 |
!!!next-token; |
!!!next-token; |
| 4522 |
next B; |
next B; |
| 4523 |
} |
} |
| 4524 |
|
## TODO: set $token->{column} appropriately |
| 4525 |
} |
} |
| 4526 |
|
|
| 4527 |
if ($self->{insertion_mode} == BEFORE_HEAD_IM) { |
if ($self->{insertion_mode} == BEFORE_HEAD_IM) { |
| 7766 |
## TODO: script stuffs |
## TODO: script stuffs |
| 7767 |
} # _tree_construct_main |
} # _tree_construct_main |
| 7768 |
|
|
| 7769 |
sub set_inner_html ($$$;$) { |
sub set_inner_html ($$$$;$) { |
| 7770 |
my $class = shift; |
my $class = shift; |
| 7771 |
my $node = shift; |
my $node = shift; |
| 7772 |
my $s = \$_[0]; |
#my $s = \$_[0]; |
| 7773 |
my $onerror = $_[1]; |
my $onerror = $_[1]; |
| 7774 |
my $get_wrapper = $_[2] || sub ($) { return $_[0] }; |
my $get_wrapper = $_[2] || sub ($) { return $_[0] }; |
| 7775 |
|
|
| 7790 |
} |
} |
| 7791 |
|
|
| 7792 |
## Step 3, 4, 5 # MUST |
## Step 3, 4, 5 # MUST |
| 7793 |
$class->parse_char_string ($$s => $node, $onerror, $get_wrapper); |
$class->parse_char_string ($_[0] => $node, $onerror, $get_wrapper); |
| 7794 |
} elsif ($nt == 1) { |
} elsif ($nt == 1) { |
| 7795 |
## TODO: If non-html element |
## TODO: If non-html element |
| 7796 |
|
|
| 7809 |
my $i = 0; |
my $i = 0; |
| 7810 |
$p->{line_prev} = $p->{line} = 1; |
$p->{line_prev} = $p->{line} = 1; |
| 7811 |
$p->{column_prev} = $p->{column} = 0; |
$p->{column_prev} = $p->{column} = 0; |
| 7812 |
|
require Whatpm::Charset::DecodeHandle; |
| 7813 |
|
my $input = Whatpm::Charset::DecodeHandle::CharString->new (\($_[0])); |
| 7814 |
|
$input = $get_wrapper->($input); |
| 7815 |
$p->{set_next_char} = sub { |
$p->{set_next_char} = sub { |
| 7816 |
my $self = shift; |
my $self = shift; |
| 7817 |
|
|
| 7818 |
pop @{$self->{prev_char}}; |
my $char = ''; |
| 7819 |
unshift @{$self->{prev_char}}, $self->{next_char}; |
if (defined $self->{next_next_char}) { |
| 7820 |
|
$char = $self->{next_next_char}; |
| 7821 |
$self->{next_char} = -1 and return if $i >= length $$s; |
delete $self->{next_next_char}; |
| 7822 |
$self->{next_char} = ord substr $$s, $i++, 1; |
$self->{next_char} = ord $char; |
| 7823 |
|
} else { |
| 7824 |
|
$self->{char_buffer} = ''; |
| 7825 |
|
$self->{char_buffer_pos} = 0; |
| 7826 |
|
|
| 7827 |
|
my $count = $input->manakai_read_until |
| 7828 |
|
($self->{char_buffer}, qr/[^\x00\x0A\x0D]/, |
| 7829 |
|
$self->{char_buffer_pos}); |
| 7830 |
|
if ($count) { |
| 7831 |
|
$self->{line_prev} = $self->{line}; |
| 7832 |
|
$self->{column_prev} = $self->{column}; |
| 7833 |
|
$self->{column}++; |
| 7834 |
|
$self->{next_char} |
| 7835 |
|
= ord substr ($self->{char_buffer}, |
| 7836 |
|
$self->{char_buffer_pos}++, 1); |
| 7837 |
|
return; |
| 7838 |
|
} |
| 7839 |
|
|
| 7840 |
|
if ($input->read ($char, 1)) { |
| 7841 |
|
$self->{next_char} = ord $char; |
| 7842 |
|
} else { |
| 7843 |
|
$self->{next_char} = -1; |
| 7844 |
|
return; |
| 7845 |
|
} |
| 7846 |
|
} |
| 7847 |
|
|
| 7848 |
($p->{line_prev}, $p->{column_prev}) = ($p->{line}, $p->{column}); |
($p->{line_prev}, $p->{column_prev}) = ($p->{line}, $p->{column}); |
| 7849 |
$p->{column}++; |
$p->{column}++; |
| 7853 |
$p->{column} = 0; |
$p->{column} = 0; |
| 7854 |
!!!cp ('i1'); |
!!!cp ('i1'); |
| 7855 |
} elsif ($self->{next_char} == 0x000D) { # CR |
} elsif ($self->{next_char} == 0x000D) { # CR |
| 7856 |
$i++ if substr ($$s, $i, 1) eq "\x0A"; |
## TODO: support for abort/streaming |
| 7857 |
|
my $next = ''; |
| 7858 |
|
if ($input->read ($next, 1) and $next ne "\x0A") { |
| 7859 |
|
$self->{next_next_char} = $next; |
| 7860 |
|
} |
| 7861 |
$self->{next_char} = 0x000A; # LF # MUST |
$self->{next_char} = 0x000A; # LF # MUST |
| 7862 |
$p->{line}++; |
$p->{line}++; |
| 7863 |
$p->{column} = 0; |
$p->{column} = 0; |
| 7864 |
!!!cp ('i2'); |
!!!cp ('i2'); |
|
} elsif ($self->{next_char} > 0x10FFFF) { |
|
|
$self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST |
|
|
!!!cp ('i3'); |
|
| 7865 |
} elsif ($self->{next_char} == 0x0000) { # NULL |
} elsif ($self->{next_char} == 0x0000) { # NULL |
| 7866 |
!!!cp ('i4'); |
!!!cp ('i4'); |
| 7867 |
!!!parse-error (type => 'NULL'); |
!!!parse-error (type => 'NULL'); |
| 7868 |
$self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST |
$self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST |
|
} elsif ($self->{next_char} <= 0x0008 or |
|
|
(0x000E <= $self->{next_char} and |
|
|
$self->{next_char} <= 0x001F) or |
|
|
(0x007F <= $self->{next_char} and |
|
|
$self->{next_char} <= 0x009F) or |
|
|
(0xD800 <= $self->{next_char} and |
|
|
$self->{next_char} <= 0xDFFF) or |
|
|
(0xFDD0 <= $self->{next_char} and |
|
|
$self->{next_char} <= 0xFDDF) or |
|
|
{ |
|
|
0xFFFE => 1, 0xFFFF => 1, 0x1FFFE => 1, 0x1FFFF => 1, |
|
|
0x2FFFE => 1, 0x2FFFF => 1, 0x3FFFE => 1, 0x3FFFF => 1, |
|
|
0x4FFFE => 1, 0x4FFFF => 1, 0x5FFFE => 1, 0x5FFFF => 1, |
|
|
0x6FFFE => 1, 0x6FFFF => 1, 0x7FFFE => 1, 0x7FFFF => 1, |
|
|
0x8FFFE => 1, 0x8FFFF => 1, 0x9FFFE => 1, 0x9FFFF => 1, |
|
|
0xAFFFE => 1, 0xAFFFF => 1, 0xBFFFE => 1, 0xBFFFF => 1, |
|
|
0xCFFFE => 1, 0xCFFFF => 1, 0xDFFFE => 1, 0xDFFFF => 1, |
|
|
0xEFFFE => 1, 0xEFFFF => 1, 0xFFFFE => 1, 0xFFFFF => 1, |
|
|
0x10FFFE => 1, 0x10FFFF => 1, |
|
|
}->{$self->{next_char}}) { |
|
|
!!!cp ('i4.1'); |
|
|
if ($self->{next_char} < 0x10000) { |
|
|
!!!parse-error (type => 'control char', |
|
|
text => (sprintf 'U+%04X', $self->{next_char})); |
|
|
} else { |
|
|
!!!parse-error (type => 'control char', |
|
|
text => (sprintf 'U-%08X', $self->{next_char})); |
|
|
} |
|
| 7869 |
} |
} |
| 7870 |
}; |
}; |
| 7871 |
$p->{prev_char} = [-1, -1, -1]; |
|
| 7872 |
$p->{next_char} = -1; |
$p->{read_until} = sub { |
| 7873 |
|
#my ($scalar, $specials_range, $offset) = @_; |
| 7874 |
|
return 0 if defined $p->{next_next_char}; |
| 7875 |
|
|
| 7876 |
|
my $pattern = qr/[^$_[1]\x00\x0A\x0D]/; |
| 7877 |
|
my $offset = $_[2] || 0; |
| 7878 |
|
|
| 7879 |
|
if ($p->{char_buffer_pos} < length $p->{char_buffer}) { |
| 7880 |
|
pos ($p->{char_buffer}) = $p->{char_buffer_pos}; |
| 7881 |
|
if ($p->{char_buffer} =~ /\G(?>$pattern)+/) { |
| 7882 |
|
substr ($_[0], $offset) |
| 7883 |
|
= substr ($p->{char_buffer}, $-[0], $+[0] - $-[0]); |
| 7884 |
|
my $count = $+[0] - $-[0]; |
| 7885 |
|
if ($count) { |
| 7886 |
|
$p->{column} += $count; |
| 7887 |
|
$p->{char_buffer_pos} += $count; |
| 7888 |
|
$p->{line_prev} = $p->{line}; |
| 7889 |
|
$p->{column_prev} = $p->{column} - 1; |
| 7890 |
|
$p->{prev_char} = [-1, -1, -1]; |
| 7891 |
|
$p->{next_char} = -1; |
| 7892 |
|
} |
| 7893 |
|
return $count; |
| 7894 |
|
} else { |
| 7895 |
|
return 0; |
| 7896 |
|
} |
| 7897 |
|
} else { |
| 7898 |
|
my $count = $input->manakai_read_until ($_[0], $pattern, $_[2]); |
| 7899 |
|
if ($count) { |
| 7900 |
|
$p->{column} += $count; |
| 7901 |
|
$p->{column_prev} += $count; |
| 7902 |
|
$p->{prev_char} = [-1, -1, -1]; |
| 7903 |
|
$p->{next_char} = -1; |
| 7904 |
|
} |
| 7905 |
|
return $count; |
| 7906 |
|
} |
| 7907 |
|
}; # $p->{read_until} |
| 7908 |
|
|
| 7909 |
my $ponerror = $onerror || sub { |
my $ponerror = $onerror || sub { |
| 7910 |
my (%opt) = @_; |
my (%opt) = @_; |
| 7911 |
my $line = $opt{line}; |
my $line = $opt{line}; |
| 7920 |
$ponerror->(line => $p->{line}, column => $p->{column}, @_); |
$ponerror->(line => $p->{line}, column => $p->{column}, @_); |
| 7921 |
}; |
}; |
| 7922 |
|
|
| 7923 |
|
my $char_onerror = sub { |
| 7924 |
|
my (undef, $type, %opt) = @_; |
| 7925 |
|
$ponerror->(layer => 'encode', |
| 7926 |
|
line => $p->{line}, column => $p->{column} + 1, |
| 7927 |
|
%opt, type => $type); |
| 7928 |
|
}; # $char_onerror |
| 7929 |
|
$input->onerror ($char_onerror); |
| 7930 |
|
|
| 7931 |
$p->_initialize_tokenizer; |
$p->_initialize_tokenizer; |
| 7932 |
$p->_initialize_tree_constructor; |
$p->_initialize_tree_constructor; |
| 7933 |
|
|