| 496 |
line => 1, column => 1, |
line => 1, column => 1, |
| 497 |
layer => 'encode'); |
layer => 'encode'); |
| 498 |
} elsif (not ($e_status & |
} elsif (not ($e_status & |
| 499 |
Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL())) { |
Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL ())) { |
| 500 |
$self->{input_encoding} = $charset->get_iana_name; |
$self->{input_encoding} = $charset->get_iana_name; |
| 501 |
!!!parse-error (type => 'chardecode:no error', |
!!!parse-error (type => 'chardecode:no error', |
| 502 |
text => $self->{input_encoding}, |
text => $self->{input_encoding}, |
| 561 |
my $char_onerror = sub { |
my $char_onerror = sub { |
| 562 |
my (undef, $type, %opt) = @_; |
my (undef, $type, %opt) = @_; |
| 563 |
!!!parse-error (layer => 'encode', |
!!!parse-error (layer => 'encode', |
| 564 |
%opt, type => $type, |
line => $self->{line}, column => $self->{column} + 1, |
| 565 |
line => $self->{line}, column => $self->{column} + 1); |
%opt, type => $type); |
| 566 |
if ($opt{octets}) { |
if ($opt{octets}) { |
| 567 |
${$opt{octets}} = "\x{FFFD}"; # relacement character |
${$opt{octets}} = "\x{FFFD}"; # relacement character |
| 568 |
} |
} |
| 586 |
line => 1, column => 1, |
line => 1, column => 1, |
| 587 |
layer => 'encode'); |
layer => 'encode'); |
| 588 |
} elsif (not ($e_status & |
} elsif (not ($e_status & |
| 589 |
Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL())) { |
Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL ())) { |
| 590 |
$self->{input_encoding} = $charset->get_iana_name; |
$self->{input_encoding} = $charset->get_iana_name; |
| 591 |
!!!parse-error (type => 'chardecode:no error', |
!!!parse-error (type => 'chardecode:no error', |
| 592 |
text => $self->{input_encoding}, |
text => $self->{input_encoding}, |
| 639 |
$self->{confident} = 1 unless exists $self->{confident}; |
$self->{confident} = 1 unless exists $self->{confident}; |
| 640 |
$self->{document}->input_encoding ($self->{input_encoding}) |
$self->{document}->input_encoding ($self->{input_encoding}) |
| 641 |
if defined $self->{input_encoding}; |
if defined $self->{input_encoding}; |
| 642 |
|
## TODO: |{input_encoding}| is needless? |
| 643 |
|
|
| 644 |
my $i = 0; |
my $i = 0; |
| 645 |
$self->{line_prev} = $self->{line} = 1; |
$self->{line_prev} = $self->{line} = 1; |
| 646 |
$self->{column_prev} = $self->{column} = 0; |
$self->{column_prev} = -1; |
| 647 |
|
$self->{column} = 0; |
| 648 |
$self->{set_next_char} = sub { |
$self->{set_next_char} = sub { |
| 649 |
my $self = shift; |
my $self = shift; |
| 650 |
|
|
| 651 |
pop @{$self->{prev_char}}; |
my $char = ''; |
|
unshift @{$self->{prev_char}}, $self->{next_char}; |
|
|
|
|
|
my $char; |
|
| 652 |
if (defined $self->{next_next_char}) { |
if (defined $self->{next_next_char}) { |
| 653 |
$char = $self->{next_next_char}; |
$char = $self->{next_next_char}; |
| 654 |
delete $self->{next_next_char}; |
delete $self->{next_next_char}; |
| 655 |
|
$self->{next_char} = ord $char; |
| 656 |
} else { |
} else { |
| 657 |
$char = $input->getc; |
$self->{char_buffer} = ''; |
| 658 |
|
$self->{char_buffer_pos} = 0; |
| 659 |
|
|
| 660 |
|
my $count = $input->manakai_read_until |
| 661 |
|
($self->{char_buffer}, |
| 662 |
|
qr/(?!\x{FDD0}-\x{FDDF}\x{FFFE}\x{FFFF}\x{1FFFE}\x{1FFFF}\x{2FFFE}\x{2FFFF}\x{3FFFE}\x{3FFFF}\x{4FFFE}\x{4FFFF}\x{5FFFE}\x{5FFFF}\x{6FFFE}\x{6FFFF}\x{7FFFE}\x{7FFFF}\x{8FFFE}\x{8FFFF}\x{9FFFE}\x{9FFFF}\x{AFFFE}\x{AFFFF}\x{BFFFE}\x{BFFFF}\x{CFFFE}\x{CFFFF}\x{DFFFE}\x{DFFFF}\x{EFFFE}\x{EFFFF}\x{FFFFE}\x{FFFFF}])[\x20-\x7E\xA0-\x{D7FF}\x{E000}-\x{10FFFD}]/, |
| 663 |
|
$self->{char_buffer_pos}); |
| 664 |
|
if ($count) { |
| 665 |
|
$self->{line_prev} = $self->{line}; |
| 666 |
|
$self->{column_prev} = $self->{column}; |
| 667 |
|
$self->{column}++; |
| 668 |
|
$self->{next_char} |
| 669 |
|
= ord substr ($self->{char_buffer}, $self->{char_buffer_pos}++, 1); |
| 670 |
|
return; |
| 671 |
|
} |
| 672 |
|
|
| 673 |
|
if ($input->read ($char, 1)) { |
| 674 |
|
$self->{next_char} = ord $char; |
| 675 |
|
} else { |
| 676 |
|
$self->{next_char} = -1; |
| 677 |
|
return; |
| 678 |
|
} |
| 679 |
} |
} |
|
$self->{next_char} = -1 and return unless defined $char; |
|
|
$self->{next_char} = ord $char; |
|
| 680 |
|
|
| 681 |
($self->{line_prev}, $self->{column_prev}) |
($self->{line_prev}, $self->{column_prev}) |
| 682 |
= ($self->{line}, $self->{column}); |
= ($self->{line}, $self->{column}); |
| 689 |
} elsif ($self->{next_char} == 0x000D) { # CR |
} elsif ($self->{next_char} == 0x000D) { # CR |
| 690 |
!!!cp ('j2'); |
!!!cp ('j2'); |
| 691 |
## TODO: support for abort/streaming |
## TODO: support for abort/streaming |
| 692 |
my $next = $input->getc; |
my $next = ''; |
| 693 |
if (defined $next and $next ne "\x0A") { |
if ($input->read ($next, 1) and $next ne "\x0A") { |
| 694 |
$self->{next_next_char} = $next; |
$self->{next_next_char} = $next; |
| 695 |
} |
} |
| 696 |
$self->{next_char} = 0x000A; # LF # MUST |
$self->{next_char} = 0x000A; # LF # MUST |
| 733 |
$self->{prev_char} = [-1, -1, -1]; |
$self->{prev_char} = [-1, -1, -1]; |
| 734 |
$self->{next_char} = -1; |
$self->{next_char} = -1; |
| 735 |
|
|
| 736 |
$self->{getc_until} = sub { return undef }; |
$self->{read_until} = sub { |
| 737 |
# if ($input->can ('manakai_getc_until')) { |
#my ($scalar, $specials_range, $offset) = @_; |
| 738 |
$self->{getc_until} = sub { |
my $specials_range = $_[1]; |
| 739 |
my $special_range = shift; |
return 0 if defined $self->{next_next_char}; |
| 740 |
return undef if defined $self->{next_next_char}; |
my $count = $input->manakai_read_until |
| 741 |
my $s = $input->manakai_getc_until |
($_[0], |
| 742 |
(qr/(?![$special_range\x{FDD0}-\x{FDDF}\x{FFFE}\x{FFFF}\x{1FFFE}\x{1FFFF}\x{2FFFE}\x{2FFFF}\x{3FFFE}\x{3FFFF}\x{4FFFE}\x{4FFFF}\x{5FFFE}\x{5FFFF}\x{6FFFE}\x{6FFFF}\x{7FFFE}\x{7FFFF}\x{8FFFE}\x{8FFFF}\x{9FFFE}\x{9FFFF}\x{AFFFE}\x{AFFFF}\x{BFFFE}\x{BFFFF}\x{CFFFE}\x{CFFFF}\x{DFFFE}\x{DFFFF}\x{EFFFE}\x{EFFFF}\x{FFFFE}\x{FFFFF}])[\x20-\x7E\xA0-\x{D7FF}\x{E000}-\x{10FFFD}]/); |
qr/(?![$specials_range\x{FDD0}-\x{FDDF}\x{FFFE}\x{FFFF}\x{1FFFE}\x{1FFFF}\x{2FFFE}\x{2FFFF}\x{3FFFE}\x{3FFFF}\x{4FFFE}\x{4FFFF}\x{5FFFE}\x{5FFFF}\x{6FFFE}\x{6FFFF}\x{7FFFE}\x{7FFFF}\x{8FFFE}\x{8FFFF}\x{9FFFE}\x{9FFFF}\x{AFFFE}\x{AFFFF}\x{BFFFE}\x{BFFFF}\x{CFFFE}\x{CFFFF}\x{DFFFE}\x{DFFFF}\x{EFFFE}\x{EFFFF}\x{FFFFE}\x{FFFFF}])[\x20-\x7E\xA0-\x{D7FF}\x{E000}-\x{10FFFD}]/, |
| 743 |
if ($s) { |
$_[2]); |
| 744 |
$self->{column} += length $$s; |
if ($count) { |
| 745 |
$self->{column_prev} += length $$s; |
$self->{column} += $count; |
| 746 |
$self->{prev_char} = [-1, -1, -1]; |
$self->{column_prev} += $count; |
| 747 |
$self->{next_char} = -1; |
$self->{prev_char} = [-1, -1, -1]; |
| 748 |
} |
$self->{next_char} = -1; |
| 749 |
return $s; |
} |
| 750 |
}; # $self->{getc_until} |
return $count; |
| 751 |
# } else { |
}; # $self->{read_until} |
| 752 |
# $self->{getc_until} = sub { |
$self->{read_until}=sub{0}; |
|
# my $special_range = shift; |
|
|
# return undef if defined $self->{next_next_char}; |
|
|
# my $c = $input->getc; |
|
|
# if ($c =~ /^(?![$special_range\x{FDD0}-\x{FDDF}\x{FFFE}\x{FFFF}\x{1FFFE}\x{1FFFF}\x{2FFFE}\x{2FFFF}\x{3FFFE}\x{3FFFF}\x{4FFFE}\x{4FFFF}\x{5FFFE}\x{5FFFF}\x{6FFFE}\x{6FFFF}\x{7FFFE}\x{7FFFF}\x{8FFFE}\x{8FFFF}\x{9FFFE}\x{9FFFF}\x{AFFFE}\x{AFFFF}\x{BFFFE}\x{BFFFF}\x{CFFFE}\x{CFFFF}\x{DFFFE}\x{DFFFF}\x{EFFFE}\x{EFFFF}\x{FFFFE}\x{FFFFF}])[\x20-\x7E\xA0-\x{D7FF}\x{E000}-\x{10FFFD}]/) { |
|
|
# $self->{column}++; |
|
|
# $self->{column_prev}++; |
|
|
# $self->{prev_char} = [-1, -1, -1]; |
|
|
# $self->{next_char} = -1; |
|
|
# return \$c; |
|
|
# } elsif (defined $c) { |
|
|
# #$input->ungetc (ord $c); |
|
|
# $self->{next_next_char} = $c; |
|
|
# return undef; |
|
|
# } else { |
|
|
# return undef; |
|
|
# } |
|
|
# }; # $self->{getc_until} |
|
|
# } |
|
| 753 |
|
|
| 754 |
my $onerror = $_[2] || sub { |
my $onerror = $_[2] || sub { |
| 755 |
my (%opt) = @_; |
my (%opt) = @_; |
| 921 |
undef $self->{last_emitted_start_tag_name}; |
undef $self->{last_emitted_start_tag_name}; |
| 922 |
#$self->{prev_state}; # initialized when used |
#$self->{prev_state}; # initialized when used |
| 923 |
delete $self->{self_closing}; |
delete $self->{self_closing}; |
| 924 |
|
$self->{char_buffer} = ''; |
| 925 |
|
$self->{char_buffer_pos} = 0; |
| 926 |
# $self->{next_char} |
# $self->{next_char} |
| 927 |
!!!next-input-character; |
!!!next-input-character; |
| 928 |
$self->{token} = []; |
$self->{token} = []; |
| 1049 |
data => chr $self->{next_char}, |
data => chr $self->{next_char}, |
| 1050 |
line => $self->{line}, column => $self->{column}, |
line => $self->{line}, column => $self->{column}, |
| 1051 |
}; |
}; |
| 1052 |
|
$self->{read_until}->($token->{data}, q[-!<>&], length $token->{data}); |
|
my $s = $self->{getc_until}->(q[-!<>&]); |
|
|
if ($s) { |
|
|
$token->{data} .= $$s; |
|
|
} |
|
| 1053 |
|
|
| 1054 |
## Stay in the data state |
## Stay in the data state |
| 1055 |
!!!next-input-character; |
!!!next-input-character; |
| 1765 |
} else { |
} else { |
| 1766 |
!!!cp (100); |
!!!cp (100); |
| 1767 |
$self->{current_attribute}->{value} .= chr ($self->{next_char}); |
$self->{current_attribute}->{value} .= chr ($self->{next_char}); |
| 1768 |
|
$self->{read_until}->($self->{current_attribute}->{value}, |
| 1769 |
|
q["&], |
| 1770 |
|
length $self->{current_attribute}->{value}); |
| 1771 |
|
|
| 1772 |
## Stay in the state |
## Stay in the state |
| 1773 |
!!!next-input-character; |
!!!next-input-character; |
| 1774 |
redo A; |
redo A; |
| 1816 |
} else { |
} else { |
| 1817 |
!!!cp (106); |
!!!cp (106); |
| 1818 |
$self->{current_attribute}->{value} .= chr ($self->{next_char}); |
$self->{current_attribute}->{value} .= chr ($self->{next_char}); |
| 1819 |
|
$self->{read_until}->($self->{current_attribute}->{value}, |
| 1820 |
|
q['&], |
| 1821 |
|
length $self->{current_attribute}->{value}); |
| 1822 |
|
|
| 1823 |
## Stay in the state |
## Stay in the state |
| 1824 |
!!!next-input-character; |
!!!next-input-character; |
| 1825 |
redo A; |
redo A; |
| 1902 |
!!!cp (116); |
!!!cp (116); |
| 1903 |
} |
} |
| 1904 |
$self->{current_attribute}->{value} .= chr ($self->{next_char}); |
$self->{current_attribute}->{value} .= chr ($self->{next_char}); |
| 1905 |
|
$self->{read_until}->($self->{current_attribute}->{value}, |
| 1906 |
|
q["'=& >], |
| 1907 |
|
length $self->{current_attribute}->{value}); |
| 1908 |
|
|
| 1909 |
## Stay in the state |
## Stay in the state |
| 1910 |
!!!next-input-character; |
!!!next-input-character; |
| 1911 |
redo A; |
redo A; |
| 2050 |
} else { |
} else { |
| 2051 |
!!!cp (126); |
!!!cp (126); |
| 2052 |
$self->{current_token}->{data} .= chr ($self->{next_char}); # comment |
$self->{current_token}->{data} .= chr ($self->{next_char}); # comment |
| 2053 |
|
$self->{read_until}->($self->{current_token}->{data}, |
| 2054 |
|
q[>], |
| 2055 |
|
length $self->{current_token}->{data}); |
| 2056 |
|
|
| 2057 |
## Stay in the state. |
## Stay in the state. |
| 2058 |
!!!next-input-character; |
!!!next-input-character; |
| 2059 |
redo A; |
redo A; |
| 2288 |
} else { |
} else { |
| 2289 |
!!!cp (147); |
!!!cp (147); |
| 2290 |
$self->{current_token}->{data} .= chr ($self->{next_char}); # comment |
$self->{current_token}->{data} .= chr ($self->{next_char}); # comment |
| 2291 |
|
$self->{read_until}->($self->{current_token}->{data}, |
| 2292 |
|
q[-], |
| 2293 |
|
length $self->{current_token}->{data}); |
| 2294 |
|
|
| 2295 |
## Stay in the state |
## Stay in the state |
| 2296 |
!!!next-input-character; |
!!!next-input-character; |
| 2297 |
redo A; |
redo A; |
| 2657 |
!!!cp (190); |
!!!cp (190); |
| 2658 |
$self->{current_token}->{public_identifier} # DOCTYPE |
$self->{current_token}->{public_identifier} # DOCTYPE |
| 2659 |
.= chr $self->{next_char}; |
.= chr $self->{next_char}; |
| 2660 |
|
$self->{read_until}->($self->{current_token}->{public_identifier}, |
| 2661 |
|
q[">], |
| 2662 |
|
length $self->{current_token}->{public_identifier}); |
| 2663 |
|
|
| 2664 |
## Stay in the state |
## Stay in the state |
| 2665 |
!!!next-input-character; |
!!!next-input-character; |
| 2666 |
redo A; |
redo A; |
| 2697 |
!!!cp (194); |
!!!cp (194); |
| 2698 |
$self->{current_token}->{public_identifier} # DOCTYPE |
$self->{current_token}->{public_identifier} # DOCTYPE |
| 2699 |
.= chr $self->{next_char}; |
.= chr $self->{next_char}; |
| 2700 |
|
$self->{read_until}->($self->{current_token}->{public_identifier}, |
| 2701 |
|
q['>], |
| 2702 |
|
length $self->{current_token}->{public_identifier}); |
| 2703 |
|
|
| 2704 |
## Stay in the state |
## Stay in the state |
| 2705 |
!!!next-input-character; |
!!!next-input-character; |
| 2706 |
redo A; |
redo A; |
| 2837 |
!!!cp (210); |
!!!cp (210); |
| 2838 |
$self->{current_token}->{system_identifier} # DOCTYPE |
$self->{current_token}->{system_identifier} # DOCTYPE |
| 2839 |
.= chr $self->{next_char}; |
.= chr $self->{next_char}; |
| 2840 |
|
$self->{read_until}->($self->{current_token}->{system_identifier}, |
| 2841 |
|
q[">], |
| 2842 |
|
length $self->{current_token}->{system_identifier}); |
| 2843 |
|
|
| 2844 |
## Stay in the state |
## Stay in the state |
| 2845 |
!!!next-input-character; |
!!!next-input-character; |
| 2846 |
redo A; |
redo A; |
| 2877 |
!!!cp (214); |
!!!cp (214); |
| 2878 |
$self->{current_token}->{system_identifier} # DOCTYPE |
$self->{current_token}->{system_identifier} # DOCTYPE |
| 2879 |
.= chr $self->{next_char}; |
.= chr $self->{next_char}; |
| 2880 |
|
$self->{read_until}->($self->{current_token}->{system_identifier}, |
| 2881 |
|
q['>], |
| 2882 |
|
length $self->{current_token}->{system_identifier}); |
| 2883 |
|
|
| 2884 |
## Stay in the state |
## Stay in the state |
| 2885 |
!!!next-input-character; |
!!!next-input-character; |
| 2886 |
redo A; |
redo A; |
| 2941 |
redo A; |
redo A; |
| 2942 |
} else { |
} else { |
| 2943 |
!!!cp (221); |
!!!cp (221); |
| 2944 |
|
my $s = ''; |
| 2945 |
|
$self->{read_until}->($s, q[>], 0); |
| 2946 |
|
|
| 2947 |
## Stay in the state |
## Stay in the state |
| 2948 |
!!!next-input-character; |
!!!next-input-character; |
| 2949 |
redo A; |
redo A; |
| 2972 |
} else { |
} else { |
| 2973 |
!!!cp (221.4); |
!!!cp (221.4); |
| 2974 |
$self->{current_token}->{data} .= chr $self->{next_char}; |
$self->{current_token}->{data} .= chr $self->{next_char}; |
| 2975 |
|
$self->{read_until}->($self->{current_token}->{data}, |
| 2976 |
|
q<]>, |
| 2977 |
|
length $self->{current_token}->{data}); |
| 2978 |
|
|
| 2979 |
## Stay in the state. |
## Stay in the state. |
| 2980 |
!!!next-input-character; |
!!!next-input-character; |
| 2981 |
redo A; |
redo A; |
| 3362 |
!!!cp (1026); |
!!!cp (1026); |
| 3363 |
!!!parse-error (type => 'bare ero', |
!!!parse-error (type => 'bare ero', |
| 3364 |
line => $self->{line_prev}, |
line => $self->{line_prev}, |
| 3365 |
column => $self->{column_prev}); |
column => $self->{column_prev} - length $self->{state_keyword}); |
| 3366 |
$data = '&' . $self->{state_keyword}; |
$data = '&' . $self->{state_keyword}; |
| 3367 |
# |
# |
| 3368 |
} |
} |
| 4500 |
unless ($self->{insertion_mode} == BEFORE_HEAD_IM) { |
unless ($self->{insertion_mode} == BEFORE_HEAD_IM) { |
| 4501 |
!!!cp ('t88.2'); |
!!!cp ('t88.2'); |
| 4502 |
$self->{open_elements}->[-1]->[0]->manakai_append_text ($1); |
$self->{open_elements}->[-1]->[0]->manakai_append_text ($1); |
| 4503 |
|
# |
| 4504 |
} else { |
} else { |
| 4505 |
!!!cp ('t88.1'); |
!!!cp ('t88.1'); |
| 4506 |
## Ignore the token. |
## Ignore the token. |
| 4507 |
!!!next-token; |
# |
|
next B; |
|
| 4508 |
} |
} |
| 4509 |
unless (length $token->{data}) { |
unless (length $token->{data}) { |
| 4510 |
!!!cp ('t88'); |
!!!cp ('t88'); |
| 4511 |
!!!next-token; |
!!!next-token; |
| 4512 |
next B; |
next B; |
| 4513 |
} |
} |
| 4514 |
|
## TODO: set $token->{column} appropriately |
| 4515 |
} |
} |
| 4516 |
|
|
| 4517 |
if ($self->{insertion_mode} == BEFORE_HEAD_IM) { |
if ($self->{insertion_mode} == BEFORE_HEAD_IM) { |
| 7756 |
## TODO: script stuffs |
## TODO: script stuffs |
| 7757 |
} # _tree_construct_main |
} # _tree_construct_main |
| 7758 |
|
|
| 7759 |
sub set_inner_html ($$$;$) { |
sub set_inner_html ($$$$;$) { |
| 7760 |
my $class = shift; |
my $class = shift; |
| 7761 |
my $node = shift; |
my $node = shift; |
| 7762 |
my $s = \$_[0]; |
#my $s = \$_[0]; |
| 7763 |
my $onerror = $_[1]; |
my $onerror = $_[1]; |
| 7764 |
my $get_wrapper = $_[2] || sub ($) { return $_[0] }; |
my $get_wrapper = $_[2] || sub ($) { return $_[0] }; |
| 7765 |
|
|
| 7780 |
} |
} |
| 7781 |
|
|
| 7782 |
## Step 3, 4, 5 # MUST |
## Step 3, 4, 5 # MUST |
| 7783 |
$class->parse_char_string ($$s => $node, $onerror, $get_wrapper); |
$class->parse_char_string ($_[0] => $node, $onerror, $get_wrapper); |
| 7784 |
} elsif ($nt == 1) { |
} elsif ($nt == 1) { |
| 7785 |
## TODO: If non-html element |
## TODO: If non-html element |
| 7786 |
|
|
| 7799 |
my $i = 0; |
my $i = 0; |
| 7800 |
$p->{line_prev} = $p->{line} = 1; |
$p->{line_prev} = $p->{line} = 1; |
| 7801 |
$p->{column_prev} = $p->{column} = 0; |
$p->{column_prev} = $p->{column} = 0; |
| 7802 |
|
require Whatpm::Charset::DecodeHandle; |
| 7803 |
|
my $input = Whatpm::Charset::DecodeHandle::CharString->new (\($_[0])); |
| 7804 |
|
$input = $get_wrapper->($input); |
| 7805 |
$p->{set_next_char} = sub { |
$p->{set_next_char} = sub { |
| 7806 |
my $self = shift; |
my $self = shift; |
| 7807 |
|
|
| 7808 |
pop @{$self->{prev_char}}; |
pop @{$self->{prev_char}}; |
| 7809 |
unshift @{$self->{prev_char}}, $self->{next_char}; |
unshift @{$self->{prev_char}}, $self->{next_char}; |
| 7810 |
|
|
| 7811 |
$self->{next_char} = -1 and return if $i >= length $$s; |
my $char = ''; |
| 7812 |
$self->{next_char} = ord substr $$s, $i++, 1; |
if (defined $self->{next_next_char}) { |
| 7813 |
|
$char = $self->{next_next_char}; |
| 7814 |
|
delete $self->{next_next_char}; |
| 7815 |
|
$self->{next_char} = ord $char; |
| 7816 |
|
} else { |
| 7817 |
|
if ($input->read ($char, 1)) { |
| 7818 |
|
$self->{next_char} = ord $char; |
| 7819 |
|
} else { |
| 7820 |
|
$self->{next_char} = -1; |
| 7821 |
|
return; |
| 7822 |
|
} |
| 7823 |
|
} |
| 7824 |
|
|
| 7825 |
($p->{line_prev}, $p->{column_prev}) = ($p->{line}, $p->{column}); |
($p->{line_prev}, $p->{column_prev}) = ($p->{line}, $p->{column}); |
| 7826 |
$p->{column}++; |
$p->{column}++; |
| 7830 |
$p->{column} = 0; |
$p->{column} = 0; |
| 7831 |
!!!cp ('i1'); |
!!!cp ('i1'); |
| 7832 |
} elsif ($self->{next_char} == 0x000D) { # CR |
} elsif ($self->{next_char} == 0x000D) { # CR |
| 7833 |
$i++ if substr ($$s, $i, 1) eq "\x0A"; |
## TODO: support for abort/streaming |
| 7834 |
|
my $next = ''; |
| 7835 |
|
if ($input->read ($next, 1) and $next ne "\x0A") { |
| 7836 |
|
$self->{next_next_char} = $next; |
| 7837 |
|
} |
| 7838 |
$self->{next_char} = 0x000A; # LF # MUST |
$self->{next_char} = 0x000A; # LF # MUST |
| 7839 |
$p->{line}++; |
$p->{line}++; |
| 7840 |
$p->{column} = 0; |
$p->{column} = 0; |
| 7879 |
$p->{prev_char} = [-1, -1, -1]; |
$p->{prev_char} = [-1, -1, -1]; |
| 7880 |
$p->{next_char} = -1; |
$p->{next_char} = -1; |
| 7881 |
|
|
| 7882 |
$p->{getc_until} = sub { |
$p->{read_until} = sub { |
| 7883 |
## TODO: ... |
#my ($scalar, $specials_range, $offset) = @_; |
| 7884 |
return undef; |
my $specials_range = $_[1]; |
| 7885 |
}; # $p->{getc_until}; |
return 0 if defined $p->{next_next_char}; |
| 7886 |
|
my $count = $input->manakai_read_until |
| 7887 |
|
($_[0], |
| 7888 |
|
qr/(?![$specials_range\x{FDD0}-\x{FDDF}\x{FFFE}\x{FFFF}\x{1FFFE}\x{1FFFF}\x{2FFFE}\x{2FFFF}\x{3FFFE}\x{3FFFF}\x{4FFFE}\x{4FFFF}\x{5FFFE}\x{5FFFF}\x{6FFFE}\x{6FFFF}\x{7FFFE}\x{7FFFF}\x{8FFFE}\x{8FFFF}\x{9FFFE}\x{9FFFF}\x{AFFFE}\x{AFFFF}\x{BFFFE}\x{BFFFF}\x{CFFFE}\x{CFFFF}\x{DFFFE}\x{DFFFF}\x{EFFFE}\x{EFFFF}\x{FFFFE}\x{FFFFF}])[\x20-\x7E\xA0-\x{D7FF}\x{E000}-\x{10FFFD}]/, |
| 7889 |
|
$_[2]); |
| 7890 |
|
if ($count) { |
| 7891 |
|
$p->{column} += $count; |
| 7892 |
|
$p->{column_prev} += $count; |
| 7893 |
|
$p->{prev_char} = [-1, -1, -1]; |
| 7894 |
|
$p->{next_char} = -1; |
| 7895 |
|
} |
| 7896 |
|
return $count; |
| 7897 |
|
}; # $p->{read_until} |
| 7898 |
|
|
| 7899 |
my $ponerror = $onerror || sub { |
my $ponerror = $onerror || sub { |
| 7900 |
my (%opt) = @_; |
my (%opt) = @_; |
| 7910 |
$ponerror->(line => $p->{line}, column => $p->{column}, @_); |
$ponerror->(line => $p->{line}, column => $p->{column}, @_); |
| 7911 |
}; |
}; |
| 7912 |
|
|
| 7913 |
|
my $char_onerror = sub { |
| 7914 |
|
my (undef, $type, %opt) = @_; |
| 7915 |
|
$ponerror->(layer => 'encode', |
| 7916 |
|
line => $p->{line}, column => $p->{column} + 1, |
| 7917 |
|
%opt, type => $type); |
| 7918 |
|
}; # $char_onerror |
| 7919 |
|
$input->onerror ($char_onerror); |
| 7920 |
|
|
| 7921 |
$p->_initialize_tokenizer; |
$p->_initialize_tokenizer; |
| 7922 |
$p->_initialize_tree_constructor; |
$p->_initialize_tree_constructor; |
| 7923 |
|
|