/[suikacvs]/markup/html/whatpm/Whatpm/HTML.pm.src
Suika

Diff of /markup/html/whatpm/Whatpm/HTML.pm.src

Parent Directory Parent Directory | Revision Log Revision Log | View Patch Patch

revision 1.172 by wakaba, Sun Sep 14 03:07:58 2008 UTC revision 1.182 by wakaba, Mon Sep 15 07:19:03 2008 UTC
# Line 3  use strict; Line 3  use strict;
3  our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r};  our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r};
4  use Error qw(:try);  use Error qw(:try);
5    
6    ## NOTE: This module don't check all HTML5 parse errors; character
7    ## encoding related parse errors are expected to be handled by relevant
8    ## modules.
9    ## Parse errors for control characters that are not allowed in HTML5
10    ## documents, for surrogate code points, and for noncharacter code
11    ## points, as well as U+FFFD substitions for characters whose code points
12    ## is higher than U+10FFFF may be detected by combining the parser with
13    ## the checker implemented by Whatpm::Charset::UnicodeChecker (for its
14    ## usage example, see |t/HTML-tree.t| in the Whatpm package or the
15    ## WebHACC::Language::HTML module in the WebHACC package).
16    
17  ## ISSUE:  ## ISSUE:
18  ## var doc = implementation.createDocument (null, null, null);  ## var doc = implementation.createDocument (null, null, null);
19  ## doc.write ('');  ## doc.write ('');
# Line 496  sub parse_byte_stream ($$$$;$$) { Line 507  sub parse_byte_stream ($$$$;$$) {
507                      line => 1, column => 1,                      line => 1, column => 1,
508                      layer => 'encode');                      layer => 'encode');
509    } elsif (not ($e_status &    } elsif (not ($e_status &
510                  Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL())) {                  Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL ())) {
511      $self->{input_encoding} = $charset->get_iana_name;      $self->{input_encoding} = $charset->get_iana_name;
512      !!!parse-error (type => 'chardecode:no error',      !!!parse-error (type => 'chardecode:no error',
513                      text => $self->{input_encoding},                      text => $self->{input_encoding},
# Line 561  sub parse_byte_stream ($$$$;$$) { Line 572  sub parse_byte_stream ($$$$;$$) {
572    my $char_onerror = sub {    my $char_onerror = sub {
573      my (undef, $type, %opt) = @_;      my (undef, $type, %opt) = @_;
574      !!!parse-error (layer => 'encode',      !!!parse-error (layer => 'encode',
575                      %opt, type => $type,                      line => $self->{line}, column => $self->{column} + 1,
576                      line => $self->{line}, column => $self->{column} + 1);                      %opt, type => $type);
577      if ($opt{octets}) {      if ($opt{octets}) {
578        ${$opt{octets}} = "\x{FFFD}"; # relacement character        ${$opt{octets}} = "\x{FFFD}"; # relacement character
579      }      }
# Line 571  sub parse_byte_stream ($$$$;$$) { Line 582  sub parse_byte_stream ($$$$;$$) {
582    my $wrapped_char_stream = $get_wrapper->($char_stream);    my $wrapped_char_stream = $get_wrapper->($char_stream);
583    $wrapped_char_stream->onerror ($char_onerror);    $wrapped_char_stream->onerror ($char_onerror);
584    
585    my @args = @_; shift @args; # $s    my @args = ($_[1], $_[2]); # $doc, $onerror - $get_wrapper = undef;
586    my $return;    my $return;
587    try {    try {
588      $return = $self->parse_char_stream ($wrapped_char_stream, @args);        $return = $self->parse_char_stream ($wrapped_char_stream, @args);  
# Line 586  sub parse_byte_stream ($$$$;$$) { Line 597  sub parse_byte_stream ($$$$;$$) {
597                        line => 1, column => 1,                        line => 1, column => 1,
598                        layer => 'encode');                        layer => 'encode');
599      } elsif (not ($e_status &      } elsif (not ($e_status &
600                    Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL())) {                    Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL ())) {
601        $self->{input_encoding} = $charset->get_iana_name;        $self->{input_encoding} = $charset->get_iana_name;
602        !!!parse-error (type => 'chardecode:no error',        !!!parse-error (type => 'chardecode:no error',
603                        text => $self->{input_encoding},                        text => $self->{input_encoding},
# Line 621  sub parse_char_string ($$$;$$) { Line 632  sub parse_char_string ($$$;$$) {
632    my $s = ref $_[0] ? $_[0] : \($_[0]);    my $s = ref $_[0] ? $_[0] : \($_[0]);
633    require Whatpm::Charset::DecodeHandle;    require Whatpm::Charset::DecodeHandle;
634    my $input = Whatpm::Charset::DecodeHandle::CharString->new ($s);    my $input = Whatpm::Charset::DecodeHandle::CharString->new ($s);
   if ($_[3]) {  
     $input = $_[3]->($input);  
   }  
635    return $self->parse_char_stream ($input, @_[1..$#_]);    return $self->parse_char_stream ($input, @_[1..$#_]);
636  } # parse_char_string  } # parse_char_string
637  *parse_string = \&parse_char_string; ## NOTE: Alias for backward compatibility.  *parse_string = \&parse_char_string; ## NOTE: Alias for backward compatibility.
638    
639  sub parse_char_stream ($$$;$) {  sub parse_char_stream ($$$;$$) {
640    my $self = ref $_[0] ? shift : shift->new;    my $self = ref $_[0] ? shift : shift->new;
641    my $input = $_[0];    my $input = $_[0];
642    $self->{document} = $_[1];    $self->{document} = $_[1];
# Line 639  sub parse_char_stream ($$$;$) { Line 647  sub parse_char_stream ($$$;$) {
647    $self->{confident} = 1 unless exists $self->{confident};    $self->{confident} = 1 unless exists $self->{confident};
648    $self->{document}->input_encoding ($self->{input_encoding})    $self->{document}->input_encoding ($self->{input_encoding})
649        if defined $self->{input_encoding};        if defined $self->{input_encoding};
650    ## TODO: |{input_encoding}| is needless?
651    
   my $i = 0;  
652    $self->{line_prev} = $self->{line} = 1;    $self->{line_prev} = $self->{line} = 1;
653    $self->{column_prev} = $self->{column} = 0;    $self->{column_prev} = -1;
654      $self->{column} = 0;
655    $self->{set_next_char} = sub {    $self->{set_next_char} = sub {
656      my $self = shift;      my $self = shift;
657    
658      pop @{$self->{prev_char}};      my $char = '';
     unshift @{$self->{prev_char}}, $self->{next_char};  
   
     my $char;  
659      if (defined $self->{next_next_char}) {      if (defined $self->{next_next_char}) {
660        $char = $self->{next_next_char};        $char = $self->{next_next_char};
661        delete $self->{next_next_char};        delete $self->{next_next_char};
662          $self->{next_char} = ord $char;
663      } else {      } else {
664        $char = $input->getc;        $self->{char_buffer} = '';
665          $self->{char_buffer_pos} = 0;
666    
667          my $count = $input->manakai_read_until
668             ($self->{char_buffer}, qr/[^\x00\x0A\x0D]/, $self->{char_buffer_pos});
669          if ($count) {
670            $self->{line_prev} = $self->{line};
671            $self->{column_prev} = $self->{column};
672            $self->{column}++;
673            $self->{next_char}
674                = ord substr ($self->{char_buffer}, $self->{char_buffer_pos}++, 1);
675            return;
676          }
677    
678          if ($input->read ($char, 1)) {
679            $self->{next_char} = ord $char;
680          } else {
681            $self->{next_char} = -1;
682            return;
683          }
684      }      }
     $self->{next_char} = -1 and return unless defined $char;  
     $self->{next_char} = ord $char;  
685    
686      ($self->{line_prev}, $self->{column_prev})      ($self->{line_prev}, $self->{column_prev})
687          = ($self->{line}, $self->{column});          = ($self->{line}, $self->{column});
# Line 670  sub parse_char_stream ($$$;$) { Line 694  sub parse_char_stream ($$$;$) {
694      } elsif ($self->{next_char} == 0x000D) { # CR      } elsif ($self->{next_char} == 0x000D) { # CR
695        !!!cp ('j2');        !!!cp ('j2');
696  ## TODO: support for abort/streaming  ## TODO: support for abort/streaming
697        my $next = $input->getc;        my $next = '';
698        if (defined $next and $next ne "\x0A") {        if ($input->read ($next, 1) and $next ne "\x0A") {
699          $self->{next_next_char} = $next;          $self->{next_next_char} = $next;
700        }        }
701        $self->{next_char} = 0x000A; # LF # MUST        $self->{next_char} = 0x000A; # LF # MUST
702        $self->{line}++;        $self->{line}++;
703        $self->{column} = 0;        $self->{column} = 0;
     } elsif ($self->{next_char} > 0x10FFFF) {  
       !!!cp ('j3');  
       $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST  
704      } elsif ($self->{next_char} == 0x0000) { # NULL      } elsif ($self->{next_char} == 0x0000) { # NULL
705        !!!cp ('j4');        !!!cp ('j4');
706        !!!parse-error (type => 'NULL');        !!!parse-error (type => 'NULL');
707        $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST        $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
     } elsif ($self->{next_char} <= 0x0008 or  
              (0x000E <= $self->{next_char} and $self->{next_char} <= 0x001F) or  
              (0x007F <= $self->{next_char} and $self->{next_char} <= 0x009F) or  
              (0xD800 <= $self->{next_char} and $self->{next_char} <= 0xDFFF) or  
              (0xFDD0 <= $self->{next_char} and $self->{next_char} <= 0xFDDF) or  
 ## ISSUE: U+FDE0-U+FDEF are not excluded  
              {  
               0xFFFE => 1, 0xFFFF => 1, 0x1FFFE => 1, 0x1FFFF => 1,  
               0x2FFFE => 1, 0x2FFFF => 1, 0x3FFFE => 1, 0x3FFFF => 1,  
               0x4FFFE => 1, 0x4FFFF => 1, 0x5FFFE => 1, 0x5FFFF => 1,  
               0x6FFFE => 1, 0x6FFFF => 1, 0x7FFFE => 1, 0x7FFFF => 1,  
               0x8FFFE => 1, 0x8FFFF => 1, 0x9FFFE => 1, 0x9FFFF => 1,  
               0xAFFFE => 1, 0xAFFFF => 1, 0xBFFFE => 1, 0xBFFFF => 1,  
               0xCFFFE => 1, 0xCFFFF => 1, 0xDFFFE => 1, 0xDFFFF => 1,  
               0xEFFFE => 1, 0xEFFFF => 1, 0xFFFFE => 1, 0xFFFFF => 1,  
               0x10FFFE => 1, 0x10FFFF => 1,  
              }->{$self->{next_char}}) {  
       !!!cp ('j5');  
       if ($self->{next_char} < 0x10000) {  
         !!!parse-error (type => 'control char',  
                         text => (sprintf 'U+%04X', $self->{next_char}));  
       } else {  
         !!!parse-error (type => 'control char',  
                         text => (sprintf 'U-%08X', $self->{next_char}));  
       }  
708      }      }
709    };    };
   $self->{prev_char} = [-1, -1, -1];  
   $self->{next_char} = -1;  
710    
711    $self->{read_until} = sub {    $self->{read_until} = sub {
712      #my ($scalar, $specials_range, $offset) = @_;      #my ($scalar, $specials_range, $offset) = @_;
     my $specials_range = $_[1];  
713      return 0 if defined $self->{next_next_char};      return 0 if defined $self->{next_next_char};
714      my $count = $input->manakai_read_until  
715         ($_[0],      my $pattern = qr/[^$_[1]\x00\x0A\x0D]/;
716          qr/(?![$specials_range\x{FDD0}-\x{FDDF}\x{FFFE}\x{FFFF}\x{1FFFE}\x{1FFFF}\x{2FFFE}\x{2FFFF}\x{3FFFE}\x{3FFFF}\x{4FFFE}\x{4FFFF}\x{5FFFE}\x{5FFFF}\x{6FFFE}\x{6FFFF}\x{7FFFE}\x{7FFFF}\x{8FFFE}\x{8FFFF}\x{9FFFE}\x{9FFFF}\x{AFFFE}\x{AFFFF}\x{BFFFE}\x{BFFFF}\x{CFFFE}\x{CFFFF}\x{DFFFE}\x{DFFFF}\x{EFFFE}\x{EFFFF}\x{FFFFE}\x{FFFFF}])[\x20-\x7E\xA0-\x{D7FF}\x{E000}-\x{10FFFD}]/,      my $offset = $_[2] || 0;
717          $_[2]);  
718      if ($count) {      if ($self->{char_buffer_pos} < length $self->{char_buffer}) {
719        $self->{column} += $count;        pos ($self->{char_buffer}) = $self->{char_buffer_pos};
720        $self->{column_prev} += $count;        if ($self->{char_buffer} =~ /\G(?>$pattern)+/) {
721        $self->{prev_char} = [-1, -1, -1];          substr ($_[0], $offset)
722        $self->{next_char} = -1;              = substr ($self->{char_buffer}, $-[0], $+[0] - $-[0]);
723            my $count = $+[0] - $-[0];
724            if ($count) {
725              $self->{column} += $count;
726              $self->{char_buffer_pos} += $count;
727              $self->{line_prev} = $self->{line};
728              $self->{column_prev} = $self->{column} - 1;
729              $self->{prev_char} = [-1, -1, -1];
730              $self->{next_char} = -1;
731            }
732            return $count;
733          } else {
734            return 0;
735          }
736        } else {
737          my $count = $input->manakai_read_until ($_[0], $pattern, $_[2]);
738          if ($count) {
739            $self->{column} += $count;
740            $self->{line_prev} = $self->{line};
741            $self->{column_prev} = $self->{column} - 1;
742            $self->{prev_char} = [-1, -1, -1];
743            $self->{next_char} = -1;
744          }
745          return $count;
746      }      }
     return $count;  
747    }; # $self->{read_until}    }; # $self->{read_until}
748    
749    my $onerror = $_[2] || sub {    my $onerror = $_[2] || sub {
# Line 741  sub parse_char_stream ($$$;$) { Line 756  sub parse_char_stream ($$$;$) {
756      $onerror->(line => $self->{line}, column => $self->{column}, @_);      $onerror->(line => $self->{line}, column => $self->{column}, @_);
757    };    };
758    
759      my $char_onerror = sub {
760        my (undef, $type, %opt) = @_;
761        !!!parse-error (layer => 'encode',
762                        line => $self->{line}, column => $self->{column} + 1,
763                        %opt, type => $type);
764      }; # $char_onerror
765    
766      if ($_[3]) {
767        $input = $_[3]->($input);
768        $input->onerror ($char_onerror);
769      } else {
770        $input->onerror ($char_onerror) unless defined $input->onerror;
771      }
772    
773    $self->_initialize_tokenizer;    $self->_initialize_tokenizer;
774    $self->_initialize_tree_constructor;    $self->_initialize_tree_constructor;
775    $self->_construct_tree;    $self->_construct_tree;
# Line 901  sub _initialize_tokenizer ($) { Line 930  sub _initialize_tokenizer ($) {
930    undef $self->{last_emitted_start_tag_name};    undef $self->{last_emitted_start_tag_name};
931    #$self->{prev_state}; # initialized when used    #$self->{prev_state}; # initialized when used
932    delete $self->{self_closing};    delete $self->{self_closing};
933    # $self->{next_char}    $self->{char_buffer} = '';
934      $self->{char_buffer_pos} = 0;
935      $self->{prev_char} = [-1, -1, -1];
936      $self->{next_char} = -1;
937    !!!next-input-character;    !!!next-input-character;
938    $self->{token} = [];    $self->{token} = [];
939    # $self->{escape}    # $self->{escape}
# Line 1743  sub _get_next_token ($) { Line 1775  sub _get_next_token ($) {
1775        } else {        } else {
1776          !!!cp (100);          !!!cp (100);
1777          $self->{current_attribute}->{value} .= chr ($self->{next_char});          $self->{current_attribute}->{value} .= chr ($self->{next_char});
1778            $self->{read_until}->($self->{current_attribute}->{value},
1779                                  q["&],
1780                                  length $self->{current_attribute}->{value});
1781    
1782          ## Stay in the state          ## Stay in the state
1783          !!!next-input-character;          !!!next-input-character;
1784          redo A;          redo A;
# Line 1790  sub _get_next_token ($) { Line 1826  sub _get_next_token ($) {
1826        } else {        } else {
1827          !!!cp (106);          !!!cp (106);
1828          $self->{current_attribute}->{value} .= chr ($self->{next_char});          $self->{current_attribute}->{value} .= chr ($self->{next_char});
1829            $self->{read_until}->($self->{current_attribute}->{value},
1830                                  q['&],
1831                                  length $self->{current_attribute}->{value});
1832    
1833          ## Stay in the state          ## Stay in the state
1834          !!!next-input-character;          !!!next-input-character;
1835          redo A;          redo A;
# Line 1872  sub _get_next_token ($) { Line 1912  sub _get_next_token ($) {
1912            !!!cp (116);            !!!cp (116);
1913          }          }
1914          $self->{current_attribute}->{value} .= chr ($self->{next_char});          $self->{current_attribute}->{value} .= chr ($self->{next_char});
1915            $self->{read_until}->($self->{current_attribute}->{value},
1916                                  q["'=& >],
1917                                  length $self->{current_attribute}->{value});
1918    
1919          ## Stay in the state          ## Stay in the state
1920          !!!next-input-character;          !!!next-input-character;
1921          redo A;          redo A;
# Line 2016  sub _get_next_token ($) { Line 2060  sub _get_next_token ($) {
2060        } else {        } else {
2061          !!!cp (126);          !!!cp (126);
2062          $self->{current_token}->{data} .= chr ($self->{next_char}); # comment          $self->{current_token}->{data} .= chr ($self->{next_char}); # comment
2063            $self->{read_until}->($self->{current_token}->{data},
2064                                  q[>],
2065                                  length $self->{current_token}->{data});
2066    
2067          ## Stay in the state.          ## Stay in the state.
2068          !!!next-input-character;          !!!next-input-character;
2069          redo A;          redo A;
# Line 2250  sub _get_next_token ($) { Line 2298  sub _get_next_token ($) {
2298        } else {        } else {
2299          !!!cp (147);          !!!cp (147);
2300          $self->{current_token}->{data} .= chr ($self->{next_char}); # comment          $self->{current_token}->{data} .= chr ($self->{next_char}); # comment
2301            $self->{read_until}->($self->{current_token}->{data},
2302                                  q[-],
2303                                  length $self->{current_token}->{data});
2304    
2305          ## Stay in the state          ## Stay in the state
2306          !!!next-input-character;          !!!next-input-character;
2307          redo A;          redo A;
# Line 2615  sub _get_next_token ($) { Line 2667  sub _get_next_token ($) {
2667          !!!cp (190);          !!!cp (190);
2668          $self->{current_token}->{public_identifier} # DOCTYPE          $self->{current_token}->{public_identifier} # DOCTYPE
2669              .= chr $self->{next_char};              .= chr $self->{next_char};
2670            $self->{read_until}->($self->{current_token}->{public_identifier},
2671                                  q[">],
2672                                  length $self->{current_token}->{public_identifier});
2673    
2674          ## Stay in the state          ## Stay in the state
2675          !!!next-input-character;          !!!next-input-character;
2676          redo A;          redo A;
# Line 2651  sub _get_next_token ($) { Line 2707  sub _get_next_token ($) {
2707          !!!cp (194);          !!!cp (194);
2708          $self->{current_token}->{public_identifier} # DOCTYPE          $self->{current_token}->{public_identifier} # DOCTYPE
2709              .= chr $self->{next_char};              .= chr $self->{next_char};
2710            $self->{read_until}->($self->{current_token}->{public_identifier},
2711                                  q['>],
2712                                  length $self->{current_token}->{public_identifier});
2713    
2714          ## Stay in the state          ## Stay in the state
2715          !!!next-input-character;          !!!next-input-character;
2716          redo A;          redo A;
# Line 2787  sub _get_next_token ($) { Line 2847  sub _get_next_token ($) {
2847          !!!cp (210);          !!!cp (210);
2848          $self->{current_token}->{system_identifier} # DOCTYPE          $self->{current_token}->{system_identifier} # DOCTYPE
2849              .= chr $self->{next_char};              .= chr $self->{next_char};
2850            $self->{read_until}->($self->{current_token}->{system_identifier},
2851                                  q[">],
2852                                  length $self->{current_token}->{system_identifier});
2853    
2854          ## Stay in the state          ## Stay in the state
2855          !!!next-input-character;          !!!next-input-character;
2856          redo A;          redo A;
# Line 2823  sub _get_next_token ($) { Line 2887  sub _get_next_token ($) {
2887          !!!cp (214);          !!!cp (214);
2888          $self->{current_token}->{system_identifier} # DOCTYPE          $self->{current_token}->{system_identifier} # DOCTYPE
2889              .= chr $self->{next_char};              .= chr $self->{next_char};
2890            $self->{read_until}->($self->{current_token}->{system_identifier},
2891                                  q['>],
2892                                  length $self->{current_token}->{system_identifier});
2893    
2894          ## Stay in the state          ## Stay in the state
2895          !!!next-input-character;          !!!next-input-character;
2896          redo A;          redo A;
# Line 2883  sub _get_next_token ($) { Line 2951  sub _get_next_token ($) {
2951          redo A;          redo A;
2952        } else {        } else {
2953          !!!cp (221);          !!!cp (221);
2954            my $s = '';
2955            $self->{read_until}->($s, q[>], 0);
2956    
2957          ## Stay in the state          ## Stay in the state
2958          !!!next-input-character;          !!!next-input-character;
2959          redo A;          redo A;
# Line 2911  sub _get_next_token ($) { Line 2982  sub _get_next_token ($) {
2982        } else {        } else {
2983          !!!cp (221.4);          !!!cp (221.4);
2984          $self->{current_token}->{data} .= chr $self->{next_char};          $self->{current_token}->{data} .= chr $self->{next_char};
2985            $self->{read_until}->($self->{current_token}->{data},
2986                                  q<]>,
2987                                  length $self->{current_token}->{data});
2988    
2989          ## Stay in the state.          ## Stay in the state.
2990          !!!next-input-character;          !!!next-input-character;
2991          redo A;          redo A;
# Line 3297  sub _get_next_token ($) { Line 3372  sub _get_next_token ($) {
3372          !!!cp (1026);          !!!cp (1026);
3373          !!!parse-error (type => 'bare ero',          !!!parse-error (type => 'bare ero',
3374                          line => $self->{line_prev},                          line => $self->{line_prev},
3375                          column => $self->{column_prev});                          column => $self->{column_prev} - length $self->{state_keyword});
3376          $data = '&' . $self->{state_keyword};          $data = '&' . $self->{state_keyword};
3377          #          #
3378        }        }
# Line 4435  sub _tree_construction_main ($) { Line 4510  sub _tree_construction_main ($) {
4510            unless ($self->{insertion_mode} == BEFORE_HEAD_IM) {            unless ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4511              !!!cp ('t88.2');              !!!cp ('t88.2');
4512              $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);              $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
4513                #
4514            } else {            } else {
4515              !!!cp ('t88.1');              !!!cp ('t88.1');
4516              ## Ignore the token.              ## Ignore the token.
4517              !!!next-token;              #
             next B;  
4518            }            }
4519            unless (length $token->{data}) {            unless (length $token->{data}) {
4520              !!!cp ('t88');              !!!cp ('t88');
4521              !!!next-token;              !!!next-token;
4522              next B;              next B;
4523            }            }
4524    ## TODO: set $token->{column} appropriately
4525          }          }
4526    
4527          if ($self->{insertion_mode} == BEFORE_HEAD_IM) {          if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
# Line 7690  sub _tree_construction_main ($) { Line 7766  sub _tree_construction_main ($) {
7766    ## TODO: script stuffs    ## TODO: script stuffs
7767  } # _tree_construct_main  } # _tree_construct_main
7768    
7769  sub set_inner_html ($$$;$) {  sub set_inner_html ($$$$;$) {
7770    my $class = shift;    my $class = shift;
7771    my $node = shift;    my $node = shift;
7772    my $s = \$_[0];    #my $s = \$_[0];
7773    my $onerror = $_[1];    my $onerror = $_[1];
7774    my $get_wrapper = $_[2] || sub ($) { return $_[0] };    my $get_wrapper = $_[2] || sub ($) { return $_[0] };
7775    
# Line 7714  sub set_inner_html ($$$;$) { Line 7790  sub set_inner_html ($$$;$) {
7790      }      }
7791    
7792      ## Step 3, 4, 5 # MUST      ## Step 3, 4, 5 # MUST
7793      $class->parse_char_string ($$s => $node, $onerror, $get_wrapper);      $class->parse_char_string ($_[0] => $node, $onerror, $get_wrapper);
7794    } elsif ($nt == 1) {    } elsif ($nt == 1) {
7795      ## TODO: If non-html element      ## TODO: If non-html element
7796    
# Line 7733  sub set_inner_html ($$$;$) { Line 7809  sub set_inner_html ($$$;$) {
7809      my $i = 0;      my $i = 0;
7810      $p->{line_prev} = $p->{line} = 1;      $p->{line_prev} = $p->{line} = 1;
7811      $p->{column_prev} = $p->{column} = 0;      $p->{column_prev} = $p->{column} = 0;
7812        require Whatpm::Charset::DecodeHandle;
7813        my $input = Whatpm::Charset::DecodeHandle::CharString->new (\($_[0]));
7814        $input = $get_wrapper->($input);
7815      $p->{set_next_char} = sub {      $p->{set_next_char} = sub {
7816        my $self = shift;        my $self = shift;
7817    
7818        pop @{$self->{prev_char}};        my $char = '';
7819        unshift @{$self->{prev_char}}, $self->{next_char};        if (defined $self->{next_next_char}) {
7820            $char = $self->{next_next_char};
7821        $self->{next_char} = -1 and return if $i >= length $$s;          delete $self->{next_next_char};
7822        $self->{next_char} = ord substr $$s, $i++, 1;          $self->{next_char} = ord $char;
7823          } else {
7824            $self->{char_buffer} = '';
7825            $self->{char_buffer_pos} = 0;
7826            
7827            my $count = $input->manakai_read_until
7828                ($self->{char_buffer}, qr/[^\x00\x0A\x0D]/,
7829                 $self->{char_buffer_pos});
7830            if ($count) {
7831              $self->{line_prev} = $self->{line};
7832              $self->{column_prev} = $self->{column};
7833              $self->{column}++;
7834              $self->{next_char}
7835                  = ord substr ($self->{char_buffer},
7836                                $self->{char_buffer_pos}++, 1);
7837              return;
7838            }
7839            
7840            if ($input->read ($char, 1)) {
7841              $self->{next_char} = ord $char;
7842            } else {
7843              $self->{next_char} = -1;
7844              return;
7845            }
7846          }
7847    
7848        ($p->{line_prev}, $p->{column_prev}) = ($p->{line}, $p->{column});        ($p->{line_prev}, $p->{column_prev}) = ($p->{line}, $p->{column});
7849        $p->{column}++;        $p->{column}++;
# Line 7750  sub set_inner_html ($$$;$) { Line 7853  sub set_inner_html ($$$;$) {
7853          $p->{column} = 0;          $p->{column} = 0;
7854          !!!cp ('i1');          !!!cp ('i1');
7855        } elsif ($self->{next_char} == 0x000D) { # CR        } elsif ($self->{next_char} == 0x000D) { # CR
7856          $i++ if substr ($$s, $i, 1) eq "\x0A";  ## TODO: support for abort/streaming
7857            my $next = '';
7858            if ($input->read ($next, 1) and $next ne "\x0A") {
7859              $self->{next_next_char} = $next;
7860            }
7861          $self->{next_char} = 0x000A; # LF # MUST          $self->{next_char} = 0x000A; # LF # MUST
7862          $p->{line}++;          $p->{line}++;
7863          $p->{column} = 0;          $p->{column} = 0;
7864          !!!cp ('i2');          !!!cp ('i2');
       } elsif ($self->{next_char} > 0x10FFFF) {  
         $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST  
         !!!cp ('i3');  
7865        } elsif ($self->{next_char} == 0x0000) { # NULL        } elsif ($self->{next_char} == 0x0000) { # NULL
7866          !!!cp ('i4');          !!!cp ('i4');
7867          !!!parse-error (type => 'NULL');          !!!parse-error (type => 'NULL');
7868          $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST          $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
       } elsif ($self->{next_char} <= 0x0008 or  
                (0x000E <= $self->{next_char} and  
                 $self->{next_char} <= 0x001F) or  
                (0x007F <= $self->{next_char} and  
                 $self->{next_char} <= 0x009F) or  
                (0xD800 <= $self->{next_char} and  
                 $self->{next_char} <= 0xDFFF) or  
                (0xFDD0 <= $self->{next_char} and  
                 $self->{next_char} <= 0xFDDF) or  
                {  
                 0xFFFE => 1, 0xFFFF => 1, 0x1FFFE => 1, 0x1FFFF => 1,  
                 0x2FFFE => 1, 0x2FFFF => 1, 0x3FFFE => 1, 0x3FFFF => 1,  
                 0x4FFFE => 1, 0x4FFFF => 1, 0x5FFFE => 1, 0x5FFFF => 1,  
                 0x6FFFE => 1, 0x6FFFF => 1, 0x7FFFE => 1, 0x7FFFF => 1,  
                 0x8FFFE => 1, 0x8FFFF => 1, 0x9FFFE => 1, 0x9FFFF => 1,  
                 0xAFFFE => 1, 0xAFFFF => 1, 0xBFFFE => 1, 0xBFFFF => 1,  
                 0xCFFFE => 1, 0xCFFFF => 1, 0xDFFFE => 1, 0xDFFFF => 1,  
                 0xEFFFE => 1, 0xEFFFF => 1, 0xFFFFE => 1, 0xFFFFF => 1,  
                 0x10FFFE => 1, 0x10FFFF => 1,  
                }->{$self->{next_char}}) {  
         !!!cp ('i4.1');  
         if ($self->{next_char} < 0x10000) {  
           !!!parse-error (type => 'control char',  
                           text => (sprintf 'U+%04X', $self->{next_char}));  
         } else {  
           !!!parse-error (type => 'control char',  
                           text => (sprintf 'U-%08X', $self->{next_char}));  
         }  
7869        }        }
7870      };      };
     $p->{prev_char} = [-1, -1, -1];  
     $p->{next_char} = -1;  
7871    
7872      $p->{read_until} = sub {      $p->{read_until} = sub {
7873        ## TODO: ...        #my ($scalar, $specials_range, $offset) = @_;
7874        return 0;        return 0 if defined $p->{next_next_char};
7875      }; # $p->{read_until};  
7876          my $pattern = qr/[^$_[1]\x00\x0A\x0D]/;
7877          my $offset = $_[2] || 0;
7878          
7879          if ($p->{char_buffer_pos} < length $p->{char_buffer}) {
7880            pos ($p->{char_buffer}) = $p->{char_buffer_pos};
7881            if ($p->{char_buffer} =~ /\G(?>$pattern)+/) {
7882              substr ($_[0], $offset)
7883                  = substr ($p->{char_buffer}, $-[0], $+[0] - $-[0]);
7884              my $count = $+[0] - $-[0];
7885              if ($count) {
7886                $p->{column} += $count;
7887                $p->{char_buffer_pos} += $count;
7888                $p->{line_prev} = $p->{line};
7889                $p->{column_prev} = $p->{column} - 1;
7890                $p->{prev_char} = [-1, -1, -1];
7891                $p->{next_char} = -1;
7892              }
7893              return $count;
7894            } else {
7895              return 0;
7896            }
7897          } else {
7898            my $count = $input->manakai_read_until ($_[0], $pattern, $_[2]);
7899            if ($count) {
7900              $p->{column} += $count;
7901              $p->{column_prev} += $count;
7902              $p->{prev_char} = [-1, -1, -1];
7903              $p->{next_char} = -1;
7904            }
7905            return $count;
7906          }
7907        }; # $p->{read_until}
7908    
7909      my $ponerror = $onerror || sub {      my $ponerror = $onerror || sub {
7910        my (%opt) = @_;        my (%opt) = @_;
# Line 7814  sub set_inner_html ($$$;$) { Line 7920  sub set_inner_html ($$$;$) {
7920        $ponerror->(line => $p->{line}, column => $p->{column}, @_);        $ponerror->(line => $p->{line}, column => $p->{column}, @_);
7921      };      };
7922            
7923        my $char_onerror = sub {
7924          my (undef, $type, %opt) = @_;
7925          $ponerror->(layer => 'encode',
7926                      line => $p->{line}, column => $p->{column} + 1,
7927                      %opt, type => $type);
7928        }; # $char_onerror
7929        $input->onerror ($char_onerror);
7930    
7931      $p->_initialize_tokenizer;      $p->_initialize_tokenizer;
7932      $p->_initialize_tree_constructor;      $p->_initialize_tree_constructor;
7933    

Legend:
Removed from v.1.172  
changed lines
  Added in v.1.182

[email protected]
ViewVC Help
Powered by ViewVC 1.1.24