/[suikacvs]/markup/html/whatpm/Whatpm/HTML.pm.src
Suika

Diff of /markup/html/whatpm/Whatpm/HTML.pm.src

Parent Directory Parent Directory | Revision Log Revision Log | View Patch Patch

revision 1.173 by wakaba, Sun Sep 14 03:59:08 2008 UTC revision 1.193 by wakaba, Sat Oct 4 04:06:33 2008 UTC
# Line 3  use strict; Line 3  use strict;
3  our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r};  our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r};
4  use Error qw(:try);  use Error qw(:try);
5    
6    ## NOTE: This module don't check all HTML5 parse errors; character
7    ## encoding related parse errors are expected to be handled by relevant
8    ## modules.
9    ## Parse errors for control characters that are not allowed in HTML5
10    ## documents, for surrogate code points, and for noncharacter code
11    ## points, as well as U+FFFD substitions for characters whose code points
12    ## is higher than U+10FFFF may be detected by combining the parser with
13    ## the checker implemented by Whatpm::Charset::UnicodeChecker (for its
14    ## usage example, see |t/HTML-tree.t| in the Whatpm package or the
15    ## WebHACC::Language::HTML module in the WebHACC package).
16    
17  ## ISSUE:  ## ISSUE:
18  ## var doc = implementation.createDocument (null, null, null);  ## var doc = implementation.createDocument (null, null, null);
19  ## doc.write ('');  ## doc.write ('');
# Line 130  my $el_category = { Line 141  my $el_category = {
141    address => ADDRESS_EL,    address => ADDRESS_EL,
142    applet => MISC_SCOPING_EL,    applet => MISC_SCOPING_EL,
143    area => MISC_SPECIAL_EL,    area => MISC_SPECIAL_EL,
144      article => MISC_SPECIAL_EL,
145      aside => MISC_SPECIAL_EL,
146    b => FORMATTING_EL,    b => FORMATTING_EL,
147    base => MISC_SPECIAL_EL,    base => MISC_SPECIAL_EL,
148    basefont => MISC_SPECIAL_EL,    basefont => MISC_SPECIAL_EL,
# Line 143  my $el_category = { Line 156  my $el_category = {
156    center => MISC_SPECIAL_EL,    center => MISC_SPECIAL_EL,
157    col => MISC_SPECIAL_EL,    col => MISC_SPECIAL_EL,
158    colgroup => MISC_SPECIAL_EL,    colgroup => MISC_SPECIAL_EL,
159      command => MISC_SPECIAL_EL,
160      datagrid => MISC_SPECIAL_EL,
161    dd => DD_EL,    dd => DD_EL,
162      details => MISC_SPECIAL_EL,
163      dialog => MISC_SPECIAL_EL,
164    dir => MISC_SPECIAL_EL,    dir => MISC_SPECIAL_EL,
165    div => DIV_EL,    div => DIV_EL,
166    dl => MISC_SPECIAL_EL,    dl => MISC_SPECIAL_EL,
167    dt => DT_EL,    dt => DT_EL,
168    em => FORMATTING_EL,    em => FORMATTING_EL,
169    embed => MISC_SPECIAL_EL,    embed => MISC_SPECIAL_EL,
170      eventsource => MISC_SPECIAL_EL,
171    fieldset => MISC_SPECIAL_EL,    fieldset => MISC_SPECIAL_EL,
172      figure => MISC_SPECIAL_EL,
173    font => FORMATTING_EL,    font => FORMATTING_EL,
174      footer => MISC_SPECIAL_EL,
175    form => FORM_EL,    form => FORM_EL,
176    frame => MISC_SPECIAL_EL,    frame => MISC_SPECIAL_EL,
177    frameset => FRAMESET_EL,    frameset => FRAMESET_EL,
# Line 162  my $el_category = { Line 182  my $el_category = {
182    h5 => HEADING_EL,    h5 => HEADING_EL,
183    h6 => HEADING_EL,    h6 => HEADING_EL,
184    head => MISC_SPECIAL_EL,    head => MISC_SPECIAL_EL,
185      header => MISC_SPECIAL_EL,
186    hr => MISC_SPECIAL_EL,    hr => MISC_SPECIAL_EL,
187    html => HTML_EL,    html => HTML_EL,
188    i => FORMATTING_EL,    i => FORMATTING_EL,
189    iframe => MISC_SPECIAL_EL,    iframe => MISC_SPECIAL_EL,
190    img => MISC_SPECIAL_EL,    img => MISC_SPECIAL_EL,
191      #image => MISC_SPECIAL_EL, ## NOTE: Commented out in the spec.
192    input => MISC_SPECIAL_EL,    input => MISC_SPECIAL_EL,
193    isindex => MISC_SPECIAL_EL,    isindex => MISC_SPECIAL_EL,
194    li => LI_EL,    li => LI_EL,
# Line 175  my $el_category = { Line 197  my $el_category = {
197    marquee => MISC_SCOPING_EL,    marquee => MISC_SCOPING_EL,
198    menu => MISC_SPECIAL_EL,    menu => MISC_SPECIAL_EL,
199    meta => MISC_SPECIAL_EL,    meta => MISC_SPECIAL_EL,
200      nav => MISC_SPECIAL_EL,
201    nobr => NOBR_EL | FORMATTING_EL,    nobr => NOBR_EL | FORMATTING_EL,
202    noembed => MISC_SPECIAL_EL,    noembed => MISC_SPECIAL_EL,
203    noframes => MISC_SPECIAL_EL,    noframes => MISC_SPECIAL_EL,
# Line 193  my $el_category = { Line 216  my $el_category = {
216    s => FORMATTING_EL,    s => FORMATTING_EL,
217    script => MISC_SPECIAL_EL,    script => MISC_SPECIAL_EL,
218    select => SELECT_EL,    select => SELECT_EL,
219      section => MISC_SPECIAL_EL,
220    small => FORMATTING_EL,    small => FORMATTING_EL,
221    spacer => MISC_SPECIAL_EL,    spacer => MISC_SPECIAL_EL,
222    strike => FORMATTING_EL,    strike => FORMATTING_EL,
# Line 312  my $foreign_attr_xname = { Line 336  my $foreign_attr_xname = {
336    
337  ## ISSUE: xmlns:xlink="non-xlink-ns" is not an error.  ## ISSUE: xmlns:xlink="non-xlink-ns" is not an error.
338    
339  my $c1_entity_char = {  my $charref_map = {
340      0x0D => 0x000A,
341    0x80 => 0x20AC,    0x80 => 0x20AC,
342    0x81 => 0xFFFD,    0x81 => 0xFFFD,
343    0x82 => 0x201A,    0x82 => 0x201A,
# Line 345  my $c1_entity_char = { Line 370  my $c1_entity_char = {
370    0x9D => 0xFFFD,    0x9D => 0xFFFD,
371    0x9E => 0x017E,    0x9E => 0x017E,
372    0x9F => 0x0178,    0x9F => 0x0178,
373  }; # $c1_entity_char  }; # $charref_map
374    $charref_map->{$_} = 0xFFFD
375        for 0x0000..0x0008, 0x000B, 0x000E..0x001F, 0x007F,
376            0xD800..0xDFFF, 0xFDD0..0xFDDF, ## ISSUE: 0xFDEF
377            0xFFFE, 0xFFFF, 0x1FFFE, 0x1FFFF, 0x2FFFE, 0x2FFFF, 0x3FFFE, 0x3FFFF,
378            0x4FFFE, 0x4FFFF, 0x5FFFE, 0x5FFFF, 0x6FFFE, 0x6FFFF, 0x7FFFE,
379            0x7FFFF, 0x8FFFE, 0x8FFFF, 0x9FFFE, 0x9FFFF, 0xAFFFE, 0xAFFFF,
380            0xBFFFE, 0xBFFFF, 0xCFFFE, 0xCFFFF, 0xDFFFE, 0xDFFFF, 0xEFFFE,
381            0xEFFFF, 0xFFFFE, 0xFFFFF, 0x10FFFE, 0x10FFFF;
382    
383    ## TODO: Invoke the reset algorithm when a resettable element is
384    ## created (cf. HTML5 revision 2259).
385    
386  sub parse_byte_string ($$$$;$) {  sub parse_byte_string ($$$$;$) {
387    my $self = shift;    my $self = shift;
# Line 390  sub parse_byte_stream ($$$$;$$) { Line 426  sub parse_byte_stream ($$$$;$$) {
426            ## TODO: Is this ok?  Transfer protocol's parameter should be            ## TODO: Is this ok?  Transfer protocol's parameter should be
427            ## interpreted in its semantics?            ## interpreted in its semantics?
428    
       ## ISSUE: Unsupported encoding is not ignored according to the spec.  
429        ($char_stream, $e_status) = $charset->get_decode_handle        ($char_stream, $e_status) = $charset->get_decode_handle
430            ($byte_stream, allow_error_reporting => 1,            ($byte_stream, allow_error_reporting => 1,
431             allow_fallback => 1);             allow_fallback => 1);
# Line 398  sub parse_byte_stream ($$$$;$$) { Line 433  sub parse_byte_stream ($$$$;$$) {
433          $self->{confident} = 1;          $self->{confident} = 1;
434          last SNIFFING;          last SNIFFING;
435        } else {        } else {
436          ## TODO: unsupported error          !!!parse-error (type => 'charset:not supported',
437                            layer => 'encode',
438                            line => 1, column => 1,
439                            value => $charset_name,
440                            level => $self->{level}->{uncertain});
441        }        }
442      }      }
443    
# Line 496  sub parse_byte_stream ($$$$;$$) { Line 535  sub parse_byte_stream ($$$$;$$) {
535                      line => 1, column => 1,                      line => 1, column => 1,
536                      layer => 'encode');                      layer => 'encode');
537    } elsif (not ($e_status &    } elsif (not ($e_status &
538                  Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL())) {                  Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL ())) {
539      $self->{input_encoding} = $charset->get_iana_name;      $self->{input_encoding} = $charset->get_iana_name;
540      !!!parse-error (type => 'chardecode:no error',      !!!parse-error (type => 'chardecode:no error',
541                      text => $self->{input_encoding},                      text => $self->{input_encoding},
# Line 561  sub parse_byte_stream ($$$$;$$) { Line 600  sub parse_byte_stream ($$$$;$$) {
600    my $char_onerror = sub {    my $char_onerror = sub {
601      my (undef, $type, %opt) = @_;      my (undef, $type, %opt) = @_;
602      !!!parse-error (layer => 'encode',      !!!parse-error (layer => 'encode',
603                      %opt, type => $type,                      line => $self->{line}, column => $self->{column} + 1,
604                      line => $self->{line}, column => $self->{column} + 1);                      %opt, type => $type);
605      if ($opt{octets}) {      if ($opt{octets}) {
606        ${$opt{octets}} = "\x{FFFD}"; # relacement character        ${$opt{octets}} = "\x{FFFD}"; # relacement character
607      }      }
# Line 571  sub parse_byte_stream ($$$$;$$) { Line 610  sub parse_byte_stream ($$$$;$$) {
610    my $wrapped_char_stream = $get_wrapper->($char_stream);    my $wrapped_char_stream = $get_wrapper->($char_stream);
611    $wrapped_char_stream->onerror ($char_onerror);    $wrapped_char_stream->onerror ($char_onerror);
612    
613    my @args = @_; shift @args; # $s    my @args = ($_[1], $_[2]); # $doc, $onerror - $get_wrapper = undef;
614    my $return;    my $return;
615    try {    try {
616      $return = $self->parse_char_stream ($wrapped_char_stream, @args);        $return = $self->parse_char_stream ($wrapped_char_stream, @args);  
# Line 586  sub parse_byte_stream ($$$$;$$) { Line 625  sub parse_byte_stream ($$$$;$$) {
625                        line => 1, column => 1,                        line => 1, column => 1,
626                        layer => 'encode');                        layer => 'encode');
627      } elsif (not ($e_status &      } elsif (not ($e_status &
628                    Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL())) {                    Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL ())) {
629        $self->{input_encoding} = $charset->get_iana_name;        $self->{input_encoding} = $charset->get_iana_name;
630        !!!parse-error (type => 'chardecode:no error',        !!!parse-error (type => 'chardecode:no error',
631                        text => $self->{input_encoding},                        text => $self->{input_encoding},
# Line 621  sub parse_char_string ($$$;$$) { Line 660  sub parse_char_string ($$$;$$) {
660    my $s = ref $_[0] ? $_[0] : \($_[0]);    my $s = ref $_[0] ? $_[0] : \($_[0]);
661    require Whatpm::Charset::DecodeHandle;    require Whatpm::Charset::DecodeHandle;
662    my $input = Whatpm::Charset::DecodeHandle::CharString->new ($s);    my $input = Whatpm::Charset::DecodeHandle::CharString->new ($s);
   if ($_[3]) {  
     $input = $_[3]->($input);  
   }  
663    return $self->parse_char_stream ($input, @_[1..$#_]);    return $self->parse_char_stream ($input, @_[1..$#_]);
664  } # parse_char_string  } # parse_char_string
665  *parse_string = \&parse_char_string; ## NOTE: Alias for backward compatibility.  *parse_string = \&parse_char_string; ## NOTE: Alias for backward compatibility.
666    
667  sub parse_char_stream ($$$;$) {  sub parse_char_stream ($$$;$$) {
668    my $self = ref $_[0] ? shift : shift->new;    my $self = ref $_[0] ? shift : shift->new;
669    my $input = $_[0];    my $input = $_[0];
670    $self->{document} = $_[1];    $self->{document} = $_[1];
# Line 639  sub parse_char_stream ($$$;$) { Line 675  sub parse_char_stream ($$$;$) {
675    $self->{confident} = 1 unless exists $self->{confident};    $self->{confident} = 1 unless exists $self->{confident};
676    $self->{document}->input_encoding ($self->{input_encoding})    $self->{document}->input_encoding ($self->{input_encoding})
677        if defined $self->{input_encoding};        if defined $self->{input_encoding};
678    ## TODO: |{input_encoding}| is needless?
679    
   my $i = 0;  
680    $self->{line_prev} = $self->{line} = 1;    $self->{line_prev} = $self->{line} = 1;
681    $self->{column_prev} = $self->{column} = 0;    $self->{column_prev} = -1;
682    $self->{set_next_char} = sub {    $self->{column} = 0;
683      $self->{set_nc} = sub {
684      my $self = shift;      my $self = shift;
685    
686      pop @{$self->{prev_char}};      my $char = '';
687      unshift @{$self->{prev_char}}, $self->{next_char};      if (defined $self->{next_nc}) {
688          $char = $self->{next_nc};
689      my $char;        delete $self->{next_nc};
690      if (defined $self->{next_next_char}) {        $self->{nc} = ord $char;
       $char = $self->{next_next_char};  
       delete $self->{next_next_char};  
691      } else {      } else {
692        $char = $input->getc;        $self->{char_buffer} = '';
693          $self->{char_buffer_pos} = 0;
694    
695          my $count = $input->manakai_read_until
696             ($self->{char_buffer}, qr/[^\x00\x0A\x0D]/, $self->{char_buffer_pos});
697          if ($count) {
698            $self->{line_prev} = $self->{line};
699            $self->{column_prev} = $self->{column};
700            $self->{column}++;
701            $self->{nc}
702                = ord substr ($self->{char_buffer}, $self->{char_buffer_pos}++, 1);
703            return;
704          }
705    
706          if ($input->read ($char, 1)) {
707            $self->{nc} = ord $char;
708          } else {
709            $self->{nc} = -1;
710            return;
711          }
712      }      }
     $self->{next_char} = -1 and return unless defined $char;  
     $self->{next_char} = ord $char;  
713    
714      ($self->{line_prev}, $self->{column_prev})      ($self->{line_prev}, $self->{column_prev})
715          = ($self->{line}, $self->{column});          = ($self->{line}, $self->{column});
716      $self->{column}++;      $self->{column}++;
717            
718      if ($self->{next_char} == 0x000A) { # LF      if ($self->{nc} == 0x000A) { # LF
719        !!!cp ('j1');        !!!cp ('j1');
720        $self->{line}++;        $self->{line}++;
721        $self->{column} = 0;        $self->{column} = 0;
722      } elsif ($self->{next_char} == 0x000D) { # CR      } elsif ($self->{nc} == 0x000D) { # CR
723        !!!cp ('j2');        !!!cp ('j2');
724  ## TODO: support for abort/streaming  ## TODO: support for abort/streaming
725        my $next = $input->getc;        my $next = '';
726        if (defined $next and $next ne "\x0A") {        if ($input->read ($next, 1) and $next ne "\x0A") {
727          $self->{next_next_char} = $next;          $self->{next_nc} = $next;
728        }        }
729        $self->{next_char} = 0x000A; # LF # MUST        $self->{nc} = 0x000A; # LF # MUST
730        $self->{line}++;        $self->{line}++;
731        $self->{column} = 0;        $self->{column} = 0;
732      } elsif ($self->{next_char} > 0x10FFFF) {      } elsif ($self->{nc} == 0x0000) { # NULL
       !!!cp ('j3');  
       $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST  
     } elsif ($self->{next_char} == 0x0000) { # NULL  
733        !!!cp ('j4');        !!!cp ('j4');
734        !!!parse-error (type => 'NULL');        !!!parse-error (type => 'NULL');
735        $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST        $self->{nc} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
     } elsif ($self->{next_char} <= 0x0008 or  
              (0x000E <= $self->{next_char} and $self->{next_char} <= 0x001F) or  
              (0x007F <= $self->{next_char} and $self->{next_char} <= 0x009F) or  
              (0xD800 <= $self->{next_char} and $self->{next_char} <= 0xDFFF) or  
              (0xFDD0 <= $self->{next_char} and $self->{next_char} <= 0xFDDF) or  
 ## ISSUE: U+FDE0-U+FDEF are not excluded  
              {  
               0xFFFE => 1, 0xFFFF => 1, 0x1FFFE => 1, 0x1FFFF => 1,  
               0x2FFFE => 1, 0x2FFFF => 1, 0x3FFFE => 1, 0x3FFFF => 1,  
               0x4FFFE => 1, 0x4FFFF => 1, 0x5FFFE => 1, 0x5FFFF => 1,  
               0x6FFFE => 1, 0x6FFFF => 1, 0x7FFFE => 1, 0x7FFFF => 1,  
               0x8FFFE => 1, 0x8FFFF => 1, 0x9FFFE => 1, 0x9FFFF => 1,  
               0xAFFFE => 1, 0xAFFFF => 1, 0xBFFFE => 1, 0xBFFFF => 1,  
               0xCFFFE => 1, 0xCFFFF => 1, 0xDFFFE => 1, 0xDFFFF => 1,  
               0xEFFFE => 1, 0xEFFFF => 1, 0xFFFFE => 1, 0xFFFFF => 1,  
               0x10FFFE => 1, 0x10FFFF => 1,  
              }->{$self->{next_char}}) {  
       !!!cp ('j5');  
       if ($self->{next_char} < 0x10000) {  
         !!!parse-error (type => 'control char',  
                         text => (sprintf 'U+%04X', $self->{next_char}));  
       } else {  
         !!!parse-error (type => 'control char',  
                         text => (sprintf 'U-%08X', $self->{next_char}));  
       }  
736      }      }
737    };    };
   $self->{prev_char} = [-1, -1, -1];  
   $self->{next_char} = -1;  
738    
739    $self->{read_until} = sub {    $self->{read_until} = sub {
740      #my ($scalar, $specials_range, $offset) = @_;      #my ($scalar, $specials_range, $offset) = @_;
741      my $specials_range = $_[1];      return 0 if defined $self->{next_nc};
742      return 0 if defined $self->{next_next_char};  
743      my $count = $input->manakai_read_until      my $pattern = qr/[^$_[1]\x00\x0A\x0D]/;
744         ($_[0],      my $offset = $_[2] || 0;
745          qr/(?![$specials_range\x{FDD0}-\x{FDDF}\x{FFFE}\x{FFFF}\x{1FFFE}\x{1FFFF}\x{2FFFE}\x{2FFFF}\x{3FFFE}\x{3FFFF}\x{4FFFE}\x{4FFFF}\x{5FFFE}\x{5FFFF}\x{6FFFE}\x{6FFFF}\x{7FFFE}\x{7FFFF}\x{8FFFE}\x{8FFFF}\x{9FFFE}\x{9FFFF}\x{AFFFE}\x{AFFFF}\x{BFFFE}\x{BFFFF}\x{CFFFE}\x{CFFFF}\x{DFFFE}\x{DFFFF}\x{EFFFE}\x{EFFFF}\x{FFFFE}\x{FFFFF}])[\x20-\x7E\xA0-\x{D7FF}\x{E000}-\x{10FFFD}]/,  
746          $_[2]);      if ($self->{char_buffer_pos} < length $self->{char_buffer}) {
747      if ($count) {        pos ($self->{char_buffer}) = $self->{char_buffer_pos};
748        $self->{column} += $count;        if ($self->{char_buffer} =~ /\G(?>$pattern)+/) {
749        $self->{column_prev} += $count;          substr ($_[0], $offset)
750        $self->{prev_char} = [-1, -1, -1];              = substr ($self->{char_buffer}, $-[0], $+[0] - $-[0]);
751        $self->{next_char} = -1;          my $count = $+[0] - $-[0];
752            if ($count) {
753              $self->{column} += $count;
754              $self->{char_buffer_pos} += $count;
755              $self->{line_prev} = $self->{line};
756              $self->{column_prev} = $self->{column} - 1;
757              $self->{nc} = -1;
758            }
759            return $count;
760          } else {
761            return 0;
762          }
763        } else {
764          my $count = $input->manakai_read_until ($_[0], $pattern, $_[2]);
765          if ($count) {
766            $self->{column} += $count;
767            $self->{line_prev} = $self->{line};
768            $self->{column_prev} = $self->{column} - 1;
769            $self->{nc} = -1;
770          }
771          return $count;
772      }      }
     return $count;  
773    }; # $self->{read_until}    }; # $self->{read_until}
774    
775    my $onerror = $_[2] || sub {    my $onerror = $_[2] || sub {
# Line 741  sub parse_char_stream ($$$;$) { Line 782  sub parse_char_stream ($$$;$) {
782      $onerror->(line => $self->{line}, column => $self->{column}, @_);      $onerror->(line => $self->{line}, column => $self->{column}, @_);
783    };    };
784    
785      my $char_onerror = sub {
786        my (undef, $type, %opt) = @_;
787        !!!parse-error (layer => 'encode',
788                        line => $self->{line}, column => $self->{column} + 1,
789                        %opt, type => $type);
790      }; # $char_onerror
791    
792      if ($_[3]) {
793        $input = $_[3]->($input);
794        $input->onerror ($char_onerror);
795      } else {
796        $input->onerror ($char_onerror) unless defined $input->onerror;
797      }
798    
799    $self->_initialize_tokenizer;    $self->_initialize_tokenizer;
800    $self->_initialize_tree_constructor;    $self->_initialize_tree_constructor;
801    $self->_construct_tree;    $self->_construct_tree;
# Line 760  sub new ($) { Line 815  sub new ($) {
815                info => 'i',                info => 'i',
816                uncertain => 'u'},                uncertain => 'u'},
817    }, $class;    }, $class;
818    $self->{set_next_char} = sub {    $self->{set_nc} = sub {
819      $self->{next_char} = -1;      $self->{nc} = -1;
820    };    };
821    $self->{parse_error} = sub {    $self->{parse_error} = sub {
822      #      #
# Line 826  sub CDATA_SECTION_STATE () { 35 } Line 881  sub CDATA_SECTION_STATE () { 35 }
881  sub MD_HYPHEN_STATE () { 36 } # "markup declaration open state" in the spec  sub MD_HYPHEN_STATE () { 36 } # "markup declaration open state" in the spec
882  sub MD_DOCTYPE_STATE () { 37 } # "markup declaration open state" in the spec  sub MD_DOCTYPE_STATE () { 37 } # "markup declaration open state" in the spec
883  sub MD_CDATA_STATE () { 38 } # "markup declaration open state" in the spec  sub MD_CDATA_STATE () { 38 } # "markup declaration open state" in the spec
884  sub CDATA_PCDATA_CLOSE_TAG_STATE () { 39 } # "close tag open state" in the spec  sub CDATA_RCDATA_CLOSE_TAG_STATE () { 39 } # "close tag open state" in the spec
885  sub CDATA_SECTION_MSE1_STATE () { 40 } # "CDATA section state" in the spec  sub CDATA_SECTION_MSE1_STATE () { 40 } # "CDATA section state" in the spec
886  sub CDATA_SECTION_MSE2_STATE () { 41 } # "CDATA section state" in the spec  sub CDATA_SECTION_MSE2_STATE () { 41 } # "CDATA section state" in the spec
887  sub PUBLIC_STATE () { 42 } # "after DOCTYPE name state" in the spec  sub PUBLIC_STATE () { 42 } # "after DOCTYPE name state" in the spec
# Line 840  sub NCR_NUM_STATE () { 46 } Line 895  sub NCR_NUM_STATE () { 46 }
895  sub HEXREF_X_STATE () { 47 }  sub HEXREF_X_STATE () { 47 }
896  sub HEXREF_HEX_STATE () { 48 }  sub HEXREF_HEX_STATE () { 48 }
897  sub ENTITY_NAME_STATE () { 49 }  sub ENTITY_NAME_STATE () { 49 }
898    sub PCDATA_STATE () { 50 } # "data state" in the spec
899    
900  sub DOCTYPE_TOKEN () { 1 }  sub DOCTYPE_TOKEN () { 1 }
901  sub COMMENT_TOKEN () { 2 }  sub COMMENT_TOKEN () { 2 }
# Line 892  sub IN_COLUMN_GROUP_IM () { 0b10 } Line 948  sub IN_COLUMN_GROUP_IM () { 0b10 }
948  sub _initialize_tokenizer ($) {  sub _initialize_tokenizer ($) {
949    my $self = shift;    my $self = shift;
950    $self->{state} = DATA_STATE; # MUST    $self->{state} = DATA_STATE; # MUST
951    #$self->{state_keyword}; # initialized when used    #$self->{s_kwd}; # state keyword - initialized when used
952    #$self->{entity__value}; # initialized when used    #$self->{entity__value}; # initialized when used
953    #$self->{entity__match}; # initialized when used    #$self->{entity__match}; # initialized when used
954    $self->{content_model} = PCDATA_CONTENT_MODEL; # be    $self->{content_model} = PCDATA_CONTENT_MODEL; # be
955    undef $self->{current_token};    undef $self->{ct}; # current token
956    undef $self->{current_attribute};    undef $self->{ca}; # current attribute
957    undef $self->{last_emitted_start_tag_name};    undef $self->{last_stag_name}; # last emitted start tag name
958    #$self->{prev_state}; # initialized when used    #$self->{prev_state}; # initialized when used
959    delete $self->{self_closing};    delete $self->{self_closing};
960    # $self->{next_char}    $self->{char_buffer} = '';
961      $self->{char_buffer_pos} = 0;
962      $self->{nc} = -1; # next input character
963      #$self->{next_nc}
964    !!!next-input-character;    !!!next-input-character;
965    $self->{token} = [];    $self->{token} = [];
966    # $self->{escape}    # $self->{escape}
# Line 912  sub _initialize_tokenizer ($) { Line 971  sub _initialize_tokenizer ($) {
971  ##       CHARACTER_TOKEN, or END_OF_FILE_TOKEN  ##       CHARACTER_TOKEN, or END_OF_FILE_TOKEN
972  ##   ->{name} (DOCTYPE_TOKEN)  ##   ->{name} (DOCTYPE_TOKEN)
973  ##   ->{tag_name} (START_TAG_TOKEN, END_TAG_TOKEN)  ##   ->{tag_name} (START_TAG_TOKEN, END_TAG_TOKEN)
974  ##   ->{public_identifier} (DOCTYPE_TOKEN)  ##   ->{pubid} (DOCTYPE_TOKEN)
975  ##   ->{system_identifier} (DOCTYPE_TOKEN)  ##   ->{sysid} (DOCTYPE_TOKEN)
976  ##   ->{quirks} == 1 or 0 (DOCTYPE_TOKEN): "force-quirks" flag  ##   ->{quirks} == 1 or 0 (DOCTYPE_TOKEN): "force-quirks" flag
977  ##   ->{attributes} isa HASH (START_TAG_TOKEN, END_TAG_TOKEN)  ##   ->{attributes} isa HASH (START_TAG_TOKEN, END_TAG_TOKEN)
978  ##        ->{name}  ##        ->{name}
# Line 935  sub _initialize_tokenizer ($) { Line 994  sub _initialize_tokenizer ($) {
994  ## TODO: Polytheistic slash SHOULD NOT be used. (Applied only to atheists.)  ## TODO: Polytheistic slash SHOULD NOT be used. (Applied only to atheists.)
995  ## (This requirement was dropped from HTML5 spec, unfortunately.)  ## (This requirement was dropped from HTML5 spec, unfortunately.)
996    
997    my $is_space = {
998      0x0009 => 1, # CHARACTER TABULATION (HT)
999      0x000A => 1, # LINE FEED (LF)
1000      #0x000B => 0, # LINE TABULATION (VT)
1001      0x000C => 1, # FORM FEED (FF)
1002      #0x000D => 1, # CARRIAGE RETURN (CR)
1003      0x0020 => 1, # SPACE (SP)
1004    };
1005    
1006  sub _get_next_token ($) {  sub _get_next_token ($) {
1007    my $self = shift;    my $self = shift;
1008    
1009    if ($self->{self_closing}) {    if ($self->{self_closing}) {
1010      !!!parse-error (type => 'nestc', token => $self->{current_token});      !!!parse-error (type => 'nestc', token => $self->{ct});
1011      ## NOTE: The |self_closing| flag is only set by start tag token.      ## NOTE: The |self_closing| flag is only set by start tag token.
1012      ## In addition, when a start tag token is emitted, it is always set to      ## In addition, when a start tag token is emitted, it is always set to
1013      ## |current_token|.      ## |ct|.
1014      delete $self->{self_closing};      delete $self->{self_closing};
1015    }    }
1016    
# Line 952  sub _get_next_token ($) { Line 1020  sub _get_next_token ($) {
1020    }    }
1021    
1022    A: {    A: {
1023      if ($self->{state} == DATA_STATE) {      if ($self->{state} == PCDATA_STATE) {
1024        if ($self->{next_char} == 0x0026) { # &        ## NOTE: Same as |DATA_STATE|, but only for |PCDATA| content model.
1025    
1026          if ($self->{nc} == 0x0026) { # &
1027            !!!cp (0.1);
1028            ## NOTE: In the spec, the tokenizer is switched to the
1029            ## "entity data state".  In this implementation, the tokenizer
1030            ## is switched to the |ENTITY_STATE|, which is an implementation
1031            ## of the "consume a character reference" algorithm.
1032            $self->{entity_add} = -1;
1033            $self->{prev_state} = DATA_STATE;
1034            $self->{state} = ENTITY_STATE;
1035            !!!next-input-character;
1036            redo A;
1037          } elsif ($self->{nc} == 0x003C) { # <
1038            !!!cp (0.2);
1039            $self->{state} = TAG_OPEN_STATE;
1040            !!!next-input-character;
1041            redo A;
1042          } elsif ($self->{nc} == -1) {
1043            !!!cp (0.3);
1044            !!!emit ({type => END_OF_FILE_TOKEN,
1045                      line => $self->{line}, column => $self->{column}});
1046            last A; ## TODO: ok?
1047          } else {
1048            !!!cp (0.4);
1049            #
1050          }
1051    
1052          # Anything else
1053          my $token = {type => CHARACTER_TOKEN,
1054                       data => chr $self->{nc},
1055                       line => $self->{line}, column => $self->{column},
1056                      };
1057          $self->{read_until}->($token->{data}, q[<&], length $token->{data});
1058    
1059          ## Stay in the state.
1060          !!!next-input-character;
1061          !!!emit ($token);
1062          redo A;
1063        } elsif ($self->{state} == DATA_STATE) {
1064          $self->{s_kwd} = '' unless defined $self->{s_kwd};
1065          if ($self->{nc} == 0x0026) { # &
1066            $self->{s_kwd} = '';
1067          if ($self->{content_model} & CM_ENTITY and # PCDATA | RCDATA          if ($self->{content_model} & CM_ENTITY and # PCDATA | RCDATA
1068              not $self->{escape}) {              not $self->{escape}) {
1069            !!!cp (1);            !!!cp (1);
# Line 961  sub _get_next_token ($) { Line 1071  sub _get_next_token ($) {
1071            ## "entity data state".  In this implementation, the tokenizer            ## "entity data state".  In this implementation, the tokenizer
1072            ## is switched to the |ENTITY_STATE|, which is an implementation            ## is switched to the |ENTITY_STATE|, which is an implementation
1073            ## of the "consume a character reference" algorithm.            ## of the "consume a character reference" algorithm.
1074            $self->{entity_additional} = -1;            $self->{entity_add} = -1;
1075            $self->{prev_state} = DATA_STATE;            $self->{prev_state} = DATA_STATE;
1076            $self->{state} = ENTITY_STATE;            $self->{state} = ENTITY_STATE;
1077            !!!next-input-character;            !!!next-input-character;
# Line 970  sub _get_next_token ($) { Line 1080  sub _get_next_token ($) {
1080            !!!cp (2);            !!!cp (2);
1081            #            #
1082          }          }
1083        } elsif ($self->{next_char} == 0x002D) { # -        } elsif ($self->{nc} == 0x002D) { # -
1084          if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA          if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
1085            unless ($self->{escape}) {            $self->{s_kwd} .= '-';
1086              if ($self->{prev_char}->[0] == 0x002D and # -            
1087                  $self->{prev_char}->[1] == 0x0021 and # !            if ($self->{s_kwd} eq '<!--') {
1088                  $self->{prev_char}->[2] == 0x003C) { # <              !!!cp (3);
1089                !!!cp (3);              $self->{escape} = 1; # unless $self->{escape};
1090                $self->{escape} = 1;              $self->{s_kwd} = '--';
1091              } else {              #
1092                !!!cp (4);            } elsif ($self->{s_kwd} eq '---') {
1093              }              !!!cp (4);
1094                $self->{s_kwd} = '--';
1095                #
1096            } else {            } else {
1097              !!!cp (5);              !!!cp (5);
1098                #
1099            }            }
1100          }          }
1101                    
1102          #          #
1103        } elsif ($self->{next_char} == 0x003C) { # <        } elsif ($self->{nc} == 0x0021) { # !
1104            if (length $self->{s_kwd}) {
1105              !!!cp (5.1);
1106              $self->{s_kwd} .= '!';
1107              #
1108            } else {
1109              !!!cp (5.2);
1110              #$self->{s_kwd} = '';
1111              #
1112            }
1113            #
1114          } elsif ($self->{nc} == 0x003C) { # <
1115          if ($self->{content_model} & CM_FULL_MARKUP or # PCDATA          if ($self->{content_model} & CM_FULL_MARKUP or # PCDATA
1116              (($self->{content_model} & CM_LIMITED_MARKUP) and # CDATA | RCDATA              (($self->{content_model} & CM_LIMITED_MARKUP) and # CDATA | RCDATA
1117               not $self->{escape})) {               not $self->{escape})) {
# Line 997  sub _get_next_token ($) { Line 1121  sub _get_next_token ($) {
1121            redo A;            redo A;
1122          } else {          } else {
1123            !!!cp (7);            !!!cp (7);
1124              $self->{s_kwd} = '';
1125            #            #
1126          }          }
1127        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1128          if ($self->{escape} and          if ($self->{escape} and
1129              ($self->{content_model} & CM_LIMITED_MARKUP)) { # RCDATA | CDATA              ($self->{content_model} & CM_LIMITED_MARKUP)) { # RCDATA | CDATA
1130            if ($self->{prev_char}->[0] == 0x002D and # -            if ($self->{s_kwd} eq '--') {
               $self->{prev_char}->[1] == 0x002D) { # -  
1131              !!!cp (8);              !!!cp (8);
1132              delete $self->{escape};              delete $self->{escape};
1133            } else {            } else {
# Line 1013  sub _get_next_token ($) { Line 1137  sub _get_next_token ($) {
1137            !!!cp (10);            !!!cp (10);
1138          }          }
1139                    
1140            $self->{s_kwd} = '';
1141          #          #
1142        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1143          !!!cp (11);          !!!cp (11);
1144            $self->{s_kwd} = '';
1145          !!!emit ({type => END_OF_FILE_TOKEN,          !!!emit ({type => END_OF_FILE_TOKEN,
1146                    line => $self->{line}, column => $self->{column}});                    line => $self->{line}, column => $self->{column}});
1147          last A; ## TODO: ok?          last A; ## TODO: ok?
1148        } else {        } else {
1149          !!!cp (12);          !!!cp (12);
1150            $self->{s_kwd} = '';
1151            #
1152        }        }
1153    
1154        # Anything else        # Anything else
1155        my $token = {type => CHARACTER_TOKEN,        my $token = {type => CHARACTER_TOKEN,
1156                     data => chr $self->{next_char},                     data => chr $self->{nc},
1157                     line => $self->{line}, column => $self->{column},                     line => $self->{line}, column => $self->{column},
1158                    };                    };
1159        $self->{read_until}->($token->{data}, q[-!<>&], length $token->{data});        if ($self->{read_until}->($token->{data}, q[-!<>&],
1160                                    length $token->{data})) {
1161            $self->{s_kwd} = '';
1162          }
1163    
1164        ## Stay in the data state        ## Stay in the data state.
1165          if ($self->{content_model} == PCDATA_CONTENT_MODEL) {
1166            !!!cp (13);
1167            $self->{state} = PCDATA_STATE;
1168          } else {
1169            !!!cp (14);
1170            ## Stay in the state.
1171          }
1172        !!!next-input-character;        !!!next-input-character;
   
1173        !!!emit ($token);        !!!emit ($token);
   
1174        redo A;        redo A;
1175      } elsif ($self->{state} == TAG_OPEN_STATE) {      } elsif ($self->{state} == TAG_OPEN_STATE) {
1176        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
1177          if ($self->{next_char} == 0x002F) { # /          if ($self->{nc} == 0x002F) { # /
1178            !!!cp (15);            !!!cp (15);
1179            !!!next-input-character;            !!!next-input-character;
1180            $self->{state} = CLOSE_TAG_OPEN_STATE;            $self->{state} = CLOSE_TAG_OPEN_STATE;
1181            redo A;            redo A;
1182            } elsif ($self->{nc} == 0x0021) { # !
1183              !!!cp (15.1);
1184              $self->{s_kwd} = '<' unless $self->{escape};
1185              #
1186          } else {          } else {
1187            !!!cp (16);            !!!cp (16);
1188            ## reconsume            #
           $self->{state} = DATA_STATE;  
   
           !!!emit ({type => CHARACTER_TOKEN, data => '<',  
                     line => $self->{line_prev},  
                     column => $self->{column_prev},  
                    });  
   
           redo A;  
1189          }          }
1190    
1191            ## reconsume
1192            $self->{state} = DATA_STATE;
1193            !!!emit ({type => CHARACTER_TOKEN, data => '<',
1194                      line => $self->{line_prev},
1195                      column => $self->{column_prev},
1196                     });
1197            redo A;
1198        } elsif ($self->{content_model} & CM_FULL_MARKUP) { # PCDATA        } elsif ($self->{content_model} & CM_FULL_MARKUP) { # PCDATA
1199          if ($self->{next_char} == 0x0021) { # !          if ($self->{nc} == 0x0021) { # !
1200            !!!cp (17);            !!!cp (17);
1201            $self->{state} = MARKUP_DECLARATION_OPEN_STATE;            $self->{state} = MARKUP_DECLARATION_OPEN_STATE;
1202            !!!next-input-character;            !!!next-input-character;
1203            redo A;            redo A;
1204          } elsif ($self->{next_char} == 0x002F) { # /          } elsif ($self->{nc} == 0x002F) { # /
1205            !!!cp (18);            !!!cp (18);
1206            $self->{state} = CLOSE_TAG_OPEN_STATE;            $self->{state} = CLOSE_TAG_OPEN_STATE;
1207            !!!next-input-character;            !!!next-input-character;
1208            redo A;            redo A;
1209          } elsif (0x0041 <= $self->{next_char} and          } elsif (0x0041 <= $self->{nc} and
1210                   $self->{next_char} <= 0x005A) { # A..Z                   $self->{nc} <= 0x005A) { # A..Z
1211            !!!cp (19);            !!!cp (19);
1212            $self->{current_token}            $self->{ct}
1213              = {type => START_TAG_TOKEN,              = {type => START_TAG_TOKEN,
1214                 tag_name => chr ($self->{next_char} + 0x0020),                 tag_name => chr ($self->{nc} + 0x0020),
1215                 line => $self->{line_prev},                 line => $self->{line_prev},
1216                 column => $self->{column_prev}};                 column => $self->{column_prev}};
1217            $self->{state} = TAG_NAME_STATE;            $self->{state} = TAG_NAME_STATE;
1218            !!!next-input-character;            !!!next-input-character;
1219            redo A;            redo A;
1220          } elsif (0x0061 <= $self->{next_char} and          } elsif (0x0061 <= $self->{nc} and
1221                   $self->{next_char} <= 0x007A) { # a..z                   $self->{nc} <= 0x007A) { # a..z
1222            !!!cp (20);            !!!cp (20);
1223            $self->{current_token} = {type => START_TAG_TOKEN,            $self->{ct} = {type => START_TAG_TOKEN,
1224                                      tag_name => chr ($self->{next_char}),                                      tag_name => chr ($self->{nc}),
1225                                      line => $self->{line_prev},                                      line => $self->{line_prev},
1226                                      column => $self->{column_prev}};                                      column => $self->{column_prev}};
1227            $self->{state} = TAG_NAME_STATE;            $self->{state} = TAG_NAME_STATE;
1228            !!!next-input-character;            !!!next-input-character;
1229            redo A;            redo A;
1230          } elsif ($self->{next_char} == 0x003E) { # >          } elsif ($self->{nc} == 0x003E) { # >
1231            !!!cp (21);            !!!cp (21);
1232            !!!parse-error (type => 'empty start tag',            !!!parse-error (type => 'empty start tag',
1233                            line => $self->{line_prev},                            line => $self->{line_prev},
# Line 1100  sub _get_next_token ($) { Line 1241  sub _get_next_token ($) {
1241                     });                     });
1242    
1243            redo A;            redo A;
1244          } elsif ($self->{next_char} == 0x003F) { # ?          } elsif ($self->{nc} == 0x003F) { # ?
1245            !!!cp (22);            !!!cp (22);
1246            !!!parse-error (type => 'pio',            !!!parse-error (type => 'pio',
1247                            line => $self->{line_prev},                            line => $self->{line_prev},
1248                            column => $self->{column_prev});                            column => $self->{column_prev});
1249            $self->{state} = BOGUS_COMMENT_STATE;            $self->{state} = BOGUS_COMMENT_STATE;
1250            $self->{current_token} = {type => COMMENT_TOKEN, data => '',            $self->{ct} = {type => COMMENT_TOKEN, data => '',
1251                                      line => $self->{line_prev},                                      line => $self->{line_prev},
1252                                      column => $self->{column_prev},                                      column => $self->{column_prev},
1253                                     };                                     };
1254            ## $self->{next_char} is intentionally left as is            ## $self->{nc} is intentionally left as is
1255            redo A;            redo A;
1256          } else {          } else {
1257            !!!cp (23);            !!!cp (23);
# Line 1132  sub _get_next_token ($) { Line 1273  sub _get_next_token ($) {
1273        }        }
1274      } elsif ($self->{state} == CLOSE_TAG_OPEN_STATE) {      } elsif ($self->{state} == CLOSE_TAG_OPEN_STATE) {
1275        ## NOTE: The "close tag open state" in the spec is implemented as        ## NOTE: The "close tag open state" in the spec is implemented as
1276        ## |CLOSE_TAG_OPEN_STATE| and |CDATA_PCDATA_CLOSE_TAG_STATE|.        ## |CLOSE_TAG_OPEN_STATE| and |CDATA_RCDATA_CLOSE_TAG_STATE|.
1277    
1278        my ($l, $c) = ($self->{line_prev}, $self->{column_prev} - 1); # "<"of"</"        my ($l, $c) = ($self->{line_prev}, $self->{column_prev} - 1); # "<"of"</"
1279        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
1280          if (defined $self->{last_emitted_start_tag_name}) {          if (defined $self->{last_stag_name}) {
1281            $self->{state} = CDATA_PCDATA_CLOSE_TAG_STATE;            $self->{state} = CDATA_RCDATA_CLOSE_TAG_STATE;
1282            $self->{state_keyword} = '';            $self->{s_kwd} = '';
1283            ## Reconsume.            ## Reconsume.
1284            redo A;            redo A;
1285          } else {          } else {
# Line 1154  sub _get_next_token ($) { Line 1295  sub _get_next_token ($) {
1295          }          }
1296        }        }
1297    
1298        if (0x0041 <= $self->{next_char} and        if (0x0041 <= $self->{nc} and
1299            $self->{next_char} <= 0x005A) { # A..Z            $self->{nc} <= 0x005A) { # A..Z
1300          !!!cp (29);          !!!cp (29);
1301          $self->{current_token}          $self->{ct}
1302              = {type => END_TAG_TOKEN,              = {type => END_TAG_TOKEN,
1303                 tag_name => chr ($self->{next_char} + 0x0020),                 tag_name => chr ($self->{nc} + 0x0020),
1304                 line => $l, column => $c};                 line => $l, column => $c};
1305          $self->{state} = TAG_NAME_STATE;          $self->{state} = TAG_NAME_STATE;
1306          !!!next-input-character;          !!!next-input-character;
1307          redo A;          redo A;
1308        } elsif (0x0061 <= $self->{next_char} and        } elsif (0x0061 <= $self->{nc} and
1309                 $self->{next_char} <= 0x007A) { # a..z                 $self->{nc} <= 0x007A) { # a..z
1310          !!!cp (30);          !!!cp (30);
1311          $self->{current_token} = {type => END_TAG_TOKEN,          $self->{ct} = {type => END_TAG_TOKEN,
1312                                    tag_name => chr ($self->{next_char}),                                    tag_name => chr ($self->{nc}),
1313                                    line => $l, column => $c};                                    line => $l, column => $c};
1314          $self->{state} = TAG_NAME_STATE;          $self->{state} = TAG_NAME_STATE;
1315          !!!next-input-character;          !!!next-input-character;
1316          redo A;          redo A;
1317        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1318          !!!cp (31);          !!!cp (31);
1319          !!!parse-error (type => 'empty end tag',          !!!parse-error (type => 'empty end tag',
1320                          line => $self->{line_prev}, ## "<" in "</>"                          line => $self->{line_prev}, ## "<" in "</>"
# Line 1181  sub _get_next_token ($) { Line 1322  sub _get_next_token ($) {
1322          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1323          !!!next-input-character;          !!!next-input-character;
1324          redo A;          redo A;
1325        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1326          !!!cp (32);          !!!cp (32);
1327          !!!parse-error (type => 'bare etago');          !!!parse-error (type => 'bare etago');
1328          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
# Line 1196  sub _get_next_token ($) { Line 1337  sub _get_next_token ($) {
1337          !!!cp (33);          !!!cp (33);
1338          !!!parse-error (type => 'bogus end tag');          !!!parse-error (type => 'bogus end tag');
1339          $self->{state} = BOGUS_COMMENT_STATE;          $self->{state} = BOGUS_COMMENT_STATE;
1340          $self->{current_token} = {type => COMMENT_TOKEN, data => '',          $self->{ct} = {type => COMMENT_TOKEN, data => '',
1341                                    line => $self->{line_prev}, # "<" of "</"                                    line => $self->{line_prev}, # "<" of "</"
1342                                    column => $self->{column_prev} - 1,                                    column => $self->{column_prev} - 1,
1343                                   };                                   };
1344          ## NOTE: $self->{next_char} is intentionally left as is.          ## NOTE: $self->{nc} is intentionally left as is.
1345          ## Although the "anything else" case of the spec not explicitly          ## Although the "anything else" case of the spec not explicitly
1346          ## states that the next input character is to be reconsumed,          ## states that the next input character is to be reconsumed,
1347          ## it will be included to the |data| of the comment token          ## it will be included to the |data| of the comment token
# Line 1208  sub _get_next_token ($) { Line 1349  sub _get_next_token ($) {
1349          ## "bogus comment state" entry.          ## "bogus comment state" entry.
1350          redo A;          redo A;
1351        }        }
1352      } elsif ($self->{state} == CDATA_PCDATA_CLOSE_TAG_STATE) {      } elsif ($self->{state} == CDATA_RCDATA_CLOSE_TAG_STATE) {
1353        my $ch = substr $self->{last_emitted_start_tag_name}, length $self->{state_keyword}, 1;        my $ch = substr $self->{last_stag_name}, length $self->{s_kwd}, 1;
1354        if (length $ch) {        if (length $ch) {
1355          my $CH = $ch;          my $CH = $ch;
1356          $ch =~ tr/a-z/A-Z/;          $ch =~ tr/a-z/A-Z/;
1357          my $nch = chr $self->{next_char};          my $nch = chr $self->{nc};
1358          if ($nch eq $ch or $nch eq $CH) {          if ($nch eq $ch or $nch eq $CH) {
1359            !!!cp (24);            !!!cp (24);
1360            ## Stay in the state.            ## Stay in the state.
1361            $self->{state_keyword} .= $nch;            $self->{s_kwd} .= $nch;
1362            !!!next-input-character;            !!!next-input-character;
1363            redo A;            redo A;
1364          } else {          } else {
# Line 1225  sub _get_next_token ($) { Line 1366  sub _get_next_token ($) {
1366            $self->{state} = DATA_STATE;            $self->{state} = DATA_STATE;
1367            ## Reconsume.            ## Reconsume.
1368            !!!emit ({type => CHARACTER_TOKEN,            !!!emit ({type => CHARACTER_TOKEN,
1369                      data => '</' . $self->{state_keyword},                      data => '</' . $self->{s_kwd},
1370                      line => $self->{line_prev},                      line => $self->{line_prev},
1371                      column => $self->{column_prev} - 1 - length $self->{state_keyword},                      column => $self->{column_prev} - 1 - length $self->{s_kwd},
1372                     });                     });
1373            redo A;            redo A;
1374          }          }
1375        } else { # after "<{tag-name}"        } else { # after "<{tag-name}"
1376          unless ({          unless ($is_space->{$self->{nc}} or
1377                   0x0009 => 1, # HT                  {
                  0x000A => 1, # LF  
                  0x000B => 1, # VT  
                  0x000C => 1, # FF  
                  0x0020 => 1, # SP  
1378                   0x003E => 1, # >                   0x003E => 1, # >
1379                   0x002F => 1, # /                   0x002F => 1, # /
1380                   -1 => 1, # EOF                   -1 => 1, # EOF
1381                  }->{$self->{next_char}}) {                  }->{$self->{nc}}) {
1382            !!!cp (26);            !!!cp (26);
1383            ## Reconsume.            ## Reconsume.
1384            $self->{state} = DATA_STATE;            $self->{state} = DATA_STATE;
1385            !!!emit ({type => CHARACTER_TOKEN,            !!!emit ({type => CHARACTER_TOKEN,
1386                      data => '</' . $self->{state_keyword},                      data => '</' . $self->{s_kwd},
1387                      line => $self->{line_prev},                      line => $self->{line_prev},
1388                      column => $self->{column_prev} - 1 - length $self->{state_keyword},                      column => $self->{column_prev} - 1 - length $self->{s_kwd},
1389                     });                     });
1390            redo A;            redo A;
1391          } else {          } else {
1392            !!!cp (27);            !!!cp (27);
1393            $self->{current_token}            $self->{ct}
1394                = {type => END_TAG_TOKEN,                = {type => END_TAG_TOKEN,
1395                   tag_name => $self->{last_emitted_start_tag_name},                   tag_name => $self->{last_stag_name},
1396                   line => $self->{line_prev},                   line => $self->{line_prev},
1397                   column => $self->{column_prev} - 1 - length $self->{state_keyword}};                   column => $self->{column_prev} - 1 - length $self->{s_kwd}};
1398            $self->{state} = TAG_NAME_STATE;            $self->{state} = TAG_NAME_STATE;
1399            ## Reconsume.            ## Reconsume.
1400            redo A;            redo A;
1401          }          }
1402        }        }
1403      } elsif ($self->{state} == TAG_NAME_STATE) {      } elsif ($self->{state} == TAG_NAME_STATE) {
1404        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
1405          !!!cp (34);          !!!cp (34);
1406          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1407          !!!next-input-character;          !!!next-input-character;
1408          redo A;          redo A;
1409        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1410          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1411            !!!cp (35);            !!!cp (35);
1412            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1413          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1414            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1415            #if ($self->{current_token}->{attributes}) {            #if ($self->{ct}->{attributes}) {
1416            #  ## NOTE: This should never be reached.            #  ## NOTE: This should never be reached.
1417            #  !!! cp (36);            #  !!! cp (36);
1418            #  !!! parse-error (type => 'end tag attribute');            #  !!! parse-error (type => 'end tag attribute');
# Line 1287  sub _get_next_token ($) { Line 1420  sub _get_next_token ($) {
1420              !!!cp (37);              !!!cp (37);
1421            #}            #}
1422          } else {          } else {
1423            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1424          }          }
1425          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1426          !!!next-input-character;          !!!next-input-character;
1427    
1428          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1429    
1430          redo A;          redo A;
1431        } elsif (0x0041 <= $self->{next_char} and        } elsif (0x0041 <= $self->{nc} and
1432                 $self->{next_char} <= 0x005A) { # A..Z                 $self->{nc} <= 0x005A) { # A..Z
1433          !!!cp (38);          !!!cp (38);
1434          $self->{current_token}->{tag_name} .= chr ($self->{next_char} + 0x0020);          $self->{ct}->{tag_name} .= chr ($self->{nc} + 0x0020);
1435            # start tag or end tag            # start tag or end tag
1436          ## Stay in this state          ## Stay in this state
1437          !!!next-input-character;          !!!next-input-character;
1438          redo A;          redo A;
1439        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1440          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1441          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1442            !!!cp (39);            !!!cp (39);
1443            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1444          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1445            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1446            #if ($self->{current_token}->{attributes}) {            #if ($self->{ct}->{attributes}) {
1447            #  ## NOTE: This state should never be reached.            #  ## NOTE: This state should never be reached.
1448            #  !!! cp (40);            #  !!! cp (40);
1449            #  !!! parse-error (type => 'end tag attribute');            #  !!! parse-error (type => 'end tag attribute');
# Line 1318  sub _get_next_token ($) { Line 1451  sub _get_next_token ($) {
1451              !!!cp (41);              !!!cp (41);
1452            #}            #}
1453          } else {          } else {
1454            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1455          }          }
1456          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1457          # reconsume          # reconsume
1458    
1459          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1460    
1461          redo A;          redo A;
1462        } elsif ($self->{next_char} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
1463          !!!cp (42);          !!!cp (42);
1464          $self->{state} = SELF_CLOSING_START_TAG_STATE;          $self->{state} = SELF_CLOSING_START_TAG_STATE;
1465          !!!next-input-character;          !!!next-input-character;
1466          redo A;          redo A;
1467        } else {        } else {
1468          !!!cp (44);          !!!cp (44);
1469          $self->{current_token}->{tag_name} .= chr $self->{next_char};          $self->{ct}->{tag_name} .= chr $self->{nc};
1470            # start tag or end tag            # start tag or end tag
1471          ## Stay in the state          ## Stay in the state
1472          !!!next-input-character;          !!!next-input-character;
1473          redo A;          redo A;
1474        }        }
1475      } elsif ($self->{state} == BEFORE_ATTRIBUTE_NAME_STATE) {      } elsif ($self->{state} == BEFORE_ATTRIBUTE_NAME_STATE) {
1476        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
1477          !!!cp (45);          !!!cp (45);
1478          ## Stay in the state          ## Stay in the state
1479          !!!next-input-character;          !!!next-input-character;
1480          redo A;          redo A;
1481        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1482          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1483            !!!cp (46);            !!!cp (46);
1484            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1485          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1486            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1487            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1488              !!!cp (47);              !!!cp (47);
1489              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1490            } else {            } else {
1491              !!!cp (48);              !!!cp (48);
1492            }            }
1493          } else {          } else {
1494            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1495          }          }
1496          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1497          !!!next-input-character;          !!!next-input-character;
1498    
1499          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1500    
1501          redo A;          redo A;
1502        } elsif (0x0041 <= $self->{next_char} and        } elsif (0x0041 <= $self->{nc} and
1503                 $self->{next_char} <= 0x005A) { # A..Z                 $self->{nc} <= 0x005A) { # A..Z
1504          !!!cp (49);          !!!cp (49);
1505          $self->{current_attribute}          $self->{ca}
1506              = {name => chr ($self->{next_char} + 0x0020),              = {name => chr ($self->{nc} + 0x0020),
1507                 value => '',                 value => '',
1508                 line => $self->{line}, column => $self->{column}};                 line => $self->{line}, column => $self->{column}};
1509          $self->{state} = ATTRIBUTE_NAME_STATE;          $self->{state} = ATTRIBUTE_NAME_STATE;
1510          !!!next-input-character;          !!!next-input-character;
1511          redo A;          redo A;
1512        } elsif ($self->{next_char} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
1513          !!!cp (50);          !!!cp (50);
1514          $self->{state} = SELF_CLOSING_START_TAG_STATE;          $self->{state} = SELF_CLOSING_START_TAG_STATE;
1515          !!!next-input-character;          !!!next-input-character;
1516          redo A;          redo A;
1517        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1518          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1519          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1520            !!!cp (52);            !!!cp (52);
1521            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1522          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1523            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1524            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1525              !!!cp (53);              !!!cp (53);
1526              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1527            } else {            } else {
1528              !!!cp (54);              !!!cp (54);
1529            }            }
1530          } else {          } else {
1531            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1532          }          }
1533          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1534          # reconsume          # reconsume
1535    
1536          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1537    
1538          redo A;          redo A;
1539        } else {        } else {
# Line 1412  sub _get_next_token ($) { Line 1541  sub _get_next_token ($) {
1541               0x0022 => 1, # "               0x0022 => 1, # "
1542               0x0027 => 1, # '               0x0027 => 1, # '
1543               0x003D => 1, # =               0x003D => 1, # =
1544              }->{$self->{next_char}}) {              }->{$self->{nc}}) {
1545            !!!cp (55);            !!!cp (55);
1546            !!!parse-error (type => 'bad attribute name');            !!!parse-error (type => 'bad attribute name');
1547          } else {          } else {
1548            !!!cp (56);            !!!cp (56);
1549          }          }
1550          $self->{current_attribute}          $self->{ca}
1551              = {name => chr ($self->{next_char}),              = {name => chr ($self->{nc}),
1552                 value => '',                 value => '',
1553                 line => $self->{line}, column => $self->{column}};                 line => $self->{line}, column => $self->{column}};
1554          $self->{state} = ATTRIBUTE_NAME_STATE;          $self->{state} = ATTRIBUTE_NAME_STATE;
# Line 1428  sub _get_next_token ($) { Line 1557  sub _get_next_token ($) {
1557        }        }
1558      } elsif ($self->{state} == ATTRIBUTE_NAME_STATE) {      } elsif ($self->{state} == ATTRIBUTE_NAME_STATE) {
1559        my $before_leave = sub {        my $before_leave = sub {
1560          if (exists $self->{current_token}->{attributes} # start tag or end tag          if (exists $self->{ct}->{attributes} # start tag or end tag
1561              ->{$self->{current_attribute}->{name}}) { # MUST              ->{$self->{ca}->{name}}) { # MUST
1562            !!!cp (57);            !!!cp (57);
1563            !!!parse-error (type => 'duplicate attribute', text => $self->{current_attribute}->{name}, line => $self->{current_attribute}->{line}, column => $self->{current_attribute}->{column});            !!!parse-error (type => 'duplicate attribute', text => $self->{ca}->{name}, line => $self->{ca}->{line}, column => $self->{ca}->{column});
1564            ## Discard $self->{current_attribute} # MUST            ## Discard $self->{ca} # MUST
1565          } else {          } else {
1566            !!!cp (58);            !!!cp (58);
1567            $self->{current_token}->{attributes}->{$self->{current_attribute}->{name}}            $self->{ct}->{attributes}->{$self->{ca}->{name}}
1568              = $self->{current_attribute};              = $self->{ca};
1569          }          }
1570        }; # $before_leave        }; # $before_leave
1571    
1572        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
1573          !!!cp (59);          !!!cp (59);
1574          $before_leave->();          $before_leave->();
1575          $self->{state} = AFTER_ATTRIBUTE_NAME_STATE;          $self->{state} = AFTER_ATTRIBUTE_NAME_STATE;
1576          !!!next-input-character;          !!!next-input-character;
1577          redo A;          redo A;
1578        } elsif ($self->{next_char} == 0x003D) { # =        } elsif ($self->{nc} == 0x003D) { # =
1579          !!!cp (60);          !!!cp (60);
1580          $before_leave->();          $before_leave->();
1581          $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;          $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;
1582          !!!next-input-character;          !!!next-input-character;
1583          redo A;          redo A;
1584        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1585          $before_leave->();          $before_leave->();
1586          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1587            !!!cp (61);            !!!cp (61);
1588            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1589          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1590            !!!cp (62);            !!!cp (62);
1591            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1592            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1593              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1594            }            }
1595          } else {          } else {
1596            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1597          }          }
1598          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1599          !!!next-input-character;          !!!next-input-character;
1600    
1601          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1602    
1603          redo A;          redo A;
1604        } elsif (0x0041 <= $self->{next_char} and        } elsif (0x0041 <= $self->{nc} and
1605                 $self->{next_char} <= 0x005A) { # A..Z                 $self->{nc} <= 0x005A) { # A..Z
1606          !!!cp (63);          !!!cp (63);
1607          $self->{current_attribute}->{name} .= chr ($self->{next_char} + 0x0020);          $self->{ca}->{name} .= chr ($self->{nc} + 0x0020);
1608          ## Stay in the state          ## Stay in the state
1609          !!!next-input-character;          !!!next-input-character;
1610          redo A;          redo A;
1611        } elsif ($self->{next_char} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
1612          !!!cp (64);          !!!cp (64);
1613          $before_leave->();          $before_leave->();
1614          $self->{state} = SELF_CLOSING_START_TAG_STATE;          $self->{state} = SELF_CLOSING_START_TAG_STATE;
1615          !!!next-input-character;          !!!next-input-character;
1616          redo A;          redo A;
1617        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1618          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1619          $before_leave->();          $before_leave->();
1620          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1621            !!!cp (66);            !!!cp (66);
1622            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1623          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1624            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1625            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1626              !!!cp (67);              !!!cp (67);
1627              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1628            } else {            } else {
# Line 1505  sub _get_next_token ($) { Line 1630  sub _get_next_token ($) {
1630              !!!cp (68);              !!!cp (68);
1631            }            }
1632          } else {          } else {
1633            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1634          }          }
1635          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1636          # reconsume          # reconsume
1637    
1638          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1639    
1640          redo A;          redo A;
1641        } else {        } else {
1642          if ($self->{next_char} == 0x0022 or # "          if ($self->{nc} == 0x0022 or # "
1643              $self->{next_char} == 0x0027) { # '              $self->{nc} == 0x0027) { # '
1644            !!!cp (69);            !!!cp (69);
1645            !!!parse-error (type => 'bad attribute name');            !!!parse-error (type => 'bad attribute name');
1646          } else {          } else {
1647            !!!cp (70);            !!!cp (70);
1648          }          }
1649          $self->{current_attribute}->{name} .= chr ($self->{next_char});          $self->{ca}->{name} .= chr ($self->{nc});
1650          ## Stay in the state          ## Stay in the state
1651          !!!next-input-character;          !!!next-input-character;
1652          redo A;          redo A;
1653        }        }
1654      } elsif ($self->{state} == AFTER_ATTRIBUTE_NAME_STATE) {      } elsif ($self->{state} == AFTER_ATTRIBUTE_NAME_STATE) {
1655        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
1656          !!!cp (71);          !!!cp (71);
1657          ## Stay in the state          ## Stay in the state
1658          !!!next-input-character;          !!!next-input-character;
1659          redo A;          redo A;
1660        } elsif ($self->{next_char} == 0x003D) { # =        } elsif ($self->{nc} == 0x003D) { # =
1661          !!!cp (72);          !!!cp (72);
1662          $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;          $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;
1663          !!!next-input-character;          !!!next-input-character;
1664          redo A;          redo A;
1665        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1666          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1667            !!!cp (73);            !!!cp (73);
1668            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1669          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1670            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1671            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1672              !!!cp (74);              !!!cp (74);
1673              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1674            } else {            } else {
# Line 1555  sub _get_next_token ($) { Line 1676  sub _get_next_token ($) {
1676              !!!cp (75);              !!!cp (75);
1677            }            }
1678          } else {          } else {
1679            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1680          }          }
1681          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1682          !!!next-input-character;          !!!next-input-character;
1683    
1684          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1685    
1686          redo A;          redo A;
1687        } elsif (0x0041 <= $self->{next_char} and        } elsif (0x0041 <= $self->{nc} and
1688                 $self->{next_char} <= 0x005A) { # A..Z                 $self->{nc} <= 0x005A) { # A..Z
1689          !!!cp (76);          !!!cp (76);
1690          $self->{current_attribute}          $self->{ca}
1691              = {name => chr ($self->{next_char} + 0x0020),              = {name => chr ($self->{nc} + 0x0020),
1692                 value => '',                 value => '',
1693                 line => $self->{line}, column => $self->{column}};                 line => $self->{line}, column => $self->{column}};
1694          $self->{state} = ATTRIBUTE_NAME_STATE;          $self->{state} = ATTRIBUTE_NAME_STATE;
1695          !!!next-input-character;          !!!next-input-character;
1696          redo A;          redo A;
1697        } elsif ($self->{next_char} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
1698          !!!cp (77);          !!!cp (77);
1699          $self->{state} = SELF_CLOSING_START_TAG_STATE;          $self->{state} = SELF_CLOSING_START_TAG_STATE;
1700          !!!next-input-character;          !!!next-input-character;
1701          redo A;          redo A;
1702        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1703          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1704          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1705            !!!cp (79);            !!!cp (79);
1706            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1707          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1708            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1709            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1710              !!!cp (80);              !!!cp (80);
1711              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1712            } else {            } else {
# Line 1593  sub _get_next_token ($) { Line 1714  sub _get_next_token ($) {
1714              !!!cp (81);              !!!cp (81);
1715            }            }
1716          } else {          } else {
1717            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1718          }          }
1719          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1720          # reconsume          # reconsume
1721    
1722          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1723    
1724          redo A;          redo A;
1725        } else {        } else {
1726          if ($self->{next_char} == 0x0022 or # "          if ($self->{nc} == 0x0022 or # "
1727              $self->{next_char} == 0x0027) { # '              $self->{nc} == 0x0027) { # '
1728            !!!cp (78);            !!!cp (78);
1729            !!!parse-error (type => 'bad attribute name');            !!!parse-error (type => 'bad attribute name');
1730          } else {          } else {
1731            !!!cp (82);            !!!cp (82);
1732          }          }
1733          $self->{current_attribute}          $self->{ca}
1734              = {name => chr ($self->{next_char}),              = {name => chr ($self->{nc}),
1735                 value => '',                 value => '',
1736                 line => $self->{line}, column => $self->{column}};                 line => $self->{line}, column => $self->{column}};
1737          $self->{state} = ATTRIBUTE_NAME_STATE;          $self->{state} = ATTRIBUTE_NAME_STATE;
# Line 1618  sub _get_next_token ($) { Line 1739  sub _get_next_token ($) {
1739          redo A;                  redo A;        
1740        }        }
1741      } elsif ($self->{state} == BEFORE_ATTRIBUTE_VALUE_STATE) {      } elsif ($self->{state} == BEFORE_ATTRIBUTE_VALUE_STATE) {
1742        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP        
1743          !!!cp (83);          !!!cp (83);
1744          ## Stay in the state          ## Stay in the state
1745          !!!next-input-character;          !!!next-input-character;
1746          redo A;          redo A;
1747        } elsif ($self->{next_char} == 0x0022) { # "        } elsif ($self->{nc} == 0x0022) { # "
1748          !!!cp (84);          !!!cp (84);
1749          $self->{state} = ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE;          $self->{state} = ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE;
1750          !!!next-input-character;          !!!next-input-character;
1751          redo A;          redo A;
1752        } elsif ($self->{next_char} == 0x0026) { # &        } elsif ($self->{nc} == 0x0026) { # &
1753          !!!cp (85);          !!!cp (85);
1754          $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;          $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;
1755          ## reconsume          ## reconsume
1756          redo A;          redo A;
1757        } elsif ($self->{next_char} == 0x0027) { # '        } elsif ($self->{nc} == 0x0027) { # '
1758          !!!cp (86);          !!!cp (86);
1759          $self->{state} = ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE;          $self->{state} = ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE;
1760          !!!next-input-character;          !!!next-input-character;
1761          redo A;          redo A;
1762        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1763          !!!parse-error (type => 'empty unquoted attribute value');          !!!parse-error (type => 'empty unquoted attribute value');
1764          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1765            !!!cp (87);            !!!cp (87);
1766            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1767          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1768            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1769            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1770              !!!cp (88);              !!!cp (88);
1771              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1772            } else {            } else {
# Line 1657  sub _get_next_token ($) { Line 1774  sub _get_next_token ($) {
1774              !!!cp (89);              !!!cp (89);
1775            }            }
1776          } else {          } else {
1777            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1778          }          }
1779          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1780          !!!next-input-character;          !!!next-input-character;
1781    
1782          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1783    
1784          redo A;          redo A;
1785        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1786          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1787          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1788            !!!cp (90);            !!!cp (90);
1789            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1790          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1791            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1792            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1793              !!!cp (91);              !!!cp (91);
1794              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1795            } else {            } else {
# Line 1680  sub _get_next_token ($) { Line 1797  sub _get_next_token ($) {
1797              !!!cp (92);              !!!cp (92);
1798            }            }
1799          } else {          } else {
1800            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1801          }          }
1802          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1803          ## reconsume          ## reconsume
1804    
1805          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1806    
1807          redo A;          redo A;
1808        } else {        } else {
1809          if ($self->{next_char} == 0x003D) { # =          if ($self->{nc} == 0x003D) { # =
1810            !!!cp (93);            !!!cp (93);
1811            !!!parse-error (type => 'bad attribute value');            !!!parse-error (type => 'bad attribute value');
1812          } else {          } else {
1813            !!!cp (94);            !!!cp (94);
1814          }          }
1815          $self->{current_attribute}->{value} .= chr ($self->{next_char});          $self->{ca}->{value} .= chr ($self->{nc});
1816          $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;          $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;
1817          !!!next-input-character;          !!!next-input-character;
1818          redo A;          redo A;
1819        }        }
1820      } elsif ($self->{state} == ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE) {      } elsif ($self->{state} == ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE) {
1821        if ($self->{next_char} == 0x0022) { # "        if ($self->{nc} == 0x0022) { # "
1822          !!!cp (95);          !!!cp (95);
1823          $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;          $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;
1824          !!!next-input-character;          !!!next-input-character;
1825          redo A;          redo A;
1826        } elsif ($self->{next_char} == 0x0026) { # &        } elsif ($self->{nc} == 0x0026) { # &
1827          !!!cp (96);          !!!cp (96);
1828          ## NOTE: In the spec, the tokenizer is switched to the          ## NOTE: In the spec, the tokenizer is switched to the
1829          ## "entity in attribute value state".  In this implementation, the          ## "entity in attribute value state".  In this implementation, the
1830          ## tokenizer is switched to the |ENTITY_STATE|, which is an          ## tokenizer is switched to the |ENTITY_STATE|, which is an
1831          ## implementation of the "consume a character reference" algorithm.          ## implementation of the "consume a character reference" algorithm.
1832          $self->{prev_state} = $self->{state};          $self->{prev_state} = $self->{state};
1833          $self->{entity_additional} = 0x0022; # "          $self->{entity_add} = 0x0022; # "
1834          $self->{state} = ENTITY_STATE;          $self->{state} = ENTITY_STATE;
1835          !!!next-input-character;          !!!next-input-character;
1836          redo A;          redo A;
1837        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1838          !!!parse-error (type => 'unclosed attribute value');          !!!parse-error (type => 'unclosed attribute value');
1839          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1840            !!!cp (97);            !!!cp (97);
1841            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1842          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1843            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1844            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1845              !!!cp (98);              !!!cp (98);
1846              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1847            } else {            } else {
# Line 1732  sub _get_next_token ($) { Line 1849  sub _get_next_token ($) {
1849              !!!cp (99);              !!!cp (99);
1850            }            }
1851          } else {          } else {
1852            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1853          }          }
1854          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1855          ## reconsume          ## reconsume
1856    
1857          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1858    
1859          redo A;          redo A;
1860        } else {        } else {
1861          !!!cp (100);          !!!cp (100);
1862          $self->{current_attribute}->{value} .= chr ($self->{next_char});          $self->{ca}->{value} .= chr ($self->{nc});
1863          $self->{read_until}->($self->{current_attribute}->{value},          $self->{read_until}->($self->{ca}->{value},
1864                                q["&],                                q["&],
1865                                length $self->{current_attribute}->{value});                                length $self->{ca}->{value});
1866    
1867          ## Stay in the state          ## Stay in the state
1868          !!!next-input-character;          !!!next-input-character;
1869          redo A;          redo A;
1870        }        }
1871      } elsif ($self->{state} == ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE) {      } elsif ($self->{state} == ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE) {
1872        if ($self->{next_char} == 0x0027) { # '        if ($self->{nc} == 0x0027) { # '
1873          !!!cp (101);          !!!cp (101);
1874          $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;          $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;
1875          !!!next-input-character;          !!!next-input-character;
1876          redo A;          redo A;
1877        } elsif ($self->{next_char} == 0x0026) { # &        } elsif ($self->{nc} == 0x0026) { # &
1878          !!!cp (102);          !!!cp (102);
1879          ## NOTE: In the spec, the tokenizer is switched to the          ## NOTE: In the spec, the tokenizer is switched to the
1880          ## "entity in attribute value state".  In this implementation, the          ## "entity in attribute value state".  In this implementation, the
1881          ## tokenizer is switched to the |ENTITY_STATE|, which is an          ## tokenizer is switched to the |ENTITY_STATE|, which is an
1882          ## implementation of the "consume a character reference" algorithm.          ## implementation of the "consume a character reference" algorithm.
1883          $self->{entity_additional} = 0x0027; # '          $self->{entity_add} = 0x0027; # '
1884          $self->{prev_state} = $self->{state};          $self->{prev_state} = $self->{state};
1885          $self->{state} = ENTITY_STATE;          $self->{state} = ENTITY_STATE;
1886          !!!next-input-character;          !!!next-input-character;
1887          redo A;          redo A;
1888        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1889          !!!parse-error (type => 'unclosed attribute value');          !!!parse-error (type => 'unclosed attribute value');
1890          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1891            !!!cp (103);            !!!cp (103);
1892            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1893          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1894            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1895            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1896              !!!cp (104);              !!!cp (104);
1897              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1898            } else {            } else {
# Line 1783  sub _get_next_token ($) { Line 1900  sub _get_next_token ($) {
1900              !!!cp (105);              !!!cp (105);
1901            }            }
1902          } else {          } else {
1903            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1904          }          }
1905          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1906          ## reconsume          ## reconsume
1907    
1908          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1909    
1910          redo A;          redo A;
1911        } else {        } else {
1912          !!!cp (106);          !!!cp (106);
1913          $self->{current_attribute}->{value} .= chr ($self->{next_char});          $self->{ca}->{value} .= chr ($self->{nc});
1914          $self->{read_until}->($self->{current_attribute}->{value},          $self->{read_until}->($self->{ca}->{value},
1915                                q['&],                                q['&],
1916                                length $self->{current_attribute}->{value});                                length $self->{ca}->{value});
1917    
1918          ## Stay in the state          ## Stay in the state
1919          !!!next-input-character;          !!!next-input-character;
1920          redo A;          redo A;
1921        }        }
1922      } elsif ($self->{state} == ATTRIBUTE_VALUE_UNQUOTED_STATE) {      } elsif ($self->{state} == ATTRIBUTE_VALUE_UNQUOTED_STATE) {
1923        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # HT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
1924          !!!cp (107);          !!!cp (107);
1925          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1926          !!!next-input-character;          !!!next-input-character;
1927          redo A;          redo A;
1928        } elsif ($self->{next_char} == 0x0026) { # &        } elsif ($self->{nc} == 0x0026) { # &
1929          !!!cp (108);          !!!cp (108);
1930          ## NOTE: In the spec, the tokenizer is switched to the          ## NOTE: In the spec, the tokenizer is switched to the
1931          ## "entity in attribute value state".  In this implementation, the          ## "entity in attribute value state".  In this implementation, the
1932          ## tokenizer is switched to the |ENTITY_STATE|, which is an          ## tokenizer is switched to the |ENTITY_STATE|, which is an
1933          ## implementation of the "consume a character reference" algorithm.          ## implementation of the "consume a character reference" algorithm.
1934          $self->{entity_additional} = -1;          $self->{entity_add} = -1;
1935          $self->{prev_state} = $self->{state};          $self->{prev_state} = $self->{state};
1936          $self->{state} = ENTITY_STATE;          $self->{state} = ENTITY_STATE;
1937          !!!next-input-character;          !!!next-input-character;
1938          redo A;          redo A;
1939        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1940          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1941            !!!cp (109);            !!!cp (109);
1942            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1943          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1944            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1945            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1946              !!!cp (110);              !!!cp (110);
1947              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1948            } else {            } else {
# Line 1837  sub _get_next_token ($) { Line 1950  sub _get_next_token ($) {
1950              !!!cp (111);              !!!cp (111);
1951            }            }
1952          } else {          } else {
1953            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1954          }          }
1955          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1956          !!!next-input-character;          !!!next-input-character;
1957    
1958          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1959    
1960          redo A;          redo A;
1961        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1962          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1963          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1964            !!!cp (112);            !!!cp (112);
1965            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1966          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1967            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1968            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1969              !!!cp (113);              !!!cp (113);
1970              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1971            } else {            } else {
# Line 1860  sub _get_next_token ($) { Line 1973  sub _get_next_token ($) {
1973              !!!cp (114);              !!!cp (114);
1974            }            }
1975          } else {          } else {
1976            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1977          }          }
1978          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1979          ## reconsume          ## reconsume
1980    
1981          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1982    
1983          redo A;          redo A;
1984        } else {        } else {
# Line 1873  sub _get_next_token ($) { Line 1986  sub _get_next_token ($) {
1986               0x0022 => 1, # "               0x0022 => 1, # "
1987               0x0027 => 1, # '               0x0027 => 1, # '
1988               0x003D => 1, # =               0x003D => 1, # =
1989              }->{$self->{next_char}}) {              }->{$self->{nc}}) {
1990            !!!cp (115);            !!!cp (115);
1991            !!!parse-error (type => 'bad attribute value');            !!!parse-error (type => 'bad attribute value');
1992          } else {          } else {
1993            !!!cp (116);            !!!cp (116);
1994          }          }
1995          $self->{current_attribute}->{value} .= chr ($self->{next_char});          $self->{ca}->{value} .= chr ($self->{nc});
1996          $self->{read_until}->($self->{current_attribute}->{value},          $self->{read_until}->($self->{ca}->{value},
1997                                q["'=& >],                                q["'=& >],
1998                                length $self->{current_attribute}->{value});                                length $self->{ca}->{value});
1999    
2000          ## Stay in the state          ## Stay in the state
2001          !!!next-input-character;          !!!next-input-character;
2002          redo A;          redo A;
2003        }        }
2004      } elsif ($self->{state} == AFTER_ATTRIBUTE_VALUE_QUOTED_STATE) {      } elsif ($self->{state} == AFTER_ATTRIBUTE_VALUE_QUOTED_STATE) {
2005        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
2006          !!!cp (118);          !!!cp (118);
2007          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
2008          !!!next-input-character;          !!!next-input-character;
2009          redo A;          redo A;
2010        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2011          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
2012            !!!cp (119);            !!!cp (119);
2013            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
2014          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
2015            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
2016            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
2017              !!!cp (120);              !!!cp (120);
2018              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
2019            } else {            } else {
# Line 1912  sub _get_next_token ($) { Line 2021  sub _get_next_token ($) {
2021              !!!cp (121);              !!!cp (121);
2022            }            }
2023          } else {          } else {
2024            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
2025          }          }
2026          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2027          !!!next-input-character;          !!!next-input-character;
2028    
2029          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
2030    
2031          redo A;          redo A;
2032        } elsif ($self->{next_char} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
2033          !!!cp (122);          !!!cp (122);
2034          $self->{state} = SELF_CLOSING_START_TAG_STATE;          $self->{state} = SELF_CLOSING_START_TAG_STATE;
2035          !!!next-input-character;          !!!next-input-character;
2036          redo A;          redo A;
2037        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2038          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
2039          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
2040            !!!cp (122.3);            !!!cp (122.3);
2041            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
2042          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
2043            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
2044              !!!cp (122.1);              !!!cp (122.1);
2045              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
2046            } else {            } else {
# Line 1939  sub _get_next_token ($) { Line 2048  sub _get_next_token ($) {
2048              !!!cp (122.2);              !!!cp (122.2);
2049            }            }
2050          } else {          } else {
2051            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
2052          }          }
2053          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2054          ## Reconsume.          ## Reconsume.
2055          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
2056          redo A;          redo A;
2057        } else {        } else {
2058          !!!cp ('124.1');          !!!cp ('124.1');
# Line 1953  sub _get_next_token ($) { Line 2062  sub _get_next_token ($) {
2062          redo A;          redo A;
2063        }        }
2064      } elsif ($self->{state} == SELF_CLOSING_START_TAG_STATE) {      } elsif ($self->{state} == SELF_CLOSING_START_TAG_STATE) {
2065        if ($self->{next_char} == 0x003E) { # >        if ($self->{nc} == 0x003E) { # >
2066          if ($self->{current_token}->{type} == END_TAG_TOKEN) {          if ($self->{ct}->{type} == END_TAG_TOKEN) {
2067            !!!cp ('124.2');            !!!cp ('124.2');
2068            !!!parse-error (type => 'nestc', token => $self->{current_token});            !!!parse-error (type => 'nestc', token => $self->{ct});
2069            ## TODO: Different type than slash in start tag            ## TODO: Different type than slash in start tag
2070            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
2071            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
2072              !!!cp ('124.4');              !!!cp ('124.4');
2073              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
2074            } else {            } else {
# Line 1974  sub _get_next_token ($) { Line 2083  sub _get_next_token ($) {
2083          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2084          !!!next-input-character;          !!!next-input-character;
2085    
2086          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
2087    
2088          redo A;          redo A;
2089        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2090          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
2091          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
2092            !!!cp (124.7);            !!!cp (124.7);
2093            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
2094          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
2095            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
2096              !!!cp (124.5);              !!!cp (124.5);
2097              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
2098            } else {            } else {
# Line 1991  sub _get_next_token ($) { Line 2100  sub _get_next_token ($) {
2100              !!!cp (124.6);              !!!cp (124.6);
2101            }            }
2102          } else {          } else {
2103            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
2104          }          }
2105          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2106          ## Reconsume.          ## Reconsume.
2107          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
2108          redo A;          redo A;
2109        } else {        } else {
2110          !!!cp ('124.4');          !!!cp ('124.4');
# Line 2011  sub _get_next_token ($) { Line 2120  sub _get_next_token ($) {
2120        ## NOTE: Unlike spec's "bogus comment state", this implementation        ## NOTE: Unlike spec's "bogus comment state", this implementation
2121        ## consumes characters one-by-one basis.        ## consumes characters one-by-one basis.
2122                
2123        if ($self->{next_char} == 0x003E) { # >        if ($self->{nc} == 0x003E) { # >
2124          !!!cp (124);          !!!cp (124);
2125          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2126          !!!next-input-character;          !!!next-input-character;
2127    
2128          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2129          redo A;          redo A;
2130        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2131          !!!cp (125);          !!!cp (125);
2132          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2133          ## reconsume          ## reconsume
2134    
2135          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2136          redo A;          redo A;
2137        } else {        } else {
2138          !!!cp (126);          !!!cp (126);
2139          $self->{current_token}->{data} .= chr ($self->{next_char}); # comment          $self->{ct}->{data} .= chr ($self->{nc}); # comment
2140          $self->{read_until}->($self->{current_token}->{data},          $self->{read_until}->($self->{ct}->{data},
2141                                q[>],                                q[>],
2142                                length $self->{current_token}->{data});                                length $self->{ct}->{data});
2143    
2144          ## Stay in the state.          ## Stay in the state.
2145          !!!next-input-character;          !!!next-input-character;
# Line 2039  sub _get_next_token ($) { Line 2148  sub _get_next_token ($) {
2148      } elsif ($self->{state} == MARKUP_DECLARATION_OPEN_STATE) {      } elsif ($self->{state} == MARKUP_DECLARATION_OPEN_STATE) {
2149        ## (only happen if PCDATA state)        ## (only happen if PCDATA state)
2150                
2151        if ($self->{next_char} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2152          !!!cp (133);          !!!cp (133);
2153          $self->{state} = MD_HYPHEN_STATE;          $self->{state} = MD_HYPHEN_STATE;
2154          !!!next-input-character;          !!!next-input-character;
2155          redo A;          redo A;
2156        } elsif ($self->{next_char} == 0x0044 or # D        } elsif ($self->{nc} == 0x0044 or # D
2157                 $self->{next_char} == 0x0064) { # d                 $self->{nc} == 0x0064) { # d
2158          ## ASCII case-insensitive.          ## ASCII case-insensitive.
2159          !!!cp (130);          !!!cp (130);
2160          $self->{state} = MD_DOCTYPE_STATE;          $self->{state} = MD_DOCTYPE_STATE;
2161          $self->{state_keyword} = chr $self->{next_char};          $self->{s_kwd} = chr $self->{nc};
2162          !!!next-input-character;          !!!next-input-character;
2163          redo A;          redo A;
2164        } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM and        } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM and
2165                 $self->{open_elements}->[-1]->[1] & FOREIGN_EL and                 $self->{open_elements}->[-1]->[1] & FOREIGN_EL and
2166                 $self->{next_char} == 0x005B) { # [                 $self->{nc} == 0x005B) { # [
2167          !!!cp (135.4);                          !!!cp (135.4);                
2168          $self->{state} = MD_CDATA_STATE;          $self->{state} = MD_CDATA_STATE;
2169          $self->{state_keyword} = '[';          $self->{s_kwd} = '[';
2170          !!!next-input-character;          !!!next-input-character;
2171          redo A;          redo A;
2172        } else {        } else {
# Line 2069  sub _get_next_token ($) { Line 2178  sub _get_next_token ($) {
2178                        column => $self->{column_prev} - 1);                        column => $self->{column_prev} - 1);
2179        ## Reconsume.        ## Reconsume.
2180        $self->{state} = BOGUS_COMMENT_STATE;        $self->{state} = BOGUS_COMMENT_STATE;
2181        $self->{current_token} = {type => COMMENT_TOKEN, data => '',        $self->{ct} = {type => COMMENT_TOKEN, data => '',
2182                                  line => $self->{line_prev},                                  line => $self->{line_prev},
2183                                  column => $self->{column_prev} - 1,                                  column => $self->{column_prev} - 1,
2184                                 };                                 };
2185        redo A;        redo A;
2186      } elsif ($self->{state} == MD_HYPHEN_STATE) {      } elsif ($self->{state} == MD_HYPHEN_STATE) {
2187        if ($self->{next_char} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2188          !!!cp (127);          !!!cp (127);
2189          $self->{current_token} = {type => COMMENT_TOKEN, data => '',          $self->{ct} = {type => COMMENT_TOKEN, data => '',
2190                                    line => $self->{line_prev},                                    line => $self->{line_prev},
2191                                    column => $self->{column_prev} - 2,                                    column => $self->{column_prev} - 2,
2192                                   };                                   };
# Line 2091  sub _get_next_token ($) { Line 2200  sub _get_next_token ($) {
2200                          column => $self->{column_prev} - 2);                          column => $self->{column_prev} - 2);
2201          $self->{state} = BOGUS_COMMENT_STATE;          $self->{state} = BOGUS_COMMENT_STATE;
2202          ## Reconsume.          ## Reconsume.
2203          $self->{current_token} = {type => COMMENT_TOKEN,          $self->{ct} = {type => COMMENT_TOKEN,
2204                                    data => '-',                                    data => '-',
2205                                    line => $self->{line_prev},                                    line => $self->{line_prev},
2206                                    column => $self->{column_prev} - 2,                                    column => $self->{column_prev} - 2,
# Line 2100  sub _get_next_token ($) { Line 2209  sub _get_next_token ($) {
2209        }        }
2210      } elsif ($self->{state} == MD_DOCTYPE_STATE) {      } elsif ($self->{state} == MD_DOCTYPE_STATE) {
2211        ## ASCII case-insensitive.        ## ASCII case-insensitive.
2212        if ($self->{next_char} == [        if ($self->{nc} == [
2213              undef,              undef,
2214              0x004F, # O              0x004F, # O
2215              0x0043, # C              0x0043, # C
2216              0x0054, # T              0x0054, # T
2217              0x0059, # Y              0x0059, # Y
2218              0x0050, # P              0x0050, # P
2219            ]->[length $self->{state_keyword}] or            ]->[length $self->{s_kwd}] or
2220            $self->{next_char} == [            $self->{nc} == [
2221              undef,              undef,
2222              0x006F, # o              0x006F, # o
2223              0x0063, # c              0x0063, # c
2224              0x0074, # t              0x0074, # t
2225              0x0079, # y              0x0079, # y
2226              0x0070, # p              0x0070, # p
2227            ]->[length $self->{state_keyword}]) {            ]->[length $self->{s_kwd}]) {
2228          !!!cp (131);          !!!cp (131);
2229          ## Stay in the state.          ## Stay in the state.
2230          $self->{state_keyword} .= chr $self->{next_char};          $self->{s_kwd} .= chr $self->{nc};
2231          !!!next-input-character;          !!!next-input-character;
2232          redo A;          redo A;
2233        } elsif ((length $self->{state_keyword}) == 6 and        } elsif ((length $self->{s_kwd}) == 6 and
2234                 ($self->{next_char} == 0x0045 or # E                 ($self->{nc} == 0x0045 or # E
2235                  $self->{next_char} == 0x0065)) { # e                  $self->{nc} == 0x0065)) { # e
2236          !!!cp (129);          !!!cp (129);
2237          $self->{state} = DOCTYPE_STATE;          $self->{state} = DOCTYPE_STATE;
2238          $self->{current_token} = {type => DOCTYPE_TOKEN,          $self->{ct} = {type => DOCTYPE_TOKEN,
2239                                    quirks => 1,                                    quirks => 1,
2240                                    line => $self->{line_prev},                                    line => $self->{line_prev},
2241                                    column => $self->{column_prev} - 7,                                    column => $self->{column_prev} - 7,
# Line 2137  sub _get_next_token ($) { Line 2246  sub _get_next_token ($) {
2246          !!!cp (132);                  !!!cp (132);        
2247          !!!parse-error (type => 'bogus comment',          !!!parse-error (type => 'bogus comment',
2248                          line => $self->{line_prev},                          line => $self->{line_prev},
2249                          column => $self->{column_prev} - 1 - length $self->{state_keyword});                          column => $self->{column_prev} - 1 - length $self->{s_kwd});
2250          $self->{state} = BOGUS_COMMENT_STATE;          $self->{state} = BOGUS_COMMENT_STATE;
2251          ## Reconsume.          ## Reconsume.
2252          $self->{current_token} = {type => COMMENT_TOKEN,          $self->{ct} = {type => COMMENT_TOKEN,
2253                                    data => $self->{state_keyword},                                    data => $self->{s_kwd},
2254                                    line => $self->{line_prev},                                    line => $self->{line_prev},
2255                                    column => $self->{column_prev} - 1 - length $self->{state_keyword},                                    column => $self->{column_prev} - 1 - length $self->{s_kwd},
2256                                   };                                   };
2257          redo A;          redo A;
2258        }        }
2259      } elsif ($self->{state} == MD_CDATA_STATE) {      } elsif ($self->{state} == MD_CDATA_STATE) {
2260        if ($self->{next_char} == {        if ($self->{nc} == {
2261              '[' => 0x0043, # C              '[' => 0x0043, # C
2262              '[C' => 0x0044, # D              '[C' => 0x0044, # D
2263              '[CD' => 0x0041, # A              '[CD' => 0x0041, # A
2264              '[CDA' => 0x0054, # T              '[CDA' => 0x0054, # T
2265              '[CDAT' => 0x0041, # A              '[CDAT' => 0x0041, # A
2266            }->{$self->{state_keyword}}) {            }->{$self->{s_kwd}}) {
2267          !!!cp (135.1);          !!!cp (135.1);
2268          ## Stay in the state.          ## Stay in the state.
2269          $self->{state_keyword} .= chr $self->{next_char};          $self->{s_kwd} .= chr $self->{nc};
2270          !!!next-input-character;          !!!next-input-character;
2271          redo A;          redo A;
2272        } elsif ($self->{state_keyword} eq '[CDATA' and        } elsif ($self->{s_kwd} eq '[CDATA' and
2273                 $self->{next_char} == 0x005B) { # [                 $self->{nc} == 0x005B) { # [
2274          !!!cp (135.2);          !!!cp (135.2);
2275          $self->{current_token} = {type => CHARACTER_TOKEN,          $self->{ct} = {type => CHARACTER_TOKEN,
2276                                    data => '',                                    data => '',
2277                                    line => $self->{line_prev},                                    line => $self->{line_prev},
2278                                    column => $self->{column_prev} - 7};                                    column => $self->{column_prev} - 7};
# Line 2174  sub _get_next_token ($) { Line 2283  sub _get_next_token ($) {
2283          !!!cp (135.3);          !!!cp (135.3);
2284          !!!parse-error (type => 'bogus comment',          !!!parse-error (type => 'bogus comment',
2285                          line => $self->{line_prev},                          line => $self->{line_prev},
2286                          column => $self->{column_prev} - 1 - length $self->{state_keyword});                          column => $self->{column_prev} - 1 - length $self->{s_kwd});
2287          $self->{state} = BOGUS_COMMENT_STATE;          $self->{state} = BOGUS_COMMENT_STATE;
2288          ## Reconsume.          ## Reconsume.
2289          $self->{current_token} = {type => COMMENT_TOKEN,          $self->{ct} = {type => COMMENT_TOKEN,
2290                                    data => $self->{state_keyword},                                    data => $self->{s_kwd},
2291                                    line => $self->{line_prev},                                    line => $self->{line_prev},
2292                                    column => $self->{column_prev} - 1 - length $self->{state_keyword},                                    column => $self->{column_prev} - 1 - length $self->{s_kwd},
2293                                   };                                   };
2294          redo A;          redo A;
2295        }        }
2296      } elsif ($self->{state} == COMMENT_START_STATE) {      } elsif ($self->{state} == COMMENT_START_STATE) {
2297        if ($self->{next_char} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2298          !!!cp (137);          !!!cp (137);
2299          $self->{state} = COMMENT_START_DASH_STATE;          $self->{state} = COMMENT_START_DASH_STATE;
2300          !!!next-input-character;          !!!next-input-character;
2301          redo A;          redo A;
2302        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2303          !!!cp (138);          !!!cp (138);
2304          !!!parse-error (type => 'bogus comment');          !!!parse-error (type => 'bogus comment');
2305          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2306          !!!next-input-character;          !!!next-input-character;
2307    
2308          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2309    
2310          redo A;          redo A;
2311        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2312          !!!cp (139);          !!!cp (139);
2313          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2314          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2315          ## reconsume          ## reconsume
2316    
2317          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2318    
2319          redo A;          redo A;
2320        } else {        } else {
2321          !!!cp (140);          !!!cp (140);
2322          $self->{current_token}->{data} # comment          $self->{ct}->{data} # comment
2323              .= chr ($self->{next_char});              .= chr ($self->{nc});
2324          $self->{state} = COMMENT_STATE;          $self->{state} = COMMENT_STATE;
2325          !!!next-input-character;          !!!next-input-character;
2326          redo A;          redo A;
2327        }        }
2328      } elsif ($self->{state} == COMMENT_START_DASH_STATE) {      } elsif ($self->{state} == COMMENT_START_DASH_STATE) {
2329        if ($self->{next_char} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2330          !!!cp (141);          !!!cp (141);
2331          $self->{state} = COMMENT_END_STATE;          $self->{state} = COMMENT_END_STATE;
2332          !!!next-input-character;          !!!next-input-character;
2333          redo A;          redo A;
2334        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2335          !!!cp (142);          !!!cp (142);
2336          !!!parse-error (type => 'bogus comment');          !!!parse-error (type => 'bogus comment');
2337          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2338          !!!next-input-character;          !!!next-input-character;
2339    
2340          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2341    
2342          redo A;          redo A;
2343        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2344          !!!cp (143);          !!!cp (143);
2345          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2346          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2347          ## reconsume          ## reconsume
2348    
2349          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2350    
2351          redo A;          redo A;
2352        } else {        } else {
2353          !!!cp (144);          !!!cp (144);
2354          $self->{current_token}->{data} # comment          $self->{ct}->{data} # comment
2355              .= '-' . chr ($self->{next_char});              .= '-' . chr ($self->{nc});
2356          $self->{state} = COMMENT_STATE;          $self->{state} = COMMENT_STATE;
2357          !!!next-input-character;          !!!next-input-character;
2358          redo A;          redo A;
2359        }        }
2360      } elsif ($self->{state} == COMMENT_STATE) {      } elsif ($self->{state} == COMMENT_STATE) {
2361        if ($self->{next_char} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2362          !!!cp (145);          !!!cp (145);
2363          $self->{state} = COMMENT_END_DASH_STATE;          $self->{state} = COMMENT_END_DASH_STATE;
2364          !!!next-input-character;          !!!next-input-character;
2365          redo A;          redo A;
2366        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2367          !!!cp (146);          !!!cp (146);
2368          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2369          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2370          ## reconsume          ## reconsume
2371    
2372          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2373    
2374          redo A;          redo A;
2375        } else {        } else {
2376          !!!cp (147);          !!!cp (147);
2377          $self->{current_token}->{data} .= chr ($self->{next_char}); # comment          $self->{ct}->{data} .= chr ($self->{nc}); # comment
2378          $self->{read_until}->($self->{current_token}->{data},          $self->{read_until}->($self->{ct}->{data},
2379                                q[-],                                q[-],
2380                                length $self->{current_token}->{data});                                length $self->{ct}->{data});
2381    
2382          ## Stay in the state          ## Stay in the state
2383          !!!next-input-character;          !!!next-input-character;
2384          redo A;          redo A;
2385        }        }
2386      } elsif ($self->{state} == COMMENT_END_DASH_STATE) {      } elsif ($self->{state} == COMMENT_END_DASH_STATE) {
2387        if ($self->{next_char} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2388          !!!cp (148);          !!!cp (148);
2389          $self->{state} = COMMENT_END_STATE;          $self->{state} = COMMENT_END_STATE;
2390          !!!next-input-character;          !!!next-input-character;
2391          redo A;          redo A;
2392        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2393          !!!cp (149);          !!!cp (149);
2394          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2395          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2396          ## reconsume          ## reconsume
2397    
2398          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2399    
2400          redo A;          redo A;
2401        } else {        } else {
2402          !!!cp (150);          !!!cp (150);
2403          $self->{current_token}->{data} .= '-' . chr ($self->{next_char}); # comment          $self->{ct}->{data} .= '-' . chr ($self->{nc}); # comment
2404          $self->{state} = COMMENT_STATE;          $self->{state} = COMMENT_STATE;
2405          !!!next-input-character;          !!!next-input-character;
2406          redo A;          redo A;
2407        }        }
2408      } elsif ($self->{state} == COMMENT_END_STATE) {      } elsif ($self->{state} == COMMENT_END_STATE) {
2409        if ($self->{next_char} == 0x003E) { # >        if ($self->{nc} == 0x003E) { # >
2410          !!!cp (151);          !!!cp (151);
2411          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2412          !!!next-input-character;          !!!next-input-character;
2413    
2414          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2415    
2416          redo A;          redo A;
2417        } elsif ($self->{next_char} == 0x002D) { # -        } elsif ($self->{nc} == 0x002D) { # -
2418          !!!cp (152);          !!!cp (152);
2419          !!!parse-error (type => 'dash in comment',          !!!parse-error (type => 'dash in comment',
2420                          line => $self->{line_prev},                          line => $self->{line_prev},
2421                          column => $self->{column_prev});                          column => $self->{column_prev});
2422          $self->{current_token}->{data} .= '-'; # comment          $self->{ct}->{data} .= '-'; # comment
2423          ## Stay in the state          ## Stay in the state
2424          !!!next-input-character;          !!!next-input-character;
2425          redo A;          redo A;
2426        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2427          !!!cp (153);          !!!cp (153);
2428          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2429          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2430          ## reconsume          ## reconsume
2431    
2432          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2433    
2434          redo A;          redo A;
2435        } else {        } else {
# Line 2328  sub _get_next_token ($) { Line 2437  sub _get_next_token ($) {
2437          !!!parse-error (type => 'dash in comment',          !!!parse-error (type => 'dash in comment',
2438                          line => $self->{line_prev},                          line => $self->{line_prev},
2439                          column => $self->{column_prev});                          column => $self->{column_prev});
2440          $self->{current_token}->{data} .= '--' . chr ($self->{next_char}); # comment          $self->{ct}->{data} .= '--' . chr ($self->{nc}); # comment
2441          $self->{state} = COMMENT_STATE;          $self->{state} = COMMENT_STATE;
2442          !!!next-input-character;          !!!next-input-character;
2443          redo A;          redo A;
2444        }        }
2445      } elsif ($self->{state} == DOCTYPE_STATE) {      } elsif ($self->{state} == DOCTYPE_STATE) {
2446        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
2447          !!!cp (155);          !!!cp (155);
2448          $self->{state} = BEFORE_DOCTYPE_NAME_STATE;          $self->{state} = BEFORE_DOCTYPE_NAME_STATE;
2449          !!!next-input-character;          !!!next-input-character;
# Line 2351  sub _get_next_token ($) { Line 2456  sub _get_next_token ($) {
2456          redo A;          redo A;
2457        }        }
2458      } elsif ($self->{state} == BEFORE_DOCTYPE_NAME_STATE) {      } elsif ($self->{state} == BEFORE_DOCTYPE_NAME_STATE) {
2459        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
2460          !!!cp (157);          !!!cp (157);
2461          ## Stay in the state          ## Stay in the state
2462          !!!next-input-character;          !!!next-input-character;
2463          redo A;          redo A;
2464        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2465          !!!cp (158);          !!!cp (158);
2466          !!!parse-error (type => 'no DOCTYPE name');          !!!parse-error (type => 'no DOCTYPE name');
2467          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2468          !!!next-input-character;          !!!next-input-character;
2469    
2470          !!!emit ($self->{current_token}); # DOCTYPE (quirks)          !!!emit ($self->{ct}); # DOCTYPE (quirks)
2471    
2472          redo A;          redo A;
2473        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2474          !!!cp (159);          !!!cp (159);
2475          !!!parse-error (type => 'no DOCTYPE name');          !!!parse-error (type => 'no DOCTYPE name');
2476          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2477          ## reconsume          ## reconsume
2478    
2479          !!!emit ($self->{current_token}); # DOCTYPE (quirks)          !!!emit ($self->{ct}); # DOCTYPE (quirks)
2480    
2481          redo A;          redo A;
2482        } else {        } else {
2483          !!!cp (160);          !!!cp (160);
2484          $self->{current_token}->{name} = chr $self->{next_char};          $self->{ct}->{name} = chr $self->{nc};
2485          delete $self->{current_token}->{quirks};          delete $self->{ct}->{quirks};
2486  ## ISSUE: "Set the token's name name to the" in the spec  ## ISSUE: "Set the token's name name to the" in the spec
2487          $self->{state} = DOCTYPE_NAME_STATE;          $self->{state} = DOCTYPE_NAME_STATE;
2488          !!!next-input-character;          !!!next-input-character;
# Line 2389  sub _get_next_token ($) { Line 2490  sub _get_next_token ($) {
2490        }        }
2491      } elsif ($self->{state} == DOCTYPE_NAME_STATE) {      } elsif ($self->{state} == DOCTYPE_NAME_STATE) {
2492  ## ISSUE: Redundant "First," in the spec.  ## ISSUE: Redundant "First," in the spec.
2493        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
2494          !!!cp (161);          !!!cp (161);
2495          $self->{state} = AFTER_DOCTYPE_NAME_STATE;          $self->{state} = AFTER_DOCTYPE_NAME_STATE;
2496          !!!next-input-character;          !!!next-input-character;
2497          redo A;          redo A;
2498        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2499          !!!cp (162);          !!!cp (162);
2500          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2501          !!!next-input-character;          !!!next-input-character;
2502    
2503          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2504    
2505          redo A;          redo A;
2506        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2507          !!!cp (163);          !!!cp (163);
2508          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2509          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2510          ## reconsume          ## reconsume
2511    
2512          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2513          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2514    
2515          redo A;          redo A;
2516        } else {        } else {
2517          !!!cp (164);          !!!cp (164);
2518          $self->{current_token}->{name}          $self->{ct}->{name}
2519            .= chr ($self->{next_char}); # DOCTYPE            .= chr ($self->{nc}); # DOCTYPE
2520          ## Stay in the state          ## Stay in the state
2521          !!!next-input-character;          !!!next-input-character;
2522          redo A;          redo A;
2523        }        }
2524      } elsif ($self->{state} == AFTER_DOCTYPE_NAME_STATE) {      } elsif ($self->{state} == AFTER_DOCTYPE_NAME_STATE) {
2525        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
2526          !!!cp (165);          !!!cp (165);
2527          ## Stay in the state          ## Stay in the state
2528          !!!next-input-character;          !!!next-input-character;
2529          redo A;          redo A;
2530        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2531          !!!cp (166);          !!!cp (166);
2532          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2533          !!!next-input-character;          !!!next-input-character;
2534    
2535          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2536    
2537          redo A;          redo A;
2538        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2539          !!!cp (167);          !!!cp (167);
2540          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2541          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2542          ## reconsume          ## reconsume
2543    
2544          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2545          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2546    
2547          redo A;          redo A;
2548        } elsif ($self->{next_char} == 0x0050 or # P        } elsif ($self->{nc} == 0x0050 or # P
2549                 $self->{next_char} == 0x0070) { # p                 $self->{nc} == 0x0070) { # p
2550          $self->{state} = PUBLIC_STATE;          $self->{state} = PUBLIC_STATE;
2551          $self->{state_keyword} = chr $self->{next_char};          $self->{s_kwd} = chr $self->{nc};
2552          !!!next-input-character;          !!!next-input-character;
2553          redo A;          redo A;
2554        } elsif ($self->{next_char} == 0x0053 or # S        } elsif ($self->{nc} == 0x0053 or # S
2555                 $self->{next_char} == 0x0073) { # s                 $self->{nc} == 0x0073) { # s
2556          $self->{state} = SYSTEM_STATE;          $self->{state} = SYSTEM_STATE;
2557          $self->{state_keyword} = chr $self->{next_char};          $self->{s_kwd} = chr $self->{nc};
2558          !!!next-input-character;          !!!next-input-character;
2559          redo A;          redo A;
2560        } else {        } else {
2561          !!!cp (180);          !!!cp (180);
2562          !!!parse-error (type => 'string after DOCTYPE name');          !!!parse-error (type => 'string after DOCTYPE name');
2563          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2564    
2565          $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2566          !!!next-input-character;          !!!next-input-character;
# Line 2475  sub _get_next_token ($) { Line 2568  sub _get_next_token ($) {
2568        }        }
2569      } elsif ($self->{state} == PUBLIC_STATE) {      } elsif ($self->{state} == PUBLIC_STATE) {
2570        ## ASCII case-insensitive        ## ASCII case-insensitive
2571        if ($self->{next_char} == [        if ($self->{nc} == [
2572              undef,              undef,
2573              0x0055, # U              0x0055, # U
2574              0x0042, # B              0x0042, # B
2575              0x004C, # L              0x004C, # L
2576              0x0049, # I              0x0049, # I
2577            ]->[length $self->{state_keyword}] or            ]->[length $self->{s_kwd}] or
2578            $self->{next_char} == [            $self->{nc} == [
2579              undef,              undef,
2580              0x0075, # u              0x0075, # u
2581              0x0062, # b              0x0062, # b
2582              0x006C, # l              0x006C, # l
2583              0x0069, # i              0x0069, # i
2584            ]->[length $self->{state_keyword}]) {            ]->[length $self->{s_kwd}]) {
2585          !!!cp (175);          !!!cp (175);
2586          ## Stay in the state.          ## Stay in the state.
2587          $self->{state_keyword} .= chr $self->{next_char};          $self->{s_kwd} .= chr $self->{nc};
2588          !!!next-input-character;          !!!next-input-character;
2589          redo A;          redo A;
2590        } elsif ((length $self->{state_keyword}) == 5 and        } elsif ((length $self->{s_kwd}) == 5 and
2591                 ($self->{next_char} == 0x0043 or # C                 ($self->{nc} == 0x0043 or # C
2592                  $self->{next_char} == 0x0063)) { # c                  $self->{nc} == 0x0063)) { # c
2593          !!!cp (168);          !!!cp (168);
2594          $self->{state} = BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE;          $self->{state} = BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2595          !!!next-input-character;          !!!next-input-character;
# Line 2505  sub _get_next_token ($) { Line 2598  sub _get_next_token ($) {
2598          !!!cp (169);          !!!cp (169);
2599          !!!parse-error (type => 'string after DOCTYPE name',          !!!parse-error (type => 'string after DOCTYPE name',
2600                          line => $self->{line_prev},                          line => $self->{line_prev},
2601                          column => $self->{column_prev} + 1 - length $self->{state_keyword});                          column => $self->{column_prev} + 1 - length $self->{s_kwd});
2602          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2603    
2604          $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2605          ## Reconsume.          ## Reconsume.
# Line 2514  sub _get_next_token ($) { Line 2607  sub _get_next_token ($) {
2607        }        }
2608      } elsif ($self->{state} == SYSTEM_STATE) {      } elsif ($self->{state} == SYSTEM_STATE) {
2609        ## ASCII case-insensitive        ## ASCII case-insensitive
2610        if ($self->{next_char} == [        if ($self->{nc} == [
2611              undef,              undef,
2612              0x0059, # Y              0x0059, # Y
2613              0x0053, # S              0x0053, # S
2614              0x0054, # T              0x0054, # T
2615              0x0045, # E              0x0045, # E
2616            ]->[length $self->{state_keyword}] or            ]->[length $self->{s_kwd}] or
2617            $self->{next_char} == [            $self->{nc} == [
2618              undef,              undef,
2619              0x0079, # y              0x0079, # y
2620              0x0073, # s              0x0073, # s
2621              0x0074, # t              0x0074, # t
2622              0x0065, # e              0x0065, # e
2623            ]->[length $self->{state_keyword}]) {            ]->[length $self->{s_kwd}]) {
2624          !!!cp (170);          !!!cp (170);
2625          ## Stay in the state.          ## Stay in the state.
2626          $self->{state_keyword} .= chr $self->{next_char};          $self->{s_kwd} .= chr $self->{nc};
2627          !!!next-input-character;          !!!next-input-character;
2628          redo A;          redo A;
2629        } elsif ((length $self->{state_keyword}) == 5 and        } elsif ((length $self->{s_kwd}) == 5 and
2630                 ($self->{next_char} == 0x004D or # M                 ($self->{nc} == 0x004D or # M
2631                  $self->{next_char} == 0x006D)) { # m                  $self->{nc} == 0x006D)) { # m
2632          !!!cp (171);          !!!cp (171);
2633          $self->{state} = BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE;          $self->{state} = BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2634          !!!next-input-character;          !!!next-input-character;
# Line 2544  sub _get_next_token ($) { Line 2637  sub _get_next_token ($) {
2637          !!!cp (172);          !!!cp (172);
2638          !!!parse-error (type => 'string after DOCTYPE name',          !!!parse-error (type => 'string after DOCTYPE name',
2639                          line => $self->{line_prev},                          line => $self->{line_prev},
2640                          column => $self->{column_prev} + 1 - length $self->{state_keyword});                          column => $self->{column_prev} + 1 - length $self->{s_kwd});
2641          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2642    
2643          $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2644          ## Reconsume.          ## Reconsume.
2645          redo A;          redo A;
2646        }        }
2647      } elsif ($self->{state} == BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {      } elsif ($self->{state} == BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {
2648        if ({        if ($is_space->{$self->{nc}}) {
             0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,  
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_char}}) {  
2649          !!!cp (181);          !!!cp (181);
2650          ## Stay in the state          ## Stay in the state
2651          !!!next-input-character;          !!!next-input-character;
2652          redo A;          redo A;
2653        } elsif ($self->{next_char} eq 0x0022) { # "        } elsif ($self->{nc} eq 0x0022) { # "
2654          !!!cp (182);          !!!cp (182);
2655          $self->{current_token}->{public_identifier} = ''; # DOCTYPE          $self->{ct}->{pubid} = ''; # DOCTYPE
2656          $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE;          $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE;
2657          !!!next-input-character;          !!!next-input-character;
2658          redo A;          redo A;
2659        } elsif ($self->{next_char} eq 0x0027) { # '        } elsif ($self->{nc} eq 0x0027) { # '
2660          !!!cp (183);          !!!cp (183);
2661          $self->{current_token}->{public_identifier} = ''; # DOCTYPE          $self->{ct}->{pubid} = ''; # DOCTYPE
2662          $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE;          $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE;
2663          !!!next-input-character;          !!!next-input-character;
2664          redo A;          redo A;
2665        } elsif ($self->{next_char} eq 0x003E) { # >        } elsif ($self->{nc} eq 0x003E) { # >
2666          !!!cp (184);          !!!cp (184);
2667          !!!parse-error (type => 'no PUBLIC literal');          !!!parse-error (type => 'no PUBLIC literal');
2668    
2669          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2670          !!!next-input-character;          !!!next-input-character;
2671    
2672          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2673          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2674    
2675          redo A;          redo A;
2676        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2677          !!!cp (185);          !!!cp (185);
2678          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2679    
2680          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2681          ## reconsume          ## reconsume
2682    
2683          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2684          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2685    
2686          redo A;          redo A;
2687        } else {        } else {
2688          !!!cp (186);          !!!cp (186);
2689          !!!parse-error (type => 'string after PUBLIC');          !!!parse-error (type => 'string after PUBLIC');
2690          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2691    
2692          $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2693          !!!next-input-character;          !!!next-input-character;
2694          redo A;          redo A;
2695        }        }
2696      } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE) {      } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE) {
2697        if ($self->{next_char} == 0x0022) { # "        if ($self->{nc} == 0x0022) { # "
2698          !!!cp (187);          !!!cp (187);
2699          $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;          $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2700          !!!next-input-character;          !!!next-input-character;
2701          redo A;          redo A;
2702        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2703          !!!cp (188);          !!!cp (188);
2704          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed PUBLIC literal');
2705    
2706          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2707          !!!next-input-character;          !!!next-input-character;
2708    
2709          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2710          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2711    
2712          redo A;          redo A;
2713        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2714          !!!cp (189);          !!!cp (189);
2715          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed PUBLIC literal');
2716    
2717          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2718          ## reconsume          ## reconsume
2719    
2720          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2721          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2722    
2723          redo A;          redo A;
2724        } else {        } else {
2725          !!!cp (190);          !!!cp (190);
2726          $self->{current_token}->{public_identifier} # DOCTYPE          $self->{ct}->{pubid} # DOCTYPE
2727              .= chr $self->{next_char};              .= chr $self->{nc};
2728          $self->{read_until}->($self->{current_token}->{public_identifier},          $self->{read_until}->($self->{ct}->{pubid}, q[">],
2729                                q[">],                                length $self->{ct}->{pubid});
                               length $self->{current_token}->{public_identifier});  
2730    
2731          ## Stay in the state          ## Stay in the state
2732          !!!next-input-character;          !!!next-input-character;
2733          redo A;          redo A;
2734        }        }
2735      } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE) {      } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE) {
2736        if ($self->{next_char} == 0x0027) { # '        if ($self->{nc} == 0x0027) { # '
2737          !!!cp (191);          !!!cp (191);
2738          $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;          $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2739          !!!next-input-character;          !!!next-input-character;
2740          redo A;          redo A;
2741        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2742          !!!cp (192);          !!!cp (192);
2743          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed PUBLIC literal');
2744    
2745          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2746          !!!next-input-character;          !!!next-input-character;
2747    
2748          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2749          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2750    
2751          redo A;          redo A;
2752        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2753          !!!cp (193);          !!!cp (193);
2754          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed PUBLIC literal');
2755    
2756          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2757          ## reconsume          ## reconsume
2758    
2759          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2760          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2761    
2762          redo A;          redo A;
2763        } else {        } else {
2764          !!!cp (194);          !!!cp (194);
2765          $self->{current_token}->{public_identifier} # DOCTYPE          $self->{ct}->{pubid} # DOCTYPE
2766              .= chr $self->{next_char};              .= chr $self->{nc};
2767          $self->{read_until}->($self->{current_token}->{public_identifier},          $self->{read_until}->($self->{ct}->{pubid}, q['>],
2768                                q['>],                                length $self->{ct}->{pubid});
                               length $self->{current_token}->{public_identifier});  
2769    
2770          ## Stay in the state          ## Stay in the state
2771          !!!next-input-character;          !!!next-input-character;
2772          redo A;          redo A;
2773        }        }
2774      } elsif ($self->{state} == AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {      } elsif ($self->{state} == AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {
2775        if ({        if ($is_space->{$self->{nc}}) {
             0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,  
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_char}}) {  
2776          !!!cp (195);          !!!cp (195);
2777          ## Stay in the state          ## Stay in the state
2778          !!!next-input-character;          !!!next-input-character;
2779          redo A;          redo A;
2780        } elsif ($self->{next_char} == 0x0022) { # "        } elsif ($self->{nc} == 0x0022) { # "
2781          !!!cp (196);          !!!cp (196);
2782          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          $self->{ct}->{sysid} = ''; # DOCTYPE
2783          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;
2784          !!!next-input-character;          !!!next-input-character;
2785          redo A;          redo A;
2786        } elsif ($self->{next_char} == 0x0027) { # '        } elsif ($self->{nc} == 0x0027) { # '
2787          !!!cp (197);          !!!cp (197);
2788          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          $self->{ct}->{sysid} = ''; # DOCTYPE
2789          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;
2790          !!!next-input-character;          !!!next-input-character;
2791          redo A;          redo A;
2792        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2793          !!!cp (198);          !!!cp (198);
2794          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2795          !!!next-input-character;          !!!next-input-character;
2796    
2797          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2798    
2799          redo A;          redo A;
2800        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2801          !!!cp (199);          !!!cp (199);
2802          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2803    
2804          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2805          ## reconsume          ## reconsume
2806    
2807          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2808          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2809    
2810          redo A;          redo A;
2811        } else {        } else {
2812          !!!cp (200);          !!!cp (200);
2813          !!!parse-error (type => 'string after PUBLIC literal');          !!!parse-error (type => 'string after PUBLIC literal');
2814          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2815    
2816          $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2817          !!!next-input-character;          !!!next-input-character;
2818          redo A;          redo A;
2819        }        }
2820      } elsif ($self->{state} == BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {      } elsif ($self->{state} == BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {
2821        if ({        if ($is_space->{$self->{nc}}) {
             0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,  
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_char}}) {  
2822          !!!cp (201);          !!!cp (201);
2823          ## Stay in the state          ## Stay in the state
2824          !!!next-input-character;          !!!next-input-character;
2825          redo A;          redo A;
2826        } elsif ($self->{next_char} == 0x0022) { # "        } elsif ($self->{nc} == 0x0022) { # "
2827          !!!cp (202);          !!!cp (202);
2828          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          $self->{ct}->{sysid} = ''; # DOCTYPE
2829          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;
2830          !!!next-input-character;          !!!next-input-character;
2831          redo A;          redo A;
2832        } elsif ($self->{next_char} == 0x0027) { # '        } elsif ($self->{nc} == 0x0027) { # '
2833          !!!cp (203);          !!!cp (203);
2834          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          $self->{ct}->{sysid} = ''; # DOCTYPE
2835          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;
2836          !!!next-input-character;          !!!next-input-character;
2837          redo A;          redo A;
2838        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2839          !!!cp (204);          !!!cp (204);
2840          !!!parse-error (type => 'no SYSTEM literal');          !!!parse-error (type => 'no SYSTEM literal');
2841          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2842          !!!next-input-character;          !!!next-input-character;
2843    
2844          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2845          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2846    
2847          redo A;          redo A;
2848        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2849          !!!cp (205);          !!!cp (205);
2850          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2851    
2852          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2853          ## reconsume          ## reconsume
2854    
2855          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2856          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2857    
2858          redo A;          redo A;
2859        } else {        } else {
2860          !!!cp (206);          !!!cp (206);
2861          !!!parse-error (type => 'string after SYSTEM');          !!!parse-error (type => 'string after SYSTEM');
2862          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2863    
2864          $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2865          !!!next-input-character;          !!!next-input-character;
2866          redo A;          redo A;
2867        }        }
2868      } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE) {      } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE) {
2869        if ($self->{next_char} == 0x0022) { # "        if ($self->{nc} == 0x0022) { # "
2870          !!!cp (207);          !!!cp (207);
2871          $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;          $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2872          !!!next-input-character;          !!!next-input-character;
2873          redo A;          redo A;
2874        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2875          !!!cp (208);          !!!cp (208);
2876          !!!parse-error (type => 'unclosed SYSTEM literal');          !!!parse-error (type => 'unclosed SYSTEM literal');
2877    
2878          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2879          !!!next-input-character;          !!!next-input-character;
2880    
2881          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2882          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2883    
2884          redo A;          redo A;
2885        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2886          !!!cp (209);          !!!cp (209);
2887          !!!parse-error (type => 'unclosed SYSTEM literal');          !!!parse-error (type => 'unclosed SYSTEM literal');
2888    
2889          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2890          ## reconsume          ## reconsume
2891    
2892          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2893          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2894    
2895          redo A;          redo A;
2896        } else {        } else {
2897          !!!cp (210);          !!!cp (210);
2898          $self->{current_token}->{system_identifier} # DOCTYPE          $self->{ct}->{sysid} # DOCTYPE
2899              .= chr $self->{next_char};              .= chr $self->{nc};
2900          $self->{read_until}->($self->{current_token}->{system_identifier},          $self->{read_until}->($self->{ct}->{sysid}, q[">],
2901                                q[">],                                length $self->{ct}->{sysid});
                               length $self->{current_token}->{system_identifier});  
2902    
2903          ## Stay in the state          ## Stay in the state
2904          !!!next-input-character;          !!!next-input-character;
2905          redo A;          redo A;
2906        }        }
2907      } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE) {      } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE) {
2908        if ($self->{next_char} == 0x0027) { # '        if ($self->{nc} == 0x0027) { # '
2909          !!!cp (211);          !!!cp (211);
2910          $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;          $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2911          !!!next-input-character;          !!!next-input-character;
2912          redo A;          redo A;
2913        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2914          !!!cp (212);          !!!cp (212);
2915          !!!parse-error (type => 'unclosed SYSTEM literal');          !!!parse-error (type => 'unclosed SYSTEM literal');
2916    
2917          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2918          !!!next-input-character;          !!!next-input-character;
2919    
2920          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2921          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2922    
2923          redo A;          redo A;
2924        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2925          !!!cp (213);          !!!cp (213);
2926          !!!parse-error (type => 'unclosed SYSTEM literal');          !!!parse-error (type => 'unclosed SYSTEM literal');
2927    
2928          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2929          ## reconsume          ## reconsume
2930    
2931          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2932          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2933    
2934          redo A;          redo A;
2935        } else {        } else {
2936          !!!cp (214);          !!!cp (214);
2937          $self->{current_token}->{system_identifier} # DOCTYPE          $self->{ct}->{sysid} # DOCTYPE
2938              .= chr $self->{next_char};              .= chr $self->{nc};
2939          $self->{read_until}->($self->{current_token}->{system_identifier},          $self->{read_until}->($self->{ct}->{sysid}, q['>],
2940                                q['>],                                length $self->{ct}->{sysid});
                               length $self->{current_token}->{system_identifier});  
2941    
2942          ## Stay in the state          ## Stay in the state
2943          !!!next-input-character;          !!!next-input-character;
2944          redo A;          redo A;
2945        }        }
2946      } elsif ($self->{state} == AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {      } elsif ($self->{state} == AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {
2947        if ({        if ($is_space->{$self->{nc}}) {
             0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,  
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_char}}) {  
2948          !!!cp (215);          !!!cp (215);
2949          ## Stay in the state          ## Stay in the state
2950          !!!next-input-character;          !!!next-input-character;
2951          redo A;          redo A;
2952        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2953          !!!cp (216);          !!!cp (216);
2954          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2955          !!!next-input-character;          !!!next-input-character;
2956    
2957          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2958    
2959          redo A;          redo A;
2960        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2961          !!!cp (217);          !!!cp (217);
2962          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2963          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2964          ## reconsume          ## reconsume
2965    
2966          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2967          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2968    
2969          redo A;          redo A;
2970        } else {        } else {
2971          !!!cp (218);          !!!cp (218);
2972          !!!parse-error (type => 'string after SYSTEM literal');          !!!parse-error (type => 'string after SYSTEM literal');
2973          #$self->{current_token}->{quirks} = 1;          #$self->{ct}->{quirks} = 1;
2974    
2975          $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2976          !!!next-input-character;          !!!next-input-character;
2977          redo A;          redo A;
2978        }        }
2979      } elsif ($self->{state} == BOGUS_DOCTYPE_STATE) {      } elsif ($self->{state} == BOGUS_DOCTYPE_STATE) {
2980        if ($self->{next_char} == 0x003E) { # >        if ($self->{nc} == 0x003E) { # >
2981          !!!cp (219);          !!!cp (219);
2982          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2983          !!!next-input-character;          !!!next-input-character;
2984    
2985          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2986    
2987          redo A;          redo A;
2988        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2989          !!!cp (220);          !!!cp (220);
2990          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2991          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2992          ## reconsume          ## reconsume
2993    
2994          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2995    
2996          redo A;          redo A;
2997        } else {        } else {
# Line 2931  sub _get_next_token ($) { Line 3008  sub _get_next_token ($) {
3008        ## by three states, |CDATA_SECTION_STATE|, |CDATA_SECTION_MSE1_STATE|,        ## by three states, |CDATA_SECTION_STATE|, |CDATA_SECTION_MSE1_STATE|,
3009        ## and |CDATA_SECTION_MSE2_STATE|.        ## and |CDATA_SECTION_MSE2_STATE|.
3010                
3011        if ($self->{next_char} == 0x005D) { # ]        if ($self->{nc} == 0x005D) { # ]
3012          !!!cp (221.1);          !!!cp (221.1);
3013          $self->{state} = CDATA_SECTION_MSE1_STATE;          $self->{state} = CDATA_SECTION_MSE1_STATE;
3014          !!!next-input-character;          !!!next-input-character;
3015          redo A;          redo A;
3016        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
3017          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
3018          !!!next-input-character;          !!!next-input-character;
3019          if (length $self->{current_token}->{data}) { # character          if (length $self->{ct}->{data}) { # character
3020            !!!cp (221.2);            !!!cp (221.2);
3021            !!!emit ($self->{current_token}); # character            !!!emit ($self->{ct}); # character
3022          } else {          } else {
3023            !!!cp (221.3);            !!!cp (221.3);
3024            ## No token to emit. $self->{current_token} is discarded.            ## No token to emit. $self->{ct} is discarded.
3025          }                  }        
3026          redo A;          redo A;
3027        } else {        } else {
3028          !!!cp (221.4);          !!!cp (221.4);
3029          $self->{current_token}->{data} .= chr $self->{next_char};          $self->{ct}->{data} .= chr $self->{nc};
3030          $self->{read_until}->($self->{current_token}->{data},          $self->{read_until}->($self->{ct}->{data},
3031                                q<]>,                                q<]>,
3032                                length $self->{current_token}->{data});                                length $self->{ct}->{data});
3033    
3034          ## Stay in the state.          ## Stay in the state.
3035          !!!next-input-character;          !!!next-input-character;
# Line 2961  sub _get_next_token ($) { Line 3038  sub _get_next_token ($) {
3038    
3039        ## ISSUE: "text tokens" in spec.        ## ISSUE: "text tokens" in spec.
3040      } elsif ($self->{state} == CDATA_SECTION_MSE1_STATE) {      } elsif ($self->{state} == CDATA_SECTION_MSE1_STATE) {
3041        if ($self->{next_char} == 0x005D) { # ]        if ($self->{nc} == 0x005D) { # ]
3042          !!!cp (221.5);          !!!cp (221.5);
3043          $self->{state} = CDATA_SECTION_MSE2_STATE;          $self->{state} = CDATA_SECTION_MSE2_STATE;
3044          !!!next-input-character;          !!!next-input-character;
3045          redo A;          redo A;
3046        } else {        } else {
3047          !!!cp (221.6);          !!!cp (221.6);
3048          $self->{current_token}->{data} .= ']';          $self->{ct}->{data} .= ']';
3049          $self->{state} = CDATA_SECTION_STATE;          $self->{state} = CDATA_SECTION_STATE;
3050          ## Reconsume.          ## Reconsume.
3051          redo A;          redo A;
3052        }        }
3053      } elsif ($self->{state} == CDATA_SECTION_MSE2_STATE) {      } elsif ($self->{state} == CDATA_SECTION_MSE2_STATE) {
3054        if ($self->{next_char} == 0x003E) { # >        if ($self->{nc} == 0x003E) { # >
3055          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
3056          !!!next-input-character;          !!!next-input-character;
3057          if (length $self->{current_token}->{data}) { # character          if (length $self->{ct}->{data}) { # character
3058            !!!cp (221.7);            !!!cp (221.7);
3059            !!!emit ($self->{current_token}); # character            !!!emit ($self->{ct}); # character
3060          } else {          } else {
3061            !!!cp (221.8);            !!!cp (221.8);
3062            ## No token to emit. $self->{current_token} is discarded.            ## No token to emit. $self->{ct} is discarded.
3063          }          }
3064          redo A;          redo A;
3065        } elsif ($self->{next_char} == 0x005D) { # ]        } elsif ($self->{nc} == 0x005D) { # ]
3066          !!!cp (221.9); # character          !!!cp (221.9); # character
3067          $self->{current_token}->{data} .= ']'; ## Add first "]" of "]]]".          $self->{ct}->{data} .= ']'; ## Add first "]" of "]]]".
3068          ## Stay in the state.          ## Stay in the state.
3069          !!!next-input-character;          !!!next-input-character;
3070          redo A;          redo A;
3071        } else {        } else {
3072          !!!cp (221.11);          !!!cp (221.11);
3073          $self->{current_token}->{data} .= ']]'; # character          $self->{ct}->{data} .= ']]'; # character
3074          $self->{state} = CDATA_SECTION_STATE;          $self->{state} = CDATA_SECTION_STATE;
3075          ## Reconsume.          ## Reconsume.
3076          redo A;          redo A;
3077        }        }
3078      } elsif ($self->{state} == ENTITY_STATE) {      } elsif ($self->{state} == ENTITY_STATE) {
3079        if ({        if ($is_space->{$self->{nc}} or
3080          0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, # HT, LF, VT, FF,            {
3081          0x0020 => 1, 0x003C => 1, 0x0026 => 1, -1 => 1, # SP, <, &              0x003C => 1, 0x0026 => 1, -1 => 1, # <, &
3082          $self->{entity_additional} => 1,              $self->{entity_add} => 1,
3083        }->{$self->{next_char}}) {            }->{$self->{nc}}) {
3084          !!!cp (1001);          !!!cp (1001);
3085          ## Don't consume          ## Don't consume
3086          ## No error          ## No error
3087          ## Return nothing.          ## Return nothing.
3088          #          #
3089        } elsif ($self->{next_char} == 0x0023) { # #        } elsif ($self->{nc} == 0x0023) { # #
3090          !!!cp (999);          !!!cp (999);
3091          $self->{state} = ENTITY_HASH_STATE;          $self->{state} = ENTITY_HASH_STATE;
3092          $self->{state_keyword} = '#';          $self->{s_kwd} = '#';
3093          !!!next-input-character;          !!!next-input-character;
3094          redo A;          redo A;
3095        } elsif ((0x0041 <= $self->{next_char} and        } elsif ((0x0041 <= $self->{nc} and
3096                  $self->{next_char} <= 0x005A) or # A..Z                  $self->{nc} <= 0x005A) or # A..Z
3097                 (0x0061 <= $self->{next_char} and                 (0x0061 <= $self->{nc} and
3098                  $self->{next_char} <= 0x007A)) { # a..z                  $self->{nc} <= 0x007A)) { # a..z
3099          !!!cp (998);          !!!cp (998);
3100          require Whatpm::_NamedEntityList;          require Whatpm::_NamedEntityList;
3101          $self->{state} = ENTITY_NAME_STATE;          $self->{state} = ENTITY_NAME_STATE;
3102          $self->{state_keyword} = chr $self->{next_char};          $self->{s_kwd} = chr $self->{nc};
3103          $self->{entity__value} = $self->{state_keyword};          $self->{entity__value} = $self->{s_kwd};
3104          $self->{entity__match} = 0;          $self->{entity__match} = 0;
3105          !!!next-input-character;          !!!next-input-character;
3106          redo A;          redo A;
# Line 3051  sub _get_next_token ($) { Line 3128  sub _get_next_token ($) {
3128          redo A;          redo A;
3129        } else {        } else {
3130          !!!cp (996);          !!!cp (996);
3131          $self->{current_attribute}->{value} .= '&';          $self->{ca}->{value} .= '&';
3132          $self->{state} = $self->{prev_state};          $self->{state} = $self->{prev_state};
3133          ## Reconsume.          ## Reconsume.
3134          redo A;          redo A;
3135        }        }
3136      } elsif ($self->{state} == ENTITY_HASH_STATE) {      } elsif ($self->{state} == ENTITY_HASH_STATE) {
3137        if ($self->{next_char} == 0x0078 or # x        if ($self->{nc} == 0x0078 or # x
3138            $self->{next_char} == 0x0058) { # X            $self->{nc} == 0x0058) { # X
3139          !!!cp (995);          !!!cp (995);
3140          $self->{state} = HEXREF_X_STATE;          $self->{state} = HEXREF_X_STATE;
3141          $self->{state_keyword} .= chr $self->{next_char};          $self->{s_kwd} .= chr $self->{nc};
3142          !!!next-input-character;          !!!next-input-character;
3143          redo A;          redo A;
3144        } elsif (0x0030 <= $self->{next_char} and        } elsif (0x0030 <= $self->{nc} and
3145                 $self->{next_char} <= 0x0039) { # 0..9                 $self->{nc} <= 0x0039) { # 0..9
3146          !!!cp (994);          !!!cp (994);
3147          $self->{state} = NCR_NUM_STATE;          $self->{state} = NCR_NUM_STATE;
3148          $self->{state_keyword} = $self->{next_char} - 0x0030;          $self->{s_kwd} = $self->{nc} - 0x0030;
3149          !!!next-input-character;          !!!next-input-character;
3150          redo A;          redo A;
3151        } else {        } else {
# Line 3092  sub _get_next_token ($) { Line 3169  sub _get_next_token ($) {
3169            redo A;            redo A;
3170          } else {          } else {
3171            !!!cp (993);            !!!cp (993);
3172            $self->{current_attribute}->{value} .= '&#';            $self->{ca}->{value} .= '&#';
3173            $self->{state} = $self->{prev_state};            $self->{state} = $self->{prev_state};
3174            ## Reconsume.            ## Reconsume.
3175            redo A;            redo A;
3176          }          }
3177        }        }
3178      } elsif ($self->{state} == NCR_NUM_STATE) {      } elsif ($self->{state} == NCR_NUM_STATE) {
3179        if (0x0030 <= $self->{next_char} and        if (0x0030 <= $self->{nc} and
3180            $self->{next_char} <= 0x0039) { # 0..9            $self->{nc} <= 0x0039) { # 0..9
3181          !!!cp (1012);          !!!cp (1012);
3182          $self->{state_keyword} *= 10;          $self->{s_kwd} *= 10;
3183          $self->{state_keyword} += $self->{next_char} - 0x0030;          $self->{s_kwd} += $self->{nc} - 0x0030;
3184                    
3185          ## Stay in the state.          ## Stay in the state.
3186          !!!next-input-character;          !!!next-input-character;
3187          redo A;          redo A;
3188        } elsif ($self->{next_char} == 0x003B) { # ;        } elsif ($self->{nc} == 0x003B) { # ;
3189          !!!cp (1013);          !!!cp (1013);
3190          !!!next-input-character;          !!!next-input-character;
3191          #          #
# Line 3119  sub _get_next_token ($) { Line 3196  sub _get_next_token ($) {
3196          #          #
3197        }        }
3198    
3199        my $code = $self->{state_keyword};        my $code = $self->{s_kwd};
3200        my $l = $self->{line_prev};        my $l = $self->{line_prev};
3201        my $c = $self->{column_prev};        my $c = $self->{column_prev};
3202        if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) {        if ($charref_map->{$code}) {
3203          !!!cp (1015);          !!!cp (1015);
3204          !!!parse-error (type => 'invalid character reference',          !!!parse-error (type => 'invalid character reference',
3205                          text => (sprintf 'U+%04X', $code),                          text => (sprintf 'U+%04X', $code),
3206                          line => $l, column => $c);                          line => $l, column => $c);
3207          $code = 0xFFFD;          $code = $charref_map->{$code};
3208        } elsif ($code > 0x10FFFF) {        } elsif ($code > 0x10FFFF) {
3209          !!!cp (1016);          !!!cp (1016);
3210          !!!parse-error (type => 'invalid character reference',          !!!parse-error (type => 'invalid character reference',
3211                          text => (sprintf 'U-%08X', $code),                          text => (sprintf 'U-%08X', $code),
3212                          line => $l, column => $c);                          line => $l, column => $c);
3213          $code = 0xFFFD;          $code = 0xFFFD;
       } elsif ($code == 0x000D) {  
         !!!cp (1017);  
         !!!parse-error (type => 'CR character reference',  
                         line => $l, column => $c);  
         $code = 0x000A;  
       } elsif (0x80 <= $code and $code <= 0x9F) {  
         !!!cp (1018);  
         !!!parse-error (type => 'C1 character reference',  
                         text => (sprintf 'U+%04X', $code),  
                         line => $l, column => $c);  
         $code = $c1_entity_char->{$code};  
3214        }        }
3215    
3216        if ($self->{prev_state} == DATA_STATE) {        if ($self->{prev_state} == DATA_STATE) {
# Line 3157  sub _get_next_token ($) { Line 3223  sub _get_next_token ($) {
3223          redo A;          redo A;
3224        } else {        } else {
3225          !!!cp (991);          !!!cp (991);
3226          $self->{current_attribute}->{value} .= chr $code;          $self->{ca}->{value} .= chr $code;
3227          $self->{current_attribute}->{has_reference} = 1;          $self->{ca}->{has_reference} = 1;
3228          $self->{state} = $self->{prev_state};          $self->{state} = $self->{prev_state};
3229          ## Reconsume.          ## Reconsume.
3230          redo A;          redo A;
3231        }        }
3232      } elsif ($self->{state} == HEXREF_X_STATE) {      } elsif ($self->{state} == HEXREF_X_STATE) {
3233        if ((0x0030 <= $self->{next_char} and $self->{next_char} <= 0x0039) or        if ((0x0030 <= $self->{nc} and $self->{nc} <= 0x0039) or
3234            (0x0041 <= $self->{next_char} and $self->{next_char} <= 0x0046) or            (0x0041 <= $self->{nc} and $self->{nc} <= 0x0046) or
3235            (0x0061 <= $self->{next_char} and $self->{next_char} <= 0x0066)) {            (0x0061 <= $self->{nc} and $self->{nc} <= 0x0066)) {
3236          # 0..9, A..F, a..f          # 0..9, A..F, a..f
3237          !!!cp (990);          !!!cp (990);
3238          $self->{state} = HEXREF_HEX_STATE;          $self->{state} = HEXREF_HEX_STATE;
3239          $self->{state_keyword} = 0;          $self->{s_kwd} = 0;
3240          ## Reconsume.          ## Reconsume.
3241          redo A;          redo A;
3242        } else {        } else {
# Line 3187  sub _get_next_token ($) { Line 3253  sub _get_next_token ($) {
3253            $self->{state} = $self->{prev_state};            $self->{state} = $self->{prev_state};
3254            ## Reconsume.            ## Reconsume.
3255            !!!emit ({type => CHARACTER_TOKEN,            !!!emit ({type => CHARACTER_TOKEN,
3256                      data => '&' . $self->{state_keyword},                      data => '&' . $self->{s_kwd},
3257                      line => $self->{line_prev},                      line => $self->{line_prev},
3258                      column => $self->{column_prev} - length $self->{state_keyword},                      column => $self->{column_prev} - length $self->{s_kwd},
3259                     });                     });
3260            redo A;            redo A;
3261          } else {          } else {
3262            !!!cp (989);            !!!cp (989);
3263            $self->{current_attribute}->{value} .= '&' . $self->{state_keyword};            $self->{ca}->{value} .= '&' . $self->{s_kwd};
3264            $self->{state} = $self->{prev_state};            $self->{state} = $self->{prev_state};
3265            ## Reconsume.            ## Reconsume.
3266            redo A;            redo A;
3267          }          }
3268        }        }
3269      } elsif ($self->{state} == HEXREF_HEX_STATE) {      } elsif ($self->{state} == HEXREF_HEX_STATE) {
3270        if (0x0030 <= $self->{next_char} and $self->{next_char} <= 0x0039) {        if (0x0030 <= $self->{nc} and $self->{nc} <= 0x0039) {
3271          # 0..9          # 0..9
3272          !!!cp (1002);          !!!cp (1002);
3273          $self->{state_keyword} *= 0x10;          $self->{s_kwd} *= 0x10;
3274          $self->{state_keyword} += $self->{next_char} - 0x0030;          $self->{s_kwd} += $self->{nc} - 0x0030;
3275          ## Stay in the state.          ## Stay in the state.
3276          !!!next-input-character;          !!!next-input-character;
3277          redo A;          redo A;
3278        } elsif (0x0061 <= $self->{next_char} and        } elsif (0x0061 <= $self->{nc} and
3279                 $self->{next_char} <= 0x0066) { # a..f                 $self->{nc} <= 0x0066) { # a..f
3280          !!!cp (1003);          !!!cp (1003);
3281          $self->{state_keyword} *= 0x10;          $self->{s_kwd} *= 0x10;
3282          $self->{state_keyword} += $self->{next_char} - 0x0060 + 9;          $self->{s_kwd} += $self->{nc} - 0x0060 + 9;
3283          ## Stay in the state.          ## Stay in the state.
3284          !!!next-input-character;          !!!next-input-character;
3285          redo A;          redo A;
3286        } elsif (0x0041 <= $self->{next_char} and        } elsif (0x0041 <= $self->{nc} and
3287                 $self->{next_char} <= 0x0046) { # A..F                 $self->{nc} <= 0x0046) { # A..F
3288          !!!cp (1004);          !!!cp (1004);
3289          $self->{state_keyword} *= 0x10;          $self->{s_kwd} *= 0x10;
3290          $self->{state_keyword} += $self->{next_char} - 0x0040 + 9;          $self->{s_kwd} += $self->{nc} - 0x0040 + 9;
3291          ## Stay in the state.          ## Stay in the state.
3292          !!!next-input-character;          !!!next-input-character;
3293          redo A;          redo A;
3294        } elsif ($self->{next_char} == 0x003B) { # ;        } elsif ($self->{nc} == 0x003B) { # ;
3295          !!!cp (1006);          !!!cp (1006);
3296          !!!next-input-character;          !!!next-input-character;
3297          #          #
# Line 3238  sub _get_next_token ($) { Line 3304  sub _get_next_token ($) {
3304          #          #
3305        }        }
3306    
3307        my $code = $self->{state_keyword};        my $code = $self->{s_kwd};
3308        my $l = $self->{line_prev};        my $l = $self->{line_prev};
3309        my $c = $self->{column_prev};        my $c = $self->{column_prev};
3310        if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) {        if ($charref_map->{$code}) {
3311          !!!cp (1008);          !!!cp (1008);
3312          !!!parse-error (type => 'invalid character reference',          !!!parse-error (type => 'invalid character reference',
3313                          text => (sprintf 'U+%04X', $code),                          text => (sprintf 'U+%04X', $code),
3314                          line => $l, column => $c);                          line => $l, column => $c);
3315          $code = 0xFFFD;          $code = $charref_map->{$code};
3316        } elsif ($code > 0x10FFFF) {        } elsif ($code > 0x10FFFF) {
3317          !!!cp (1009);          !!!cp (1009);
3318          !!!parse-error (type => 'invalid character reference',          !!!parse-error (type => 'invalid character reference',
3319                          text => (sprintf 'U-%08X', $code),                          text => (sprintf 'U-%08X', $code),
3320                          line => $l, column => $c);                          line => $l, column => $c);
3321          $code = 0xFFFD;          $code = 0xFFFD;
       } elsif ($code == 0x000D) {  
         !!!cp (1010);  
         !!!parse-error (type => 'CR character reference', line => $l, column => $c);  
         $code = 0x000A;  
       } elsif (0x80 <= $code and $code <= 0x9F) {  
         !!!cp (1011);  
         !!!parse-error (type => 'C1 character reference', text => (sprintf 'U+%04X', $code), line => $l, column => $c);  
         $code = $c1_entity_char->{$code};  
3322        }        }
3323    
3324        if ($self->{prev_state} == DATA_STATE) {        if ($self->{prev_state} == DATA_STATE) {
# Line 3273  sub _get_next_token ($) { Line 3331  sub _get_next_token ($) {
3331          redo A;          redo A;
3332        } else {        } else {
3333          !!!cp (987);          !!!cp (987);
3334          $self->{current_attribute}->{value} .= chr $code;          $self->{ca}->{value} .= chr $code;
3335          $self->{current_attribute}->{has_reference} = 1;          $self->{ca}->{has_reference} = 1;
3336          $self->{state} = $self->{prev_state};          $self->{state} = $self->{prev_state};
3337          ## Reconsume.          ## Reconsume.
3338          redo A;          redo A;
3339        }        }
3340      } elsif ($self->{state} == ENTITY_NAME_STATE) {      } elsif ($self->{state} == ENTITY_NAME_STATE) {
3341        if (length $self->{state_keyword} < 30 and        if (length $self->{s_kwd} < 30 and
3342            ## NOTE: Some number greater than the maximum length of entity name            ## NOTE: Some number greater than the maximum length of entity name
3343            ((0x0041 <= $self->{next_char} and # a            ((0x0041 <= $self->{nc} and # a
3344              $self->{next_char} <= 0x005A) or # x              $self->{nc} <= 0x005A) or # x
3345             (0x0061 <= $self->{next_char} and # a             (0x0061 <= $self->{nc} and # a
3346              $self->{next_char} <= 0x007A) or # z              $self->{nc} <= 0x007A) or # z
3347             (0x0030 <= $self->{next_char} and # 0             (0x0030 <= $self->{nc} and # 0
3348              $self->{next_char} <= 0x0039) or # 9              $self->{nc} <= 0x0039) or # 9
3349             $self->{next_char} == 0x003B)) { # ;             $self->{nc} == 0x003B)) { # ;
3350          our $EntityChar;          our $EntityChar;
3351          $self->{state_keyword} .= chr $self->{next_char};          $self->{s_kwd} .= chr $self->{nc};
3352          if (defined $EntityChar->{$self->{state_keyword}}) {          if (defined $EntityChar->{$self->{s_kwd}}) {
3353            if ($self->{next_char} == 0x003B) { # ;            if ($self->{nc} == 0x003B) { # ;
3354              !!!cp (1020);              !!!cp (1020);
3355              $self->{entity__value} = $EntityChar->{$self->{state_keyword}};              $self->{entity__value} = $EntityChar->{$self->{s_kwd}};
3356              $self->{entity__match} = 1;              $self->{entity__match} = 1;
3357              !!!next-input-character;              !!!next-input-character;
3358              #              #
3359            } else {            } else {
3360              !!!cp (1021);              !!!cp (1021);
3361              $self->{entity__value} = $EntityChar->{$self->{state_keyword}};              $self->{entity__value} = $EntityChar->{$self->{s_kwd}};
3362              $self->{entity__match} = -1;              $self->{entity__match} = -1;
3363              ## Stay in the state.              ## Stay in the state.
3364              !!!next-input-character;              !!!next-input-character;
# Line 3308  sub _get_next_token ($) { Line 3366  sub _get_next_token ($) {
3366            }            }
3367          } else {          } else {
3368            !!!cp (1022);            !!!cp (1022);
3369            $self->{entity__value} .= chr $self->{next_char};            $self->{entity__value} .= chr $self->{nc};
3370            $self->{entity__match} *= 2;            $self->{entity__match} *= 2;
3371            ## Stay in the state.            ## Stay in the state.
3372            !!!next-input-character;            !!!next-input-character;
# Line 3328  sub _get_next_token ($) { Line 3386  sub _get_next_token ($) {
3386          if ($self->{prev_state} != DATA_STATE and # in attribute          if ($self->{prev_state} != DATA_STATE and # in attribute
3387              $self->{entity__match} < -1) {              $self->{entity__match} < -1) {
3388            !!!cp (1024);            !!!cp (1024);
3389            $data = '&' . $self->{state_keyword};            $data = '&' . $self->{s_kwd};
3390            #            #
3391          } else {          } else {
3392            !!!cp (1025);            !!!cp (1025);
# Line 3340  sub _get_next_token ($) { Line 3398  sub _get_next_token ($) {
3398          !!!cp (1026);          !!!cp (1026);
3399          !!!parse-error (type => 'bare ero',          !!!parse-error (type => 'bare ero',
3400                          line => $self->{line_prev},                          line => $self->{line_prev},
3401                          column => $self->{column_prev});                          column => $self->{column_prev} - length $self->{s_kwd});
3402          $data = '&' . $self->{state_keyword};          $data = '&' . $self->{s_kwd};
3403          #          #
3404        }        }
3405        
# Line 3362  sub _get_next_token ($) { Line 3420  sub _get_next_token ($) {
3420          !!!emit ({type => CHARACTER_TOKEN,          !!!emit ({type => CHARACTER_TOKEN,
3421                    data => $data,                    data => $data,
3422                    line => $self->{line_prev},                    line => $self->{line_prev},
3423                    column => $self->{column_prev} + 1 - length $self->{state_keyword},                    column => $self->{column_prev} + 1 - length $self->{s_kwd},
3424                   });                   });
3425          redo A;          redo A;
3426        } else {        } else {
3427          !!!cp (985);          !!!cp (985);
3428          $self->{current_attribute}->{value} .= $data;          $self->{ca}->{value} .= $data;
3429          $self->{current_attribute}->{has_reference} = 1 if $has_ref;          $self->{ca}->{has_reference} = 1 if $has_ref;
3430          $self->{state} = $self->{prev_state};          $self->{state} = $self->{prev_state};
3431          ## Reconsume.          ## Reconsume.
3432          redo A;          redo A;
# Line 3446  sub _tree_construction_initial ($) { Line 3504  sub _tree_construction_initial ($) {
3504        $doctype_name = '' unless defined $doctype_name;        $doctype_name = '' unless defined $doctype_name;
3505        $doctype_name =~ tr/a-z/A-Z/; # ASCII case-insensitive        $doctype_name =~ tr/a-z/A-Z/; # ASCII case-insensitive
3506        if (not defined $token->{name} or # <!DOCTYPE>        if (not defined $token->{name} or # <!DOCTYPE>
3507            defined $token->{system_identifier}) {            defined $token->{sysid}) {
3508          !!!cp ('t1');          !!!cp ('t1');
3509          !!!parse-error (type => 'not HTML5', token => $token);          !!!parse-error (type => 'not HTML5', token => $token);
3510        } elsif ($doctype_name ne 'HTML') {        } elsif ($doctype_name ne 'HTML') {
3511          !!!cp ('t2');          !!!cp ('t2');
3512          !!!parse-error (type => 'not HTML5', token => $token);          !!!parse-error (type => 'not HTML5', token => $token);
3513        } elsif (defined $token->{public_identifier}) {        } elsif (defined $token->{pubid}) {
3514          if ($token->{public_identifier} eq 'XSLT-compat') {          if ($token->{pubid} eq 'XSLT-compat') {
3515            !!!cp ('t1.2');            !!!cp ('t1.2');
3516            !!!parse-error (type => 'XSLT-compat', token => $token,            !!!parse-error (type => 'XSLT-compat', token => $token,
3517                            level => $self->{level}->{should});                            level => $self->{level}->{should});
# Line 3469  sub _tree_construction_initial ($) { Line 3527  sub _tree_construction_initial ($) {
3527          ($token->{name}); ## ISSUE: If name is missing (e.g. <!DOCTYPE>)?          ($token->{name}); ## ISSUE: If name is missing (e.g. <!DOCTYPE>)?
3528        ## NOTE: Default value for both |public_id| and |system_id| attributes        ## NOTE: Default value for both |public_id| and |system_id| attributes
3529        ## are empty strings, so that we don't set any value in missing cases.        ## are empty strings, so that we don't set any value in missing cases.
3530        $doctype->public_id ($token->{public_identifier})        $doctype->public_id ($token->{pubid}) if defined $token->{pubid};
3531            if defined $token->{public_identifier};        $doctype->system_id ($token->{sysid}) if defined $token->{sysid};
       $doctype->system_id ($token->{system_identifier})  
           if defined $token->{system_identifier};  
3532        ## NOTE: Other DocumentType attributes are null or empty lists.        ## NOTE: Other DocumentType attributes are null or empty lists.
3533        ## ISSUE: internalSubset = null??        ## ISSUE: internalSubset = null??
3534        $self->{document}->append_child ($doctype);        $self->{document}->append_child ($doctype);
# Line 3480  sub _tree_construction_initial ($) { Line 3536  sub _tree_construction_initial ($) {
3536        if ($token->{quirks} or $doctype_name ne 'HTML') {        if ($token->{quirks} or $doctype_name ne 'HTML') {
3537          !!!cp ('t4');          !!!cp ('t4');
3538          $self->{document}->manakai_compat_mode ('quirks');          $self->{document}->manakai_compat_mode ('quirks');
3539        } elsif (defined $token->{public_identifier}) {        } elsif (defined $token->{pubid}) {
3540          my $pubid = $token->{public_identifier};          my $pubid = $token->{pubid};
3541          $pubid =~ tr/a-z/A-z/;          $pubid =~ tr/a-z/A-z/;
3542          my $prefix = [          my $prefix = [
3543            "+//SILMARIL//DTD HTML PRO V0R11 19970101//",            "+//SILMARIL//DTD HTML PRO V0R11 19970101//",
# Line 3555  sub _tree_construction_initial ($) { Line 3611  sub _tree_construction_initial ($) {
3611            $self->{document}->manakai_compat_mode ('quirks');            $self->{document}->manakai_compat_mode ('quirks');
3612          } elsif ($pubid =~ m[^-//W3C//DTD HTML 4.01 FRAMESET//] or          } elsif ($pubid =~ m[^-//W3C//DTD HTML 4.01 FRAMESET//] or
3613                   $pubid =~ m[^-//W3C//DTD HTML 4.01 TRANSITIONAL//]) {                   $pubid =~ m[^-//W3C//DTD HTML 4.01 TRANSITIONAL//]) {
3614            if (defined $token->{system_identifier}) {            if (defined $token->{sysid}) {
3615              !!!cp ('t6');              !!!cp ('t6');
3616              $self->{document}->manakai_compat_mode ('quirks');              $self->{document}->manakai_compat_mode ('quirks');
3617            } else {            } else {
# Line 3572  sub _tree_construction_initial ($) { Line 3628  sub _tree_construction_initial ($) {
3628        } else {        } else {
3629          !!!cp ('t10');          !!!cp ('t10');
3630        }        }
3631        if (defined $token->{system_identifier}) {        if (defined $token->{sysid}) {
3632          my $sysid = $token->{system_identifier};          my $sysid = $token->{sysid};
3633          $sysid =~ tr/A-Z/a-z/;          $sysid =~ tr/A-Z/a-z/;
3634          if ($sysid eq "http://www.ibm.com/data/dtd/v11/ibmxhtml1-transitional.dtd") {          if ($sysid eq "http://www.ibm.com/data/dtd/v11/ibmxhtml1-transitional.dtd") {
3635            ## NOTE: Ensure that |PUBLIC "(limited quirks)" "(quirks)"| is            ## NOTE: Ensure that |PUBLIC "(limited quirks)" "(quirks)"| is
# Line 3603  sub _tree_construction_initial ($) { Line 3659  sub _tree_construction_initial ($) {
3659        !!!ack-later;        !!!ack-later;
3660        return;        return;
3661      } elsif ($token->{type} == CHARACTER_TOKEN) {      } elsif ($token->{type} == CHARACTER_TOKEN) {
3662        if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D        if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
3663          ## Ignore the token          ## Ignore the token
3664    
3665          unless (length $token->{data}) {          unless (length $token->{data}) {
# Line 3660  sub _tree_construction_root_element ($) Line 3716  sub _tree_construction_root_element ($)
3716          !!!next-token;          !!!next-token;
3717          redo B;          redo B;
3718        } elsif ($token->{type} == CHARACTER_TOKEN) {        } elsif ($token->{type} == CHARACTER_TOKEN) {
3719          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D          if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
3720            ## Ignore the token.            ## Ignore the token.
3721    
3722            unless (length $token->{data}) {            unless (length $token->{data}) {
# Line 4474  sub _tree_construction_main ($) { Line 4530  sub _tree_construction_main ($) {
4530    
4531      if ($self->{insertion_mode} & HEAD_IMS) {      if ($self->{insertion_mode} & HEAD_IMS) {
4532        if ($token->{type} == CHARACTER_TOKEN) {        if ($token->{type} == CHARACTER_TOKEN) {
4533          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {          if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
4534            unless ($self->{insertion_mode} == BEFORE_HEAD_IM) {            unless ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4535              !!!cp ('t88.2');              !!!cp ('t88.2');
4536              $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);              $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
4537                #
4538            } else {            } else {
4539              !!!cp ('t88.1');              !!!cp ('t88.1');
4540              ## Ignore the token.              ## Ignore the token.
4541              !!!next-token;              #
             next B;  
4542            }            }
4543            unless (length $token->{data}) {            unless (length $token->{data}) {
4544              !!!cp ('t88');              !!!cp ('t88');
4545              !!!next-token;              !!!next-token;
4546              next B;              next B;
4547            }            }
4548    ## TODO: set $token->{column} appropriately
4549          }          }
4550    
4551          if ($self->{insertion_mode} == BEFORE_HEAD_IM) {          if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
# Line 4652  sub _tree_construction_main ($) { Line 4709  sub _tree_construction_main ($) {
4709                  } elsif ($token->{attributes}->{content}) {                  } elsif ($token->{attributes}->{content}) {
4710                    if ($token->{attributes}->{content}->{value}                    if ($token->{attributes}->{content}->{value}
4711                        =~ /[Cc][Hh][Aa][Rr][Ss][Ee][Tt]                        =~ /[Cc][Hh][Aa][Rr][Ss][Ee][Tt]
4712                            [\x09-\x0D\x20]*=                            [\x09\x0A\x0C\x0D\x20]*=
4713                            [\x09-\x0D\x20]*(?>"([^"]*)"|'([^']*)'|                            [\x09\x0A\x0C\x0D\x20]*(?>"([^"]*)"|'([^']*)'|
4714                            ([^"'\x09-\x0D\x20][^\x09-\x0D\x20\x3B]*))/x) {                            ([^"'\x09\x0A\x0C\x0D\x20]
4715                               [^\x09\x0A\x0C\x0D\x20\x3B]*))/x) {
4716                      !!!cp ('t107');                      !!!cp ('t107');
4717                      ## NOTE: Whether the encoding is supported or not is handled                      ## NOTE: Whether the encoding is supported or not is handled
4718                      ## in the {change_encoding} callback.                      ## in the {change_encoding} callback.
# Line 5463  sub _tree_construction_main ($) { Line 5521  sub _tree_construction_main ($) {
5521      } elsif ($self->{insertion_mode} & TABLE_IMS) {      } elsif ($self->{insertion_mode} & TABLE_IMS) {
5522        if ($token->{type} == CHARACTER_TOKEN) {        if ($token->{type} == CHARACTER_TOKEN) {
5523          if (not $open_tables->[-1]->[1] and # tainted          if (not $open_tables->[-1]->[1] and # tainted
5524              $token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {              $token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
5525            $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);            $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
5526                                
5527            unless (length $token->{data}) {            unless (length $token->{data}) {
# Line 6147  sub _tree_construction_main ($) { Line 6205  sub _tree_construction_main ($) {
6205        }        }
6206      } elsif ($self->{insertion_mode} == IN_COLUMN_GROUP_IM) {      } elsif ($self->{insertion_mode} == IN_COLUMN_GROUP_IM) {
6207            if ($token->{type} == CHARACTER_TOKEN) {            if ($token->{type} == CHARACTER_TOKEN) {
6208              if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {              if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
6209                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
6210                unless (length $token->{data}) {                unless (length $token->{data}) {
6211                  !!!cp ('t260');                  !!!cp ('t260');
# Line 6488  sub _tree_construction_main ($) { Line 6546  sub _tree_construction_main ($) {
6546        }        }
6547      } elsif ($self->{insertion_mode} & BODY_AFTER_IMS) {      } elsif ($self->{insertion_mode} & BODY_AFTER_IMS) {
6548        if ($token->{type} == CHARACTER_TOKEN) {        if ($token->{type} == CHARACTER_TOKEN) {
6549          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {          if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
6550            my $data = $1;            my $data = $1;
6551            ## As if in body            ## As if in body
6552            $reconstruct_active_formatting_elements->($insert_to_current);            $reconstruct_active_formatting_elements->($insert_to_current);
# Line 6505  sub _tree_construction_main ($) { Line 6563  sub _tree_construction_main ($) {
6563          if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {          if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
6564            !!!cp ('t301');            !!!cp ('t301');
6565            !!!parse-error (type => 'after html:#text', token => $token);            !!!parse-error (type => 'after html:#text', token => $token);
6566              #
           ## Reprocess in the "after body" insertion mode.  
6567          } else {          } else {
6568            !!!cp ('t302');            !!!cp ('t302');
6569              ## "after body" insertion mode
6570              !!!parse-error (type => 'after body:#text', token => $token);
6571              #
6572          }          }
           
         ## "after body" insertion mode  
         !!!parse-error (type => 'after body:#text', token => $token);  
6573    
6574          $self->{insertion_mode} = IN_BODY_IM;          $self->{insertion_mode} = IN_BODY_IM;
6575          ## reprocess          ## reprocess
# Line 6522  sub _tree_construction_main ($) { Line 6579  sub _tree_construction_main ($) {
6579            !!!cp ('t303');            !!!cp ('t303');
6580            !!!parse-error (type => 'after html',            !!!parse-error (type => 'after html',
6581                            text => $token->{tag_name}, token => $token);                            text => $token->{tag_name}, token => $token);
6582                        #
           ## Reprocess in the "after body" insertion mode.  
6583          } else {          } else {
6584            !!!cp ('t304');            !!!cp ('t304');
6585              ## "after body" insertion mode
6586              !!!parse-error (type => 'after body',
6587                              text => $token->{tag_name}, token => $token);
6588              #
6589          }          }
6590    
         ## "after body" insertion mode  
         !!!parse-error (type => 'after body',  
                         text => $token->{tag_name}, token => $token);  
   
6591          $self->{insertion_mode} = IN_BODY_IM;          $self->{insertion_mode} = IN_BODY_IM;
6592          !!!ack-later;          !!!ack-later;
6593          ## reprocess          ## reprocess
# Line 6542  sub _tree_construction_main ($) { Line 6598  sub _tree_construction_main ($) {
6598            !!!parse-error (type => 'after html:/',            !!!parse-error (type => 'after html:/',
6599                            text => $token->{tag_name}, token => $token);                            text => $token->{tag_name}, token => $token);
6600                        
6601            $self->{insertion_mode} = AFTER_BODY_IM;            $self->{insertion_mode} = IN_BODY_IM;
6602            ## Reprocess in the "after body" insertion mode.            ## Reprocess.
6603              next B;
6604          } else {          } else {
6605            !!!cp ('t306');            !!!cp ('t306');
6606          }          }
# Line 6581  sub _tree_construction_main ($) { Line 6638  sub _tree_construction_main ($) {
6638        }        }
6639      } elsif ($self->{insertion_mode} & FRAME_IMS) {      } elsif ($self->{insertion_mode} & FRAME_IMS) {
6640        if ($token->{type} == CHARACTER_TOKEN) {        if ($token->{type} == CHARACTER_TOKEN) {
6641          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {          if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
6642            $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);            $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
6643                        
6644            unless (length $token->{data}) {            unless (length $token->{data}) {
# Line 6591  sub _tree_construction_main ($) { Line 6648  sub _tree_construction_main ($) {
6648            }            }
6649          }          }
6650                    
6651          if ($token->{data} =~ s/^[^\x09\x0A\x0B\x0C\x20]+//) {          if ($token->{data} =~ s/^[^\x09\x0A\x0C\x20]+//) {
6652            if ($self->{insertion_mode} == IN_FRAMESET_IM) {            if ($self->{insertion_mode} == IN_FRAMESET_IM) {
6653              !!!cp ('t311');              !!!cp ('t311');
6654              !!!parse-error (type => 'in frameset:#text', token => $token);              !!!parse-error (type => 'in frameset:#text', token => $token);
# Line 6768  sub _tree_construction_main ($) { Line 6825  sub _tree_construction_main ($) {
6825            } elsif ($token->{attributes}->{content}) {            } elsif ($token->{attributes}->{content}) {
6826              if ($token->{attributes}->{content}->{value}              if ($token->{attributes}->{content}->{value}
6827                  =~ /[Cc][Hh][Aa][Rr][Ss][Ee][Tt]                  =~ /[Cc][Hh][Aa][Rr][Ss][Ee][Tt]
6828                      [\x09-\x0D\x20]*=                      [\x09\x0A\x0C\x0D\x20]*=
6829                      [\x09-\x0D\x20]*(?>"([^"]*)"|'([^']*)'|                      [\x09\x0A\x0C\x0D\x20]*(?>"([^"]*)"|'([^']*)'|
6830                      ([^"'\x09-\x0D\x20][^\x09-\x0D\x20\x3B]*))/x) {                      ([^"'\x09\x0A\x0C\x0D\x20][^\x09\x0A\x0C\x0D\x20\x3B]*))
6831                       /x) {
6832                !!!cp ('t336');                !!!cp ('t336');
6833                ## NOTE: Whether the encoding is supported or not is handled                ## NOTE: Whether the encoding is supported or not is handled
6834                ## in the {change_encoding} callback.                ## in the {change_encoding} callback.
# Line 6915  sub _tree_construction_main ($) { Line 6973  sub _tree_construction_main ($) {
6973              last INSCOPE;              last INSCOPE;
6974            }            }
6975          } # INSCOPE          } # INSCOPE
6976    
6977            ## NOTE: Special, Scope (<li><foo><li> == <li><foo><li/></foo></li>)
6978              ## Interpreted as <li><foo/></li><li/> (non-conforming)
6979              ## blockquote (O9.27), center (O), dd (Fx3, O, S3.1.2, IE7),
6980              ## dt (Fx, O, S, IE), dl (O), fieldset (O, S, IE), form (Fx, O, S),
6981              ## hn (O), pre (O), applet (O, S), button (O, S), marquee (Fx, O, S),
6982              ## object (Fx)
6983              ## Generate non-tree (non-conforming)
6984              ## basefont (IE7 (where basefont is non-void)), center (IE),
6985              ## form (IE), hn (IE)
6986            ## address, div, p (<li><foo><li> == <li><foo/></li><li/>)
6987              ## Interpreted as <li><foo><li/></foo></li> (non-conforming)
6988              ## div (Fx, S)
6989                        
6990          ## Step 1          ## Step 1
6991          my $i = -1;          my $i = -1;
# Line 7295  sub _tree_construction_main ($) { Line 7366  sub _tree_construction_main ($) {
7366            !!!nack ('t380.1');            !!!nack ('t380.1');
7367          } elsif ({          } elsif ({
7368                    b => 1, big => 1, em => 1, font => 1, i => 1,                    b => 1, big => 1, em => 1, font => 1, i => 1,
7369                    s => 1, small => 1, strile => 1,                    s => 1, small => 1, strike => 1,
7370                    strong => 1, tt => 1, u => 1,                    strong => 1, tt => 1, u => 1,
7371                   }->{$token->{tag_name}}) {                   }->{$token->{tag_name}}) {
7372            !!!cp ('t375');            !!!cp ('t375');
# Line 7600  sub _tree_construction_main ($) { Line 7671  sub _tree_construction_main ($) {
7671        } elsif ({        } elsif ({
7672                  a => 1,                  a => 1,
7673                  b => 1, big => 1, em => 1, font => 1, i => 1,                  b => 1, big => 1, em => 1, font => 1, i => 1,
7674                  nobr => 1, s => 1, small => 1, strile => 1,                  nobr => 1, s => 1, small => 1, strike => 1,
7675                  strong => 1, tt => 1, u => 1,                  strong => 1, tt => 1, u => 1,
7676                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
7677          !!!cp ('t427');          !!!cp ('t427');
# Line 7691  sub _tree_construction_main ($) { Line 7762  sub _tree_construction_main ($) {
7762                ## Ignore the token                ## Ignore the token
7763                !!!next-token;                !!!next-token;
7764                last S2;                last S2;
             }  
7765    
7766                  ## NOTE: |<span><dd></span>a|: In Safari 3.1.2 and Opera
7767                  ## 9.27, "a" is a child of <dd> (conforming).  In
7768                  ## Firefox 3.0.2, "a" is a child of <body>.  In WinIE 7,
7769                  ## "a" is a child of both <body> and <dd>.
7770                }
7771                
7772              !!!cp ('t434');              !!!cp ('t434');
7773            }            }
7774                        
# Line 7733  sub _tree_construction_main ($) { Line 7809  sub _tree_construction_main ($) {
7809    ## TODO: script stuffs    ## TODO: script stuffs
7810  } # _tree_construct_main  } # _tree_construct_main
7811    
7812  sub set_inner_html ($$$;$) {  sub set_inner_html ($$$$;$) {
7813    my $class = shift;    my $class = shift;
7814    my $node = shift;    my $node = shift;
7815    my $s = \$_[0];    #my $s = \$_[0];
7816    my $onerror = $_[1];    my $onerror = $_[1];
7817    my $get_wrapper = $_[2] || sub ($) { return $_[0] };    my $get_wrapper = $_[2] || sub ($) { return $_[0] };
7818    
# Line 7757  sub set_inner_html ($$$;$) { Line 7833  sub set_inner_html ($$$;$) {
7833      }      }
7834    
7835      ## Step 3, 4, 5 # MUST      ## Step 3, 4, 5 # MUST
7836      $class->parse_char_string ($$s => $node, $onerror, $get_wrapper);      $class->parse_char_string ($_[0] => $node, $onerror, $get_wrapper);
7837    } elsif ($nt == 1) {    } elsif ($nt == 1) {
7838      ## TODO: If non-html element      ## TODO: If non-html element
7839    
# Line 7776  sub set_inner_html ($$$;$) { Line 7852  sub set_inner_html ($$$;$) {
7852      my $i = 0;      my $i = 0;
7853      $p->{line_prev} = $p->{line} = 1;      $p->{line_prev} = $p->{line} = 1;
7854      $p->{column_prev} = $p->{column} = 0;      $p->{column_prev} = $p->{column} = 0;
7855      $p->{set_next_char} = sub {      require Whatpm::Charset::DecodeHandle;
7856        my $input = Whatpm::Charset::DecodeHandle::CharString->new (\($_[0]));
7857        $input = $get_wrapper->($input);
7858        $p->{set_nc} = sub {
7859        my $self = shift;        my $self = shift;
7860    
7861        pop @{$self->{prev_char}};        my $char = '';
7862        unshift @{$self->{prev_char}}, $self->{next_char};        if (defined $self->{next_nc}) {
7863            $char = $self->{next_nc};
7864        $self->{next_char} = -1 and return if $i >= length $$s;          delete $self->{next_nc};
7865        $self->{next_char} = ord substr $$s, $i++, 1;          $self->{nc} = ord $char;
7866          } else {
7867            $self->{char_buffer} = '';
7868            $self->{char_buffer_pos} = 0;
7869            
7870            my $count = $input->manakai_read_until
7871                ($self->{char_buffer}, qr/[^\x00\x0A\x0D]/,
7872                 $self->{char_buffer_pos});
7873            if ($count) {
7874              $self->{line_prev} = $self->{line};
7875              $self->{column_prev} = $self->{column};
7876              $self->{column}++;
7877              $self->{nc}
7878                  = ord substr ($self->{char_buffer},
7879                                $self->{char_buffer_pos}++, 1);
7880              return;
7881            }
7882            
7883            if ($input->read ($char, 1)) {
7884              $self->{nc} = ord $char;
7885            } else {
7886              $self->{nc} = -1;
7887              return;
7888            }
7889          }
7890    
7891        ($p->{line_prev}, $p->{column_prev}) = ($p->{line}, $p->{column});        ($p->{line_prev}, $p->{column_prev}) = ($p->{line}, $p->{column});
7892        $p->{column}++;        $p->{column}++;
7893    
7894        if ($self->{next_char} == 0x000A) { # LF        if ($self->{nc} == 0x000A) { # LF
7895          $p->{line}++;          $p->{line}++;
7896          $p->{column} = 0;          $p->{column} = 0;
7897          !!!cp ('i1');          !!!cp ('i1');
7898        } elsif ($self->{next_char} == 0x000D) { # CR        } elsif ($self->{nc} == 0x000D) { # CR
7899          $i++ if substr ($$s, $i, 1) eq "\x0A";  ## TODO: support for abort/streaming
7900          $self->{next_char} = 0x000A; # LF # MUST          my $next = '';
7901            if ($input->read ($next, 1) and $next ne "\x0A") {
7902              $self->{next_nc} = $next;
7903            }
7904            $self->{nc} = 0x000A; # LF # MUST
7905          $p->{line}++;          $p->{line}++;
7906          $p->{column} = 0;          $p->{column} = 0;
7907          !!!cp ('i2');          !!!cp ('i2');
7908        } elsif ($self->{next_char} > 0x10FFFF) {        } elsif ($self->{nc} == 0x0000) { # NULL
         $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST  
         !!!cp ('i3');  
       } elsif ($self->{next_char} == 0x0000) { # NULL  
7909          !!!cp ('i4');          !!!cp ('i4');
7910          !!!parse-error (type => 'NULL');          !!!parse-error (type => 'NULL');
7911          $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST          $self->{nc} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
       } elsif ($self->{next_char} <= 0x0008 or  
                (0x000E <= $self->{next_char} and  
                 $self->{next_char} <= 0x001F) or  
                (0x007F <= $self->{next_char} and  
                 $self->{next_char} <= 0x009F) or  
                (0xD800 <= $self->{next_char} and  
                 $self->{next_char} <= 0xDFFF) or  
                (0xFDD0 <= $self->{next_char} and  
                 $self->{next_char} <= 0xFDDF) or  
                {  
                 0xFFFE => 1, 0xFFFF => 1, 0x1FFFE => 1, 0x1FFFF => 1,  
                 0x2FFFE => 1, 0x2FFFF => 1, 0x3FFFE => 1, 0x3FFFF => 1,  
                 0x4FFFE => 1, 0x4FFFF => 1, 0x5FFFE => 1, 0x5FFFF => 1,  
                 0x6FFFE => 1, 0x6FFFF => 1, 0x7FFFE => 1, 0x7FFFF => 1,  
                 0x8FFFE => 1, 0x8FFFF => 1, 0x9FFFE => 1, 0x9FFFF => 1,  
                 0xAFFFE => 1, 0xAFFFF => 1, 0xBFFFE => 1, 0xBFFFF => 1,  
                 0xCFFFE => 1, 0xCFFFF => 1, 0xDFFFE => 1, 0xDFFFF => 1,  
                 0xEFFFE => 1, 0xEFFFF => 1, 0xFFFFE => 1, 0xFFFFF => 1,  
                 0x10FFFE => 1, 0x10FFFF => 1,  
                }->{$self->{next_char}}) {  
         !!!cp ('i4.1');  
         if ($self->{next_char} < 0x10000) {  
           !!!parse-error (type => 'control char',  
                           text => (sprintf 'U+%04X', $self->{next_char}));  
         } else {  
           !!!parse-error (type => 'control char',  
                           text => (sprintf 'U-%08X', $self->{next_char}));  
         }  
7912        }        }
7913      };      };
     $p->{prev_char} = [-1, -1, -1];  
     $p->{next_char} = -1;  
7914    
7915      $p->{read_until} = sub {      $p->{read_until} = sub {
7916        ## TODO: ...        #my ($scalar, $specials_range, $offset) = @_;
7917        return 0;        return 0 if defined $p->{next_nc};
7918      }; # $p->{read_until};  
7919          my $pattern = qr/[^$_[1]\x00\x0A\x0D]/;
7920          my $offset = $_[2] || 0;
7921          
7922          if ($p->{char_buffer_pos} < length $p->{char_buffer}) {
7923            pos ($p->{char_buffer}) = $p->{char_buffer_pos};
7924            if ($p->{char_buffer} =~ /\G(?>$pattern)+/) {
7925              substr ($_[0], $offset)
7926                  = substr ($p->{char_buffer}, $-[0], $+[0] - $-[0]);
7927              my $count = $+[0] - $-[0];
7928              if ($count) {
7929                $p->{column} += $count;
7930                $p->{char_buffer_pos} += $count;
7931                $p->{line_prev} = $p->{line};
7932                $p->{column_prev} = $p->{column} - 1;
7933                $p->{nc} = -1;
7934              }
7935              return $count;
7936            } else {
7937              return 0;
7938            }
7939          } else {
7940            my $count = $input->manakai_read_until ($_[0], $pattern, $_[2]);
7941            if ($count) {
7942              $p->{column} += $count;
7943              $p->{column_prev} += $count;
7944              $p->{nc} = -1;
7945            }
7946            return $count;
7947          }
7948        }; # $p->{read_until}
7949    
7950      my $ponerror = $onerror || sub {      my $ponerror = $onerror || sub {
7951        my (%opt) = @_;        my (%opt) = @_;
# Line 7857  sub set_inner_html ($$$;$) { Line 7961  sub set_inner_html ($$$;$) {
7961        $ponerror->(line => $p->{line}, column => $p->{column}, @_);        $ponerror->(line => $p->{line}, column => $p->{column}, @_);
7962      };      };
7963            
7964        my $char_onerror = sub {
7965          my (undef, $type, %opt) = @_;
7966          $ponerror->(layer => 'encode',
7967                      line => $p->{line}, column => $p->{column} + 1,
7968                      %opt, type => $type);
7969        }; # $char_onerror
7970        $input->onerror ($char_onerror);
7971    
7972      $p->_initialize_tokenizer;      $p->_initialize_tokenizer;
7973      $p->_initialize_tree_constructor;      $p->_initialize_tree_constructor;
7974    

Legend:
Removed from v.1.173  
changed lines
  Added in v.1.193

[email protected]
ViewVC Help
Powered by ViewVC 1.1.24