/[suikacvs]/markup/html/whatpm/Whatpm/HTML.pm.src
Suika

Diff of /markup/html/whatpm/Whatpm/HTML.pm.src

Parent Directory Parent Directory | Revision Log Revision Log | View Patch Patch

revision 1.179 by wakaba, Sun Sep 14 13:09:01 2008 UTC revision 1.192 by wakaba, Thu Oct 2 10:59:04 2008 UTC
# Line 3  use strict; Line 3  use strict;
3  our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r};  our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r};
4  use Error qw(:try);  use Error qw(:try);
5    
6    ## NOTE: This module don't check all HTML5 parse errors; character
7    ## encoding related parse errors are expected to be handled by relevant
8    ## modules.
9    ## Parse errors for control characters that are not allowed in HTML5
10    ## documents, for surrogate code points, and for noncharacter code
11    ## points, as well as U+FFFD substitions for characters whose code points
12    ## is higher than U+10FFFF may be detected by combining the parser with
13    ## the checker implemented by Whatpm::Charset::UnicodeChecker (for its
14    ## usage example, see |t/HTML-tree.t| in the Whatpm package or the
15    ## WebHACC::Language::HTML module in the WebHACC package).
16    
17  ## ISSUE:  ## ISSUE:
18  ## var doc = implementation.createDocument (null, null, null);  ## var doc = implementation.createDocument (null, null, null);
19  ## doc.write ('');  ## doc.write ('');
# Line 312  my $foreign_attr_xname = { Line 323  my $foreign_attr_xname = {
323    
324  ## ISSUE: xmlns:xlink="non-xlink-ns" is not an error.  ## ISSUE: xmlns:xlink="non-xlink-ns" is not an error.
325    
326  my $c1_entity_char = {  my $charref_map = {
327      0x0D => 0x000A,
328    0x80 => 0x20AC,    0x80 => 0x20AC,
329    0x81 => 0xFFFD,    0x81 => 0xFFFD,
330    0x82 => 0x201A,    0x82 => 0x201A,
# Line 345  my $c1_entity_char = { Line 357  my $c1_entity_char = {
357    0x9D => 0xFFFD,    0x9D => 0xFFFD,
358    0x9E => 0x017E,    0x9E => 0x017E,
359    0x9F => 0x0178,    0x9F => 0x0178,
360  }; # $c1_entity_char  }; # $charref_map
361    $charref_map->{$_} = 0xFFFD
362        for 0x0000..0x0008, 0x000B, 0x000E..0x001F, 0x007F,
363            0xD800..0xDFFF, 0xFDD0..0xFDDF, ## ISSUE: 0xFDEF
364            0xFFFE, 0xFFFF, 0x1FFFE, 0x1FFFF, 0x2FFFE, 0x2FFFF, 0x3FFFE, 0x3FFFF,
365            0x4FFFE, 0x4FFFF, 0x5FFFE, 0x5FFFF, 0x6FFFE, 0x6FFFF, 0x7FFFE,
366            0x7FFFF, 0x8FFFE, 0x8FFFF, 0x9FFFE, 0x9FFFF, 0xAFFFE, 0xAFFFF,
367            0xBFFFE, 0xBFFFF, 0xCFFFE, 0xCFFFF, 0xDFFFE, 0xDFFFF, 0xEFFFE,
368            0xEFFFF, 0xFFFFE, 0xFFFFF, 0x10FFFE, 0x10FFFF;
369    
370    ## TODO: Invoke the reset algorithm when a resettable element is
371    ## created (cf. HTML5 revision 2259).
372    
373  sub parse_byte_string ($$$$;$) {  sub parse_byte_string ($$$$;$) {
374    my $self = shift;    my $self = shift;
# Line 390  sub parse_byte_stream ($$$$;$$) { Line 413  sub parse_byte_stream ($$$$;$$) {
413            ## TODO: Is this ok?  Transfer protocol's parameter should be            ## TODO: Is this ok?  Transfer protocol's parameter should be
414            ## interpreted in its semantics?            ## interpreted in its semantics?
415    
       ## ISSUE: Unsupported encoding is not ignored according to the spec.  
416        ($char_stream, $e_status) = $charset->get_decode_handle        ($char_stream, $e_status) = $charset->get_decode_handle
417            ($byte_stream, allow_error_reporting => 1,            ($byte_stream, allow_error_reporting => 1,
418             allow_fallback => 1);             allow_fallback => 1);
# Line 398  sub parse_byte_stream ($$$$;$$) { Line 420  sub parse_byte_stream ($$$$;$$) {
420          $self->{confident} = 1;          $self->{confident} = 1;
421          last SNIFFING;          last SNIFFING;
422        } else {        } else {
423          ## TODO: unsupported error          !!!parse-error (type => 'charset:not supported',
424                            layer => 'encode',
425                            line => 1, column => 1,
426                            value => $charset_name,
427                            level => $self->{level}->{uncertain});
428        }        }
429      }      }
430    
# Line 571  sub parse_byte_stream ($$$$;$$) { Line 597  sub parse_byte_stream ($$$$;$$) {
597    my $wrapped_char_stream = $get_wrapper->($char_stream);    my $wrapped_char_stream = $get_wrapper->($char_stream);
598    $wrapped_char_stream->onerror ($char_onerror);    $wrapped_char_stream->onerror ($char_onerror);
599    
600    my @args = @_; shift @args; # $s    my @args = ($_[1], $_[2]); # $doc, $onerror - $get_wrapper = undef;
601    my $return;    my $return;
602    try {    try {
603      $return = $self->parse_char_stream ($wrapped_char_stream, @args);        $return = $self->parse_char_stream ($wrapped_char_stream, @args);  
# Line 621  sub parse_char_string ($$$;$$) { Line 647  sub parse_char_string ($$$;$$) {
647    my $s = ref $_[0] ? $_[0] : \($_[0]);    my $s = ref $_[0] ? $_[0] : \($_[0]);
648    require Whatpm::Charset::DecodeHandle;    require Whatpm::Charset::DecodeHandle;
649    my $input = Whatpm::Charset::DecodeHandle::CharString->new ($s);    my $input = Whatpm::Charset::DecodeHandle::CharString->new ($s);
   if ($_[3]) {  
     $input = $_[3]->($input);  
   }  
650    return $self->parse_char_stream ($input, @_[1..$#_]);    return $self->parse_char_stream ($input, @_[1..$#_]);
651  } # parse_char_string  } # parse_char_string
652  *parse_string = \&parse_char_string; ## NOTE: Alias for backward compatibility.  *parse_string = \&parse_char_string; ## NOTE: Alias for backward compatibility.
653    
654  sub parse_char_stream ($$$;$) {  sub parse_char_stream ($$$;$$) {
655    my $self = ref $_[0] ? shift : shift->new;    my $self = ref $_[0] ? shift : shift->new;
656    my $input = $_[0];    my $input = $_[0];
657    $self->{document} = $_[1];    $self->{document} = $_[1];
# Line 641  sub parse_char_stream ($$$;$) { Line 664  sub parse_char_stream ($$$;$) {
664        if defined $self->{input_encoding};        if defined $self->{input_encoding};
665  ## TODO: |{input_encoding}| is needless?  ## TODO: |{input_encoding}| is needless?
666    
   my $i = 0;  
667    $self->{line_prev} = $self->{line} = 1;    $self->{line_prev} = $self->{line} = 1;
668    $self->{column_prev} = -1;    $self->{column_prev} = -1;
669    $self->{column} = 0;    $self->{column} = 0;
670    $self->{set_next_char} = sub {    $self->{set_nc} = sub {
671      my $self = shift;      my $self = shift;
672    
673      my $char = '';      my $char = '';
674      if (defined $self->{next_next_char}) {      if (defined $self->{next_nc}) {
675        $char = $self->{next_next_char};        $char = $self->{next_nc};
676        delete $self->{next_next_char};        delete $self->{next_nc};
677        $self->{next_char} = ord $char;        $self->{nc} = ord $char;
678      } else {      } else {
679        $self->{char_buffer} = '';        $self->{char_buffer} = '';
680        $self->{char_buffer_pos} = 0;        $self->{char_buffer_pos} = 0;
681    
682        my $count = $input->manakai_read_until        my $count = $input->manakai_read_until
683           ($self->{char_buffer},           ($self->{char_buffer}, qr/[^\x00\x0A\x0D]/, $self->{char_buffer_pos});
           qr/(?!\x{FDD0}-\x{FDDF}\x{FFFE}\x{FFFF}\x{1FFFE}\x{1FFFF}\x{2FFFE}\x{2FFFF}\x{3FFFE}\x{3FFFF}\x{4FFFE}\x{4FFFF}\x{5FFFE}\x{5FFFF}\x{6FFFE}\x{6FFFF}\x{7FFFE}\x{7FFFF}\x{8FFFE}\x{8FFFF}\x{9FFFE}\x{9FFFF}\x{AFFFE}\x{AFFFF}\x{BFFFE}\x{BFFFF}\x{CFFFE}\x{CFFFF}\x{DFFFE}\x{DFFFF}\x{EFFFE}\x{EFFFF}\x{FFFFE}\x{FFFFF}])[\x20-\x7E\xA0-\x{D7FF}\x{E000}-\x{10FFFD}]/,  
           $self->{char_buffer_pos});  
684        if ($count) {        if ($count) {
685          $self->{line_prev} = $self->{line};          $self->{line_prev} = $self->{line};
686          $self->{column_prev} = $self->{column};          $self->{column_prev} = $self->{column};
687          $self->{column}++;          $self->{column}++;
688          $self->{next_char}          $self->{nc}
689              = ord substr ($self->{char_buffer}, $self->{char_buffer_pos}++, 1);              = ord substr ($self->{char_buffer}, $self->{char_buffer_pos}++, 1);
690          return;          return;
691        }        }
692    
693        if ($input->read ($char, 1)) {        if ($input->read ($char, 1)) {
694          $self->{next_char} = ord $char;          $self->{nc} = ord $char;
695        } else {        } else {
696          $self->{next_char} = -1;          $self->{nc} = -1;
697          return;          return;
698        }        }
699      }      }
# Line 682  sub parse_char_stream ($$$;$) { Line 702  sub parse_char_stream ($$$;$) {
702          = ($self->{line}, $self->{column});          = ($self->{line}, $self->{column});
703      $self->{column}++;      $self->{column}++;
704            
705      if ($self->{next_char} == 0x000A) { # LF      if ($self->{nc} == 0x000A) { # LF
706        !!!cp ('j1');        !!!cp ('j1');
707        $self->{line}++;        $self->{line}++;
708        $self->{column} = 0;        $self->{column} = 0;
709      } elsif ($self->{next_char} == 0x000D) { # CR      } elsif ($self->{nc} == 0x000D) { # CR
710        !!!cp ('j2');        !!!cp ('j2');
711  ## TODO: support for abort/streaming  ## TODO: support for abort/streaming
712        my $next = '';        my $next = '';
713        if ($input->read ($next, 1) and $next ne "\x0A") {        if ($input->read ($next, 1) and $next ne "\x0A") {
714          $self->{next_next_char} = $next;          $self->{next_nc} = $next;
715        }        }
716        $self->{next_char} = 0x000A; # LF # MUST        $self->{nc} = 0x000A; # LF # MUST
717        $self->{line}++;        $self->{line}++;
718        $self->{column} = 0;        $self->{column} = 0;
719      } elsif ($self->{next_char} > 0x10FFFF) {      } elsif ($self->{nc} == 0x0000) { # NULL
       !!!cp ('j3');  
       $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST  
     } elsif ($self->{next_char} == 0x0000) { # NULL  
720        !!!cp ('j4');        !!!cp ('j4');
721        !!!parse-error (type => 'NULL');        !!!parse-error (type => 'NULL');
722        $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST        $self->{nc} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
     } elsif ($self->{next_char} <= 0x0008 or  
              (0x000E <= $self->{next_char} and $self->{next_char} <= 0x001F) or  
              (0x007F <= $self->{next_char} and $self->{next_char} <= 0x009F) or  
              (0xD800 <= $self->{next_char} and $self->{next_char} <= 0xDFFF) or  
              (0xFDD0 <= $self->{next_char} and $self->{next_char} <= 0xFDDF) or  
 ## ISSUE: U+FDE0-U+FDEF are not excluded  
              {  
               0xFFFE => 1, 0xFFFF => 1, 0x1FFFE => 1, 0x1FFFF => 1,  
               0x2FFFE => 1, 0x2FFFF => 1, 0x3FFFE => 1, 0x3FFFF => 1,  
               0x4FFFE => 1, 0x4FFFF => 1, 0x5FFFE => 1, 0x5FFFF => 1,  
               0x6FFFE => 1, 0x6FFFF => 1, 0x7FFFE => 1, 0x7FFFF => 1,  
               0x8FFFE => 1, 0x8FFFF => 1, 0x9FFFE => 1, 0x9FFFF => 1,  
               0xAFFFE => 1, 0xAFFFF => 1, 0xBFFFE => 1, 0xBFFFF => 1,  
               0xCFFFE => 1, 0xCFFFF => 1, 0xDFFFE => 1, 0xDFFFF => 1,  
               0xEFFFE => 1, 0xEFFFF => 1, 0xFFFFE => 1, 0xFFFFF => 1,  
               0x10FFFE => 1, 0x10FFFF => 1,  
              }->{$self->{next_char}}) {  
       !!!cp ('j5');  
       if ($self->{next_char} < 0x10000) {  
         !!!parse-error (type => 'control char',  
                         text => (sprintf 'U+%04X', $self->{next_char}));  
       } else {  
         !!!parse-error (type => 'control char',  
                         text => (sprintf 'U-%08X', $self->{next_char}));  
       }  
723      }      }
724    };    };
   $self->{prev_char} = [-1, -1, -1];  
   $self->{next_char} = -1;  
725    
726    $self->{read_until} = sub {    $self->{read_until} = sub {
727      #my ($scalar, $specials_range, $offset) = @_;      #my ($scalar, $specials_range, $offset) = @_;
728      my $specials_range = $_[1];      return 0 if defined $self->{next_nc};
729      return 0 if defined $self->{next_next_char};  
730      my $count = $input->manakai_read_until      my $pattern = qr/[^$_[1]\x00\x0A\x0D]/;
731         ($_[0],      my $offset = $_[2] || 0;
732          qr/(?![$specials_range\x{FDD0}-\x{FDDF}\x{FFFE}\x{FFFF}\x{1FFFE}\x{1FFFF}\x{2FFFE}\x{2FFFF}\x{3FFFE}\x{3FFFF}\x{4FFFE}\x{4FFFF}\x{5FFFE}\x{5FFFF}\x{6FFFE}\x{6FFFF}\x{7FFFE}\x{7FFFF}\x{8FFFE}\x{8FFFF}\x{9FFFE}\x{9FFFF}\x{AFFFE}\x{AFFFF}\x{BFFFE}\x{BFFFF}\x{CFFFE}\x{CFFFF}\x{DFFFE}\x{DFFFF}\x{EFFFE}\x{EFFFF}\x{FFFFE}\x{FFFFF}])[\x20-\x7E\xA0-\x{D7FF}\x{E000}-\x{10FFFD}]/,  
733          $_[2]);      if ($self->{char_buffer_pos} < length $self->{char_buffer}) {
734      if ($count) {        pos ($self->{char_buffer}) = $self->{char_buffer_pos};
735        $self->{column} += $count;        if ($self->{char_buffer} =~ /\G(?>$pattern)+/) {
736        $self->{column_prev} += $count;          substr ($_[0], $offset)
737        $self->{prev_char} = [-1, -1, -1];              = substr ($self->{char_buffer}, $-[0], $+[0] - $-[0]);
738        $self->{next_char} = -1;          my $count = $+[0] - $-[0];
739            if ($count) {
740              $self->{column} += $count;
741              $self->{char_buffer_pos} += $count;
742              $self->{line_prev} = $self->{line};
743              $self->{column_prev} = $self->{column} - 1;
744              $self->{nc} = -1;
745            }
746            return $count;
747          } else {
748            return 0;
749          }
750        } else {
751          my $count = $input->manakai_read_until ($_[0], $pattern, $_[2]);
752          if ($count) {
753            $self->{column} += $count;
754            $self->{line_prev} = $self->{line};
755            $self->{column_prev} = $self->{column} - 1;
756            $self->{nc} = -1;
757          }
758          return $count;
759      }      }
     return $count;  
760    }; # $self->{read_until}    }; # $self->{read_until}
 $self->{read_until}=sub{0};  
761    
762    my $onerror = $_[2] || sub {    my $onerror = $_[2] || sub {
763      my (%opt) = @_;      my (%opt) = @_;
# Line 761  $self->{read_until}=sub{0}; Line 769  $self->{read_until}=sub{0};
769      $onerror->(line => $self->{line}, column => $self->{column}, @_);      $onerror->(line => $self->{line}, column => $self->{column}, @_);
770    };    };
771    
772      my $char_onerror = sub {
773        my (undef, $type, %opt) = @_;
774        !!!parse-error (layer => 'encode',
775                        line => $self->{line}, column => $self->{column} + 1,
776                        %opt, type => $type);
777      }; # $char_onerror
778    
779      if ($_[3]) {
780        $input = $_[3]->($input);
781        $input->onerror ($char_onerror);
782      } else {
783        $input->onerror ($char_onerror) unless defined $input->onerror;
784      }
785    
786    $self->_initialize_tokenizer;    $self->_initialize_tokenizer;
787    $self->_initialize_tree_constructor;    $self->_initialize_tree_constructor;
788    $self->_construct_tree;    $self->_construct_tree;
# Line 780  sub new ($) { Line 802  sub new ($) {
802                info => 'i',                info => 'i',
803                uncertain => 'u'},                uncertain => 'u'},
804    }, $class;    }, $class;
805    $self->{set_next_char} = sub {    $self->{set_nc} = sub {
806      $self->{next_char} = -1;      $self->{nc} = -1;
807    };    };
808    $self->{parse_error} = sub {    $self->{parse_error} = sub {
809      #      #
# Line 846  sub CDATA_SECTION_STATE () { 35 } Line 868  sub CDATA_SECTION_STATE () { 35 }
868  sub MD_HYPHEN_STATE () { 36 } # "markup declaration open state" in the spec  sub MD_HYPHEN_STATE () { 36 } # "markup declaration open state" in the spec
869  sub MD_DOCTYPE_STATE () { 37 } # "markup declaration open state" in the spec  sub MD_DOCTYPE_STATE () { 37 } # "markup declaration open state" in the spec
870  sub MD_CDATA_STATE () { 38 } # "markup declaration open state" in the spec  sub MD_CDATA_STATE () { 38 } # "markup declaration open state" in the spec
871  sub CDATA_PCDATA_CLOSE_TAG_STATE () { 39 } # "close tag open state" in the spec  sub CDATA_RCDATA_CLOSE_TAG_STATE () { 39 } # "close tag open state" in the spec
872  sub CDATA_SECTION_MSE1_STATE () { 40 } # "CDATA section state" in the spec  sub CDATA_SECTION_MSE1_STATE () { 40 } # "CDATA section state" in the spec
873  sub CDATA_SECTION_MSE2_STATE () { 41 } # "CDATA section state" in the spec  sub CDATA_SECTION_MSE2_STATE () { 41 } # "CDATA section state" in the spec
874  sub PUBLIC_STATE () { 42 } # "after DOCTYPE name state" in the spec  sub PUBLIC_STATE () { 42 } # "after DOCTYPE name state" in the spec
# Line 860  sub NCR_NUM_STATE () { 46 } Line 882  sub NCR_NUM_STATE () { 46 }
882  sub HEXREF_X_STATE () { 47 }  sub HEXREF_X_STATE () { 47 }
883  sub HEXREF_HEX_STATE () { 48 }  sub HEXREF_HEX_STATE () { 48 }
884  sub ENTITY_NAME_STATE () { 49 }  sub ENTITY_NAME_STATE () { 49 }
885    sub PCDATA_STATE () { 50 } # "data state" in the spec
886    
887  sub DOCTYPE_TOKEN () { 1 }  sub DOCTYPE_TOKEN () { 1 }
888  sub COMMENT_TOKEN () { 2 }  sub COMMENT_TOKEN () { 2 }
# Line 912  sub IN_COLUMN_GROUP_IM () { 0b10 } Line 935  sub IN_COLUMN_GROUP_IM () { 0b10 }
935  sub _initialize_tokenizer ($) {  sub _initialize_tokenizer ($) {
936    my $self = shift;    my $self = shift;
937    $self->{state} = DATA_STATE; # MUST    $self->{state} = DATA_STATE; # MUST
938    #$self->{state_keyword}; # initialized when used    #$self->{s_kwd}; # state keyword - initialized when used
939    #$self->{entity__value}; # initialized when used    #$self->{entity__value}; # initialized when used
940    #$self->{entity__match}; # initialized when used    #$self->{entity__match}; # initialized when used
941    $self->{content_model} = PCDATA_CONTENT_MODEL; # be    $self->{content_model} = PCDATA_CONTENT_MODEL; # be
942    undef $self->{current_token};    undef $self->{ct}; # current token
943    undef $self->{current_attribute};    undef $self->{ca}; # current attribute
944    undef $self->{last_emitted_start_tag_name};    undef $self->{last_stag_name}; # last emitted start tag name
945    #$self->{prev_state}; # initialized when used    #$self->{prev_state}; # initialized when used
946    delete $self->{self_closing};    delete $self->{self_closing};
947    $self->{char_buffer} = '';    $self->{char_buffer} = '';
948    $self->{char_buffer_pos} = 0;    $self->{char_buffer_pos} = 0;
949    # $self->{next_char}    $self->{nc} = -1; # next input character
950      #$self->{next_nc}
951    !!!next-input-character;    !!!next-input-character;
952    $self->{token} = [];    $self->{token} = [];
953    # $self->{escape}    # $self->{escape}
# Line 934  sub _initialize_tokenizer ($) { Line 958  sub _initialize_tokenizer ($) {
958  ##       CHARACTER_TOKEN, or END_OF_FILE_TOKEN  ##       CHARACTER_TOKEN, or END_OF_FILE_TOKEN
959  ##   ->{name} (DOCTYPE_TOKEN)  ##   ->{name} (DOCTYPE_TOKEN)
960  ##   ->{tag_name} (START_TAG_TOKEN, END_TAG_TOKEN)  ##   ->{tag_name} (START_TAG_TOKEN, END_TAG_TOKEN)
961  ##   ->{public_identifier} (DOCTYPE_TOKEN)  ##   ->{pubid} (DOCTYPE_TOKEN)
962  ##   ->{system_identifier} (DOCTYPE_TOKEN)  ##   ->{sysid} (DOCTYPE_TOKEN)
963  ##   ->{quirks} == 1 or 0 (DOCTYPE_TOKEN): "force-quirks" flag  ##   ->{quirks} == 1 or 0 (DOCTYPE_TOKEN): "force-quirks" flag
964  ##   ->{attributes} isa HASH (START_TAG_TOKEN, END_TAG_TOKEN)  ##   ->{attributes} isa HASH (START_TAG_TOKEN, END_TAG_TOKEN)
965  ##        ->{name}  ##        ->{name}
# Line 957  sub _initialize_tokenizer ($) { Line 981  sub _initialize_tokenizer ($) {
981  ## TODO: Polytheistic slash SHOULD NOT be used. (Applied only to atheists.)  ## TODO: Polytheistic slash SHOULD NOT be used. (Applied only to atheists.)
982  ## (This requirement was dropped from HTML5 spec, unfortunately.)  ## (This requirement was dropped from HTML5 spec, unfortunately.)
983    
984    my $is_space = {
985      0x0009 => 1, # CHARACTER TABULATION (HT)
986      0x000A => 1, # LINE FEED (LF)
987      #0x000B => 0, # LINE TABULATION (VT)
988      0x000C => 1, # FORM FEED (FF)
989      #0x000D => 1, # CARRIAGE RETURN (CR)
990      0x0020 => 1, # SPACE (SP)
991    };
992    
993  sub _get_next_token ($) {  sub _get_next_token ($) {
994    my $self = shift;    my $self = shift;
995    
996    if ($self->{self_closing}) {    if ($self->{self_closing}) {
997      !!!parse-error (type => 'nestc', token => $self->{current_token});      !!!parse-error (type => 'nestc', token => $self->{ct});
998      ## NOTE: The |self_closing| flag is only set by start tag token.      ## NOTE: The |self_closing| flag is only set by start tag token.
999      ## In addition, when a start tag token is emitted, it is always set to      ## In addition, when a start tag token is emitted, it is always set to
1000      ## |current_token|.      ## |ct|.
1001      delete $self->{self_closing};      delete $self->{self_closing};
1002    }    }
1003    
# Line 974  sub _get_next_token ($) { Line 1007  sub _get_next_token ($) {
1007    }    }
1008    
1009    A: {    A: {
1010      if ($self->{state} == DATA_STATE) {      if ($self->{state} == PCDATA_STATE) {
1011        if ($self->{next_char} == 0x0026) { # &        ## NOTE: Same as |DATA_STATE|, but only for |PCDATA| content model.
1012    
1013          if ($self->{nc} == 0x0026) { # &
1014            !!!cp (0.1);
1015            ## NOTE: In the spec, the tokenizer is switched to the
1016            ## "entity data state".  In this implementation, the tokenizer
1017            ## is switched to the |ENTITY_STATE|, which is an implementation
1018            ## of the "consume a character reference" algorithm.
1019            $self->{entity_add} = -1;
1020            $self->{prev_state} = DATA_STATE;
1021            $self->{state} = ENTITY_STATE;
1022            !!!next-input-character;
1023            redo A;
1024          } elsif ($self->{nc} == 0x003C) { # <
1025            !!!cp (0.2);
1026            $self->{state} = TAG_OPEN_STATE;
1027            !!!next-input-character;
1028            redo A;
1029          } elsif ($self->{nc} == -1) {
1030            !!!cp (0.3);
1031            !!!emit ({type => END_OF_FILE_TOKEN,
1032                      line => $self->{line}, column => $self->{column}});
1033            last A; ## TODO: ok?
1034          } else {
1035            !!!cp (0.4);
1036            #
1037          }
1038    
1039          # Anything else
1040          my $token = {type => CHARACTER_TOKEN,
1041                       data => chr $self->{nc},
1042                       line => $self->{line}, column => $self->{column},
1043                      };
1044          $self->{read_until}->($token->{data}, q[<&], length $token->{data});
1045    
1046          ## Stay in the state.
1047          !!!next-input-character;
1048          !!!emit ($token);
1049          redo A;
1050        } elsif ($self->{state} == DATA_STATE) {
1051          $self->{s_kwd} = '' unless defined $self->{s_kwd};
1052          if ($self->{nc} == 0x0026) { # &
1053            $self->{s_kwd} = '';
1054          if ($self->{content_model} & CM_ENTITY and # PCDATA | RCDATA          if ($self->{content_model} & CM_ENTITY and # PCDATA | RCDATA
1055              not $self->{escape}) {              not $self->{escape}) {
1056            !!!cp (1);            !!!cp (1);
# Line 983  sub _get_next_token ($) { Line 1058  sub _get_next_token ($) {
1058            ## "entity data state".  In this implementation, the tokenizer            ## "entity data state".  In this implementation, the tokenizer
1059            ## is switched to the |ENTITY_STATE|, which is an implementation            ## is switched to the |ENTITY_STATE|, which is an implementation
1060            ## of the "consume a character reference" algorithm.            ## of the "consume a character reference" algorithm.
1061            $self->{entity_additional} = -1;            $self->{entity_add} = -1;
1062            $self->{prev_state} = DATA_STATE;            $self->{prev_state} = DATA_STATE;
1063            $self->{state} = ENTITY_STATE;            $self->{state} = ENTITY_STATE;
1064            !!!next-input-character;            !!!next-input-character;
# Line 992  sub _get_next_token ($) { Line 1067  sub _get_next_token ($) {
1067            !!!cp (2);            !!!cp (2);
1068            #            #
1069          }          }
1070        } elsif ($self->{next_char} == 0x002D) { # -        } elsif ($self->{nc} == 0x002D) { # -
1071          if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA          if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
1072            unless ($self->{escape}) {            $self->{s_kwd} .= '-';
1073              if ($self->{prev_char}->[0] == 0x002D and # -            
1074                  $self->{prev_char}->[1] == 0x0021 and # !            if ($self->{s_kwd} eq '<!--') {
1075                  $self->{prev_char}->[2] == 0x003C) { # <              !!!cp (3);
1076                !!!cp (3);              $self->{escape} = 1; # unless $self->{escape};
1077                $self->{escape} = 1;              $self->{s_kwd} = '--';
1078              } else {              #
1079                !!!cp (4);            } elsif ($self->{s_kwd} eq '---') {
1080              }              !!!cp (4);
1081                $self->{s_kwd} = '--';
1082                #
1083            } else {            } else {
1084              !!!cp (5);              !!!cp (5);
1085                #
1086            }            }
1087          }          }
1088                    
1089          #          #
1090        } elsif ($self->{next_char} == 0x003C) { # <        } elsif ($self->{nc} == 0x0021) { # !
1091            if (length $self->{s_kwd}) {
1092              !!!cp (5.1);
1093              $self->{s_kwd} .= '!';
1094              #
1095            } else {
1096              !!!cp (5.2);
1097              #$self->{s_kwd} = '';
1098              #
1099            }
1100            #
1101          } elsif ($self->{nc} == 0x003C) { # <
1102          if ($self->{content_model} & CM_FULL_MARKUP or # PCDATA          if ($self->{content_model} & CM_FULL_MARKUP or # PCDATA
1103              (($self->{content_model} & CM_LIMITED_MARKUP) and # CDATA | RCDATA              (($self->{content_model} & CM_LIMITED_MARKUP) and # CDATA | RCDATA
1104               not $self->{escape})) {               not $self->{escape})) {
# Line 1019  sub _get_next_token ($) { Line 1108  sub _get_next_token ($) {
1108            redo A;            redo A;
1109          } else {          } else {
1110            !!!cp (7);            !!!cp (7);
1111              $self->{s_kwd} = '';
1112            #            #
1113          }          }
1114        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1115          if ($self->{escape} and          if ($self->{escape} and
1116              ($self->{content_model} & CM_LIMITED_MARKUP)) { # RCDATA | CDATA              ($self->{content_model} & CM_LIMITED_MARKUP)) { # RCDATA | CDATA
1117            if ($self->{prev_char}->[0] == 0x002D and # -            if ($self->{s_kwd} eq '--') {
               $self->{prev_char}->[1] == 0x002D) { # -  
1118              !!!cp (8);              !!!cp (8);
1119              delete $self->{escape};              delete $self->{escape};
1120            } else {            } else {
# Line 1035  sub _get_next_token ($) { Line 1124  sub _get_next_token ($) {
1124            !!!cp (10);            !!!cp (10);
1125          }          }
1126                    
1127            $self->{s_kwd} = '';
1128          #          #
1129        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1130          !!!cp (11);          !!!cp (11);
1131            $self->{s_kwd} = '';
1132          !!!emit ({type => END_OF_FILE_TOKEN,          !!!emit ({type => END_OF_FILE_TOKEN,
1133                    line => $self->{line}, column => $self->{column}});                    line => $self->{line}, column => $self->{column}});
1134          last A; ## TODO: ok?          last A; ## TODO: ok?
1135        } else {        } else {
1136          !!!cp (12);          !!!cp (12);
1137            $self->{s_kwd} = '';
1138            #
1139        }        }
1140    
1141        # Anything else        # Anything else
1142        my $token = {type => CHARACTER_TOKEN,        my $token = {type => CHARACTER_TOKEN,
1143                     data => chr $self->{next_char},                     data => chr $self->{nc},
1144                     line => $self->{line}, column => $self->{column},                     line => $self->{line}, column => $self->{column},
1145                    };                    };
1146        $self->{read_until}->($token->{data}, q[-!<>&], length $token->{data});        if ($self->{read_until}->($token->{data}, q[-!<>&],
1147                                    length $token->{data})) {
1148            $self->{s_kwd} = '';
1149          }
1150    
1151        ## Stay in the data state        ## Stay in the data state.
1152          if ($self->{content_model} == PCDATA_CONTENT_MODEL) {
1153            !!!cp (13);
1154            $self->{state} = PCDATA_STATE;
1155          } else {
1156            !!!cp (14);
1157            ## Stay in the state.
1158          }
1159        !!!next-input-character;        !!!next-input-character;
   
1160        !!!emit ($token);        !!!emit ($token);
   
1161        redo A;        redo A;
1162      } elsif ($self->{state} == TAG_OPEN_STATE) {      } elsif ($self->{state} == TAG_OPEN_STATE) {
1163        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
1164          if ($self->{next_char} == 0x002F) { # /          if ($self->{nc} == 0x002F) { # /
1165            !!!cp (15);            !!!cp (15);
1166            !!!next-input-character;            !!!next-input-character;
1167            $self->{state} = CLOSE_TAG_OPEN_STATE;            $self->{state} = CLOSE_TAG_OPEN_STATE;
1168            redo A;            redo A;
1169            } elsif ($self->{nc} == 0x0021) { # !
1170              !!!cp (15.1);
1171              $self->{s_kwd} = '<' unless $self->{escape};
1172              #
1173          } else {          } else {
1174            !!!cp (16);            !!!cp (16);
1175            ## reconsume            #
           $self->{state} = DATA_STATE;  
   
           !!!emit ({type => CHARACTER_TOKEN, data => '<',  
                     line => $self->{line_prev},  
                     column => $self->{column_prev},  
                    });  
   
           redo A;  
1176          }          }
1177    
1178            ## reconsume
1179            $self->{state} = DATA_STATE;
1180            !!!emit ({type => CHARACTER_TOKEN, data => '<',
1181                      line => $self->{line_prev},
1182                      column => $self->{column_prev},
1183                     });
1184            redo A;
1185        } elsif ($self->{content_model} & CM_FULL_MARKUP) { # PCDATA        } elsif ($self->{content_model} & CM_FULL_MARKUP) { # PCDATA
1186          if ($self->{next_char} == 0x0021) { # !          if ($self->{nc} == 0x0021) { # !
1187            !!!cp (17);            !!!cp (17);
1188            $self->{state} = MARKUP_DECLARATION_OPEN_STATE;            $self->{state} = MARKUP_DECLARATION_OPEN_STATE;
1189            !!!next-input-character;            !!!next-input-character;
1190            redo A;            redo A;
1191          } elsif ($self->{next_char} == 0x002F) { # /          } elsif ($self->{nc} == 0x002F) { # /
1192            !!!cp (18);            !!!cp (18);
1193            $self->{state} = CLOSE_TAG_OPEN_STATE;            $self->{state} = CLOSE_TAG_OPEN_STATE;
1194            !!!next-input-character;            !!!next-input-character;
1195            redo A;            redo A;
1196          } elsif (0x0041 <= $self->{next_char} and          } elsif (0x0041 <= $self->{nc} and
1197                   $self->{next_char} <= 0x005A) { # A..Z                   $self->{nc} <= 0x005A) { # A..Z
1198            !!!cp (19);            !!!cp (19);
1199            $self->{current_token}            $self->{ct}
1200              = {type => START_TAG_TOKEN,              = {type => START_TAG_TOKEN,
1201                 tag_name => chr ($self->{next_char} + 0x0020),                 tag_name => chr ($self->{nc} + 0x0020),
1202                 line => $self->{line_prev},                 line => $self->{line_prev},
1203                 column => $self->{column_prev}};                 column => $self->{column_prev}};
1204            $self->{state} = TAG_NAME_STATE;            $self->{state} = TAG_NAME_STATE;
1205            !!!next-input-character;            !!!next-input-character;
1206            redo A;            redo A;
1207          } elsif (0x0061 <= $self->{next_char} and          } elsif (0x0061 <= $self->{nc} and
1208                   $self->{next_char} <= 0x007A) { # a..z                   $self->{nc} <= 0x007A) { # a..z
1209            !!!cp (20);            !!!cp (20);
1210            $self->{current_token} = {type => START_TAG_TOKEN,            $self->{ct} = {type => START_TAG_TOKEN,
1211                                      tag_name => chr ($self->{next_char}),                                      tag_name => chr ($self->{nc}),
1212                                      line => $self->{line_prev},                                      line => $self->{line_prev},
1213                                      column => $self->{column_prev}};                                      column => $self->{column_prev}};
1214            $self->{state} = TAG_NAME_STATE;            $self->{state} = TAG_NAME_STATE;
1215            !!!next-input-character;            !!!next-input-character;
1216            redo A;            redo A;
1217          } elsif ($self->{next_char} == 0x003E) { # >          } elsif ($self->{nc} == 0x003E) { # >
1218            !!!cp (21);            !!!cp (21);
1219            !!!parse-error (type => 'empty start tag',            !!!parse-error (type => 'empty start tag',
1220                            line => $self->{line_prev},                            line => $self->{line_prev},
# Line 1122  sub _get_next_token ($) { Line 1228  sub _get_next_token ($) {
1228                     });                     });
1229    
1230            redo A;            redo A;
1231          } elsif ($self->{next_char} == 0x003F) { # ?          } elsif ($self->{nc} == 0x003F) { # ?
1232            !!!cp (22);            !!!cp (22);
1233            !!!parse-error (type => 'pio',            !!!parse-error (type => 'pio',
1234                            line => $self->{line_prev},                            line => $self->{line_prev},
1235                            column => $self->{column_prev});                            column => $self->{column_prev});
1236            $self->{state} = BOGUS_COMMENT_STATE;            $self->{state} = BOGUS_COMMENT_STATE;
1237            $self->{current_token} = {type => COMMENT_TOKEN, data => '',            $self->{ct} = {type => COMMENT_TOKEN, data => '',
1238                                      line => $self->{line_prev},                                      line => $self->{line_prev},
1239                                      column => $self->{column_prev},                                      column => $self->{column_prev},
1240                                     };                                     };
1241            ## $self->{next_char} is intentionally left as is            ## $self->{nc} is intentionally left as is
1242            redo A;            redo A;
1243          } else {          } else {
1244            !!!cp (23);            !!!cp (23);
# Line 1154  sub _get_next_token ($) { Line 1260  sub _get_next_token ($) {
1260        }        }
1261      } elsif ($self->{state} == CLOSE_TAG_OPEN_STATE) {      } elsif ($self->{state} == CLOSE_TAG_OPEN_STATE) {
1262        ## NOTE: The "close tag open state" in the spec is implemented as        ## NOTE: The "close tag open state" in the spec is implemented as
1263        ## |CLOSE_TAG_OPEN_STATE| and |CDATA_PCDATA_CLOSE_TAG_STATE|.        ## |CLOSE_TAG_OPEN_STATE| and |CDATA_RCDATA_CLOSE_TAG_STATE|.
1264    
1265        my ($l, $c) = ($self->{line_prev}, $self->{column_prev} - 1); # "<"of"</"        my ($l, $c) = ($self->{line_prev}, $self->{column_prev} - 1); # "<"of"</"
1266        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
1267          if (defined $self->{last_emitted_start_tag_name}) {          if (defined $self->{last_stag_name}) {
1268            $self->{state} = CDATA_PCDATA_CLOSE_TAG_STATE;            $self->{state} = CDATA_RCDATA_CLOSE_TAG_STATE;
1269            $self->{state_keyword} = '';            $self->{s_kwd} = '';
1270            ## Reconsume.            ## Reconsume.
1271            redo A;            redo A;
1272          } else {          } else {
# Line 1176  sub _get_next_token ($) { Line 1282  sub _get_next_token ($) {
1282          }          }
1283        }        }
1284    
1285        if (0x0041 <= $self->{next_char} and        if (0x0041 <= $self->{nc} and
1286            $self->{next_char} <= 0x005A) { # A..Z            $self->{nc} <= 0x005A) { # A..Z
1287          !!!cp (29);          !!!cp (29);
1288          $self->{current_token}          $self->{ct}
1289              = {type => END_TAG_TOKEN,              = {type => END_TAG_TOKEN,
1290                 tag_name => chr ($self->{next_char} + 0x0020),                 tag_name => chr ($self->{nc} + 0x0020),
1291                 line => $l, column => $c};                 line => $l, column => $c};
1292          $self->{state} = TAG_NAME_STATE;          $self->{state} = TAG_NAME_STATE;
1293          !!!next-input-character;          !!!next-input-character;
1294          redo A;          redo A;
1295        } elsif (0x0061 <= $self->{next_char} and        } elsif (0x0061 <= $self->{nc} and
1296                 $self->{next_char} <= 0x007A) { # a..z                 $self->{nc} <= 0x007A) { # a..z
1297          !!!cp (30);          !!!cp (30);
1298          $self->{current_token} = {type => END_TAG_TOKEN,          $self->{ct} = {type => END_TAG_TOKEN,
1299                                    tag_name => chr ($self->{next_char}),                                    tag_name => chr ($self->{nc}),
1300                                    line => $l, column => $c};                                    line => $l, column => $c};
1301          $self->{state} = TAG_NAME_STATE;          $self->{state} = TAG_NAME_STATE;
1302          !!!next-input-character;          !!!next-input-character;
1303          redo A;          redo A;
1304        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1305          !!!cp (31);          !!!cp (31);
1306          !!!parse-error (type => 'empty end tag',          !!!parse-error (type => 'empty end tag',
1307                          line => $self->{line_prev}, ## "<" in "</>"                          line => $self->{line_prev}, ## "<" in "</>"
# Line 1203  sub _get_next_token ($) { Line 1309  sub _get_next_token ($) {
1309          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1310          !!!next-input-character;          !!!next-input-character;
1311          redo A;          redo A;
1312        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1313          !!!cp (32);          !!!cp (32);
1314          !!!parse-error (type => 'bare etago');          !!!parse-error (type => 'bare etago');
1315          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
# Line 1218  sub _get_next_token ($) { Line 1324  sub _get_next_token ($) {
1324          !!!cp (33);          !!!cp (33);
1325          !!!parse-error (type => 'bogus end tag');          !!!parse-error (type => 'bogus end tag');
1326          $self->{state} = BOGUS_COMMENT_STATE;          $self->{state} = BOGUS_COMMENT_STATE;
1327          $self->{current_token} = {type => COMMENT_TOKEN, data => '',          $self->{ct} = {type => COMMENT_TOKEN, data => '',
1328                                    line => $self->{line_prev}, # "<" of "</"                                    line => $self->{line_prev}, # "<" of "</"
1329                                    column => $self->{column_prev} - 1,                                    column => $self->{column_prev} - 1,
1330                                   };                                   };
1331          ## NOTE: $self->{next_char} is intentionally left as is.          ## NOTE: $self->{nc} is intentionally left as is.
1332          ## Although the "anything else" case of the spec not explicitly          ## Although the "anything else" case of the spec not explicitly
1333          ## states that the next input character is to be reconsumed,          ## states that the next input character is to be reconsumed,
1334          ## it will be included to the |data| of the comment token          ## it will be included to the |data| of the comment token
# Line 1230  sub _get_next_token ($) { Line 1336  sub _get_next_token ($) {
1336          ## "bogus comment state" entry.          ## "bogus comment state" entry.
1337          redo A;          redo A;
1338        }        }
1339      } elsif ($self->{state} == CDATA_PCDATA_CLOSE_TAG_STATE) {      } elsif ($self->{state} == CDATA_RCDATA_CLOSE_TAG_STATE) {
1340        my $ch = substr $self->{last_emitted_start_tag_name}, length $self->{state_keyword}, 1;        my $ch = substr $self->{last_stag_name}, length $self->{s_kwd}, 1;
1341        if (length $ch) {        if (length $ch) {
1342          my $CH = $ch;          my $CH = $ch;
1343          $ch =~ tr/a-z/A-Z/;          $ch =~ tr/a-z/A-Z/;
1344          my $nch = chr $self->{next_char};          my $nch = chr $self->{nc};
1345          if ($nch eq $ch or $nch eq $CH) {          if ($nch eq $ch or $nch eq $CH) {
1346            !!!cp (24);            !!!cp (24);
1347            ## Stay in the state.            ## Stay in the state.
1348            $self->{state_keyword} .= $nch;            $self->{s_kwd} .= $nch;
1349            !!!next-input-character;            !!!next-input-character;
1350            redo A;            redo A;
1351          } else {          } else {
# Line 1247  sub _get_next_token ($) { Line 1353  sub _get_next_token ($) {
1353            $self->{state} = DATA_STATE;            $self->{state} = DATA_STATE;
1354            ## Reconsume.            ## Reconsume.
1355            !!!emit ({type => CHARACTER_TOKEN,            !!!emit ({type => CHARACTER_TOKEN,
1356                      data => '</' . $self->{state_keyword},                      data => '</' . $self->{s_kwd},
1357                      line => $self->{line_prev},                      line => $self->{line_prev},
1358                      column => $self->{column_prev} - 1 - length $self->{state_keyword},                      column => $self->{column_prev} - 1 - length $self->{s_kwd},
1359                     });                     });
1360            redo A;            redo A;
1361          }          }
1362        } else { # after "<{tag-name}"        } else { # after "<{tag-name}"
1363          unless ({          unless ($is_space->{$self->{nc}} or
1364                   0x0009 => 1, # HT                  {
                  0x000A => 1, # LF  
                  0x000B => 1, # VT  
                  0x000C => 1, # FF  
                  0x0020 => 1, # SP  
1365                   0x003E => 1, # >                   0x003E => 1, # >
1366                   0x002F => 1, # /                   0x002F => 1, # /
1367                   -1 => 1, # EOF                   -1 => 1, # EOF
1368                  }->{$self->{next_char}}) {                  }->{$self->{nc}}) {
1369            !!!cp (26);            !!!cp (26);
1370            ## Reconsume.            ## Reconsume.
1371            $self->{state} = DATA_STATE;            $self->{state} = DATA_STATE;
1372            !!!emit ({type => CHARACTER_TOKEN,            !!!emit ({type => CHARACTER_TOKEN,
1373                      data => '</' . $self->{state_keyword},                      data => '</' . $self->{s_kwd},
1374                      line => $self->{line_prev},                      line => $self->{line_prev},
1375                      column => $self->{column_prev} - 1 - length $self->{state_keyword},                      column => $self->{column_prev} - 1 - length $self->{s_kwd},
1376                     });                     });
1377            redo A;            redo A;
1378          } else {          } else {
1379            !!!cp (27);            !!!cp (27);
1380            $self->{current_token}            $self->{ct}
1381                = {type => END_TAG_TOKEN,                = {type => END_TAG_TOKEN,
1382                   tag_name => $self->{last_emitted_start_tag_name},                   tag_name => $self->{last_stag_name},
1383                   line => $self->{line_prev},                   line => $self->{line_prev},
1384                   column => $self->{column_prev} - 1 - length $self->{state_keyword}};                   column => $self->{column_prev} - 1 - length $self->{s_kwd}};
1385            $self->{state} = TAG_NAME_STATE;            $self->{state} = TAG_NAME_STATE;
1386            ## Reconsume.            ## Reconsume.
1387            redo A;            redo A;
1388          }          }
1389        }        }
1390      } elsif ($self->{state} == TAG_NAME_STATE) {      } elsif ($self->{state} == TAG_NAME_STATE) {
1391        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
1392          !!!cp (34);          !!!cp (34);
1393          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1394          !!!next-input-character;          !!!next-input-character;
1395          redo A;          redo A;
1396        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1397          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1398            !!!cp (35);            !!!cp (35);
1399            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1400          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1401            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1402            #if ($self->{current_token}->{attributes}) {            #if ($self->{ct}->{attributes}) {
1403            #  ## NOTE: This should never be reached.            #  ## NOTE: This should never be reached.
1404            #  !!! cp (36);            #  !!! cp (36);
1405            #  !!! parse-error (type => 'end tag attribute');            #  !!! parse-error (type => 'end tag attribute');
# Line 1309  sub _get_next_token ($) { Line 1407  sub _get_next_token ($) {
1407              !!!cp (37);              !!!cp (37);
1408            #}            #}
1409          } else {          } else {
1410            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1411          }          }
1412          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1413          !!!next-input-character;          !!!next-input-character;
1414    
1415          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1416    
1417          redo A;          redo A;
1418        } elsif (0x0041 <= $self->{next_char} and        } elsif (0x0041 <= $self->{nc} and
1419                 $self->{next_char} <= 0x005A) { # A..Z                 $self->{nc} <= 0x005A) { # A..Z
1420          !!!cp (38);          !!!cp (38);
1421          $self->{current_token}->{tag_name} .= chr ($self->{next_char} + 0x0020);          $self->{ct}->{tag_name} .= chr ($self->{nc} + 0x0020);
1422            # start tag or end tag            # start tag or end tag
1423          ## Stay in this state          ## Stay in this state
1424          !!!next-input-character;          !!!next-input-character;
1425          redo A;          redo A;
1426        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1427          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1428          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1429            !!!cp (39);            !!!cp (39);
1430            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1431          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1432            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1433            #if ($self->{current_token}->{attributes}) {            #if ($self->{ct}->{attributes}) {
1434            #  ## NOTE: This state should never be reached.            #  ## NOTE: This state should never be reached.
1435            #  !!! cp (40);            #  !!! cp (40);
1436            #  !!! parse-error (type => 'end tag attribute');            #  !!! parse-error (type => 'end tag attribute');
# Line 1340  sub _get_next_token ($) { Line 1438  sub _get_next_token ($) {
1438              !!!cp (41);              !!!cp (41);
1439            #}            #}
1440          } else {          } else {
1441            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1442          }          }
1443          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1444          # reconsume          # reconsume
1445    
1446          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1447    
1448          redo A;          redo A;
1449        } elsif ($self->{next_char} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
1450          !!!cp (42);          !!!cp (42);
1451          $self->{state} = SELF_CLOSING_START_TAG_STATE;          $self->{state} = SELF_CLOSING_START_TAG_STATE;
1452          !!!next-input-character;          !!!next-input-character;
1453          redo A;          redo A;
1454        } else {        } else {
1455          !!!cp (44);          !!!cp (44);
1456          $self->{current_token}->{tag_name} .= chr $self->{next_char};          $self->{ct}->{tag_name} .= chr $self->{nc};
1457            # start tag or end tag            # start tag or end tag
1458          ## Stay in the state          ## Stay in the state
1459          !!!next-input-character;          !!!next-input-character;
1460          redo A;          redo A;
1461        }        }
1462      } elsif ($self->{state} == BEFORE_ATTRIBUTE_NAME_STATE) {      } elsif ($self->{state} == BEFORE_ATTRIBUTE_NAME_STATE) {
1463        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
1464          !!!cp (45);          !!!cp (45);
1465          ## Stay in the state          ## Stay in the state
1466          !!!next-input-character;          !!!next-input-character;
1467          redo A;          redo A;
1468        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1469          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1470            !!!cp (46);            !!!cp (46);
1471            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1472          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1473            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1474            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1475              !!!cp (47);              !!!cp (47);
1476              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1477            } else {            } else {
1478              !!!cp (48);              !!!cp (48);
1479            }            }
1480          } else {          } else {
1481            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1482          }          }
1483          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1484          !!!next-input-character;          !!!next-input-character;
1485    
1486          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1487    
1488          redo A;          redo A;
1489        } elsif (0x0041 <= $self->{next_char} and        } elsif (0x0041 <= $self->{nc} and
1490                 $self->{next_char} <= 0x005A) { # A..Z                 $self->{nc} <= 0x005A) { # A..Z
1491          !!!cp (49);          !!!cp (49);
1492          $self->{current_attribute}          $self->{ca}
1493              = {name => chr ($self->{next_char} + 0x0020),              = {name => chr ($self->{nc} + 0x0020),
1494                 value => '',                 value => '',
1495                 line => $self->{line}, column => $self->{column}};                 line => $self->{line}, column => $self->{column}};
1496          $self->{state} = ATTRIBUTE_NAME_STATE;          $self->{state} = ATTRIBUTE_NAME_STATE;
1497          !!!next-input-character;          !!!next-input-character;
1498          redo A;          redo A;
1499        } elsif ($self->{next_char} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
1500          !!!cp (50);          !!!cp (50);
1501          $self->{state} = SELF_CLOSING_START_TAG_STATE;          $self->{state} = SELF_CLOSING_START_TAG_STATE;
1502          !!!next-input-character;          !!!next-input-character;
1503          redo A;          redo A;
1504        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1505          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1506          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1507            !!!cp (52);            !!!cp (52);
1508            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1509          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1510            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1511            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1512              !!!cp (53);              !!!cp (53);
1513              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1514            } else {            } else {
1515              !!!cp (54);              !!!cp (54);
1516            }            }
1517          } else {          } else {
1518            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1519          }          }
1520          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1521          # reconsume          # reconsume
1522    
1523          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1524    
1525          redo A;          redo A;
1526        } else {        } else {
# Line 1434  sub _get_next_token ($) { Line 1528  sub _get_next_token ($) {
1528               0x0022 => 1, # "               0x0022 => 1, # "
1529               0x0027 => 1, # '               0x0027 => 1, # '
1530               0x003D => 1, # =               0x003D => 1, # =
1531              }->{$self->{next_char}}) {              }->{$self->{nc}}) {
1532            !!!cp (55);            !!!cp (55);
1533            !!!parse-error (type => 'bad attribute name');            !!!parse-error (type => 'bad attribute name');
1534          } else {          } else {
1535            !!!cp (56);            !!!cp (56);
1536          }          }
1537          $self->{current_attribute}          $self->{ca}
1538              = {name => chr ($self->{next_char}),              = {name => chr ($self->{nc}),
1539                 value => '',                 value => '',
1540                 line => $self->{line}, column => $self->{column}};                 line => $self->{line}, column => $self->{column}};
1541          $self->{state} = ATTRIBUTE_NAME_STATE;          $self->{state} = ATTRIBUTE_NAME_STATE;
# Line 1450  sub _get_next_token ($) { Line 1544  sub _get_next_token ($) {
1544        }        }
1545      } elsif ($self->{state} == ATTRIBUTE_NAME_STATE) {      } elsif ($self->{state} == ATTRIBUTE_NAME_STATE) {
1546        my $before_leave = sub {        my $before_leave = sub {
1547          if (exists $self->{current_token}->{attributes} # start tag or end tag          if (exists $self->{ct}->{attributes} # start tag or end tag
1548              ->{$self->{current_attribute}->{name}}) { # MUST              ->{$self->{ca}->{name}}) { # MUST
1549            !!!cp (57);            !!!cp (57);
1550            !!!parse-error (type => 'duplicate attribute', text => $self->{current_attribute}->{name}, line => $self->{current_attribute}->{line}, column => $self->{current_attribute}->{column});            !!!parse-error (type => 'duplicate attribute', text => $self->{ca}->{name}, line => $self->{ca}->{line}, column => $self->{ca}->{column});
1551            ## Discard $self->{current_attribute} # MUST            ## Discard $self->{ca} # MUST
1552          } else {          } else {
1553            !!!cp (58);            !!!cp (58);
1554            $self->{current_token}->{attributes}->{$self->{current_attribute}->{name}}            $self->{ct}->{attributes}->{$self->{ca}->{name}}
1555              = $self->{current_attribute};              = $self->{ca};
1556          }          }
1557        }; # $before_leave        }; # $before_leave
1558    
1559        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
1560          !!!cp (59);          !!!cp (59);
1561          $before_leave->();          $before_leave->();
1562          $self->{state} = AFTER_ATTRIBUTE_NAME_STATE;          $self->{state} = AFTER_ATTRIBUTE_NAME_STATE;
1563          !!!next-input-character;          !!!next-input-character;
1564          redo A;          redo A;
1565        } elsif ($self->{next_char} == 0x003D) { # =        } elsif ($self->{nc} == 0x003D) { # =
1566          !!!cp (60);          !!!cp (60);
1567          $before_leave->();          $before_leave->();
1568          $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;          $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;
1569          !!!next-input-character;          !!!next-input-character;
1570          redo A;          redo A;
1571        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1572          $before_leave->();          $before_leave->();
1573          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1574            !!!cp (61);            !!!cp (61);
1575            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1576          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1577            !!!cp (62);            !!!cp (62);
1578            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1579            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1580              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1581            }            }
1582          } else {          } else {
1583            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1584          }          }
1585          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1586          !!!next-input-character;          !!!next-input-character;
1587    
1588          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1589    
1590          redo A;          redo A;
1591        } elsif (0x0041 <= $self->{next_char} and        } elsif (0x0041 <= $self->{nc} and
1592                 $self->{next_char} <= 0x005A) { # A..Z                 $self->{nc} <= 0x005A) { # A..Z
1593          !!!cp (63);          !!!cp (63);
1594          $self->{current_attribute}->{name} .= chr ($self->{next_char} + 0x0020);          $self->{ca}->{name} .= chr ($self->{nc} + 0x0020);
1595          ## Stay in the state          ## Stay in the state
1596          !!!next-input-character;          !!!next-input-character;
1597          redo A;          redo A;
1598        } elsif ($self->{next_char} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
1599          !!!cp (64);          !!!cp (64);
1600          $before_leave->();          $before_leave->();
1601          $self->{state} = SELF_CLOSING_START_TAG_STATE;          $self->{state} = SELF_CLOSING_START_TAG_STATE;
1602          !!!next-input-character;          !!!next-input-character;
1603          redo A;          redo A;
1604        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1605          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1606          $before_leave->();          $before_leave->();
1607          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1608            !!!cp (66);            !!!cp (66);
1609            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1610          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1611            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1612            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1613              !!!cp (67);              !!!cp (67);
1614              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1615            } else {            } else {
# Line 1527  sub _get_next_token ($) { Line 1617  sub _get_next_token ($) {
1617              !!!cp (68);              !!!cp (68);
1618            }            }
1619          } else {          } else {
1620            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1621          }          }
1622          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1623          # reconsume          # reconsume
1624    
1625          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1626    
1627          redo A;          redo A;
1628        } else {        } else {
1629          if ($self->{next_char} == 0x0022 or # "          if ($self->{nc} == 0x0022 or # "
1630              $self->{next_char} == 0x0027) { # '              $self->{nc} == 0x0027) { # '
1631            !!!cp (69);            !!!cp (69);
1632            !!!parse-error (type => 'bad attribute name');            !!!parse-error (type => 'bad attribute name');
1633          } else {          } else {
1634            !!!cp (70);            !!!cp (70);
1635          }          }
1636          $self->{current_attribute}->{name} .= chr ($self->{next_char});          $self->{ca}->{name} .= chr ($self->{nc});
1637          ## Stay in the state          ## Stay in the state
1638          !!!next-input-character;          !!!next-input-character;
1639          redo A;          redo A;
1640        }        }
1641      } elsif ($self->{state} == AFTER_ATTRIBUTE_NAME_STATE) {      } elsif ($self->{state} == AFTER_ATTRIBUTE_NAME_STATE) {
1642        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
1643          !!!cp (71);          !!!cp (71);
1644          ## Stay in the state          ## Stay in the state
1645          !!!next-input-character;          !!!next-input-character;
1646          redo A;          redo A;
1647        } elsif ($self->{next_char} == 0x003D) { # =        } elsif ($self->{nc} == 0x003D) { # =
1648          !!!cp (72);          !!!cp (72);
1649          $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;          $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;
1650          !!!next-input-character;          !!!next-input-character;
1651          redo A;          redo A;
1652        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1653          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1654            !!!cp (73);            !!!cp (73);
1655            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1656          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1657            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1658            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1659              !!!cp (74);              !!!cp (74);
1660              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1661            } else {            } else {
# Line 1577  sub _get_next_token ($) { Line 1663  sub _get_next_token ($) {
1663              !!!cp (75);              !!!cp (75);
1664            }            }
1665          } else {          } else {
1666            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1667          }          }
1668          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1669          !!!next-input-character;          !!!next-input-character;
1670    
1671          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1672    
1673          redo A;          redo A;
1674        } elsif (0x0041 <= $self->{next_char} and        } elsif (0x0041 <= $self->{nc} and
1675                 $self->{next_char} <= 0x005A) { # A..Z                 $self->{nc} <= 0x005A) { # A..Z
1676          !!!cp (76);          !!!cp (76);
1677          $self->{current_attribute}          $self->{ca}
1678              = {name => chr ($self->{next_char} + 0x0020),              = {name => chr ($self->{nc} + 0x0020),
1679                 value => '',                 value => '',
1680                 line => $self->{line}, column => $self->{column}};                 line => $self->{line}, column => $self->{column}};
1681          $self->{state} = ATTRIBUTE_NAME_STATE;          $self->{state} = ATTRIBUTE_NAME_STATE;
1682          !!!next-input-character;          !!!next-input-character;
1683          redo A;          redo A;
1684        } elsif ($self->{next_char} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
1685          !!!cp (77);          !!!cp (77);
1686          $self->{state} = SELF_CLOSING_START_TAG_STATE;          $self->{state} = SELF_CLOSING_START_TAG_STATE;
1687          !!!next-input-character;          !!!next-input-character;
1688          redo A;          redo A;
1689        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1690          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1691          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1692            !!!cp (79);            !!!cp (79);
1693            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1694          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1695            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1696            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1697              !!!cp (80);              !!!cp (80);
1698              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1699            } else {            } else {
# Line 1615  sub _get_next_token ($) { Line 1701  sub _get_next_token ($) {
1701              !!!cp (81);              !!!cp (81);
1702            }            }
1703          } else {          } else {
1704            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1705          }          }
1706          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1707          # reconsume          # reconsume
1708    
1709          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1710    
1711          redo A;          redo A;
1712        } else {        } else {
1713          if ($self->{next_char} == 0x0022 or # "          if ($self->{nc} == 0x0022 or # "
1714              $self->{next_char} == 0x0027) { # '              $self->{nc} == 0x0027) { # '
1715            !!!cp (78);            !!!cp (78);
1716            !!!parse-error (type => 'bad attribute name');            !!!parse-error (type => 'bad attribute name');
1717          } else {          } else {
1718            !!!cp (82);            !!!cp (82);
1719          }          }
1720          $self->{current_attribute}          $self->{ca}
1721              = {name => chr ($self->{next_char}),              = {name => chr ($self->{nc}),
1722                 value => '',                 value => '',
1723                 line => $self->{line}, column => $self->{column}};                 line => $self->{line}, column => $self->{column}};
1724          $self->{state} = ATTRIBUTE_NAME_STATE;          $self->{state} = ATTRIBUTE_NAME_STATE;
# Line 1640  sub _get_next_token ($) { Line 1726  sub _get_next_token ($) {
1726          redo A;                  redo A;        
1727        }        }
1728      } elsif ($self->{state} == BEFORE_ATTRIBUTE_VALUE_STATE) {      } elsif ($self->{state} == BEFORE_ATTRIBUTE_VALUE_STATE) {
1729        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP        
1730          !!!cp (83);          !!!cp (83);
1731          ## Stay in the state          ## Stay in the state
1732          !!!next-input-character;          !!!next-input-character;
1733          redo A;          redo A;
1734        } elsif ($self->{next_char} == 0x0022) { # "        } elsif ($self->{nc} == 0x0022) { # "
1735          !!!cp (84);          !!!cp (84);
1736          $self->{state} = ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE;          $self->{state} = ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE;
1737          !!!next-input-character;          !!!next-input-character;
1738          redo A;          redo A;
1739        } elsif ($self->{next_char} == 0x0026) { # &        } elsif ($self->{nc} == 0x0026) { # &
1740          !!!cp (85);          !!!cp (85);
1741          $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;          $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;
1742          ## reconsume          ## reconsume
1743          redo A;          redo A;
1744        } elsif ($self->{next_char} == 0x0027) { # '        } elsif ($self->{nc} == 0x0027) { # '
1745          !!!cp (86);          !!!cp (86);
1746          $self->{state} = ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE;          $self->{state} = ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE;
1747          !!!next-input-character;          !!!next-input-character;
1748          redo A;          redo A;
1749        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1750          !!!parse-error (type => 'empty unquoted attribute value');          !!!parse-error (type => 'empty unquoted attribute value');
1751          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1752            !!!cp (87);            !!!cp (87);
1753            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1754          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1755            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1756            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1757              !!!cp (88);              !!!cp (88);
1758              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1759            } else {            } else {
# Line 1679  sub _get_next_token ($) { Line 1761  sub _get_next_token ($) {
1761              !!!cp (89);              !!!cp (89);
1762            }            }
1763          } else {          } else {
1764            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1765          }          }
1766          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1767          !!!next-input-character;          !!!next-input-character;
1768    
1769          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1770    
1771          redo A;          redo A;
1772        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1773          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1774          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1775            !!!cp (90);            !!!cp (90);
1776            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1777          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1778            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1779            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1780              !!!cp (91);              !!!cp (91);
1781              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1782            } else {            } else {
# Line 1702  sub _get_next_token ($) { Line 1784  sub _get_next_token ($) {
1784              !!!cp (92);              !!!cp (92);
1785            }            }
1786          } else {          } else {
1787            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1788          }          }
1789          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1790          ## reconsume          ## reconsume
1791    
1792          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1793    
1794          redo A;          redo A;
1795        } else {        } else {
1796          if ($self->{next_char} == 0x003D) { # =          if ($self->{nc} == 0x003D) { # =
1797            !!!cp (93);            !!!cp (93);
1798            !!!parse-error (type => 'bad attribute value');            !!!parse-error (type => 'bad attribute value');
1799          } else {          } else {
1800            !!!cp (94);            !!!cp (94);
1801          }          }
1802          $self->{current_attribute}->{value} .= chr ($self->{next_char});          $self->{ca}->{value} .= chr ($self->{nc});
1803          $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;          $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;
1804          !!!next-input-character;          !!!next-input-character;
1805          redo A;          redo A;
1806        }        }
1807      } elsif ($self->{state} == ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE) {      } elsif ($self->{state} == ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE) {
1808        if ($self->{next_char} == 0x0022) { # "        if ($self->{nc} == 0x0022) { # "
1809          !!!cp (95);          !!!cp (95);
1810          $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;          $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;
1811          !!!next-input-character;          !!!next-input-character;
1812          redo A;          redo A;
1813        } elsif ($self->{next_char} == 0x0026) { # &        } elsif ($self->{nc} == 0x0026) { # &
1814          !!!cp (96);          !!!cp (96);
1815          ## NOTE: In the spec, the tokenizer is switched to the          ## NOTE: In the spec, the tokenizer is switched to the
1816          ## "entity in attribute value state".  In this implementation, the          ## "entity in attribute value state".  In this implementation, the
1817          ## tokenizer is switched to the |ENTITY_STATE|, which is an          ## tokenizer is switched to the |ENTITY_STATE|, which is an
1818          ## implementation of the "consume a character reference" algorithm.          ## implementation of the "consume a character reference" algorithm.
1819          $self->{prev_state} = $self->{state};          $self->{prev_state} = $self->{state};
1820          $self->{entity_additional} = 0x0022; # "          $self->{entity_add} = 0x0022; # "
1821          $self->{state} = ENTITY_STATE;          $self->{state} = ENTITY_STATE;
1822          !!!next-input-character;          !!!next-input-character;
1823          redo A;          redo A;
1824        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1825          !!!parse-error (type => 'unclosed attribute value');          !!!parse-error (type => 'unclosed attribute value');
1826          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1827            !!!cp (97);            !!!cp (97);
1828            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1829          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1830            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1831            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1832              !!!cp (98);              !!!cp (98);
1833              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1834            } else {            } else {
# Line 1754  sub _get_next_token ($) { Line 1836  sub _get_next_token ($) {
1836              !!!cp (99);              !!!cp (99);
1837            }            }
1838          } else {          } else {
1839            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1840          }          }
1841          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1842          ## reconsume          ## reconsume
1843    
1844          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1845    
1846          redo A;          redo A;
1847        } else {        } else {
1848          !!!cp (100);          !!!cp (100);
1849          $self->{current_attribute}->{value} .= chr ($self->{next_char});          $self->{ca}->{value} .= chr ($self->{nc});
1850          $self->{read_until}->($self->{current_attribute}->{value},          $self->{read_until}->($self->{ca}->{value},
1851                                q["&],                                q["&],
1852                                length $self->{current_attribute}->{value});                                length $self->{ca}->{value});
1853    
1854          ## Stay in the state          ## Stay in the state
1855          !!!next-input-character;          !!!next-input-character;
1856          redo A;          redo A;
1857        }        }
1858      } elsif ($self->{state} == ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE) {      } elsif ($self->{state} == ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE) {
1859        if ($self->{next_char} == 0x0027) { # '        if ($self->{nc} == 0x0027) { # '
1860          !!!cp (101);          !!!cp (101);
1861          $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;          $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;
1862          !!!next-input-character;          !!!next-input-character;
1863          redo A;          redo A;
1864        } elsif ($self->{next_char} == 0x0026) { # &        } elsif ($self->{nc} == 0x0026) { # &
1865          !!!cp (102);          !!!cp (102);
1866          ## NOTE: In the spec, the tokenizer is switched to the          ## NOTE: In the spec, the tokenizer is switched to the
1867          ## "entity in attribute value state".  In this implementation, the          ## "entity in attribute value state".  In this implementation, the
1868          ## tokenizer is switched to the |ENTITY_STATE|, which is an          ## tokenizer is switched to the |ENTITY_STATE|, which is an
1869          ## implementation of the "consume a character reference" algorithm.          ## implementation of the "consume a character reference" algorithm.
1870          $self->{entity_additional} = 0x0027; # '          $self->{entity_add} = 0x0027; # '
1871          $self->{prev_state} = $self->{state};          $self->{prev_state} = $self->{state};
1872          $self->{state} = ENTITY_STATE;          $self->{state} = ENTITY_STATE;
1873          !!!next-input-character;          !!!next-input-character;
1874          redo A;          redo A;
1875        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1876          !!!parse-error (type => 'unclosed attribute value');          !!!parse-error (type => 'unclosed attribute value');
1877          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1878            !!!cp (103);            !!!cp (103);
1879            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1880          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1881            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1882            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1883              !!!cp (104);              !!!cp (104);
1884              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1885            } else {            } else {
# Line 1805  sub _get_next_token ($) { Line 1887  sub _get_next_token ($) {
1887              !!!cp (105);              !!!cp (105);
1888            }            }
1889          } else {          } else {
1890            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1891          }          }
1892          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1893          ## reconsume          ## reconsume
1894    
1895          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1896    
1897          redo A;          redo A;
1898        } else {        } else {
1899          !!!cp (106);          !!!cp (106);
1900          $self->{current_attribute}->{value} .= chr ($self->{next_char});          $self->{ca}->{value} .= chr ($self->{nc});
1901          $self->{read_until}->($self->{current_attribute}->{value},          $self->{read_until}->($self->{ca}->{value},
1902                                q['&],                                q['&],
1903                                length $self->{current_attribute}->{value});                                length $self->{ca}->{value});
1904    
1905          ## Stay in the state          ## Stay in the state
1906          !!!next-input-character;          !!!next-input-character;
1907          redo A;          redo A;
1908        }        }
1909      } elsif ($self->{state} == ATTRIBUTE_VALUE_UNQUOTED_STATE) {      } elsif ($self->{state} == ATTRIBUTE_VALUE_UNQUOTED_STATE) {
1910        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # HT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
1911          !!!cp (107);          !!!cp (107);
1912          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1913          !!!next-input-character;          !!!next-input-character;
1914          redo A;          redo A;
1915        } elsif ($self->{next_char} == 0x0026) { # &        } elsif ($self->{nc} == 0x0026) { # &
1916          !!!cp (108);          !!!cp (108);
1917          ## NOTE: In the spec, the tokenizer is switched to the          ## NOTE: In the spec, the tokenizer is switched to the
1918          ## "entity in attribute value state".  In this implementation, the          ## "entity in attribute value state".  In this implementation, the
1919          ## tokenizer is switched to the |ENTITY_STATE|, which is an          ## tokenizer is switched to the |ENTITY_STATE|, which is an
1920          ## implementation of the "consume a character reference" algorithm.          ## implementation of the "consume a character reference" algorithm.
1921          $self->{entity_additional} = -1;          $self->{entity_add} = -1;
1922          $self->{prev_state} = $self->{state};          $self->{prev_state} = $self->{state};
1923          $self->{state} = ENTITY_STATE;          $self->{state} = ENTITY_STATE;
1924          !!!next-input-character;          !!!next-input-character;
1925          redo A;          redo A;
1926        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1927          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1928            !!!cp (109);            !!!cp (109);
1929            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1930          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1931            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1932            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1933              !!!cp (110);              !!!cp (110);
1934              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1935            } else {            } else {
# Line 1859  sub _get_next_token ($) { Line 1937  sub _get_next_token ($) {
1937              !!!cp (111);              !!!cp (111);
1938            }            }
1939          } else {          } else {
1940            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1941          }          }
1942          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1943          !!!next-input-character;          !!!next-input-character;
1944    
1945          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1946    
1947          redo A;          redo A;
1948        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1949          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1950          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1951            !!!cp (112);            !!!cp (112);
1952            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1953          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1954            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1955            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1956              !!!cp (113);              !!!cp (113);
1957              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1958            } else {            } else {
# Line 1882  sub _get_next_token ($) { Line 1960  sub _get_next_token ($) {
1960              !!!cp (114);              !!!cp (114);
1961            }            }
1962          } else {          } else {
1963            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1964          }          }
1965          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1966          ## reconsume          ## reconsume
1967    
1968          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1969    
1970          redo A;          redo A;
1971        } else {        } else {
# Line 1895  sub _get_next_token ($) { Line 1973  sub _get_next_token ($) {
1973               0x0022 => 1, # "               0x0022 => 1, # "
1974               0x0027 => 1, # '               0x0027 => 1, # '
1975               0x003D => 1, # =               0x003D => 1, # =
1976              }->{$self->{next_char}}) {              }->{$self->{nc}}) {
1977            !!!cp (115);            !!!cp (115);
1978            !!!parse-error (type => 'bad attribute value');            !!!parse-error (type => 'bad attribute value');
1979          } else {          } else {
1980            !!!cp (116);            !!!cp (116);
1981          }          }
1982          $self->{current_attribute}->{value} .= chr ($self->{next_char});          $self->{ca}->{value} .= chr ($self->{nc});
1983          $self->{read_until}->($self->{current_attribute}->{value},          $self->{read_until}->($self->{ca}->{value},
1984                                q["'=& >],                                q["'=& >],
1985                                length $self->{current_attribute}->{value});                                length $self->{ca}->{value});
1986    
1987          ## Stay in the state          ## Stay in the state
1988          !!!next-input-character;          !!!next-input-character;
1989          redo A;          redo A;
1990        }        }
1991      } elsif ($self->{state} == AFTER_ATTRIBUTE_VALUE_QUOTED_STATE) {      } elsif ($self->{state} == AFTER_ATTRIBUTE_VALUE_QUOTED_STATE) {
1992        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
1993          !!!cp (118);          !!!cp (118);
1994          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1995          !!!next-input-character;          !!!next-input-character;
1996          redo A;          redo A;
1997        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1998          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1999            !!!cp (119);            !!!cp (119);
2000            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
2001          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
2002            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
2003            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
2004              !!!cp (120);              !!!cp (120);
2005              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
2006            } else {            } else {
# Line 1934  sub _get_next_token ($) { Line 2008  sub _get_next_token ($) {
2008              !!!cp (121);              !!!cp (121);
2009            }            }
2010          } else {          } else {
2011            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
2012          }          }
2013          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2014          !!!next-input-character;          !!!next-input-character;
2015    
2016          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
2017    
2018          redo A;          redo A;
2019        } elsif ($self->{next_char} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
2020          !!!cp (122);          !!!cp (122);
2021          $self->{state} = SELF_CLOSING_START_TAG_STATE;          $self->{state} = SELF_CLOSING_START_TAG_STATE;
2022          !!!next-input-character;          !!!next-input-character;
2023          redo A;          redo A;
2024        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2025          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
2026          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
2027            !!!cp (122.3);            !!!cp (122.3);
2028            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
2029          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
2030            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
2031              !!!cp (122.1);              !!!cp (122.1);
2032              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
2033            } else {            } else {
# Line 1961  sub _get_next_token ($) { Line 2035  sub _get_next_token ($) {
2035              !!!cp (122.2);              !!!cp (122.2);
2036            }            }
2037          } else {          } else {
2038            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
2039          }          }
2040          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2041          ## Reconsume.          ## Reconsume.
2042          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
2043          redo A;          redo A;
2044        } else {        } else {
2045          !!!cp ('124.1');          !!!cp ('124.1');
# Line 1975  sub _get_next_token ($) { Line 2049  sub _get_next_token ($) {
2049          redo A;          redo A;
2050        }        }
2051      } elsif ($self->{state} == SELF_CLOSING_START_TAG_STATE) {      } elsif ($self->{state} == SELF_CLOSING_START_TAG_STATE) {
2052        if ($self->{next_char} == 0x003E) { # >        if ($self->{nc} == 0x003E) { # >
2053          if ($self->{current_token}->{type} == END_TAG_TOKEN) {          if ($self->{ct}->{type} == END_TAG_TOKEN) {
2054            !!!cp ('124.2');            !!!cp ('124.2');
2055            !!!parse-error (type => 'nestc', token => $self->{current_token});            !!!parse-error (type => 'nestc', token => $self->{ct});
2056            ## TODO: Different type than slash in start tag            ## TODO: Different type than slash in start tag
2057            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
2058            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
2059              !!!cp ('124.4');              !!!cp ('124.4');
2060              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
2061            } else {            } else {
# Line 1996  sub _get_next_token ($) { Line 2070  sub _get_next_token ($) {
2070          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2071          !!!next-input-character;          !!!next-input-character;
2072    
2073          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
2074    
2075          redo A;          redo A;
2076        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2077          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
2078          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
2079            !!!cp (124.7);            !!!cp (124.7);
2080            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
2081          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
2082            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
2083              !!!cp (124.5);              !!!cp (124.5);
2084              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
2085            } else {            } else {
# Line 2013  sub _get_next_token ($) { Line 2087  sub _get_next_token ($) {
2087              !!!cp (124.6);              !!!cp (124.6);
2088            }            }
2089          } else {          } else {
2090            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
2091          }          }
2092          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2093          ## Reconsume.          ## Reconsume.
2094          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
2095          redo A;          redo A;
2096        } else {        } else {
2097          !!!cp ('124.4');          !!!cp ('124.4');
# Line 2033  sub _get_next_token ($) { Line 2107  sub _get_next_token ($) {
2107        ## NOTE: Unlike spec's "bogus comment state", this implementation        ## NOTE: Unlike spec's "bogus comment state", this implementation
2108        ## consumes characters one-by-one basis.        ## consumes characters one-by-one basis.
2109                
2110        if ($self->{next_char} == 0x003E) { # >        if ($self->{nc} == 0x003E) { # >
2111          !!!cp (124);          !!!cp (124);
2112          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2113          !!!next-input-character;          !!!next-input-character;
2114    
2115          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2116          redo A;          redo A;
2117        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2118          !!!cp (125);          !!!cp (125);
2119          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2120          ## reconsume          ## reconsume
2121    
2122          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2123          redo A;          redo A;
2124        } else {        } else {
2125          !!!cp (126);          !!!cp (126);
2126          $self->{current_token}->{data} .= chr ($self->{next_char}); # comment          $self->{ct}->{data} .= chr ($self->{nc}); # comment
2127          $self->{read_until}->($self->{current_token}->{data},          $self->{read_until}->($self->{ct}->{data},
2128                                q[>],                                q[>],
2129                                length $self->{current_token}->{data});                                length $self->{ct}->{data});
2130    
2131          ## Stay in the state.          ## Stay in the state.
2132          !!!next-input-character;          !!!next-input-character;
# Line 2061  sub _get_next_token ($) { Line 2135  sub _get_next_token ($) {
2135      } elsif ($self->{state} == MARKUP_DECLARATION_OPEN_STATE) {      } elsif ($self->{state} == MARKUP_DECLARATION_OPEN_STATE) {
2136        ## (only happen if PCDATA state)        ## (only happen if PCDATA state)
2137                
2138        if ($self->{next_char} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2139          !!!cp (133);          !!!cp (133);
2140          $self->{state} = MD_HYPHEN_STATE;          $self->{state} = MD_HYPHEN_STATE;
2141          !!!next-input-character;          !!!next-input-character;
2142          redo A;          redo A;
2143        } elsif ($self->{next_char} == 0x0044 or # D        } elsif ($self->{nc} == 0x0044 or # D
2144                 $self->{next_char} == 0x0064) { # d                 $self->{nc} == 0x0064) { # d
2145          ## ASCII case-insensitive.          ## ASCII case-insensitive.
2146          !!!cp (130);          !!!cp (130);
2147          $self->{state} = MD_DOCTYPE_STATE;          $self->{state} = MD_DOCTYPE_STATE;
2148          $self->{state_keyword} = chr $self->{next_char};          $self->{s_kwd} = chr $self->{nc};
2149          !!!next-input-character;          !!!next-input-character;
2150          redo A;          redo A;
2151        } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM and        } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM and
2152                 $self->{open_elements}->[-1]->[1] & FOREIGN_EL and                 $self->{open_elements}->[-1]->[1] & FOREIGN_EL and
2153                 $self->{next_char} == 0x005B) { # [                 $self->{nc} == 0x005B) { # [
2154          !!!cp (135.4);                          !!!cp (135.4);                
2155          $self->{state} = MD_CDATA_STATE;          $self->{state} = MD_CDATA_STATE;
2156          $self->{state_keyword} = '[';          $self->{s_kwd} = '[';
2157          !!!next-input-character;          !!!next-input-character;
2158          redo A;          redo A;
2159        } else {        } else {
# Line 2091  sub _get_next_token ($) { Line 2165  sub _get_next_token ($) {
2165                        column => $self->{column_prev} - 1);                        column => $self->{column_prev} - 1);
2166        ## Reconsume.        ## Reconsume.
2167        $self->{state} = BOGUS_COMMENT_STATE;        $self->{state} = BOGUS_COMMENT_STATE;
2168        $self->{current_token} = {type => COMMENT_TOKEN, data => '',        $self->{ct} = {type => COMMENT_TOKEN, data => '',
2169                                  line => $self->{line_prev},                                  line => $self->{line_prev},
2170                                  column => $self->{column_prev} - 1,                                  column => $self->{column_prev} - 1,
2171                                 };                                 };
2172        redo A;        redo A;
2173      } elsif ($self->{state} == MD_HYPHEN_STATE) {      } elsif ($self->{state} == MD_HYPHEN_STATE) {
2174        if ($self->{next_char} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2175          !!!cp (127);          !!!cp (127);
2176          $self->{current_token} = {type => COMMENT_TOKEN, data => '',          $self->{ct} = {type => COMMENT_TOKEN, data => '',
2177                                    line => $self->{line_prev},                                    line => $self->{line_prev},
2178                                    column => $self->{column_prev} - 2,                                    column => $self->{column_prev} - 2,
2179                                   };                                   };
# Line 2113  sub _get_next_token ($) { Line 2187  sub _get_next_token ($) {
2187                          column => $self->{column_prev} - 2);                          column => $self->{column_prev} - 2);
2188          $self->{state} = BOGUS_COMMENT_STATE;          $self->{state} = BOGUS_COMMENT_STATE;
2189          ## Reconsume.          ## Reconsume.
2190          $self->{current_token} = {type => COMMENT_TOKEN,          $self->{ct} = {type => COMMENT_TOKEN,
2191                                    data => '-',                                    data => '-',
2192                                    line => $self->{line_prev},                                    line => $self->{line_prev},
2193                                    column => $self->{column_prev} - 2,                                    column => $self->{column_prev} - 2,
# Line 2122  sub _get_next_token ($) { Line 2196  sub _get_next_token ($) {
2196        }        }
2197      } elsif ($self->{state} == MD_DOCTYPE_STATE) {      } elsif ($self->{state} == MD_DOCTYPE_STATE) {
2198        ## ASCII case-insensitive.        ## ASCII case-insensitive.
2199        if ($self->{next_char} == [        if ($self->{nc} == [
2200              undef,              undef,
2201              0x004F, # O              0x004F, # O
2202              0x0043, # C              0x0043, # C
2203              0x0054, # T              0x0054, # T
2204              0x0059, # Y              0x0059, # Y
2205              0x0050, # P              0x0050, # P
2206            ]->[length $self->{state_keyword}] or            ]->[length $self->{s_kwd}] or
2207            $self->{next_char} == [            $self->{nc} == [
2208              undef,              undef,
2209              0x006F, # o              0x006F, # o
2210              0x0063, # c              0x0063, # c
2211              0x0074, # t              0x0074, # t
2212              0x0079, # y              0x0079, # y
2213              0x0070, # p              0x0070, # p
2214            ]->[length $self->{state_keyword}]) {            ]->[length $self->{s_kwd}]) {
2215          !!!cp (131);          !!!cp (131);
2216          ## Stay in the state.          ## Stay in the state.
2217          $self->{state_keyword} .= chr $self->{next_char};          $self->{s_kwd} .= chr $self->{nc};
2218          !!!next-input-character;          !!!next-input-character;
2219          redo A;          redo A;
2220        } elsif ((length $self->{state_keyword}) == 6 and        } elsif ((length $self->{s_kwd}) == 6 and
2221                 ($self->{next_char} == 0x0045 or # E                 ($self->{nc} == 0x0045 or # E
2222                  $self->{next_char} == 0x0065)) { # e                  $self->{nc} == 0x0065)) { # e
2223          !!!cp (129);          !!!cp (129);
2224          $self->{state} = DOCTYPE_STATE;          $self->{state} = DOCTYPE_STATE;
2225          $self->{current_token} = {type => DOCTYPE_TOKEN,          $self->{ct} = {type => DOCTYPE_TOKEN,
2226                                    quirks => 1,                                    quirks => 1,
2227                                    line => $self->{line_prev},                                    line => $self->{line_prev},
2228                                    column => $self->{column_prev} - 7,                                    column => $self->{column_prev} - 7,
# Line 2159  sub _get_next_token ($) { Line 2233  sub _get_next_token ($) {
2233          !!!cp (132);                  !!!cp (132);        
2234          !!!parse-error (type => 'bogus comment',          !!!parse-error (type => 'bogus comment',
2235                          line => $self->{line_prev},                          line => $self->{line_prev},
2236                          column => $self->{column_prev} - 1 - length $self->{state_keyword});                          column => $self->{column_prev} - 1 - length $self->{s_kwd});
2237          $self->{state} = BOGUS_COMMENT_STATE;          $self->{state} = BOGUS_COMMENT_STATE;
2238          ## Reconsume.          ## Reconsume.
2239          $self->{current_token} = {type => COMMENT_TOKEN,          $self->{ct} = {type => COMMENT_TOKEN,
2240                                    data => $self->{state_keyword},                                    data => $self->{s_kwd},
2241                                    line => $self->{line_prev},                                    line => $self->{line_prev},
2242                                    column => $self->{column_prev} - 1 - length $self->{state_keyword},                                    column => $self->{column_prev} - 1 - length $self->{s_kwd},
2243                                   };                                   };
2244          redo A;          redo A;
2245        }        }
2246      } elsif ($self->{state} == MD_CDATA_STATE) {      } elsif ($self->{state} == MD_CDATA_STATE) {
2247        if ($self->{next_char} == {        if ($self->{nc} == {
2248              '[' => 0x0043, # C              '[' => 0x0043, # C
2249              '[C' => 0x0044, # D              '[C' => 0x0044, # D
2250              '[CD' => 0x0041, # A              '[CD' => 0x0041, # A
2251              '[CDA' => 0x0054, # T              '[CDA' => 0x0054, # T
2252              '[CDAT' => 0x0041, # A              '[CDAT' => 0x0041, # A
2253            }->{$self->{state_keyword}}) {            }->{$self->{s_kwd}}) {
2254          !!!cp (135.1);          !!!cp (135.1);
2255          ## Stay in the state.          ## Stay in the state.
2256          $self->{state_keyword} .= chr $self->{next_char};          $self->{s_kwd} .= chr $self->{nc};
2257          !!!next-input-character;          !!!next-input-character;
2258          redo A;          redo A;
2259        } elsif ($self->{state_keyword} eq '[CDATA' and        } elsif ($self->{s_kwd} eq '[CDATA' and
2260                 $self->{next_char} == 0x005B) { # [                 $self->{nc} == 0x005B) { # [
2261          !!!cp (135.2);          !!!cp (135.2);
2262          $self->{current_token} = {type => CHARACTER_TOKEN,          $self->{ct} = {type => CHARACTER_TOKEN,
2263                                    data => '',                                    data => '',
2264                                    line => $self->{line_prev},                                    line => $self->{line_prev},
2265                                    column => $self->{column_prev} - 7};                                    column => $self->{column_prev} - 7};
# Line 2196  sub _get_next_token ($) { Line 2270  sub _get_next_token ($) {
2270          !!!cp (135.3);          !!!cp (135.3);
2271          !!!parse-error (type => 'bogus comment',          !!!parse-error (type => 'bogus comment',
2272                          line => $self->{line_prev},                          line => $self->{line_prev},
2273                          column => $self->{column_prev} - 1 - length $self->{state_keyword});                          column => $self->{column_prev} - 1 - length $self->{s_kwd});
2274          $self->{state} = BOGUS_COMMENT_STATE;          $self->{state} = BOGUS_COMMENT_STATE;
2275          ## Reconsume.          ## Reconsume.
2276          $self->{current_token} = {type => COMMENT_TOKEN,          $self->{ct} = {type => COMMENT_TOKEN,
2277                                    data => $self->{state_keyword},                                    data => $self->{s_kwd},
2278                                    line => $self->{line_prev},                                    line => $self->{line_prev},
2279                                    column => $self->{column_prev} - 1 - length $self->{state_keyword},                                    column => $self->{column_prev} - 1 - length $self->{s_kwd},
2280                                   };                                   };
2281          redo A;          redo A;
2282        }        }
2283      } elsif ($self->{state} == COMMENT_START_STATE) {      } elsif ($self->{state} == COMMENT_START_STATE) {
2284        if ($self->{next_char} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2285          !!!cp (137);          !!!cp (137);
2286          $self->{state} = COMMENT_START_DASH_STATE;          $self->{state} = COMMENT_START_DASH_STATE;
2287          !!!next-input-character;          !!!next-input-character;
2288          redo A;          redo A;
2289        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2290          !!!cp (138);          !!!cp (138);
2291          !!!parse-error (type => 'bogus comment');          !!!parse-error (type => 'bogus comment');
2292          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2293          !!!next-input-character;          !!!next-input-character;
2294    
2295          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2296    
2297          redo A;          redo A;
2298        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2299          !!!cp (139);          !!!cp (139);
2300          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2301          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2302          ## reconsume          ## reconsume
2303    
2304          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2305    
2306          redo A;          redo A;
2307        } else {        } else {
2308          !!!cp (140);          !!!cp (140);
2309          $self->{current_token}->{data} # comment          $self->{ct}->{data} # comment
2310              .= chr ($self->{next_char});              .= chr ($self->{nc});
2311          $self->{state} = COMMENT_STATE;          $self->{state} = COMMENT_STATE;
2312          !!!next-input-character;          !!!next-input-character;
2313          redo A;          redo A;
2314        }        }
2315      } elsif ($self->{state} == COMMENT_START_DASH_STATE) {      } elsif ($self->{state} == COMMENT_START_DASH_STATE) {
2316        if ($self->{next_char} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2317          !!!cp (141);          !!!cp (141);
2318          $self->{state} = COMMENT_END_STATE;          $self->{state} = COMMENT_END_STATE;
2319          !!!next-input-character;          !!!next-input-character;
2320          redo A;          redo A;
2321        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2322          !!!cp (142);          !!!cp (142);
2323          !!!parse-error (type => 'bogus comment');          !!!parse-error (type => 'bogus comment');
2324          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2325          !!!next-input-character;          !!!next-input-character;
2326    
2327          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2328    
2329          redo A;          redo A;
2330        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2331          !!!cp (143);          !!!cp (143);
2332          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2333          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2334          ## reconsume          ## reconsume
2335    
2336          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2337    
2338          redo A;          redo A;
2339        } else {        } else {
2340          !!!cp (144);          !!!cp (144);
2341          $self->{current_token}->{data} # comment          $self->{ct}->{data} # comment
2342              .= '-' . chr ($self->{next_char});              .= '-' . chr ($self->{nc});
2343          $self->{state} = COMMENT_STATE;          $self->{state} = COMMENT_STATE;
2344          !!!next-input-character;          !!!next-input-character;
2345          redo A;          redo A;
2346        }        }
2347      } elsif ($self->{state} == COMMENT_STATE) {      } elsif ($self->{state} == COMMENT_STATE) {
2348        if ($self->{next_char} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2349          !!!cp (145);          !!!cp (145);
2350          $self->{state} = COMMENT_END_DASH_STATE;          $self->{state} = COMMENT_END_DASH_STATE;
2351          !!!next-input-character;          !!!next-input-character;
2352          redo A;          redo A;
2353        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2354          !!!cp (146);          !!!cp (146);
2355          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2356          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2357          ## reconsume          ## reconsume
2358    
2359          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2360    
2361          redo A;          redo A;
2362        } else {        } else {
2363          !!!cp (147);          !!!cp (147);
2364          $self->{current_token}->{data} .= chr ($self->{next_char}); # comment          $self->{ct}->{data} .= chr ($self->{nc}); # comment
2365          $self->{read_until}->($self->{current_token}->{data},          $self->{read_until}->($self->{ct}->{data},
2366                                q[-],                                q[-],
2367                                length $self->{current_token}->{data});                                length $self->{ct}->{data});
2368    
2369          ## Stay in the state          ## Stay in the state
2370          !!!next-input-character;          !!!next-input-character;
2371          redo A;          redo A;
2372        }        }
2373      } elsif ($self->{state} == COMMENT_END_DASH_STATE) {      } elsif ($self->{state} == COMMENT_END_DASH_STATE) {
2374        if ($self->{next_char} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2375          !!!cp (148);          !!!cp (148);
2376          $self->{state} = COMMENT_END_STATE;          $self->{state} = COMMENT_END_STATE;
2377          !!!next-input-character;          !!!next-input-character;
2378          redo A;          redo A;
2379        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2380          !!!cp (149);          !!!cp (149);
2381          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2382          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2383          ## reconsume          ## reconsume
2384    
2385          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2386    
2387          redo A;          redo A;
2388        } else {        } else {
2389          !!!cp (150);          !!!cp (150);
2390          $self->{current_token}->{data} .= '-' . chr ($self->{next_char}); # comment          $self->{ct}->{data} .= '-' . chr ($self->{nc}); # comment
2391          $self->{state} = COMMENT_STATE;          $self->{state} = COMMENT_STATE;
2392          !!!next-input-character;          !!!next-input-character;
2393          redo A;          redo A;
2394        }        }
2395      } elsif ($self->{state} == COMMENT_END_STATE) {      } elsif ($self->{state} == COMMENT_END_STATE) {
2396        if ($self->{next_char} == 0x003E) { # >        if ($self->{nc} == 0x003E) { # >
2397          !!!cp (151);          !!!cp (151);
2398          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2399          !!!next-input-character;          !!!next-input-character;
2400    
2401          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2402    
2403          redo A;          redo A;
2404        } elsif ($self->{next_char} == 0x002D) { # -        } elsif ($self->{nc} == 0x002D) { # -
2405          !!!cp (152);          !!!cp (152);
2406          !!!parse-error (type => 'dash in comment',          !!!parse-error (type => 'dash in comment',
2407                          line => $self->{line_prev},                          line => $self->{line_prev},
2408                          column => $self->{column_prev});                          column => $self->{column_prev});
2409          $self->{current_token}->{data} .= '-'; # comment          $self->{ct}->{data} .= '-'; # comment
2410          ## Stay in the state          ## Stay in the state
2411          !!!next-input-character;          !!!next-input-character;
2412          redo A;          redo A;
2413        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2414          !!!cp (153);          !!!cp (153);
2415          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2416          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2417          ## reconsume          ## reconsume
2418    
2419          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2420    
2421          redo A;          redo A;
2422        } else {        } else {
# Line 2350  sub _get_next_token ($) { Line 2424  sub _get_next_token ($) {
2424          !!!parse-error (type => 'dash in comment',          !!!parse-error (type => 'dash in comment',
2425                          line => $self->{line_prev},                          line => $self->{line_prev},
2426                          column => $self->{column_prev});                          column => $self->{column_prev});
2427          $self->{current_token}->{data} .= '--' . chr ($self->{next_char}); # comment          $self->{ct}->{data} .= '--' . chr ($self->{nc}); # comment
2428          $self->{state} = COMMENT_STATE;          $self->{state} = COMMENT_STATE;
2429          !!!next-input-character;          !!!next-input-character;
2430          redo A;          redo A;
2431        }        }
2432      } elsif ($self->{state} == DOCTYPE_STATE) {      } elsif ($self->{state} == DOCTYPE_STATE) {
2433        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
2434          !!!cp (155);          !!!cp (155);
2435          $self->{state} = BEFORE_DOCTYPE_NAME_STATE;          $self->{state} = BEFORE_DOCTYPE_NAME_STATE;
2436          !!!next-input-character;          !!!next-input-character;
# Line 2373  sub _get_next_token ($) { Line 2443  sub _get_next_token ($) {
2443          redo A;          redo A;
2444        }        }
2445      } elsif ($self->{state} == BEFORE_DOCTYPE_NAME_STATE) {      } elsif ($self->{state} == BEFORE_DOCTYPE_NAME_STATE) {
2446        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
2447          !!!cp (157);          !!!cp (157);
2448          ## Stay in the state          ## Stay in the state
2449          !!!next-input-character;          !!!next-input-character;
2450          redo A;          redo A;
2451        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2452          !!!cp (158);          !!!cp (158);
2453          !!!parse-error (type => 'no DOCTYPE name');          !!!parse-error (type => 'no DOCTYPE name');
2454          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2455          !!!next-input-character;          !!!next-input-character;
2456    
2457          !!!emit ($self->{current_token}); # DOCTYPE (quirks)          !!!emit ($self->{ct}); # DOCTYPE (quirks)
2458    
2459          redo A;          redo A;
2460        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2461          !!!cp (159);          !!!cp (159);
2462          !!!parse-error (type => 'no DOCTYPE name');          !!!parse-error (type => 'no DOCTYPE name');
2463          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2464          ## reconsume          ## reconsume
2465    
2466          !!!emit ($self->{current_token}); # DOCTYPE (quirks)          !!!emit ($self->{ct}); # DOCTYPE (quirks)
2467    
2468          redo A;          redo A;
2469        } else {        } else {
2470          !!!cp (160);          !!!cp (160);
2471          $self->{current_token}->{name} = chr $self->{next_char};          $self->{ct}->{name} = chr $self->{nc};
2472          delete $self->{current_token}->{quirks};          delete $self->{ct}->{quirks};
2473  ## ISSUE: "Set the token's name name to the" in the spec  ## ISSUE: "Set the token's name name to the" in the spec
2474          $self->{state} = DOCTYPE_NAME_STATE;          $self->{state} = DOCTYPE_NAME_STATE;
2475          !!!next-input-character;          !!!next-input-character;
# Line 2411  sub _get_next_token ($) { Line 2477  sub _get_next_token ($) {
2477        }        }
2478      } elsif ($self->{state} == DOCTYPE_NAME_STATE) {      } elsif ($self->{state} == DOCTYPE_NAME_STATE) {
2479  ## ISSUE: Redundant "First," in the spec.  ## ISSUE: Redundant "First," in the spec.
2480        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
2481          !!!cp (161);          !!!cp (161);
2482          $self->{state} = AFTER_DOCTYPE_NAME_STATE;          $self->{state} = AFTER_DOCTYPE_NAME_STATE;
2483          !!!next-input-character;          !!!next-input-character;
2484          redo A;          redo A;
2485        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2486          !!!cp (162);          !!!cp (162);
2487          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2488          !!!next-input-character;          !!!next-input-character;
2489    
2490          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2491    
2492          redo A;          redo A;
2493        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2494          !!!cp (163);          !!!cp (163);
2495          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2496          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2497          ## reconsume          ## reconsume
2498    
2499          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2500          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2501    
2502          redo A;          redo A;
2503        } else {        } else {
2504          !!!cp (164);          !!!cp (164);
2505          $self->{current_token}->{name}          $self->{ct}->{name}
2506            .= chr ($self->{next_char}); # DOCTYPE            .= chr ($self->{nc}); # DOCTYPE
2507          ## Stay in the state          ## Stay in the state
2508          !!!next-input-character;          !!!next-input-character;
2509          redo A;          redo A;
2510        }        }
2511      } elsif ($self->{state} == AFTER_DOCTYPE_NAME_STATE) {      } elsif ($self->{state} == AFTER_DOCTYPE_NAME_STATE) {
2512        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
2513          !!!cp (165);          !!!cp (165);
2514          ## Stay in the state          ## Stay in the state
2515          !!!next-input-character;          !!!next-input-character;
2516          redo A;          redo A;
2517        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2518          !!!cp (166);          !!!cp (166);
2519          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2520          !!!next-input-character;          !!!next-input-character;
2521    
2522          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2523    
2524          redo A;          redo A;
2525        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2526          !!!cp (167);          !!!cp (167);
2527          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2528          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2529          ## reconsume          ## reconsume
2530    
2531          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2532          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2533    
2534          redo A;          redo A;
2535        } elsif ($self->{next_char} == 0x0050 or # P        } elsif ($self->{nc} == 0x0050 or # P
2536                 $self->{next_char} == 0x0070) { # p                 $self->{nc} == 0x0070) { # p
2537          $self->{state} = PUBLIC_STATE;          $self->{state} = PUBLIC_STATE;
2538          $self->{state_keyword} = chr $self->{next_char};          $self->{s_kwd} = chr $self->{nc};
2539          !!!next-input-character;          !!!next-input-character;
2540          redo A;          redo A;
2541        } elsif ($self->{next_char} == 0x0053 or # S        } elsif ($self->{nc} == 0x0053 or # S
2542                 $self->{next_char} == 0x0073) { # s                 $self->{nc} == 0x0073) { # s
2543          $self->{state} = SYSTEM_STATE;          $self->{state} = SYSTEM_STATE;
2544          $self->{state_keyword} = chr $self->{next_char};          $self->{s_kwd} = chr $self->{nc};
2545          !!!next-input-character;          !!!next-input-character;
2546          redo A;          redo A;
2547        } else {        } else {
2548          !!!cp (180);          !!!cp (180);
2549          !!!parse-error (type => 'string after DOCTYPE name');          !!!parse-error (type => 'string after DOCTYPE name');
2550          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2551    
2552          $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2553          !!!next-input-character;          !!!next-input-character;
# Line 2497  sub _get_next_token ($) { Line 2555  sub _get_next_token ($) {
2555        }        }
2556      } elsif ($self->{state} == PUBLIC_STATE) {      } elsif ($self->{state} == PUBLIC_STATE) {
2557        ## ASCII case-insensitive        ## ASCII case-insensitive
2558        if ($self->{next_char} == [        if ($self->{nc} == [
2559              undef,              undef,
2560              0x0055, # U              0x0055, # U
2561              0x0042, # B              0x0042, # B
2562              0x004C, # L              0x004C, # L
2563              0x0049, # I              0x0049, # I
2564            ]->[length $self->{state_keyword}] or            ]->[length $self->{s_kwd}] or
2565            $self->{next_char} == [            $self->{nc} == [
2566              undef,              undef,
2567              0x0075, # u              0x0075, # u
2568              0x0062, # b              0x0062, # b
2569              0x006C, # l              0x006C, # l
2570              0x0069, # i              0x0069, # i
2571            ]->[length $self->{state_keyword}]) {            ]->[length $self->{s_kwd}]) {
2572          !!!cp (175);          !!!cp (175);
2573          ## Stay in the state.          ## Stay in the state.
2574          $self->{state_keyword} .= chr $self->{next_char};          $self->{s_kwd} .= chr $self->{nc};
2575          !!!next-input-character;          !!!next-input-character;
2576          redo A;          redo A;
2577        } elsif ((length $self->{state_keyword}) == 5 and        } elsif ((length $self->{s_kwd}) == 5 and
2578                 ($self->{next_char} == 0x0043 or # C                 ($self->{nc} == 0x0043 or # C
2579                  $self->{next_char} == 0x0063)) { # c                  $self->{nc} == 0x0063)) { # c
2580          !!!cp (168);          !!!cp (168);
2581          $self->{state} = BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE;          $self->{state} = BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2582          !!!next-input-character;          !!!next-input-character;
# Line 2527  sub _get_next_token ($) { Line 2585  sub _get_next_token ($) {
2585          !!!cp (169);          !!!cp (169);
2586          !!!parse-error (type => 'string after DOCTYPE name',          !!!parse-error (type => 'string after DOCTYPE name',
2587                          line => $self->{line_prev},                          line => $self->{line_prev},
2588                          column => $self->{column_prev} + 1 - length $self->{state_keyword});                          column => $self->{column_prev} + 1 - length $self->{s_kwd});
2589          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2590    
2591          $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2592          ## Reconsume.          ## Reconsume.
# Line 2536  sub _get_next_token ($) { Line 2594  sub _get_next_token ($) {
2594        }        }
2595      } elsif ($self->{state} == SYSTEM_STATE) {      } elsif ($self->{state} == SYSTEM_STATE) {
2596        ## ASCII case-insensitive        ## ASCII case-insensitive
2597        if ($self->{next_char} == [        if ($self->{nc} == [
2598              undef,              undef,
2599              0x0059, # Y              0x0059, # Y
2600              0x0053, # S              0x0053, # S
2601              0x0054, # T              0x0054, # T
2602              0x0045, # E              0x0045, # E
2603            ]->[length $self->{state_keyword}] or            ]->[length $self->{s_kwd}] or
2604            $self->{next_char} == [            $self->{nc} == [
2605              undef,              undef,
2606              0x0079, # y              0x0079, # y
2607              0x0073, # s              0x0073, # s
2608              0x0074, # t              0x0074, # t
2609              0x0065, # e              0x0065, # e
2610            ]->[length $self->{state_keyword}]) {            ]->[length $self->{s_kwd}]) {
2611          !!!cp (170);          !!!cp (170);
2612          ## Stay in the state.          ## Stay in the state.
2613          $self->{state_keyword} .= chr $self->{next_char};          $self->{s_kwd} .= chr $self->{nc};
2614          !!!next-input-character;          !!!next-input-character;
2615          redo A;          redo A;
2616        } elsif ((length $self->{state_keyword}) == 5 and        } elsif ((length $self->{s_kwd}) == 5 and
2617                 ($self->{next_char} == 0x004D or # M                 ($self->{nc} == 0x004D or # M
2618                  $self->{next_char} == 0x006D)) { # m                  $self->{nc} == 0x006D)) { # m
2619          !!!cp (171);          !!!cp (171);
2620          $self->{state} = BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE;          $self->{state} = BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2621          !!!next-input-character;          !!!next-input-character;
# Line 2566  sub _get_next_token ($) { Line 2624  sub _get_next_token ($) {
2624          !!!cp (172);          !!!cp (172);
2625          !!!parse-error (type => 'string after DOCTYPE name',          !!!parse-error (type => 'string after DOCTYPE name',
2626                          line => $self->{line_prev},                          line => $self->{line_prev},
2627                          column => $self->{column_prev} + 1 - length $self->{state_keyword});                          column => $self->{column_prev} + 1 - length $self->{s_kwd});
2628          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2629    
2630          $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2631          ## Reconsume.          ## Reconsume.
2632          redo A;          redo A;
2633        }        }
2634      } elsif ($self->{state} == BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {      } elsif ($self->{state} == BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {
2635        if ({        if ($is_space->{$self->{nc}}) {
             0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,  
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_char}}) {  
2636          !!!cp (181);          !!!cp (181);
2637          ## Stay in the state          ## Stay in the state
2638          !!!next-input-character;          !!!next-input-character;
2639          redo A;          redo A;
2640        } elsif ($self->{next_char} eq 0x0022) { # "        } elsif ($self->{nc} eq 0x0022) { # "
2641          !!!cp (182);          !!!cp (182);
2642          $self->{current_token}->{public_identifier} = ''; # DOCTYPE          $self->{ct}->{pubid} = ''; # DOCTYPE
2643          $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE;          $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE;
2644          !!!next-input-character;          !!!next-input-character;
2645          redo A;          redo A;
2646        } elsif ($self->{next_char} eq 0x0027) { # '        } elsif ($self->{nc} eq 0x0027) { # '
2647          !!!cp (183);          !!!cp (183);
2648          $self->{current_token}->{public_identifier} = ''; # DOCTYPE          $self->{ct}->{pubid} = ''; # DOCTYPE
2649          $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE;          $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE;
2650          !!!next-input-character;          !!!next-input-character;
2651          redo A;          redo A;
2652        } elsif ($self->{next_char} eq 0x003E) { # >        } elsif ($self->{nc} eq 0x003E) { # >
2653          !!!cp (184);          !!!cp (184);
2654          !!!parse-error (type => 'no PUBLIC literal');          !!!parse-error (type => 'no PUBLIC literal');
2655    
2656          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2657          !!!next-input-character;          !!!next-input-character;
2658    
2659          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2660          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2661    
2662          redo A;          redo A;
2663        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2664          !!!cp (185);          !!!cp (185);
2665          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2666    
2667          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2668          ## reconsume          ## reconsume
2669    
2670          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2671          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2672    
2673          redo A;          redo A;
2674        } else {        } else {
2675          !!!cp (186);          !!!cp (186);
2676          !!!parse-error (type => 'string after PUBLIC');          !!!parse-error (type => 'string after PUBLIC');
2677          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2678    
2679          $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2680          !!!next-input-character;          !!!next-input-character;
2681          redo A;          redo A;
2682        }        }
2683      } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE) {      } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE) {
2684        if ($self->{next_char} == 0x0022) { # "        if ($self->{nc} == 0x0022) { # "
2685          !!!cp (187);          !!!cp (187);
2686          $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;          $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2687          !!!next-input-character;          !!!next-input-character;
2688          redo A;          redo A;
2689        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2690          !!!cp (188);          !!!cp (188);
2691          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed PUBLIC literal');
2692    
2693          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2694          !!!next-input-character;          !!!next-input-character;
2695    
2696          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2697          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2698    
2699          redo A;          redo A;
2700        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2701          !!!cp (189);          !!!cp (189);
2702          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed PUBLIC literal');
2703    
2704          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2705          ## reconsume          ## reconsume
2706    
2707          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2708          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2709    
2710          redo A;          redo A;
2711        } else {        } else {
2712          !!!cp (190);          !!!cp (190);
2713          $self->{current_token}->{public_identifier} # DOCTYPE          $self->{ct}->{pubid} # DOCTYPE
2714              .= chr $self->{next_char};              .= chr $self->{nc};
2715          $self->{read_until}->($self->{current_token}->{public_identifier},          $self->{read_until}->($self->{ct}->{pubid}, q[">],
2716                                q[">],                                length $self->{ct}->{pubid});
                               length $self->{current_token}->{public_identifier});  
2717    
2718          ## Stay in the state          ## Stay in the state
2719          !!!next-input-character;          !!!next-input-character;
2720          redo A;          redo A;
2721        }        }
2722      } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE) {      } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE) {
2723        if ($self->{next_char} == 0x0027) { # '        if ($self->{nc} == 0x0027) { # '
2724          !!!cp (191);          !!!cp (191);
2725          $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;          $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2726          !!!next-input-character;          !!!next-input-character;
2727          redo A;          redo A;
2728        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2729          !!!cp (192);          !!!cp (192);
2730          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed PUBLIC literal');
2731    
2732          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2733          !!!next-input-character;          !!!next-input-character;
2734    
2735          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2736          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2737    
2738          redo A;          redo A;
2739        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2740          !!!cp (193);          !!!cp (193);
2741          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed PUBLIC literal');
2742    
2743          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2744          ## reconsume          ## reconsume
2745    
2746          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2747          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2748    
2749          redo A;          redo A;
2750        } else {        } else {
2751          !!!cp (194);          !!!cp (194);
2752          $self->{current_token}->{public_identifier} # DOCTYPE          $self->{ct}->{pubid} # DOCTYPE
2753              .= chr $self->{next_char};              .= chr $self->{nc};
2754          $self->{read_until}->($self->{current_token}->{public_identifier},          $self->{read_until}->($self->{ct}->{pubid}, q['>],
2755                                q['>],                                length $self->{ct}->{pubid});
                               length $self->{current_token}->{public_identifier});  
2756    
2757          ## Stay in the state          ## Stay in the state
2758          !!!next-input-character;          !!!next-input-character;
2759          redo A;          redo A;
2760        }        }
2761      } elsif ($self->{state} == AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {      } elsif ($self->{state} == AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {
2762        if ({        if ($is_space->{$self->{nc}}) {
             0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,  
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_char}}) {  
2763          !!!cp (195);          !!!cp (195);
2764          ## Stay in the state          ## Stay in the state
2765          !!!next-input-character;          !!!next-input-character;
2766          redo A;          redo A;
2767        } elsif ($self->{next_char} == 0x0022) { # "        } elsif ($self->{nc} == 0x0022) { # "
2768          !!!cp (196);          !!!cp (196);
2769          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          $self->{ct}->{sysid} = ''; # DOCTYPE
2770          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;
2771          !!!next-input-character;          !!!next-input-character;
2772          redo A;          redo A;
2773        } elsif ($self->{next_char} == 0x0027) { # '        } elsif ($self->{nc} == 0x0027) { # '
2774          !!!cp (197);          !!!cp (197);
2775          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          $self->{ct}->{sysid} = ''; # DOCTYPE
2776          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;
2777          !!!next-input-character;          !!!next-input-character;
2778          redo A;          redo A;
2779        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2780          !!!cp (198);          !!!cp (198);
2781          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2782          !!!next-input-character;          !!!next-input-character;
2783    
2784          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2785    
2786          redo A;          redo A;
2787        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2788          !!!cp (199);          !!!cp (199);
2789          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2790    
2791          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2792          ## reconsume          ## reconsume
2793    
2794          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2795          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2796    
2797          redo A;          redo A;
2798        } else {        } else {
2799          !!!cp (200);          !!!cp (200);
2800          !!!parse-error (type => 'string after PUBLIC literal');          !!!parse-error (type => 'string after PUBLIC literal');
2801          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2802    
2803          $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2804          !!!next-input-character;          !!!next-input-character;
2805          redo A;          redo A;
2806        }        }
2807      } elsif ($self->{state} == BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {      } elsif ($self->{state} == BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {
2808        if ({        if ($is_space->{$self->{nc}}) {
             0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,  
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_char}}) {  
2809          !!!cp (201);          !!!cp (201);
2810          ## Stay in the state          ## Stay in the state
2811          !!!next-input-character;          !!!next-input-character;
2812          redo A;          redo A;
2813        } elsif ($self->{next_char} == 0x0022) { # "        } elsif ($self->{nc} == 0x0022) { # "
2814          !!!cp (202);          !!!cp (202);
2815          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          $self->{ct}->{sysid} = ''; # DOCTYPE
2816          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;
2817          !!!next-input-character;          !!!next-input-character;
2818          redo A;          redo A;
2819        } elsif ($self->{next_char} == 0x0027) { # '        } elsif ($self->{nc} == 0x0027) { # '
2820          !!!cp (203);          !!!cp (203);
2821          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          $self->{ct}->{sysid} = ''; # DOCTYPE
2822          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;
2823          !!!next-input-character;          !!!next-input-character;
2824          redo A;          redo A;
2825        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2826          !!!cp (204);          !!!cp (204);
2827          !!!parse-error (type => 'no SYSTEM literal');          !!!parse-error (type => 'no SYSTEM literal');
2828          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2829          !!!next-input-character;          !!!next-input-character;
2830    
2831          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2832          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2833    
2834          redo A;          redo A;
2835        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2836          !!!cp (205);          !!!cp (205);
2837          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2838    
2839          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2840          ## reconsume          ## reconsume
2841    
2842          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2843          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2844    
2845          redo A;          redo A;
2846        } else {        } else {
2847          !!!cp (206);          !!!cp (206);
2848          !!!parse-error (type => 'string after SYSTEM');          !!!parse-error (type => 'string after SYSTEM');
2849          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2850    
2851          $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2852          !!!next-input-character;          !!!next-input-character;
2853          redo A;          redo A;
2854        }        }
2855      } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE) {      } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE) {
2856        if ($self->{next_char} == 0x0022) { # "        if ($self->{nc} == 0x0022) { # "
2857          !!!cp (207);          !!!cp (207);
2858          $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;          $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2859          !!!next-input-character;          !!!next-input-character;
2860          redo A;          redo A;
2861        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2862          !!!cp (208);          !!!cp (208);
2863          !!!parse-error (type => 'unclosed SYSTEM literal');          !!!parse-error (type => 'unclosed SYSTEM literal');
2864    
2865          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2866          !!!next-input-character;          !!!next-input-character;
2867    
2868          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2869          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2870    
2871          redo A;          redo A;
2872        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2873          !!!cp (209);          !!!cp (209);
2874          !!!parse-error (type => 'unclosed SYSTEM literal');          !!!parse-error (type => 'unclosed SYSTEM literal');
2875    
2876          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2877          ## reconsume          ## reconsume
2878    
2879          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2880          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2881    
2882          redo A;          redo A;
2883        } else {        } else {
2884          !!!cp (210);          !!!cp (210);
2885          $self->{current_token}->{system_identifier} # DOCTYPE          $self->{ct}->{sysid} # DOCTYPE
2886              .= chr $self->{next_char};              .= chr $self->{nc};
2887          $self->{read_until}->($self->{current_token}->{system_identifier},          $self->{read_until}->($self->{ct}->{sysid}, q[">],
2888                                q[">],                                length $self->{ct}->{sysid});
                               length $self->{current_token}->{system_identifier});  
2889    
2890          ## Stay in the state          ## Stay in the state
2891          !!!next-input-character;          !!!next-input-character;
2892          redo A;          redo A;
2893        }        }
2894      } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE) {      } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE) {
2895        if ($self->{next_char} == 0x0027) { # '        if ($self->{nc} == 0x0027) { # '
2896          !!!cp (211);          !!!cp (211);
2897          $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;          $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2898          !!!next-input-character;          !!!next-input-character;
2899          redo A;          redo A;
2900        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2901          !!!cp (212);          !!!cp (212);
2902          !!!parse-error (type => 'unclosed SYSTEM literal');          !!!parse-error (type => 'unclosed SYSTEM literal');
2903    
2904          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2905          !!!next-input-character;          !!!next-input-character;
2906    
2907          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2908          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2909    
2910          redo A;          redo A;
2911        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2912          !!!cp (213);          !!!cp (213);
2913          !!!parse-error (type => 'unclosed SYSTEM literal');          !!!parse-error (type => 'unclosed SYSTEM literal');
2914    
2915          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2916          ## reconsume          ## reconsume
2917    
2918          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2919          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2920    
2921          redo A;          redo A;
2922        } else {        } else {
2923          !!!cp (214);          !!!cp (214);
2924          $self->{current_token}->{system_identifier} # DOCTYPE          $self->{ct}->{sysid} # DOCTYPE
2925              .= chr $self->{next_char};              .= chr $self->{nc};
2926          $self->{read_until}->($self->{current_token}->{system_identifier},          $self->{read_until}->($self->{ct}->{sysid}, q['>],
2927                                q['>],                                length $self->{ct}->{sysid});
                               length $self->{current_token}->{system_identifier});  
2928    
2929          ## Stay in the state          ## Stay in the state
2930          !!!next-input-character;          !!!next-input-character;
2931          redo A;          redo A;
2932        }        }
2933      } elsif ($self->{state} == AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {      } elsif ($self->{state} == AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {
2934        if ({        if ($is_space->{$self->{nc}}) {
             0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,  
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_char}}) {  
2935          !!!cp (215);          !!!cp (215);
2936          ## Stay in the state          ## Stay in the state
2937          !!!next-input-character;          !!!next-input-character;
2938          redo A;          redo A;
2939        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2940          !!!cp (216);          !!!cp (216);
2941          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2942          !!!next-input-character;          !!!next-input-character;
2943    
2944          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2945    
2946          redo A;          redo A;
2947        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2948          !!!cp (217);          !!!cp (217);
2949          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2950          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2951          ## reconsume          ## reconsume
2952    
2953          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2954          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2955    
2956          redo A;          redo A;
2957        } else {        } else {
2958          !!!cp (218);          !!!cp (218);
2959          !!!parse-error (type => 'string after SYSTEM literal');          !!!parse-error (type => 'string after SYSTEM literal');
2960          #$self->{current_token}->{quirks} = 1;          #$self->{ct}->{quirks} = 1;
2961    
2962          $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2963          !!!next-input-character;          !!!next-input-character;
2964          redo A;          redo A;
2965        }        }
2966      } elsif ($self->{state} == BOGUS_DOCTYPE_STATE) {      } elsif ($self->{state} == BOGUS_DOCTYPE_STATE) {
2967        if ($self->{next_char} == 0x003E) { # >        if ($self->{nc} == 0x003E) { # >
2968          !!!cp (219);          !!!cp (219);
2969          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2970          !!!next-input-character;          !!!next-input-character;
2971    
2972          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2973    
2974          redo A;          redo A;
2975        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2976          !!!cp (220);          !!!cp (220);
2977          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2978          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2979          ## reconsume          ## reconsume
2980    
2981          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2982    
2983          redo A;          redo A;
2984        } else {        } else {
# Line 2953  sub _get_next_token ($) { Line 2995  sub _get_next_token ($) {
2995        ## by three states, |CDATA_SECTION_STATE|, |CDATA_SECTION_MSE1_STATE|,        ## by three states, |CDATA_SECTION_STATE|, |CDATA_SECTION_MSE1_STATE|,
2996        ## and |CDATA_SECTION_MSE2_STATE|.        ## and |CDATA_SECTION_MSE2_STATE|.
2997                
2998        if ($self->{next_char} == 0x005D) { # ]        if ($self->{nc} == 0x005D) { # ]
2999          !!!cp (221.1);          !!!cp (221.1);
3000          $self->{state} = CDATA_SECTION_MSE1_STATE;          $self->{state} = CDATA_SECTION_MSE1_STATE;
3001          !!!next-input-character;          !!!next-input-character;
3002          redo A;          redo A;
3003        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
3004          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
3005          !!!next-input-character;          !!!next-input-character;
3006          if (length $self->{current_token}->{data}) { # character          if (length $self->{ct}->{data}) { # character
3007            !!!cp (221.2);            !!!cp (221.2);
3008            !!!emit ($self->{current_token}); # character            !!!emit ($self->{ct}); # character
3009          } else {          } else {
3010            !!!cp (221.3);            !!!cp (221.3);
3011            ## No token to emit. $self->{current_token} is discarded.            ## No token to emit. $self->{ct} is discarded.
3012          }                  }        
3013          redo A;          redo A;
3014        } else {        } else {
3015          !!!cp (221.4);          !!!cp (221.4);
3016          $self->{current_token}->{data} .= chr $self->{next_char};          $self->{ct}->{data} .= chr $self->{nc};
3017          $self->{read_until}->($self->{current_token}->{data},          $self->{read_until}->($self->{ct}->{data},
3018                                q<]>,                                q<]>,
3019                                length $self->{current_token}->{data});                                length $self->{ct}->{data});
3020    
3021          ## Stay in the state.          ## Stay in the state.
3022          !!!next-input-character;          !!!next-input-character;
# Line 2983  sub _get_next_token ($) { Line 3025  sub _get_next_token ($) {
3025    
3026        ## ISSUE: "text tokens" in spec.        ## ISSUE: "text tokens" in spec.
3027      } elsif ($self->{state} == CDATA_SECTION_MSE1_STATE) {      } elsif ($self->{state} == CDATA_SECTION_MSE1_STATE) {
3028        if ($self->{next_char} == 0x005D) { # ]        if ($self->{nc} == 0x005D) { # ]
3029          !!!cp (221.5);          !!!cp (221.5);
3030          $self->{state} = CDATA_SECTION_MSE2_STATE;          $self->{state} = CDATA_SECTION_MSE2_STATE;
3031          !!!next-input-character;          !!!next-input-character;
3032          redo A;          redo A;
3033        } else {        } else {
3034          !!!cp (221.6);          !!!cp (221.6);
3035          $self->{current_token}->{data} .= ']';          $self->{ct}->{data} .= ']';
3036          $self->{state} = CDATA_SECTION_STATE;          $self->{state} = CDATA_SECTION_STATE;
3037          ## Reconsume.          ## Reconsume.
3038          redo A;          redo A;
3039        }        }
3040      } elsif ($self->{state} == CDATA_SECTION_MSE2_STATE) {      } elsif ($self->{state} == CDATA_SECTION_MSE2_STATE) {
3041        if ($self->{next_char} == 0x003E) { # >        if ($self->{nc} == 0x003E) { # >
3042          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
3043          !!!next-input-character;          !!!next-input-character;
3044          if (length $self->{current_token}->{data}) { # character          if (length $self->{ct}->{data}) { # character
3045            !!!cp (221.7);            !!!cp (221.7);
3046            !!!emit ($self->{current_token}); # character            !!!emit ($self->{ct}); # character
3047          } else {          } else {
3048            !!!cp (221.8);            !!!cp (221.8);
3049            ## No token to emit. $self->{current_token} is discarded.            ## No token to emit. $self->{ct} is discarded.
3050          }          }
3051          redo A;          redo A;
3052        } elsif ($self->{next_char} == 0x005D) { # ]        } elsif ($self->{nc} == 0x005D) { # ]
3053          !!!cp (221.9); # character          !!!cp (221.9); # character
3054          $self->{current_token}->{data} .= ']'; ## Add first "]" of "]]]".          $self->{ct}->{data} .= ']'; ## Add first "]" of "]]]".
3055          ## Stay in the state.          ## Stay in the state.
3056          !!!next-input-character;          !!!next-input-character;
3057          redo A;          redo A;
3058        } else {        } else {
3059          !!!cp (221.11);          !!!cp (221.11);
3060          $self->{current_token}->{data} .= ']]'; # character          $self->{ct}->{data} .= ']]'; # character
3061          $self->{state} = CDATA_SECTION_STATE;          $self->{state} = CDATA_SECTION_STATE;
3062          ## Reconsume.          ## Reconsume.
3063          redo A;          redo A;
3064        }        }
3065      } elsif ($self->{state} == ENTITY_STATE) {      } elsif ($self->{state} == ENTITY_STATE) {
3066        if ({        if ($is_space->{$self->{nc}} or
3067          0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, # HT, LF, VT, FF,            {
3068          0x0020 => 1, 0x003C => 1, 0x0026 => 1, -1 => 1, # SP, <, &              0x003C => 1, 0x0026 => 1, -1 => 1, # <, &
3069          $self->{entity_additional} => 1,              $self->{entity_add} => 1,
3070        }->{$self->{next_char}}) {            }->{$self->{nc}}) {
3071          !!!cp (1001);          !!!cp (1001);
3072          ## Don't consume          ## Don't consume
3073          ## No error          ## No error
3074          ## Return nothing.          ## Return nothing.
3075          #          #
3076        } elsif ($self->{next_char} == 0x0023) { # #        } elsif ($self->{nc} == 0x0023) { # #
3077          !!!cp (999);          !!!cp (999);
3078          $self->{state} = ENTITY_HASH_STATE;          $self->{state} = ENTITY_HASH_STATE;
3079          $self->{state_keyword} = '#';          $self->{s_kwd} = '#';
3080          !!!next-input-character;          !!!next-input-character;
3081          redo A;          redo A;
3082        } elsif ((0x0041 <= $self->{next_char} and        } elsif ((0x0041 <= $self->{nc} and
3083                  $self->{next_char} <= 0x005A) or # A..Z                  $self->{nc} <= 0x005A) or # A..Z
3084                 (0x0061 <= $self->{next_char} and                 (0x0061 <= $self->{nc} and
3085                  $self->{next_char} <= 0x007A)) { # a..z                  $self->{nc} <= 0x007A)) { # a..z
3086          !!!cp (998);          !!!cp (998);
3087          require Whatpm::_NamedEntityList;          require Whatpm::_NamedEntityList;
3088          $self->{state} = ENTITY_NAME_STATE;          $self->{state} = ENTITY_NAME_STATE;
3089          $self->{state_keyword} = chr $self->{next_char};          $self->{s_kwd} = chr $self->{nc};
3090          $self->{entity__value} = $self->{state_keyword};          $self->{entity__value} = $self->{s_kwd};
3091          $self->{entity__match} = 0;          $self->{entity__match} = 0;
3092          !!!next-input-character;          !!!next-input-character;
3093          redo A;          redo A;
# Line 3073  sub _get_next_token ($) { Line 3115  sub _get_next_token ($) {
3115          redo A;          redo A;
3116        } else {        } else {
3117          !!!cp (996);          !!!cp (996);
3118          $self->{current_attribute}->{value} .= '&';          $self->{ca}->{value} .= '&';
3119          $self->{state} = $self->{prev_state};          $self->{state} = $self->{prev_state};
3120          ## Reconsume.          ## Reconsume.
3121          redo A;          redo A;
3122        }        }
3123      } elsif ($self->{state} == ENTITY_HASH_STATE) {      } elsif ($self->{state} == ENTITY_HASH_STATE) {
3124        if ($self->{next_char} == 0x0078 or # x        if ($self->{nc} == 0x0078 or # x
3125            $self->{next_char} == 0x0058) { # X            $self->{nc} == 0x0058) { # X
3126          !!!cp (995);          !!!cp (995);
3127          $self->{state} = HEXREF_X_STATE;          $self->{state} = HEXREF_X_STATE;
3128          $self->{state_keyword} .= chr $self->{next_char};          $self->{s_kwd} .= chr $self->{nc};
3129          !!!next-input-character;          !!!next-input-character;
3130          redo A;          redo A;
3131        } elsif (0x0030 <= $self->{next_char} and        } elsif (0x0030 <= $self->{nc} and
3132                 $self->{next_char} <= 0x0039) { # 0..9                 $self->{nc} <= 0x0039) { # 0..9
3133          !!!cp (994);          !!!cp (994);
3134          $self->{state} = NCR_NUM_STATE;          $self->{state} = NCR_NUM_STATE;
3135          $self->{state_keyword} = $self->{next_char} - 0x0030;          $self->{s_kwd} = $self->{nc} - 0x0030;
3136          !!!next-input-character;          !!!next-input-character;
3137          redo A;          redo A;
3138        } else {        } else {
# Line 3114  sub _get_next_token ($) { Line 3156  sub _get_next_token ($) {
3156            redo A;            redo A;
3157          } else {          } else {
3158            !!!cp (993);            !!!cp (993);
3159            $self->{current_attribute}->{value} .= '&#';            $self->{ca}->{value} .= '&#';
3160            $self->{state} = $self->{prev_state};            $self->{state} = $self->{prev_state};
3161            ## Reconsume.            ## Reconsume.
3162            redo A;            redo A;
3163          }          }
3164        }        }
3165      } elsif ($self->{state} == NCR_NUM_STATE) {      } elsif ($self->{state} == NCR_NUM_STATE) {
3166        if (0x0030 <= $self->{next_char} and        if (0x0030 <= $self->{nc} and
3167            $self->{next_char} <= 0x0039) { # 0..9            $self->{nc} <= 0x0039) { # 0..9
3168          !!!cp (1012);          !!!cp (1012);
3169          $self->{state_keyword} *= 10;          $self->{s_kwd} *= 10;
3170          $self->{state_keyword} += $self->{next_char} - 0x0030;          $self->{s_kwd} += $self->{nc} - 0x0030;
3171                    
3172          ## Stay in the state.          ## Stay in the state.
3173          !!!next-input-character;          !!!next-input-character;
3174          redo A;          redo A;
3175        } elsif ($self->{next_char} == 0x003B) { # ;        } elsif ($self->{nc} == 0x003B) { # ;
3176          !!!cp (1013);          !!!cp (1013);
3177          !!!next-input-character;          !!!next-input-character;
3178          #          #
# Line 3141  sub _get_next_token ($) { Line 3183  sub _get_next_token ($) {
3183          #          #
3184        }        }
3185    
3186        my $code = $self->{state_keyword};        my $code = $self->{s_kwd};
3187        my $l = $self->{line_prev};        my $l = $self->{line_prev};
3188        my $c = $self->{column_prev};        my $c = $self->{column_prev};
3189        if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) {        if ($charref_map->{$code}) {
3190          !!!cp (1015);          !!!cp (1015);
3191          !!!parse-error (type => 'invalid character reference',          !!!parse-error (type => 'invalid character reference',
3192                          text => (sprintf 'U+%04X', $code),                          text => (sprintf 'U+%04X', $code),
3193                          line => $l, column => $c);                          line => $l, column => $c);
3194          $code = 0xFFFD;          $code = $charref_map->{$code};
3195        } elsif ($code > 0x10FFFF) {        } elsif ($code > 0x10FFFF) {
3196          !!!cp (1016);          !!!cp (1016);
3197          !!!parse-error (type => 'invalid character reference',          !!!parse-error (type => 'invalid character reference',
3198                          text => (sprintf 'U-%08X', $code),                          text => (sprintf 'U-%08X', $code),
3199                          line => $l, column => $c);                          line => $l, column => $c);
3200          $code = 0xFFFD;          $code = 0xFFFD;
       } elsif ($code == 0x000D) {  
         !!!cp (1017);  
         !!!parse-error (type => 'CR character reference',  
                         line => $l, column => $c);  
         $code = 0x000A;  
       } elsif (0x80 <= $code and $code <= 0x9F) {  
         !!!cp (1018);  
         !!!parse-error (type => 'C1 character reference',  
                         text => (sprintf 'U+%04X', $code),  
                         line => $l, column => $c);  
         $code = $c1_entity_char->{$code};  
3201        }        }
3202    
3203        if ($self->{prev_state} == DATA_STATE) {        if ($self->{prev_state} == DATA_STATE) {
# Line 3179  sub _get_next_token ($) { Line 3210  sub _get_next_token ($) {
3210          redo A;          redo A;
3211        } else {        } else {
3212          !!!cp (991);          !!!cp (991);
3213          $self->{current_attribute}->{value} .= chr $code;          $self->{ca}->{value} .= chr $code;
3214          $self->{current_attribute}->{has_reference} = 1;          $self->{ca}->{has_reference} = 1;
3215          $self->{state} = $self->{prev_state};          $self->{state} = $self->{prev_state};
3216          ## Reconsume.          ## Reconsume.
3217          redo A;          redo A;
3218        }        }
3219      } elsif ($self->{state} == HEXREF_X_STATE) {      } elsif ($self->{state} == HEXREF_X_STATE) {
3220        if ((0x0030 <= $self->{next_char} and $self->{next_char} <= 0x0039) or        if ((0x0030 <= $self->{nc} and $self->{nc} <= 0x0039) or
3221            (0x0041 <= $self->{next_char} and $self->{next_char} <= 0x0046) or            (0x0041 <= $self->{nc} and $self->{nc} <= 0x0046) or
3222            (0x0061 <= $self->{next_char} and $self->{next_char} <= 0x0066)) {            (0x0061 <= $self->{nc} and $self->{nc} <= 0x0066)) {
3223          # 0..9, A..F, a..f          # 0..9, A..F, a..f
3224          !!!cp (990);          !!!cp (990);
3225          $self->{state} = HEXREF_HEX_STATE;          $self->{state} = HEXREF_HEX_STATE;
3226          $self->{state_keyword} = 0;          $self->{s_kwd} = 0;
3227          ## Reconsume.          ## Reconsume.
3228          redo A;          redo A;
3229        } else {        } else {
# Line 3209  sub _get_next_token ($) { Line 3240  sub _get_next_token ($) {
3240            $self->{state} = $self->{prev_state};            $self->{state} = $self->{prev_state};
3241            ## Reconsume.            ## Reconsume.
3242            !!!emit ({type => CHARACTER_TOKEN,            !!!emit ({type => CHARACTER_TOKEN,
3243                      data => '&' . $self->{state_keyword},                      data => '&' . $self->{s_kwd},
3244                      line => $self->{line_prev},                      line => $self->{line_prev},
3245                      column => $self->{column_prev} - length $self->{state_keyword},                      column => $self->{column_prev} - length $self->{s_kwd},
3246                     });                     });
3247            redo A;            redo A;
3248          } else {          } else {
3249            !!!cp (989);            !!!cp (989);
3250            $self->{current_attribute}->{value} .= '&' . $self->{state_keyword};            $self->{ca}->{value} .= '&' . $self->{s_kwd};
3251            $self->{state} = $self->{prev_state};            $self->{state} = $self->{prev_state};
3252            ## Reconsume.            ## Reconsume.
3253            redo A;            redo A;
3254          }          }
3255        }        }
3256      } elsif ($self->{state} == HEXREF_HEX_STATE) {      } elsif ($self->{state} == HEXREF_HEX_STATE) {
3257        if (0x0030 <= $self->{next_char} and $self->{next_char} <= 0x0039) {        if (0x0030 <= $self->{nc} and $self->{nc} <= 0x0039) {
3258          # 0..9          # 0..9
3259          !!!cp (1002);          !!!cp (1002);
3260          $self->{state_keyword} *= 0x10;          $self->{s_kwd} *= 0x10;
3261          $self->{state_keyword} += $self->{next_char} - 0x0030;          $self->{s_kwd} += $self->{nc} - 0x0030;
3262          ## Stay in the state.          ## Stay in the state.
3263          !!!next-input-character;          !!!next-input-character;
3264          redo A;          redo A;
3265        } elsif (0x0061 <= $self->{next_char} and        } elsif (0x0061 <= $self->{nc} and
3266                 $self->{next_char} <= 0x0066) { # a..f                 $self->{nc} <= 0x0066) { # a..f
3267          !!!cp (1003);          !!!cp (1003);
3268          $self->{state_keyword} *= 0x10;          $self->{s_kwd} *= 0x10;
3269          $self->{state_keyword} += $self->{next_char} - 0x0060 + 9;          $self->{s_kwd} += $self->{nc} - 0x0060 + 9;
3270          ## Stay in the state.          ## Stay in the state.
3271          !!!next-input-character;          !!!next-input-character;
3272          redo A;          redo A;
3273        } elsif (0x0041 <= $self->{next_char} and        } elsif (0x0041 <= $self->{nc} and
3274                 $self->{next_char} <= 0x0046) { # A..F                 $self->{nc} <= 0x0046) { # A..F
3275          !!!cp (1004);          !!!cp (1004);
3276          $self->{state_keyword} *= 0x10;          $self->{s_kwd} *= 0x10;
3277          $self->{state_keyword} += $self->{next_char} - 0x0040 + 9;          $self->{s_kwd} += $self->{nc} - 0x0040 + 9;
3278          ## Stay in the state.          ## Stay in the state.
3279          !!!next-input-character;          !!!next-input-character;
3280          redo A;          redo A;
3281        } elsif ($self->{next_char} == 0x003B) { # ;        } elsif ($self->{nc} == 0x003B) { # ;
3282          !!!cp (1006);          !!!cp (1006);
3283          !!!next-input-character;          !!!next-input-character;
3284          #          #
# Line 3260  sub _get_next_token ($) { Line 3291  sub _get_next_token ($) {
3291          #          #
3292        }        }
3293    
3294        my $code = $self->{state_keyword};        my $code = $self->{s_kwd};
3295        my $l = $self->{line_prev};        my $l = $self->{line_prev};
3296        my $c = $self->{column_prev};        my $c = $self->{column_prev};
3297        if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) {        if ($charref_map->{$code}) {
3298          !!!cp (1008);          !!!cp (1008);
3299          !!!parse-error (type => 'invalid character reference',          !!!parse-error (type => 'invalid character reference',
3300                          text => (sprintf 'U+%04X', $code),                          text => (sprintf 'U+%04X', $code),
3301                          line => $l, column => $c);                          line => $l, column => $c);
3302          $code = 0xFFFD;          $code = $charref_map->{$code};
3303        } elsif ($code > 0x10FFFF) {        } elsif ($code > 0x10FFFF) {
3304          !!!cp (1009);          !!!cp (1009);
3305          !!!parse-error (type => 'invalid character reference',          !!!parse-error (type => 'invalid character reference',
3306                          text => (sprintf 'U-%08X', $code),                          text => (sprintf 'U-%08X', $code),
3307                          line => $l, column => $c);                          line => $l, column => $c);
3308          $code = 0xFFFD;          $code = 0xFFFD;
       } elsif ($code == 0x000D) {  
         !!!cp (1010);  
         !!!parse-error (type => 'CR character reference', line => $l, column => $c);  
         $code = 0x000A;  
       } elsif (0x80 <= $code and $code <= 0x9F) {  
         !!!cp (1011);  
         !!!parse-error (type => 'C1 character reference', text => (sprintf 'U+%04X', $code), line => $l, column => $c);  
         $code = $c1_entity_char->{$code};  
3309        }        }
3310    
3311        if ($self->{prev_state} == DATA_STATE) {        if ($self->{prev_state} == DATA_STATE) {
# Line 3295  sub _get_next_token ($) { Line 3318  sub _get_next_token ($) {
3318          redo A;          redo A;
3319        } else {        } else {
3320          !!!cp (987);          !!!cp (987);
3321          $self->{current_attribute}->{value} .= chr $code;          $self->{ca}->{value} .= chr $code;
3322          $self->{current_attribute}->{has_reference} = 1;          $self->{ca}->{has_reference} = 1;
3323          $self->{state} = $self->{prev_state};          $self->{state} = $self->{prev_state};
3324          ## Reconsume.          ## Reconsume.
3325          redo A;          redo A;
3326        }        }
3327      } elsif ($self->{state} == ENTITY_NAME_STATE) {      } elsif ($self->{state} == ENTITY_NAME_STATE) {
3328        if (length $self->{state_keyword} < 30 and        if (length $self->{s_kwd} < 30 and
3329            ## NOTE: Some number greater than the maximum length of entity name            ## NOTE: Some number greater than the maximum length of entity name
3330            ((0x0041 <= $self->{next_char} and # a            ((0x0041 <= $self->{nc} and # a
3331              $self->{next_char} <= 0x005A) or # x              $self->{nc} <= 0x005A) or # x
3332             (0x0061 <= $self->{next_char} and # a             (0x0061 <= $self->{nc} and # a
3333              $self->{next_char} <= 0x007A) or # z              $self->{nc} <= 0x007A) or # z
3334             (0x0030 <= $self->{next_char} and # 0             (0x0030 <= $self->{nc} and # 0
3335              $self->{next_char} <= 0x0039) or # 9              $self->{nc} <= 0x0039) or # 9
3336             $self->{next_char} == 0x003B)) { # ;             $self->{nc} == 0x003B)) { # ;
3337          our $EntityChar;          our $EntityChar;
3338          $self->{state_keyword} .= chr $self->{next_char};          $self->{s_kwd} .= chr $self->{nc};
3339          if (defined $EntityChar->{$self->{state_keyword}}) {          if (defined $EntityChar->{$self->{s_kwd}}) {
3340            if ($self->{next_char} == 0x003B) { # ;            if ($self->{nc} == 0x003B) { # ;
3341              !!!cp (1020);              !!!cp (1020);
3342              $self->{entity__value} = $EntityChar->{$self->{state_keyword}};              $self->{entity__value} = $EntityChar->{$self->{s_kwd}};
3343              $self->{entity__match} = 1;              $self->{entity__match} = 1;
3344              !!!next-input-character;              !!!next-input-character;
3345              #              #
3346            } else {            } else {
3347              !!!cp (1021);              !!!cp (1021);
3348              $self->{entity__value} = $EntityChar->{$self->{state_keyword}};              $self->{entity__value} = $EntityChar->{$self->{s_kwd}};
3349              $self->{entity__match} = -1;              $self->{entity__match} = -1;
3350              ## Stay in the state.              ## Stay in the state.
3351              !!!next-input-character;              !!!next-input-character;
# Line 3330  sub _get_next_token ($) { Line 3353  sub _get_next_token ($) {
3353            }            }
3354          } else {          } else {
3355            !!!cp (1022);            !!!cp (1022);
3356            $self->{entity__value} .= chr $self->{next_char};            $self->{entity__value} .= chr $self->{nc};
3357            $self->{entity__match} *= 2;            $self->{entity__match} *= 2;
3358            ## Stay in the state.            ## Stay in the state.
3359            !!!next-input-character;            !!!next-input-character;
# Line 3350  sub _get_next_token ($) { Line 3373  sub _get_next_token ($) {
3373          if ($self->{prev_state} != DATA_STATE and # in attribute          if ($self->{prev_state} != DATA_STATE and # in attribute
3374              $self->{entity__match} < -1) {              $self->{entity__match} < -1) {
3375            !!!cp (1024);            !!!cp (1024);
3376            $data = '&' . $self->{state_keyword};            $data = '&' . $self->{s_kwd};
3377            #            #
3378          } else {          } else {
3379            !!!cp (1025);            !!!cp (1025);
# Line 3362  sub _get_next_token ($) { Line 3385  sub _get_next_token ($) {
3385          !!!cp (1026);          !!!cp (1026);
3386          !!!parse-error (type => 'bare ero',          !!!parse-error (type => 'bare ero',
3387                          line => $self->{line_prev},                          line => $self->{line_prev},
3388                          column => $self->{column_prev} - length $self->{state_keyword});                          column => $self->{column_prev} - length $self->{s_kwd});
3389          $data = '&' . $self->{state_keyword};          $data = '&' . $self->{s_kwd};
3390          #          #
3391        }        }
3392        
# Line 3384  sub _get_next_token ($) { Line 3407  sub _get_next_token ($) {
3407          !!!emit ({type => CHARACTER_TOKEN,          !!!emit ({type => CHARACTER_TOKEN,
3408                    data => $data,                    data => $data,
3409                    line => $self->{line_prev},                    line => $self->{line_prev},
3410                    column => $self->{column_prev} + 1 - length $self->{state_keyword},                    column => $self->{column_prev} + 1 - length $self->{s_kwd},
3411                   });                   });
3412          redo A;          redo A;
3413        } else {        } else {
3414          !!!cp (985);          !!!cp (985);
3415          $self->{current_attribute}->{value} .= $data;          $self->{ca}->{value} .= $data;
3416          $self->{current_attribute}->{has_reference} = 1 if $has_ref;          $self->{ca}->{has_reference} = 1 if $has_ref;
3417          $self->{state} = $self->{prev_state};          $self->{state} = $self->{prev_state};
3418          ## Reconsume.          ## Reconsume.
3419          redo A;          redo A;
# Line 3468  sub _tree_construction_initial ($) { Line 3491  sub _tree_construction_initial ($) {
3491        $doctype_name = '' unless defined $doctype_name;        $doctype_name = '' unless defined $doctype_name;
3492        $doctype_name =~ tr/a-z/A-Z/; # ASCII case-insensitive        $doctype_name =~ tr/a-z/A-Z/; # ASCII case-insensitive
3493        if (not defined $token->{name} or # <!DOCTYPE>        if (not defined $token->{name} or # <!DOCTYPE>
3494            defined $token->{system_identifier}) {            defined $token->{sysid}) {
3495          !!!cp ('t1');          !!!cp ('t1');
3496          !!!parse-error (type => 'not HTML5', token => $token);          !!!parse-error (type => 'not HTML5', token => $token);
3497        } elsif ($doctype_name ne 'HTML') {        } elsif ($doctype_name ne 'HTML') {
3498          !!!cp ('t2');          !!!cp ('t2');
3499          !!!parse-error (type => 'not HTML5', token => $token);          !!!parse-error (type => 'not HTML5', token => $token);
3500        } elsif (defined $token->{public_identifier}) {        } elsif (defined $token->{pubid}) {
3501          if ($token->{public_identifier} eq 'XSLT-compat') {          if ($token->{pubid} eq 'XSLT-compat') {
3502            !!!cp ('t1.2');            !!!cp ('t1.2');
3503            !!!parse-error (type => 'XSLT-compat', token => $token,            !!!parse-error (type => 'XSLT-compat', token => $token,
3504                            level => $self->{level}->{should});                            level => $self->{level}->{should});
# Line 3491  sub _tree_construction_initial ($) { Line 3514  sub _tree_construction_initial ($) {
3514          ($token->{name}); ## ISSUE: If name is missing (e.g. <!DOCTYPE>)?          ($token->{name}); ## ISSUE: If name is missing (e.g. <!DOCTYPE>)?
3515        ## NOTE: Default value for both |public_id| and |system_id| attributes        ## NOTE: Default value for both |public_id| and |system_id| attributes
3516        ## are empty strings, so that we don't set any value in missing cases.        ## are empty strings, so that we don't set any value in missing cases.
3517        $doctype->public_id ($token->{public_identifier})        $doctype->public_id ($token->{pubid}) if defined $token->{pubid};
3518            if defined $token->{public_identifier};        $doctype->system_id ($token->{sysid}) if defined $token->{sysid};
       $doctype->system_id ($token->{system_identifier})  
           if defined $token->{system_identifier};  
3519        ## NOTE: Other DocumentType attributes are null or empty lists.        ## NOTE: Other DocumentType attributes are null or empty lists.
3520        ## ISSUE: internalSubset = null??        ## ISSUE: internalSubset = null??
3521        $self->{document}->append_child ($doctype);        $self->{document}->append_child ($doctype);
# Line 3502  sub _tree_construction_initial ($) { Line 3523  sub _tree_construction_initial ($) {
3523        if ($token->{quirks} or $doctype_name ne 'HTML') {        if ($token->{quirks} or $doctype_name ne 'HTML') {
3524          !!!cp ('t4');          !!!cp ('t4');
3525          $self->{document}->manakai_compat_mode ('quirks');          $self->{document}->manakai_compat_mode ('quirks');
3526        } elsif (defined $token->{public_identifier}) {        } elsif (defined $token->{pubid}) {
3527          my $pubid = $token->{public_identifier};          my $pubid = $token->{pubid};
3528          $pubid =~ tr/a-z/A-z/;          $pubid =~ tr/a-z/A-z/;
3529          my $prefix = [          my $prefix = [
3530            "+//SILMARIL//DTD HTML PRO V0R11 19970101//",            "+//SILMARIL//DTD HTML PRO V0R11 19970101//",
# Line 3577  sub _tree_construction_initial ($) { Line 3598  sub _tree_construction_initial ($) {
3598            $self->{document}->manakai_compat_mode ('quirks');            $self->{document}->manakai_compat_mode ('quirks');
3599          } elsif ($pubid =~ m[^-//W3C//DTD HTML 4.01 FRAMESET//] or          } elsif ($pubid =~ m[^-//W3C//DTD HTML 4.01 FRAMESET//] or
3600                   $pubid =~ m[^-//W3C//DTD HTML 4.01 TRANSITIONAL//]) {                   $pubid =~ m[^-//W3C//DTD HTML 4.01 TRANSITIONAL//]) {
3601            if (defined $token->{system_identifier}) {            if (defined $token->{sysid}) {
3602              !!!cp ('t6');              !!!cp ('t6');
3603              $self->{document}->manakai_compat_mode ('quirks');              $self->{document}->manakai_compat_mode ('quirks');
3604            } else {            } else {
# Line 3594  sub _tree_construction_initial ($) { Line 3615  sub _tree_construction_initial ($) {
3615        } else {        } else {
3616          !!!cp ('t10');          !!!cp ('t10');
3617        }        }
3618        if (defined $token->{system_identifier}) {        if (defined $token->{sysid}) {
3619          my $sysid = $token->{system_identifier};          my $sysid = $token->{sysid};
3620          $sysid =~ tr/A-Z/a-z/;          $sysid =~ tr/A-Z/a-z/;
3621          if ($sysid eq "http://www.ibm.com/data/dtd/v11/ibmxhtml1-transitional.dtd") {          if ($sysid eq "http://www.ibm.com/data/dtd/v11/ibmxhtml1-transitional.dtd") {
3622            ## NOTE: Ensure that |PUBLIC "(limited quirks)" "(quirks)"| is            ## NOTE: Ensure that |PUBLIC "(limited quirks)" "(quirks)"| is
# Line 3625  sub _tree_construction_initial ($) { Line 3646  sub _tree_construction_initial ($) {
3646        !!!ack-later;        !!!ack-later;
3647        return;        return;
3648      } elsif ($token->{type} == CHARACTER_TOKEN) {      } elsif ($token->{type} == CHARACTER_TOKEN) {
3649        if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D        if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
3650          ## Ignore the token          ## Ignore the token
3651    
3652          unless (length $token->{data}) {          unless (length $token->{data}) {
# Line 3682  sub _tree_construction_root_element ($) Line 3703  sub _tree_construction_root_element ($)
3703          !!!next-token;          !!!next-token;
3704          redo B;          redo B;
3705        } elsif ($token->{type} == CHARACTER_TOKEN) {        } elsif ($token->{type} == CHARACTER_TOKEN) {
3706          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D          if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
3707            ## Ignore the token.            ## Ignore the token.
3708    
3709            unless (length $token->{data}) {            unless (length $token->{data}) {
# Line 4496  sub _tree_construction_main ($) { Line 4517  sub _tree_construction_main ($) {
4517    
4518      if ($self->{insertion_mode} & HEAD_IMS) {      if ($self->{insertion_mode} & HEAD_IMS) {
4519        if ($token->{type} == CHARACTER_TOKEN) {        if ($token->{type} == CHARACTER_TOKEN) {
4520          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {          if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
4521            unless ($self->{insertion_mode} == BEFORE_HEAD_IM) {            unless ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4522              !!!cp ('t88.2');              !!!cp ('t88.2');
4523              $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);              $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
# Line 4675  sub _tree_construction_main ($) { Line 4696  sub _tree_construction_main ($) {
4696                  } elsif ($token->{attributes}->{content}) {                  } elsif ($token->{attributes}->{content}) {
4697                    if ($token->{attributes}->{content}->{value}                    if ($token->{attributes}->{content}->{value}
4698                        =~ /[Cc][Hh][Aa][Rr][Ss][Ee][Tt]                        =~ /[Cc][Hh][Aa][Rr][Ss][Ee][Tt]
4699                            [\x09-\x0D\x20]*=                            [\x09\x0A\x0C\x0D\x20]*=
4700                            [\x09-\x0D\x20]*(?>"([^"]*)"|'([^']*)'|                            [\x09\x0A\x0C\x0D\x20]*(?>"([^"]*)"|'([^']*)'|
4701                            ([^"'\x09-\x0D\x20][^\x09-\x0D\x20\x3B]*))/x) {                            ([^"'\x09\x0A\x0C\x0D\x20]
4702                               [^\x09\x0A\x0C\x0D\x20\x3B]*))/x) {
4703                      !!!cp ('t107');                      !!!cp ('t107');
4704                      ## NOTE: Whether the encoding is supported or not is handled                      ## NOTE: Whether the encoding is supported or not is handled
4705                      ## in the {change_encoding} callback.                      ## in the {change_encoding} callback.
# Line 5486  sub _tree_construction_main ($) { Line 5508  sub _tree_construction_main ($) {
5508      } elsif ($self->{insertion_mode} & TABLE_IMS) {      } elsif ($self->{insertion_mode} & TABLE_IMS) {
5509        if ($token->{type} == CHARACTER_TOKEN) {        if ($token->{type} == CHARACTER_TOKEN) {
5510          if (not $open_tables->[-1]->[1] and # tainted          if (not $open_tables->[-1]->[1] and # tainted
5511              $token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {              $token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
5512            $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);            $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
5513                                
5514            unless (length $token->{data}) {            unless (length $token->{data}) {
# Line 6170  sub _tree_construction_main ($) { Line 6192  sub _tree_construction_main ($) {
6192        }        }
6193      } elsif ($self->{insertion_mode} == IN_COLUMN_GROUP_IM) {      } elsif ($self->{insertion_mode} == IN_COLUMN_GROUP_IM) {
6194            if ($token->{type} == CHARACTER_TOKEN) {            if ($token->{type} == CHARACTER_TOKEN) {
6195              if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {              if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
6196                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
6197                unless (length $token->{data}) {                unless (length $token->{data}) {
6198                  !!!cp ('t260');                  !!!cp ('t260');
# Line 6511  sub _tree_construction_main ($) { Line 6533  sub _tree_construction_main ($) {
6533        }        }
6534      } elsif ($self->{insertion_mode} & BODY_AFTER_IMS) {      } elsif ($self->{insertion_mode} & BODY_AFTER_IMS) {
6535        if ($token->{type} == CHARACTER_TOKEN) {        if ($token->{type} == CHARACTER_TOKEN) {
6536          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {          if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
6537            my $data = $1;            my $data = $1;
6538            ## As if in body            ## As if in body
6539            $reconstruct_active_formatting_elements->($insert_to_current);            $reconstruct_active_formatting_elements->($insert_to_current);
# Line 6528  sub _tree_construction_main ($) { Line 6550  sub _tree_construction_main ($) {
6550          if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {          if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
6551            !!!cp ('t301');            !!!cp ('t301');
6552            !!!parse-error (type => 'after html:#text', token => $token);            !!!parse-error (type => 'after html:#text', token => $token);
6553              #
           ## Reprocess in the "after body" insertion mode.  
6554          } else {          } else {
6555            !!!cp ('t302');            !!!cp ('t302');
6556              ## "after body" insertion mode
6557              !!!parse-error (type => 'after body:#text', token => $token);
6558              #
6559          }          }
           
         ## "after body" insertion mode  
         !!!parse-error (type => 'after body:#text', token => $token);  
6560    
6561          $self->{insertion_mode} = IN_BODY_IM;          $self->{insertion_mode} = IN_BODY_IM;
6562          ## reprocess          ## reprocess
# Line 6545  sub _tree_construction_main ($) { Line 6566  sub _tree_construction_main ($) {
6566            !!!cp ('t303');            !!!cp ('t303');
6567            !!!parse-error (type => 'after html',            !!!parse-error (type => 'after html',
6568                            text => $token->{tag_name}, token => $token);                            text => $token->{tag_name}, token => $token);
6569                        #
           ## Reprocess in the "after body" insertion mode.  
6570          } else {          } else {
6571            !!!cp ('t304');            !!!cp ('t304');
6572              ## "after body" insertion mode
6573              !!!parse-error (type => 'after body',
6574                              text => $token->{tag_name}, token => $token);
6575              #
6576          }          }
6577    
         ## "after body" insertion mode  
         !!!parse-error (type => 'after body',  
                         text => $token->{tag_name}, token => $token);  
   
6578          $self->{insertion_mode} = IN_BODY_IM;          $self->{insertion_mode} = IN_BODY_IM;
6579          !!!ack-later;          !!!ack-later;
6580          ## reprocess          ## reprocess
# Line 6565  sub _tree_construction_main ($) { Line 6585  sub _tree_construction_main ($) {
6585            !!!parse-error (type => 'after html:/',            !!!parse-error (type => 'after html:/',
6586                            text => $token->{tag_name}, token => $token);                            text => $token->{tag_name}, token => $token);
6587                        
6588            $self->{insertion_mode} = AFTER_BODY_IM;            $self->{insertion_mode} = IN_BODY_IM;
6589            ## Reprocess in the "after body" insertion mode.            ## Reprocess.
6590              next B;
6591          } else {          } else {
6592            !!!cp ('t306');            !!!cp ('t306');
6593          }          }
# Line 6604  sub _tree_construction_main ($) { Line 6625  sub _tree_construction_main ($) {
6625        }        }
6626      } elsif ($self->{insertion_mode} & FRAME_IMS) {      } elsif ($self->{insertion_mode} & FRAME_IMS) {
6627        if ($token->{type} == CHARACTER_TOKEN) {        if ($token->{type} == CHARACTER_TOKEN) {
6628          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {          if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
6629            $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);            $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
6630                        
6631            unless (length $token->{data}) {            unless (length $token->{data}) {
# Line 6614  sub _tree_construction_main ($) { Line 6635  sub _tree_construction_main ($) {
6635            }            }
6636          }          }
6637                    
6638          if ($token->{data} =~ s/^[^\x09\x0A\x0B\x0C\x20]+//) {          if ($token->{data} =~ s/^[^\x09\x0A\x0C\x20]+//) {
6639            if ($self->{insertion_mode} == IN_FRAMESET_IM) {            if ($self->{insertion_mode} == IN_FRAMESET_IM) {
6640              !!!cp ('t311');              !!!cp ('t311');
6641              !!!parse-error (type => 'in frameset:#text', token => $token);              !!!parse-error (type => 'in frameset:#text', token => $token);
# Line 6791  sub _tree_construction_main ($) { Line 6812  sub _tree_construction_main ($) {
6812            } elsif ($token->{attributes}->{content}) {            } elsif ($token->{attributes}->{content}) {
6813              if ($token->{attributes}->{content}->{value}              if ($token->{attributes}->{content}->{value}
6814                  =~ /[Cc][Hh][Aa][Rr][Ss][Ee][Tt]                  =~ /[Cc][Hh][Aa][Rr][Ss][Ee][Tt]
6815                      [\x09-\x0D\x20]*=                      [\x09\x0A\x0C\x0D\x20]*=
6816                      [\x09-\x0D\x20]*(?>"([^"]*)"|'([^']*)'|                      [\x09\x0A\x0C\x0D\x20]*(?>"([^"]*)"|'([^']*)'|
6817                      ([^"'\x09-\x0D\x20][^\x09-\x0D\x20\x3B]*))/x) {                      ([^"'\x09\x0A\x0C\x0D\x20][^\x09\x0A\x0C\x0D\x20\x3B]*))
6818                       /x) {
6819                !!!cp ('t336');                !!!cp ('t336');
6820                ## NOTE: Whether the encoding is supported or not is handled                ## NOTE: Whether the encoding is supported or not is handled
6821                ## in the {change_encoding} callback.                ## in the {change_encoding} callback.
# Line 7802  sub set_inner_html ($$$$;$) { Line 7824  sub set_inner_html ($$$$;$) {
7824      require Whatpm::Charset::DecodeHandle;      require Whatpm::Charset::DecodeHandle;
7825      my $input = Whatpm::Charset::DecodeHandle::CharString->new (\($_[0]));      my $input = Whatpm::Charset::DecodeHandle::CharString->new (\($_[0]));
7826      $input = $get_wrapper->($input);      $input = $get_wrapper->($input);
7827      $p->{set_next_char} = sub {      $p->{set_nc} = sub {
7828        my $self = shift;        my $self = shift;
7829    
       pop @{$self->{prev_char}};  
       unshift @{$self->{prev_char}}, $self->{next_char};  
   
7830        my $char = '';        my $char = '';
7831        if (defined $self->{next_next_char}) {        if (defined $self->{next_nc}) {
7832          $char = $self->{next_next_char};          $char = $self->{next_nc};
7833          delete $self->{next_next_char};          delete $self->{next_nc};
7834          $self->{next_char} = ord $char;          $self->{nc} = ord $char;
7835        } else {        } else {
7836            $self->{char_buffer} = '';
7837            $self->{char_buffer_pos} = 0;
7838            
7839            my $count = $input->manakai_read_until
7840                ($self->{char_buffer}, qr/[^\x00\x0A\x0D]/,
7841                 $self->{char_buffer_pos});
7842            if ($count) {
7843              $self->{line_prev} = $self->{line};
7844              $self->{column_prev} = $self->{column};
7845              $self->{column}++;
7846              $self->{nc}
7847                  = ord substr ($self->{char_buffer},
7848                                $self->{char_buffer_pos}++, 1);
7849              return;
7850            }
7851            
7852          if ($input->read ($char, 1)) {          if ($input->read ($char, 1)) {
7853            $self->{next_char} = ord $char;            $self->{nc} = ord $char;
7854          } else {          } else {
7855            $self->{next_char} = -1;            $self->{nc} = -1;
7856            return;            return;
7857          }          }
7858        }        }
# Line 7825  sub set_inner_html ($$$$;$) { Line 7860  sub set_inner_html ($$$$;$) {
7860        ($p->{line_prev}, $p->{column_prev}) = ($p->{line}, $p->{column});        ($p->{line_prev}, $p->{column_prev}) = ($p->{line}, $p->{column});
7861        $p->{column}++;        $p->{column}++;
7862    
7863        if ($self->{next_char} == 0x000A) { # LF        if ($self->{nc} == 0x000A) { # LF
7864          $p->{line}++;          $p->{line}++;
7865          $p->{column} = 0;          $p->{column} = 0;
7866          !!!cp ('i1');          !!!cp ('i1');
7867        } elsif ($self->{next_char} == 0x000D) { # CR        } elsif ($self->{nc} == 0x000D) { # CR
7868  ## TODO: support for abort/streaming  ## TODO: support for abort/streaming
7869          my $next = '';          my $next = '';
7870          if ($input->read ($next, 1) and $next ne "\x0A") {          if ($input->read ($next, 1) and $next ne "\x0A") {
7871            $self->{next_next_char} = $next;            $self->{next_nc} = $next;
7872          }          }
7873          $self->{next_char} = 0x000A; # LF # MUST          $self->{nc} = 0x000A; # LF # MUST
7874          $p->{line}++;          $p->{line}++;
7875          $p->{column} = 0;          $p->{column} = 0;
7876          !!!cp ('i2');          !!!cp ('i2');
7877        } elsif ($self->{next_char} > 0x10FFFF) {        } elsif ($self->{nc} == 0x0000) { # NULL
         $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST  
         !!!cp ('i3');  
       } elsif ($self->{next_char} == 0x0000) { # NULL  
7878          !!!cp ('i4');          !!!cp ('i4');
7879          !!!parse-error (type => 'NULL');          !!!parse-error (type => 'NULL');
7880          $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST          $self->{nc} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
       } elsif ($self->{next_char} <= 0x0008 or  
                (0x000E <= $self->{next_char} and  
                 $self->{next_char} <= 0x001F) or  
                (0x007F <= $self->{next_char} and  
                 $self->{next_char} <= 0x009F) or  
                (0xD800 <= $self->{next_char} and  
                 $self->{next_char} <= 0xDFFF) or  
                (0xFDD0 <= $self->{next_char} and  
                 $self->{next_char} <= 0xFDDF) or  
                {  
                 0xFFFE => 1, 0xFFFF => 1, 0x1FFFE => 1, 0x1FFFF => 1,  
                 0x2FFFE => 1, 0x2FFFF => 1, 0x3FFFE => 1, 0x3FFFF => 1,  
                 0x4FFFE => 1, 0x4FFFF => 1, 0x5FFFE => 1, 0x5FFFF => 1,  
                 0x6FFFE => 1, 0x6FFFF => 1, 0x7FFFE => 1, 0x7FFFF => 1,  
                 0x8FFFE => 1, 0x8FFFF => 1, 0x9FFFE => 1, 0x9FFFF => 1,  
                 0xAFFFE => 1, 0xAFFFF => 1, 0xBFFFE => 1, 0xBFFFF => 1,  
                 0xCFFFE => 1, 0xCFFFF => 1, 0xDFFFE => 1, 0xDFFFF => 1,  
                 0xEFFFE => 1, 0xEFFFF => 1, 0xFFFFE => 1, 0xFFFFF => 1,  
                 0x10FFFE => 1, 0x10FFFF => 1,  
                }->{$self->{next_char}}) {  
         !!!cp ('i4.1');  
         if ($self->{next_char} < 0x10000) {  
           !!!parse-error (type => 'control char',  
                           text => (sprintf 'U+%04X', $self->{next_char}));  
         } else {  
           !!!parse-error (type => 'control char',  
                           text => (sprintf 'U-%08X', $self->{next_char}));  
         }  
7881        }        }
7882      };      };
     $p->{prev_char} = [-1, -1, -1];  
     $p->{next_char} = -1;  
7883    
7884      $p->{read_until} = sub {      $p->{read_until} = sub {
7885        #my ($scalar, $specials_range, $offset) = @_;        #my ($scalar, $specials_range, $offset) = @_;
7886        my $specials_range = $_[1];        return 0 if defined $p->{next_nc};
7887        return 0 if defined $p->{next_next_char};  
7888        my $count = $input->manakai_read_until        my $pattern = qr/[^$_[1]\x00\x0A\x0D]/;
7889          ($_[0],        my $offset = $_[2] || 0;
7890           qr/(?![$specials_range\x{FDD0}-\x{FDDF}\x{FFFE}\x{FFFF}\x{1FFFE}\x{1FFFF}\x{2FFFE}\x{2FFFF}\x{3FFFE}\x{3FFFF}\x{4FFFE}\x{4FFFF}\x{5FFFE}\x{5FFFF}\x{6FFFE}\x{6FFFF}\x{7FFFE}\x{7FFFF}\x{8FFFE}\x{8FFFF}\x{9FFFE}\x{9FFFF}\x{AFFFE}\x{AFFFF}\x{BFFFE}\x{BFFFF}\x{CFFFE}\x{CFFFF}\x{DFFFE}\x{DFFFF}\x{EFFFE}\x{EFFFF}\x{FFFFE}\x{FFFFF}])[\x20-\x7E\xA0-\x{D7FF}\x{E000}-\x{10FFFD}]/,        
7891           $_[2]);        if ($p->{char_buffer_pos} < length $p->{char_buffer}) {
7892        if ($count) {          pos ($p->{char_buffer}) = $p->{char_buffer_pos};
7893          $p->{column} += $count;          if ($p->{char_buffer} =~ /\G(?>$pattern)+/) {
7894          $p->{column_prev} += $count;            substr ($_[0], $offset)
7895          $p->{prev_char} = [-1, -1, -1];                = substr ($p->{char_buffer}, $-[0], $+[0] - $-[0]);
7896          $p->{next_char} = -1;            my $count = $+[0] - $-[0];
7897              if ($count) {
7898                $p->{column} += $count;
7899                $p->{char_buffer_pos} += $count;
7900                $p->{line_prev} = $p->{line};
7901                $p->{column_prev} = $p->{column} - 1;
7902                $p->{nc} = -1;
7903              }
7904              return $count;
7905            } else {
7906              return 0;
7907            }
7908          } else {
7909            my $count = $input->manakai_read_until ($_[0], $pattern, $_[2]);
7910            if ($count) {
7911              $p->{column} += $count;
7912              $p->{column_prev} += $count;
7913              $p->{nc} = -1;
7914            }
7915            return $count;
7916        }        }
       return $count;  
7917      }; # $p->{read_until}      }; # $p->{read_until}
7918    
7919      my $ponerror = $onerror || sub {      my $ponerror = $onerror || sub {

Legend:
Removed from v.1.179  
changed lines
  Added in v.1.192

[email protected]
ViewVC Help
Powered by ViewVC 1.1.24