/[suikacvs]/markup/html/whatpm/Whatpm/HTML.pm.src
Suika

Diff of /markup/html/whatpm/Whatpm/HTML.pm.src

Parent Directory Parent Directory | Revision Log Revision Log | View Patch Patch

revision 1.162 by wakaba, Thu Sep 11 09:12:27 2008 UTC revision 1.205 by wakaba, Mon Oct 13 06:18:31 2008 UTC
# Line 3  use strict; Line 3  use strict;
3  our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r};  our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r};
4  use Error qw(:try);  use Error qw(:try);
5    
6    ## NOTE: This module don't check all HTML5 parse errors; character
7    ## encoding related parse errors are expected to be handled by relevant
8    ## modules.
9    ## Parse errors for control characters that are not allowed in HTML5
10    ## documents, for surrogate code points, and for noncharacter code
11    ## points, as well as U+FFFD substitions for characters whose code points
12    ## is higher than U+10FFFF may be detected by combining the parser with
13    ## the checker implemented by Whatpm::Charset::UnicodeChecker (for its
14    ## usage example, see |t/HTML-tree.t| in the Whatpm package or the
15    ## WebHACC::Language::HTML module in the WebHACC package).
16    
17  ## ISSUE:  ## ISSUE:
18  ## var doc = implementation.createDocument (null, null, null);  ## var doc = implementation.createDocument (null, null, null);
19  ## doc.write ('');  ## doc.write ('');
# Line 55  sub TABLE_ROWS_EL () { Line 66  sub TABLE_ROWS_EL () {
66  }  }
67    
68  ## NOTE: Used in "generate implied end tags" algorithm.  ## NOTE: Used in "generate implied end tags" algorithm.
69  ## NOTE: There is a code where a modified version of END_TAG_OPTIONAL_EL  ## NOTE: There is a code where a modified version of
70  ## is used in "generate implied end tags" implementation (search for the  ## END_TAG_OPTIONAL_EL is used in "generate implied end tags"
71  ## function mae).  ## implementation (search for the algorithm name).
72  sub END_TAG_OPTIONAL_EL () {  sub END_TAG_OPTIONAL_EL () {
73    DD_EL |    DD_EL |
74    DT_EL |    DT_EL |
75    LI_EL |    LI_EL |
76      OPTION_EL |
77      OPTGROUP_EL |
78    P_EL |    P_EL |
79    RUBY_COMPONENT_EL    RUBY_COMPONENT_EL
80  }  }
# Line 73  sub ALL_END_TAG_OPTIONAL_EL () { Line 86  sub ALL_END_TAG_OPTIONAL_EL () {
86    LI_EL |    LI_EL |
87    P_EL |    P_EL |
88    
89      ## ISSUE: option, optgroup, rt, rp?
90    
91    BODY_EL |    BODY_EL |
92    HTML_EL |    HTML_EL |
93    TABLE_CELL_EL |    TABLE_CELL_EL |
# Line 117  sub SPECIAL_EL () { Line 132  sub SPECIAL_EL () {
132    FORM_EL |    FORM_EL |
133    FRAMESET_EL |    FRAMESET_EL |
134    HEADING_EL |    HEADING_EL |
   OPTION_EL |  
   OPTGROUP_EL |  
135    SELECT_EL |    SELECT_EL |
136    TABLE_ROW_EL |    TABLE_ROW_EL |
137    TABLE_ROW_GROUP_EL |    TABLE_ROW_GROUP_EL |
# Line 130  my $el_category = { Line 143  my $el_category = {
143    address => ADDRESS_EL,    address => ADDRESS_EL,
144    applet => MISC_SCOPING_EL,    applet => MISC_SCOPING_EL,
145    area => MISC_SPECIAL_EL,    area => MISC_SPECIAL_EL,
146      article => MISC_SPECIAL_EL,
147      aside => MISC_SPECIAL_EL,
148    b => FORMATTING_EL,    b => FORMATTING_EL,
149    base => MISC_SPECIAL_EL,    base => MISC_SPECIAL_EL,
150    basefont => MISC_SPECIAL_EL,    basefont => MISC_SPECIAL_EL,
# Line 143  my $el_category = { Line 158  my $el_category = {
158    center => MISC_SPECIAL_EL,    center => MISC_SPECIAL_EL,
159    col => MISC_SPECIAL_EL,    col => MISC_SPECIAL_EL,
160    colgroup => MISC_SPECIAL_EL,    colgroup => MISC_SPECIAL_EL,
161      command => MISC_SPECIAL_EL,
162      datagrid => MISC_SPECIAL_EL,
163    dd => DD_EL,    dd => DD_EL,
164      details => MISC_SPECIAL_EL,
165      dialog => MISC_SPECIAL_EL,
166    dir => MISC_SPECIAL_EL,    dir => MISC_SPECIAL_EL,
167    div => DIV_EL,    div => DIV_EL,
168    dl => MISC_SPECIAL_EL,    dl => MISC_SPECIAL_EL,
169    dt => DT_EL,    dt => DT_EL,
170    em => FORMATTING_EL,    em => FORMATTING_EL,
171    embed => MISC_SPECIAL_EL,    embed => MISC_SPECIAL_EL,
172      eventsource => MISC_SPECIAL_EL,
173    fieldset => MISC_SPECIAL_EL,    fieldset => MISC_SPECIAL_EL,
174      figure => MISC_SPECIAL_EL,
175    font => FORMATTING_EL,    font => FORMATTING_EL,
176      footer => MISC_SPECIAL_EL,
177    form => FORM_EL,    form => FORM_EL,
178    frame => MISC_SPECIAL_EL,    frame => MISC_SPECIAL_EL,
179    frameset => FRAMESET_EL,    frameset => FRAMESET_EL,
# Line 162  my $el_category = { Line 184  my $el_category = {
184    h5 => HEADING_EL,    h5 => HEADING_EL,
185    h6 => HEADING_EL,    h6 => HEADING_EL,
186    head => MISC_SPECIAL_EL,    head => MISC_SPECIAL_EL,
187      header => MISC_SPECIAL_EL,
188    hr => MISC_SPECIAL_EL,    hr => MISC_SPECIAL_EL,
189    html => HTML_EL,    html => HTML_EL,
190    i => FORMATTING_EL,    i => FORMATTING_EL,
191    iframe => MISC_SPECIAL_EL,    iframe => MISC_SPECIAL_EL,
192    img => MISC_SPECIAL_EL,    img => MISC_SPECIAL_EL,
193      #image => MISC_SPECIAL_EL, ## NOTE: Commented out in the spec.
194    input => MISC_SPECIAL_EL,    input => MISC_SPECIAL_EL,
195    isindex => MISC_SPECIAL_EL,    isindex => MISC_SPECIAL_EL,
196    li => LI_EL,    li => LI_EL,
# Line 175  my $el_category = { Line 199  my $el_category = {
199    marquee => MISC_SCOPING_EL,    marquee => MISC_SCOPING_EL,
200    menu => MISC_SPECIAL_EL,    menu => MISC_SPECIAL_EL,
201    meta => MISC_SPECIAL_EL,    meta => MISC_SPECIAL_EL,
202      nav => MISC_SPECIAL_EL,
203    nobr => NOBR_EL | FORMATTING_EL,    nobr => NOBR_EL | FORMATTING_EL,
204    noembed => MISC_SPECIAL_EL,    noembed => MISC_SPECIAL_EL,
205    noframes => MISC_SPECIAL_EL,    noframes => MISC_SPECIAL_EL,
# Line 193  my $el_category = { Line 218  my $el_category = {
218    s => FORMATTING_EL,    s => FORMATTING_EL,
219    script => MISC_SPECIAL_EL,    script => MISC_SPECIAL_EL,
220    select => SELECT_EL,    select => SELECT_EL,
221      section => MISC_SPECIAL_EL,
222    small => FORMATTING_EL,    small => FORMATTING_EL,
223    spacer => MISC_SPECIAL_EL,    spacer => MISC_SPECIAL_EL,
224    strike => FORMATTING_EL,    strike => FORMATTING_EL,
# Line 223  my $el_category_f = { Line 249  my $el_category_f = {
249      mtext => FOREIGN_FLOW_CONTENT_EL,      mtext => FOREIGN_FLOW_CONTENT_EL,
250    },    },
251    $SVG_NS => {    $SVG_NS => {
252      foreignObject => FOREIGN_FLOW_CONTENT_EL,      foreignObject => FOREIGN_FLOW_CONTENT_EL | MISC_SCOPING_EL,
253      desc => FOREIGN_FLOW_CONTENT_EL,      desc => FOREIGN_FLOW_CONTENT_EL,
254      title => FOREIGN_FLOW_CONTENT_EL,      title => FOREIGN_FLOW_CONTENT_EL,
255    },    },
# Line 312  my $foreign_attr_xname = { Line 338  my $foreign_attr_xname = {
338    
339  ## ISSUE: xmlns:xlink="non-xlink-ns" is not an error.  ## ISSUE: xmlns:xlink="non-xlink-ns" is not an error.
340    
341  my $c1_entity_char = {  my $charref_map = {
342      0x0D => 0x000A,
343    0x80 => 0x20AC,    0x80 => 0x20AC,
344    0x81 => 0xFFFD,    0x81 => 0xFFFD,
345    0x82 => 0x201A,    0x82 => 0x201A,
# Line 345  my $c1_entity_char = { Line 372  my $c1_entity_char = {
372    0x9D => 0xFFFD,    0x9D => 0xFFFD,
373    0x9E => 0x017E,    0x9E => 0x017E,
374    0x9F => 0x0178,    0x9F => 0x0178,
375  }; # $c1_entity_char  }; # $charref_map
376    $charref_map->{$_} = 0xFFFD
377        for 0x0000..0x0008, 0x000B, 0x000E..0x001F, 0x007F,
378            0xD800..0xDFFF, 0xFDD0..0xFDDF, ## ISSUE: 0xFDEF
379            0xFFFE, 0xFFFF, 0x1FFFE, 0x1FFFF, 0x2FFFE, 0x2FFFF, 0x3FFFE, 0x3FFFF,
380            0x4FFFE, 0x4FFFF, 0x5FFFE, 0x5FFFF, 0x6FFFE, 0x6FFFF, 0x7FFFE,
381            0x7FFFF, 0x8FFFE, 0x8FFFF, 0x9FFFE, 0x9FFFF, 0xAFFFE, 0xAFFFF,
382            0xBFFFE, 0xBFFFF, 0xCFFFE, 0xCFFFF, 0xDFFFE, 0xDFFFF, 0xEFFFE,
383            0xEFFFF, 0xFFFFE, 0xFFFFF, 0x10FFFE, 0x10FFFF;
384    
385    ## TODO: Invoke the reset algorithm when a resettable element is
386    ## created (cf. HTML5 revision 2259).
387    
388  sub parse_byte_string ($$$$;$) {  sub parse_byte_string ($$$$;$) {
389    my $self = shift;    my $self = shift;
# Line 390  sub parse_byte_stream ($$$$;$$) { Line 428  sub parse_byte_stream ($$$$;$$) {
428            ## TODO: Is this ok?  Transfer protocol's parameter should be            ## TODO: Is this ok?  Transfer protocol's parameter should be
429            ## interpreted in its semantics?            ## interpreted in its semantics?
430    
       ## ISSUE: Unsupported encoding is not ignored according to the spec.  
431        ($char_stream, $e_status) = $charset->get_decode_handle        ($char_stream, $e_status) = $charset->get_decode_handle
432            ($byte_stream, allow_error_reporting => 1,            ($byte_stream, allow_error_reporting => 1,
433             allow_fallback => 1);             allow_fallback => 1);
# Line 398  sub parse_byte_stream ($$$$;$$) { Line 435  sub parse_byte_stream ($$$$;$$) {
435          $self->{confident} = 1;          $self->{confident} = 1;
436          last SNIFFING;          last SNIFFING;
437        } else {        } else {
438          ## TODO: unsupported error          !!!parse-error (type => 'charset:not supported',
439                            layer => 'encode',
440                            line => 1, column => 1,
441                            value => $charset_name,
442                            level => $self->{level}->{uncertain});
443        }        }
444      }      }
445    
# Line 447  sub parse_byte_stream ($$$$;$$) { Line 488  sub parse_byte_stream ($$$$;$$) {
488      if (defined $charset_name) {      if (defined $charset_name) {
489        $charset = Message::Charset::Info->get_by_html_name ($charset_name);        $charset = Message::Charset::Info->get_by_html_name ($charset_name);
490    
       ## ISSUE: Unsupported encoding is not ignored according to the spec.  
491        require Whatpm::Charset::DecodeHandle;        require Whatpm::Charset::DecodeHandle;
492        $buffer = Whatpm::Charset::DecodeHandle::ByteBuffer->new        $buffer = Whatpm::Charset::DecodeHandle::ByteBuffer->new
493            ($byte_stream);            ($byte_stream);
# Line 496  sub parse_byte_stream ($$$$;$$) { Line 536  sub parse_byte_stream ($$$$;$$) {
536                      line => 1, column => 1,                      line => 1, column => 1,
537                      layer => 'encode');                      layer => 'encode');
538    } elsif (not ($e_status &    } elsif (not ($e_status &
539                  Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL())) {                  Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL ())) {
540      $self->{input_encoding} = $charset->get_iana_name;      $self->{input_encoding} = $charset->get_iana_name;
541      !!!parse-error (type => 'chardecode:no error',      !!!parse-error (type => 'chardecode:no error',
542                      text => $self->{input_encoding},                      text => $self->{input_encoding},
# Line 561  sub parse_byte_stream ($$$$;$$) { Line 601  sub parse_byte_stream ($$$$;$$) {
601    my $char_onerror = sub {    my $char_onerror = sub {
602      my (undef, $type, %opt) = @_;      my (undef, $type, %opt) = @_;
603      !!!parse-error (layer => 'encode',      !!!parse-error (layer => 'encode',
604                      %opt, type => $type,                      line => $self->{line}, column => $self->{column} + 1,
605                      line => $self->{line}, column => $self->{column} + 1);                      %opt, type => $type);
606      if ($opt{octets}) {      if ($opt{octets}) {
607        ${$opt{octets}} = "\x{FFFD}"; # relacement character        ${$opt{octets}} = "\x{FFFD}"; # relacement character
608      }      }
# Line 571  sub parse_byte_stream ($$$$;$$) { Line 611  sub parse_byte_stream ($$$$;$$) {
611    my $wrapped_char_stream = $get_wrapper->($char_stream);    my $wrapped_char_stream = $get_wrapper->($char_stream);
612    $wrapped_char_stream->onerror ($char_onerror);    $wrapped_char_stream->onerror ($char_onerror);
613    
614    my @args = @_; shift @args; # $s    my @args = ($_[1], $_[2]); # $doc, $onerror - $get_wrapper = undef;
615    my $return;    my $return;
616    try {    try {
617      $return = $self->parse_char_stream ($wrapped_char_stream, @args);        $return = $self->parse_char_stream ($wrapped_char_stream, @args);  
# Line 586  sub parse_byte_stream ($$$$;$$) { Line 626  sub parse_byte_stream ($$$$;$$) {
626                        line => 1, column => 1,                        line => 1, column => 1,
627                        layer => 'encode');                        layer => 'encode');
628      } elsif (not ($e_status &      } elsif (not ($e_status &
629                    Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL())) {                    Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL ())) {
630        $self->{input_encoding} = $charset->get_iana_name;        $self->{input_encoding} = $charset->get_iana_name;
631        !!!parse-error (type => 'chardecode:no error',        !!!parse-error (type => 'chardecode:no error',
632                        text => $self->{input_encoding},                        text => $self->{input_encoding},
# Line 618  sub parse_byte_stream ($$$$;$$) { Line 658  sub parse_byte_stream ($$$$;$$) {
658  sub parse_char_string ($$$;$$) {  sub parse_char_string ($$$;$$) {
659    #my ($self, $s, $doc, $onerror, $get_wrapper) = @_;    #my ($self, $s, $doc, $onerror, $get_wrapper) = @_;
660    my $self = shift;    my $self = shift;
   require utf8;  
661    my $s = ref $_[0] ? $_[0] : \($_[0]);    my $s = ref $_[0] ? $_[0] : \($_[0]);
662    open my $input, '<' . (utf8::is_utf8 ($$s) ? ':utf8' : ''), $s;    require Whatpm::Charset::DecodeHandle;
663    if ($_[3]) {    my $input = Whatpm::Charset::DecodeHandle::CharString->new ($s);
     $input = $_[3]->($input);  
   }  
664    return $self->parse_char_stream ($input, @_[1..$#_]);    return $self->parse_char_stream ($input, @_[1..$#_]);
665  } # parse_char_string  } # parse_char_string
666  *parse_string = \&parse_char_string; ## NOTE: Alias for backward compatibility.  *parse_string = \&parse_char_string; ## NOTE: Alias for backward compatibility.
667    
668  sub parse_char_stream ($$$;$) {  sub parse_char_stream ($$$;$$) {
669    my $self = ref $_[0] ? shift : shift->new;    my $self = ref $_[0] ? shift : shift->new;
670    my $input = $_[0];    my $input = $_[0];
671    $self->{document} = $_[1];    $self->{document} = $_[1];
# Line 639  sub parse_char_stream ($$$;$) { Line 676  sub parse_char_stream ($$$;$) {
676    $self->{confident} = 1 unless exists $self->{confident};    $self->{confident} = 1 unless exists $self->{confident};
677    $self->{document}->input_encoding ($self->{input_encoding})    $self->{document}->input_encoding ($self->{input_encoding})
678        if defined $self->{input_encoding};        if defined $self->{input_encoding};
679    ## TODO: |{input_encoding}| is needless?
680    
   my $i = 0;  
681    $self->{line_prev} = $self->{line} = 1;    $self->{line_prev} = $self->{line} = 1;
682    $self->{column_prev} = $self->{column} = 0;    $self->{column_prev} = -1;
683    $self->{set_next_char} = sub {    $self->{column} = 0;
684      $self->{set_nc} = sub {
685      my $self = shift;      my $self = shift;
686    
687      pop @{$self->{prev_char}};      my $char = '';
688      unshift @{$self->{prev_char}}, $self->{next_char};      if (defined $self->{next_nc}) {
689          $char = $self->{next_nc};
690      my $char;        delete $self->{next_nc};
691      if (defined $self->{next_next_char}) {        $self->{nc} = ord $char;
       $char = $self->{next_next_char};  
       delete $self->{next_next_char};  
692      } else {      } else {
693        $char = $input->getc;        $self->{char_buffer} = '';
694          $self->{char_buffer_pos} = 0;
695    
696          my $count = $input->manakai_read_until
697             ($self->{char_buffer}, qr/[^\x00\x0A\x0D]/, $self->{char_buffer_pos});
698          if ($count) {
699            $self->{line_prev} = $self->{line};
700            $self->{column_prev} = $self->{column};
701            $self->{column}++;
702            $self->{nc}
703                = ord substr ($self->{char_buffer}, $self->{char_buffer_pos}++, 1);
704            return;
705          }
706    
707          if ($input->read ($char, 1)) {
708            $self->{nc} = ord $char;
709          } else {
710            $self->{nc} = -1;
711            return;
712          }
713      }      }
     $self->{next_char} = -1 and return unless defined $char;  
     $self->{next_char} = ord $char;  
714    
715      ($self->{line_prev}, $self->{column_prev})      ($self->{line_prev}, $self->{column_prev})
716          = ($self->{line}, $self->{column});          = ($self->{line}, $self->{column});
717      $self->{column}++;      $self->{column}++;
718            
719      if ($self->{next_char} == 0x000A) { # LF      if ($self->{nc} == 0x000A) { # LF
720        !!!cp ('j1');        !!!cp ('j1');
721        $self->{line}++;        $self->{line}++;
722        $self->{column} = 0;        $self->{column} = 0;
723      } elsif ($self->{next_char} == 0x000D) { # CR      } elsif ($self->{nc} == 0x000D) { # CR
724        !!!cp ('j2');        !!!cp ('j2');
725        my $next = $input->getc;  ## TODO: support for abort/streaming
726        if (defined $next and $next ne "\x0A") {        my $next = '';
727          $self->{next_next_char} = $next;        if ($input->read ($next, 1) and $next ne "\x0A") {
728            $self->{next_nc} = $next;
729        }        }
730        $self->{next_char} = 0x000A; # LF # MUST        $self->{nc} = 0x000A; # LF # MUST
731        $self->{line}++;        $self->{line}++;
732        $self->{column} = 0;        $self->{column} = 0;
733      } elsif ($self->{next_char} > 0x10FFFF) {      } elsif ($self->{nc} == 0x0000) { # NULL
       !!!cp ('j3');  
       $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST  
     } elsif ($self->{next_char} == 0x0000) { # NULL  
734        !!!cp ('j4');        !!!cp ('j4');
735        !!!parse-error (type => 'NULL');        !!!parse-error (type => 'NULL');
736        $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST        $self->{nc} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
737      } elsif ($self->{next_char} <= 0x0008 or      }
738               (0x000E <= $self->{next_char} and $self->{next_char} <= 0x001F) or    };
739               (0x007F <= $self->{next_char} and $self->{next_char} <= 0x009F) or  
740               (0xD800 <= $self->{next_char} and $self->{next_char} <= 0xDFFF) or    $self->{read_until} = sub {
741               (0xFDD0 <= $self->{next_char} and $self->{next_char} <= 0xFDDF) or      #my ($scalar, $specials_range, $offset) = @_;
742               {      return 0 if defined $self->{next_nc};
743                0xFFFE => 1, 0xFFFF => 1, 0x1FFFE => 1, 0x1FFFF => 1,  
744                0x2FFFE => 1, 0x2FFFF => 1, 0x3FFFE => 1, 0x3FFFF => 1,      my $pattern = qr/[^$_[1]\x00\x0A\x0D]/;
745                0x4FFFE => 1, 0x4FFFF => 1, 0x5FFFE => 1, 0x5FFFF => 1,      my $offset = $_[2] || 0;
746                0x6FFFE => 1, 0x6FFFF => 1, 0x7FFFE => 1, 0x7FFFF => 1,  
747                0x8FFFE => 1, 0x8FFFF => 1, 0x9FFFE => 1, 0x9FFFF => 1,      if ($self->{char_buffer_pos} < length $self->{char_buffer}) {
748                0xAFFFE => 1, 0xAFFFF => 1, 0xBFFFE => 1, 0xBFFFF => 1,        pos ($self->{char_buffer}) = $self->{char_buffer_pos};
749                0xCFFFE => 1, 0xCFFFF => 1, 0xDFFFE => 1, 0xDFFFF => 1,        if ($self->{char_buffer} =~ /\G(?>$pattern)+/) {
750                0xEFFFE => 1, 0xEFFFF => 1, 0xFFFFE => 1, 0xFFFFF => 1,          substr ($_[0], $offset)
751                0x10FFFE => 1, 0x10FFFF => 1,              = substr ($self->{char_buffer}, $-[0], $+[0] - $-[0]);
752               }->{$self->{next_char}}) {          my $count = $+[0] - $-[0];
753        !!!cp ('j5');          if ($count) {
754        if ($self->{next_char} < 0x10000) {            $self->{column} += $count;
755          !!!parse-error (type => 'control char',            $self->{char_buffer_pos} += $count;
756                          text => (sprintf 'U+%04X', $self->{next_char}));            $self->{line_prev} = $self->{line};
757              $self->{column_prev} = $self->{column} - 1;
758              $self->{nc} = -1;
759            }
760            return $count;
761        } else {        } else {
762          !!!parse-error (type => 'control char',          return 0;
                         text => (sprintf 'U-%08X', $self->{next_char}));  
763        }        }
764        } else {
765          my $count = $input->manakai_read_until ($_[0], $pattern, $_[2]);
766          if ($count) {
767            $self->{column} += $count;
768            $self->{line_prev} = $self->{line};
769            $self->{column_prev} = $self->{column} - 1;
770            $self->{nc} = -1;
771          }
772          return $count;
773      }      }
774    };    }; # $self->{read_until}
   $self->{prev_char} = [-1, -1, -1];  
   $self->{next_char} = -1;  
775    
776    my $onerror = $_[2] || sub {    my $onerror = $_[2] || sub {
777      my (%opt) = @_;      my (%opt) = @_;
# Line 722  sub parse_char_stream ($$$;$) { Line 783  sub parse_char_stream ($$$;$) {
783      $onerror->(line => $self->{line}, column => $self->{column}, @_);      $onerror->(line => $self->{line}, column => $self->{column}, @_);
784    };    };
785    
786      my $char_onerror = sub {
787        my (undef, $type, %opt) = @_;
788        !!!parse-error (layer => 'encode',
789                        line => $self->{line}, column => $self->{column} + 1,
790                        %opt, type => $type);
791      }; # $char_onerror
792    
793      if ($_[3]) {
794        $input = $_[3]->($input);
795        $input->onerror ($char_onerror);
796      } else {
797        $input->onerror ($char_onerror) unless defined $input->onerror;
798      }
799    
800    $self->_initialize_tokenizer;    $self->_initialize_tokenizer;
801    $self->_initialize_tree_constructor;    $self->_initialize_tree_constructor;
802    $self->_construct_tree;    $self->_construct_tree;
# Line 741  sub new ($) { Line 816  sub new ($) {
816                info => 'i',                info => 'i',
817                uncertain => 'u'},                uncertain => 'u'},
818    }, $class;    }, $class;
819    $self->{set_next_char} = sub {    $self->{set_nc} = sub {
820      $self->{next_char} = -1;      $self->{nc} = -1;
821    };    };
822    $self->{parse_error} = sub {    $self->{parse_error} = sub {
823      #      #
# Line 769  sub RCDATA_CONTENT_MODEL () { CM_ENTITY Line 844  sub RCDATA_CONTENT_MODEL () { CM_ENTITY
844  sub PCDATA_CONTENT_MODEL () { CM_ENTITY | CM_FULL_MARKUP }  sub PCDATA_CONTENT_MODEL () { CM_ENTITY | CM_FULL_MARKUP }
845    
846  sub DATA_STATE () { 0 }  sub DATA_STATE () { 0 }
847  sub ENTITY_DATA_STATE () { 1 }  #sub ENTITY_DATA_STATE () { 1 }
848  sub TAG_OPEN_STATE () { 2 }  sub TAG_OPEN_STATE () { 2 }
849  sub CLOSE_TAG_OPEN_STATE () { 3 }  sub CLOSE_TAG_OPEN_STATE () { 3 }
850  sub TAG_NAME_STATE () { 4 }  sub TAG_NAME_STATE () { 4 }
# Line 780  sub BEFORE_ATTRIBUTE_VALUE_STATE () { 8 Line 855  sub BEFORE_ATTRIBUTE_VALUE_STATE () { 8
855  sub ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE () { 9 }  sub ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE () { 9 }
856  sub ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE () { 10 }  sub ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE () { 10 }
857  sub ATTRIBUTE_VALUE_UNQUOTED_STATE () { 11 }  sub ATTRIBUTE_VALUE_UNQUOTED_STATE () { 11 }
858  sub ENTITY_IN_ATTRIBUTE_VALUE_STATE () { 12 }  #sub ENTITY_IN_ATTRIBUTE_VALUE_STATE () { 12 }
859  sub MARKUP_DECLARATION_OPEN_STATE () { 13 }  sub MARKUP_DECLARATION_OPEN_STATE () { 13 }
860  sub COMMENT_START_STATE () { 14 }  sub COMMENT_START_STATE () { 14 }
861  sub COMMENT_START_DASH_STATE () { 15 }  sub COMMENT_START_DASH_STATE () { 15 }
# Line 803  sub AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STAT Line 878  sub AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STAT
878  sub BOGUS_DOCTYPE_STATE () { 32 }  sub BOGUS_DOCTYPE_STATE () { 32 }
879  sub AFTER_ATTRIBUTE_VALUE_QUOTED_STATE () { 33 }  sub AFTER_ATTRIBUTE_VALUE_QUOTED_STATE () { 33 }
880  sub SELF_CLOSING_START_TAG_STATE () { 34 }  sub SELF_CLOSING_START_TAG_STATE () { 34 }
881  sub CDATA_BLOCK_STATE () { 35 }  sub CDATA_SECTION_STATE () { 35 }
882    sub MD_HYPHEN_STATE () { 36 } # "markup declaration open state" in the spec
883    sub MD_DOCTYPE_STATE () { 37 } # "markup declaration open state" in the spec
884    sub MD_CDATA_STATE () { 38 } # "markup declaration open state" in the spec
885    sub CDATA_RCDATA_CLOSE_TAG_STATE () { 39 } # "close tag open state" in the spec
886    sub CDATA_SECTION_MSE1_STATE () { 40 } # "CDATA section state" in the spec
887    sub CDATA_SECTION_MSE2_STATE () { 41 } # "CDATA section state" in the spec
888    sub PUBLIC_STATE () { 42 } # "after DOCTYPE name state" in the spec
889    sub SYSTEM_STATE () { 43 } # "after DOCTYPE name state" in the spec
890    ## NOTE: "Entity data state", "entity in attribute value state", and
891    ## "consume a character reference" algorithm are jointly implemented
892    ## using the following six states:
893    sub ENTITY_STATE () { 44 }
894    sub ENTITY_HASH_STATE () { 45 }
895    sub NCR_NUM_STATE () { 46 }
896    sub HEXREF_X_STATE () { 47 }
897    sub HEXREF_HEX_STATE () { 48 }
898    sub ENTITY_NAME_STATE () { 49 }
899    sub PCDATA_STATE () { 50 } # "data state" in the spec
900    
901  sub DOCTYPE_TOKEN () { 1 }  sub DOCTYPE_TOKEN () { 1 }
902  sub COMMENT_TOKEN () { 2 }  sub COMMENT_TOKEN () { 2 }
# Line 825  sub IN_FOREIGN_CONTENT_IM () { 0b1000000 Line 918  sub IN_FOREIGN_CONTENT_IM () { 0b1000000
918      ## NOTE: "in foreign content" insertion mode is special; it is combined      ## NOTE: "in foreign content" insertion mode is special; it is combined
919      ## with the secondary insertion mode.  In this parser, they are stored      ## with the secondary insertion mode.  In this parser, they are stored
920      ## together in the bit-or'ed form.      ## together in the bit-or'ed form.
921    sub IN_CDATA_RCDATA_IM () { 0b1000000000000 }
922        ## NOTE: "in CDATA/RCDATA" insertion mode is also special; it is
923        ## combined with the original insertion mode.  In thie parser,
924        ## they are stored together in the bit-or'ed form.
925    
926  ## NOTE: "initial" and "before html" insertion modes have no constants.  ## NOTE: "initial" and "before html" insertion modes have no constants.
927    
# Line 856  sub IN_COLUMN_GROUP_IM () { 0b10 } Line 953  sub IN_COLUMN_GROUP_IM () { 0b10 }
953  sub _initialize_tokenizer ($) {  sub _initialize_tokenizer ($) {
954    my $self = shift;    my $self = shift;
955    $self->{state} = DATA_STATE; # MUST    $self->{state} = DATA_STATE; # MUST
956      #$self->{s_kwd}; # state keyword - initialized when used
957      #$self->{entity__value}; # initialized when used
958      #$self->{entity__match}; # initialized when used
959    $self->{content_model} = PCDATA_CONTENT_MODEL; # be    $self->{content_model} = PCDATA_CONTENT_MODEL; # be
960    undef $self->{current_token}; # start tag, end tag, comment, or DOCTYPE    undef $self->{ct}; # current token
961    undef $self->{current_attribute};    undef $self->{ca}; # current attribute
962    undef $self->{last_emitted_start_tag_name};    undef $self->{last_stag_name}; # last emitted start tag name
963    undef $self->{last_attribute_value_state};    #$self->{prev_state}; # initialized when used
964    delete $self->{self_closing};    delete $self->{self_closing};
965    $self->{char} = [];    $self->{char_buffer} = '';
966    # $self->{next_char}    $self->{char_buffer_pos} = 0;
967      $self->{nc} = -1; # next input character
968      #$self->{next_nc}
969    !!!next-input-character;    !!!next-input-character;
970    $self->{token} = [];    $self->{token} = [];
971    # $self->{escape}    # $self->{escape}
# Line 874  sub _initialize_tokenizer ($) { Line 976  sub _initialize_tokenizer ($) {
976  ##       CHARACTER_TOKEN, or END_OF_FILE_TOKEN  ##       CHARACTER_TOKEN, or END_OF_FILE_TOKEN
977  ##   ->{name} (DOCTYPE_TOKEN)  ##   ->{name} (DOCTYPE_TOKEN)
978  ##   ->{tag_name} (START_TAG_TOKEN, END_TAG_TOKEN)  ##   ->{tag_name} (START_TAG_TOKEN, END_TAG_TOKEN)
979  ##   ->{public_identifier} (DOCTYPE_TOKEN)  ##   ->{pubid} (DOCTYPE_TOKEN)
980  ##   ->{system_identifier} (DOCTYPE_TOKEN)  ##   ->{sysid} (DOCTYPE_TOKEN)
981  ##   ->{quirks} == 1 or 0 (DOCTYPE_TOKEN): "force-quirks" flag  ##   ->{quirks} == 1 or 0 (DOCTYPE_TOKEN): "force-quirks" flag
982  ##   ->{attributes} isa HASH (START_TAG_TOKEN, END_TAG_TOKEN)  ##   ->{attributes} isa HASH (START_TAG_TOKEN, END_TAG_TOKEN)
983  ##        ->{name}  ##        ->{name}
# Line 894  sub _initialize_tokenizer ($) { Line 996  sub _initialize_tokenizer ($) {
996  ## has completed loading.  If one has, then it MUST be executed  ## has completed loading.  If one has, then it MUST be executed
997  ## and removed from the list.  ## and removed from the list.
998    
999  ## NOTE: HTML5 "Writing HTML documents" section, applied to  ## TODO: Polytheistic slash SHOULD NOT be used. (Applied only to atheists.)
1000  ## documents and not to user agents and conformance checkers,  ## (This requirement was dropped from HTML5 spec, unfortunately.)
1001  ## contains some requirements that are not detected by the  
1002  ## parsing algorithm:  my $is_space = {
1003  ## - Some requirements on character encoding declarations. ## TODO    0x0009 => 1, # CHARACTER TABULATION (HT)
1004  ## - "Elements MUST NOT contain content that their content model disallows."    0x000A => 1, # LINE FEED (LF)
1005  ##   ... Some are parse error, some are not (will be reported by c.c.).    #0x000B => 0, # LINE TABULATION (VT)
1006  ## - Polytheistic slash SHOULD NOT be used. (Applied only to atheists.) ## TODO    0x000C => 1, # FORM FEED (FF)
1007  ## - Text (in elements, attributes, and comments) SHOULD NOT contain    #0x000D => 1, # CARRIAGE RETURN (CR)
1008  ##   control characters other than space characters. ## TODO: (what is control character? C0, C1 and DEL?  Unicode control character?)    0x0020 => 1, # SPACE (SP)
1009    };
 ## TODO: HTML5 poses authors two SHOULD-level requirements that cannot  
 ## be detected by the HTML5 parsing algorithm:  
 ## - Text,  
1010    
1011  sub _get_next_token ($) {  sub _get_next_token ($) {
1012    my $self = shift;    my $self = shift;
1013    
1014    if ($self->{self_closing}) {    if ($self->{self_closing}) {
1015      !!!parse-error (type => 'nestc', token => $self->{current_token});      !!!parse-error (type => 'nestc', token => $self->{ct});
1016      ## NOTE: The |self_closing| flag is only set by start tag token.      ## NOTE: The |self_closing| flag is only set by start tag token.
1017      ## In addition, when a start tag token is emitted, it is always set to      ## In addition, when a start tag token is emitted, it is always set to
1018      ## |current_token|.      ## |ct|.
1019      delete $self->{self_closing};      delete $self->{self_closing};
1020    }    }
1021    
# Line 926  sub _get_next_token ($) { Line 1025  sub _get_next_token ($) {
1025    }    }
1026    
1027    A: {    A: {
1028      if ($self->{state} == DATA_STATE) {      if ($self->{state} == PCDATA_STATE) {
1029        if ($self->{next_char} == 0x0026) { # &        ## NOTE: Same as |DATA_STATE|, but only for |PCDATA| content model.
1030    
1031          if ($self->{nc} == 0x0026) { # &
1032            !!!cp (0.1);
1033            ## NOTE: In the spec, the tokenizer is switched to the
1034            ## "entity data state".  In this implementation, the tokenizer
1035            ## is switched to the |ENTITY_STATE|, which is an implementation
1036            ## of the "consume a character reference" algorithm.
1037            $self->{entity_add} = -1;
1038            $self->{prev_state} = DATA_STATE;
1039            $self->{state} = ENTITY_STATE;
1040            !!!next-input-character;
1041            redo A;
1042          } elsif ($self->{nc} == 0x003C) { # <
1043            !!!cp (0.2);
1044            $self->{state} = TAG_OPEN_STATE;
1045            !!!next-input-character;
1046            redo A;
1047          } elsif ($self->{nc} == -1) {
1048            !!!cp (0.3);
1049            !!!emit ({type => END_OF_FILE_TOKEN,
1050                      line => $self->{line}, column => $self->{column}});
1051            last A; ## TODO: ok?
1052          } else {
1053            !!!cp (0.4);
1054            #
1055          }
1056    
1057          # Anything else
1058          my $token = {type => CHARACTER_TOKEN,
1059                       data => chr $self->{nc},
1060                       line => $self->{line}, column => $self->{column},
1061                      };
1062          $self->{read_until}->($token->{data}, q[<&], length $token->{data});
1063    
1064          ## Stay in the state.
1065          !!!next-input-character;
1066          !!!emit ($token);
1067          redo A;
1068        } elsif ($self->{state} == DATA_STATE) {
1069          $self->{s_kwd} = '' unless defined $self->{s_kwd};
1070          if ($self->{nc} == 0x0026) { # &
1071            $self->{s_kwd} = '';
1072          if ($self->{content_model} & CM_ENTITY and # PCDATA | RCDATA          if ($self->{content_model} & CM_ENTITY and # PCDATA | RCDATA
1073              not $self->{escape}) {              not $self->{escape}) {
1074            !!!cp (1);            !!!cp (1);
1075            $self->{state} = ENTITY_DATA_STATE;            ## NOTE: In the spec, the tokenizer is switched to the
1076              ## "entity data state".  In this implementation, the tokenizer
1077              ## is switched to the |ENTITY_STATE|, which is an implementation
1078              ## of the "consume a character reference" algorithm.
1079              $self->{entity_add} = -1;
1080              $self->{prev_state} = DATA_STATE;
1081              $self->{state} = ENTITY_STATE;
1082            !!!next-input-character;            !!!next-input-character;
1083            redo A;            redo A;
1084          } else {          } else {
1085            !!!cp (2);            !!!cp (2);
1086            #            #
1087          }          }
1088        } elsif ($self->{next_char} == 0x002D) { # -        } elsif ($self->{nc} == 0x002D) { # -
1089          if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA          if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
1090            unless ($self->{escape}) {            $self->{s_kwd} .= '-';
1091              if ($self->{prev_char}->[0] == 0x002D and # -            
1092                  $self->{prev_char}->[1] == 0x0021 and # !            if ($self->{s_kwd} eq '<!--') {
1093                  $self->{prev_char}->[2] == 0x003C) { # <              !!!cp (3);
1094                !!!cp (3);              $self->{escape} = 1; # unless $self->{escape};
1095                $self->{escape} = 1;              $self->{s_kwd} = '--';
1096              } else {              #
1097                !!!cp (4);            } elsif ($self->{s_kwd} eq '---') {
1098              }              !!!cp (4);
1099                $self->{s_kwd} = '--';
1100                #
1101            } else {            } else {
1102              !!!cp (5);              !!!cp (5);
1103                #
1104            }            }
1105          }          }
1106                    
1107          #          #
1108        } elsif ($self->{next_char} == 0x003C) { # <        } elsif ($self->{nc} == 0x0021) { # !
1109            if (length $self->{s_kwd}) {
1110              !!!cp (5.1);
1111              $self->{s_kwd} .= '!';
1112              #
1113            } else {
1114              !!!cp (5.2);
1115              #$self->{s_kwd} = '';
1116              #
1117            }
1118            #
1119          } elsif ($self->{nc} == 0x003C) { # <
1120          if ($self->{content_model} & CM_FULL_MARKUP or # PCDATA          if ($self->{content_model} & CM_FULL_MARKUP or # PCDATA
1121              (($self->{content_model} & CM_LIMITED_MARKUP) and # CDATA | RCDATA              (($self->{content_model} & CM_LIMITED_MARKUP) and # CDATA | RCDATA
1122               not $self->{escape})) {               not $self->{escape})) {
# Line 965  sub _get_next_token ($) { Line 1126  sub _get_next_token ($) {
1126            redo A;            redo A;
1127          } else {          } else {
1128            !!!cp (7);            !!!cp (7);
1129              $self->{s_kwd} = '';
1130            #            #
1131          }          }
1132        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1133          if ($self->{escape} and          if ($self->{escape} and
1134              ($self->{content_model} & CM_LIMITED_MARKUP)) { # RCDATA | CDATA              ($self->{content_model} & CM_LIMITED_MARKUP)) { # RCDATA | CDATA
1135            if ($self->{prev_char}->[0] == 0x002D and # -            if ($self->{s_kwd} eq '--') {
               $self->{prev_char}->[1] == 0x002D) { # -  
1136              !!!cp (8);              !!!cp (8);
1137              delete $self->{escape};              delete $self->{escape};
1138            } else {            } else {
# Line 981  sub _get_next_token ($) { Line 1142  sub _get_next_token ($) {
1142            !!!cp (10);            !!!cp (10);
1143          }          }
1144                    
1145            $self->{s_kwd} = '';
1146          #          #
1147        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1148          !!!cp (11);          !!!cp (11);
1149            $self->{s_kwd} = '';
1150          !!!emit ({type => END_OF_FILE_TOKEN,          !!!emit ({type => END_OF_FILE_TOKEN,
1151                    line => $self->{line}, column => $self->{column}});                    line => $self->{line}, column => $self->{column}});
1152          last A; ## TODO: ok?          last A; ## TODO: ok?
1153        } else {        } else {
1154          !!!cp (12);          !!!cp (12);
1155            $self->{s_kwd} = '';
1156            #
1157        }        }
1158    
1159        # Anything else        # Anything else
1160        my $token = {type => CHARACTER_TOKEN,        my $token = {type => CHARACTER_TOKEN,
1161                     data => chr $self->{next_char},                     data => chr $self->{nc},
1162                     line => $self->{line}, column => $self->{column},                     line => $self->{line}, column => $self->{column},
1163                    };                    };
1164        ## Stay in the data state        if ($self->{read_until}->($token->{data}, q[-!<>&],
1165        !!!next-input-character;                                  length $token->{data})) {
1166            $self->{s_kwd} = '';
1167        !!!emit ($token);        }
   
       redo A;  
     } elsif ($self->{state} == ENTITY_DATA_STATE) {  
       ## (cannot happen in CDATA state)  
   
       my ($l, $c) = ($self->{line_prev}, $self->{column_prev});  
         
       my $token = $self->_tokenize_attempt_to_consume_an_entity (0, -1);  
   
       $self->{state} = DATA_STATE;  
       # next-input-character is already done  
1168    
1169        unless (defined $token) {        ## Stay in the data state.
1170          if ($self->{content_model} == PCDATA_CONTENT_MODEL) {
1171          !!!cp (13);          !!!cp (13);
1172          !!!emit ({type => CHARACTER_TOKEN, data => '&',          $self->{state} = PCDATA_STATE;
                   line => $l, column => $c,  
                  });  
1173        } else {        } else {
1174          !!!cp (14);          !!!cp (14);
1175          !!!emit ($token);          ## Stay in the state.
1176        }        }
1177          !!!next-input-character;
1178          !!!emit ($token);
1179        redo A;        redo A;
1180      } elsif ($self->{state} == TAG_OPEN_STATE) {      } elsif ($self->{state} == TAG_OPEN_STATE) {
1181        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
1182          if ($self->{next_char} == 0x002F) { # /          if ($self->{nc} == 0x002F) { # /
1183            !!!cp (15);            !!!cp (15);
1184            !!!next-input-character;            !!!next-input-character;
1185            $self->{state} = CLOSE_TAG_OPEN_STATE;            $self->{state} = CLOSE_TAG_OPEN_STATE;
1186            redo A;            redo A;
1187            } elsif ($self->{nc} == 0x0021) { # !
1188              !!!cp (15.1);
1189              $self->{s_kwd} = '<' unless $self->{escape};
1190              #
1191          } else {          } else {
1192            !!!cp (16);            !!!cp (16);
1193            ## reconsume            #
           $self->{state} = DATA_STATE;  
   
           !!!emit ({type => CHARACTER_TOKEN, data => '<',  
                     line => $self->{line_prev},  
                     column => $self->{column_prev},  
                    });  
   
           redo A;  
1194          }          }
1195    
1196            ## reconsume
1197            $self->{state} = DATA_STATE;
1198            !!!emit ({type => CHARACTER_TOKEN, data => '<',
1199                      line => $self->{line_prev},
1200                      column => $self->{column_prev},
1201                     });
1202            redo A;
1203        } elsif ($self->{content_model} & CM_FULL_MARKUP) { # PCDATA        } elsif ($self->{content_model} & CM_FULL_MARKUP) { # PCDATA
1204          if ($self->{next_char} == 0x0021) { # !          if ($self->{nc} == 0x0021) { # !
1205            !!!cp (17);            !!!cp (17);
1206            $self->{state} = MARKUP_DECLARATION_OPEN_STATE;            $self->{state} = MARKUP_DECLARATION_OPEN_STATE;
1207            !!!next-input-character;            !!!next-input-character;
1208            redo A;            redo A;
1209          } elsif ($self->{next_char} == 0x002F) { # /          } elsif ($self->{nc} == 0x002F) { # /
1210            !!!cp (18);            !!!cp (18);
1211            $self->{state} = CLOSE_TAG_OPEN_STATE;            $self->{state} = CLOSE_TAG_OPEN_STATE;
1212            !!!next-input-character;            !!!next-input-character;
1213            redo A;            redo A;
1214          } elsif (0x0041 <= $self->{next_char} and          } elsif (0x0041 <= $self->{nc} and
1215                   $self->{next_char} <= 0x005A) { # A..Z                   $self->{nc} <= 0x005A) { # A..Z
1216            !!!cp (19);            !!!cp (19);
1217            $self->{current_token}            $self->{ct}
1218              = {type => START_TAG_TOKEN,              = {type => START_TAG_TOKEN,
1219                 tag_name => chr ($self->{next_char} + 0x0020),                 tag_name => chr ($self->{nc} + 0x0020),
1220                 line => $self->{line_prev},                 line => $self->{line_prev},
1221                 column => $self->{column_prev}};                 column => $self->{column_prev}};
1222            $self->{state} = TAG_NAME_STATE;            $self->{state} = TAG_NAME_STATE;
1223            !!!next-input-character;            !!!next-input-character;
1224            redo A;            redo A;
1225          } elsif (0x0061 <= $self->{next_char} and          } elsif (0x0061 <= $self->{nc} and
1226                   $self->{next_char} <= 0x007A) { # a..z                   $self->{nc} <= 0x007A) { # a..z
1227            !!!cp (20);            !!!cp (20);
1228            $self->{current_token} = {type => START_TAG_TOKEN,            $self->{ct} = {type => START_TAG_TOKEN,
1229                                      tag_name => chr ($self->{next_char}),                                      tag_name => chr ($self->{nc}),
1230                                      line => $self->{line_prev},                                      line => $self->{line_prev},
1231                                      column => $self->{column_prev}};                                      column => $self->{column_prev}};
1232            $self->{state} = TAG_NAME_STATE;            $self->{state} = TAG_NAME_STATE;
1233            !!!next-input-character;            !!!next-input-character;
1234            redo A;            redo A;
1235          } elsif ($self->{next_char} == 0x003E) { # >          } elsif ($self->{nc} == 0x003E) { # >
1236            !!!cp (21);            !!!cp (21);
1237            !!!parse-error (type => 'empty start tag',            !!!parse-error (type => 'empty start tag',
1238                            line => $self->{line_prev},                            line => $self->{line_prev},
# Line 1087  sub _get_next_token ($) { Line 1246  sub _get_next_token ($) {
1246                     });                     });
1247    
1248            redo A;            redo A;
1249          } elsif ($self->{next_char} == 0x003F) { # ?          } elsif ($self->{nc} == 0x003F) { # ?
1250            !!!cp (22);            !!!cp (22);
1251            !!!parse-error (type => 'pio',            !!!parse-error (type => 'pio',
1252                            line => $self->{line_prev},                            line => $self->{line_prev},
1253                            column => $self->{column_prev});                            column => $self->{column_prev});
1254            $self->{state} = BOGUS_COMMENT_STATE;            $self->{state} = BOGUS_COMMENT_STATE;
1255            $self->{current_token} = {type => COMMENT_TOKEN, data => '',            $self->{ct} = {type => COMMENT_TOKEN, data => '',
1256                                      line => $self->{line_prev},                                      line => $self->{line_prev},
1257                                      column => $self->{column_prev},                                      column => $self->{column_prev},
1258                                     };                                     };
1259            ## $self->{next_char} is intentionally left as is            ## $self->{nc} is intentionally left as is
1260            redo A;            redo A;
1261          } else {          } else {
1262            !!!cp (23);            !!!cp (23);
# Line 1118  sub _get_next_token ($) { Line 1277  sub _get_next_token ($) {
1277          die "$0: $self->{content_model} in tag open";          die "$0: $self->{content_model} in tag open";
1278        }        }
1279      } elsif ($self->{state} == CLOSE_TAG_OPEN_STATE) {      } elsif ($self->{state} == CLOSE_TAG_OPEN_STATE) {
1280          ## NOTE: The "close tag open state" in the spec is implemented as
1281          ## |CLOSE_TAG_OPEN_STATE| and |CDATA_RCDATA_CLOSE_TAG_STATE|.
1282    
1283        my ($l, $c) = ($self->{line_prev}, $self->{column_prev} - 1); # "<"of"</"        my ($l, $c) = ($self->{line_prev}, $self->{column_prev} - 1); # "<"of"</"
1284        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
1285          if (defined $self->{last_emitted_start_tag_name}) {          if (defined $self->{last_stag_name}) {
1286              $self->{state} = CDATA_RCDATA_CLOSE_TAG_STATE;
1287            ## NOTE: <http://krijnhoetmer.nl/irc-logs/whatwg/20070626#l-564>            $self->{s_kwd} = '';
1288            my @next_char;            ## Reconsume.
1289            TAGNAME: for (my $i = 0; $i < length $self->{last_emitted_start_tag_name}; $i++) {            redo A;
             push @next_char, $self->{next_char};  
             my $c = ord substr ($self->{last_emitted_start_tag_name}, $i, 1);  
             my $C = 0x0061 <= $c && $c <= 0x007A ? $c - 0x0020 : $c;  
             if ($self->{next_char} == $c or $self->{next_char} == $C) {  
               !!!cp (24);  
               !!!next-input-character;  
               next TAGNAME;  
             } else {  
               !!!cp (25);  
               $self->{next_char} = shift @next_char; # reconsume  
               !!!back-next-input-character (@next_char);  
               $self->{state} = DATA_STATE;  
   
               !!!emit ({type => CHARACTER_TOKEN, data => '</',  
                         line => $l, column => $c,  
                        });  
     
               redo A;  
             }  
           }  
           push @next_char, $self->{next_char};  
         
           unless ($self->{next_char} == 0x0009 or # HT  
                   $self->{next_char} == 0x000A or # LF  
                   $self->{next_char} == 0x000B or # VT  
                   $self->{next_char} == 0x000C or # FF  
                   $self->{next_char} == 0x0020 or # SP  
                   $self->{next_char} == 0x003E or # >  
                   $self->{next_char} == 0x002F or # /  
                   $self->{next_char} == -1) {  
             !!!cp (26);  
             $self->{next_char} = shift @next_char; # reconsume  
             !!!back-next-input-character (@next_char);  
             $self->{state} = DATA_STATE;  
             !!!emit ({type => CHARACTER_TOKEN, data => '</',  
                       line => $l, column => $c,  
                      });  
             redo A;  
           } else {  
             !!!cp (27);  
             $self->{next_char} = shift @next_char;  
             !!!back-next-input-character (@next_char);  
             # and consume...  
           }  
1290          } else {          } else {
1291            ## No start tag token has ever been emitted            ## No start tag token has ever been emitted
1292              ## NOTE: See <http://krijnhoetmer.nl/irc-logs/whatwg/20070626#l-564>.
1293            !!!cp (28);            !!!cp (28);
           # next-input-character is already done  
1294            $self->{state} = DATA_STATE;            $self->{state} = DATA_STATE;
1295              ## Reconsume.
1296            !!!emit ({type => CHARACTER_TOKEN, data => '</',            !!!emit ({type => CHARACTER_TOKEN, data => '</',
1297                      line => $l, column => $c,                      line => $l, column => $c,
1298                     });                     });
1299            redo A;            redo A;
1300          }          }
1301        }        }
1302          
1303        if (0x0041 <= $self->{next_char} and        if (0x0041 <= $self->{nc} and
1304            $self->{next_char} <= 0x005A) { # A..Z            $self->{nc} <= 0x005A) { # A..Z
1305          !!!cp (29);          !!!cp (29);
1306          $self->{current_token}          $self->{ct}
1307              = {type => END_TAG_TOKEN,              = {type => END_TAG_TOKEN,
1308                 tag_name => chr ($self->{next_char} + 0x0020),                 tag_name => chr ($self->{nc} + 0x0020),
1309                 line => $l, column => $c};                 line => $l, column => $c};
1310          $self->{state} = TAG_NAME_STATE;          $self->{state} = TAG_NAME_STATE;
1311          !!!next-input-character;          !!!next-input-character;
1312          redo A;          redo A;
1313        } elsif (0x0061 <= $self->{next_char} and        } elsif (0x0061 <= $self->{nc} and
1314                 $self->{next_char} <= 0x007A) { # a..z                 $self->{nc} <= 0x007A) { # a..z
1315          !!!cp (30);          !!!cp (30);
1316          $self->{current_token} = {type => END_TAG_TOKEN,          $self->{ct} = {type => END_TAG_TOKEN,
1317                                    tag_name => chr ($self->{next_char}),                                    tag_name => chr ($self->{nc}),
1318                                    line => $l, column => $c};                                    line => $l, column => $c};
1319          $self->{state} = TAG_NAME_STATE;          $self->{state} = TAG_NAME_STATE;
1320          !!!next-input-character;          !!!next-input-character;
1321          redo A;          redo A;
1322        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1323          !!!cp (31);          !!!cp (31);
1324          !!!parse-error (type => 'empty end tag',          !!!parse-error (type => 'empty end tag',
1325                          line => $self->{line_prev}, ## "<" in "</>"                          line => $self->{line_prev}, ## "<" in "</>"
# Line 1208  sub _get_next_token ($) { Line 1327  sub _get_next_token ($) {
1327          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1328          !!!next-input-character;          !!!next-input-character;
1329          redo A;          redo A;
1330        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1331          !!!cp (32);          !!!cp (32);
1332          !!!parse-error (type => 'bare etago');          !!!parse-error (type => 'bare etago');
1333          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
# Line 1223  sub _get_next_token ($) { Line 1342  sub _get_next_token ($) {
1342          !!!cp (33);          !!!cp (33);
1343          !!!parse-error (type => 'bogus end tag');          !!!parse-error (type => 'bogus end tag');
1344          $self->{state} = BOGUS_COMMENT_STATE;          $self->{state} = BOGUS_COMMENT_STATE;
1345          $self->{current_token} = {type => COMMENT_TOKEN, data => '',          $self->{ct} = {type => COMMENT_TOKEN, data => '',
1346                                    line => $self->{line_prev}, # "<" of "</"                                    line => $self->{line_prev}, # "<" of "</"
1347                                    column => $self->{column_prev} - 1,                                    column => $self->{column_prev} - 1,
1348                                   };                                   };
1349          ## $self->{next_char} is intentionally left as is          ## NOTE: $self->{nc} is intentionally left as is.
1350          redo A;          ## Although the "anything else" case of the spec not explicitly
1351            ## states that the next input character is to be reconsumed,
1352            ## it will be included to the |data| of the comment token
1353            ## generated from the bogus end tag, as defined in the
1354            ## "bogus comment state" entry.
1355            redo A;
1356          }
1357        } elsif ($self->{state} == CDATA_RCDATA_CLOSE_TAG_STATE) {
1358          my $ch = substr $self->{last_stag_name}, length $self->{s_kwd}, 1;
1359          if (length $ch) {
1360            my $CH = $ch;
1361            $ch =~ tr/a-z/A-Z/;
1362            my $nch = chr $self->{nc};
1363            if ($nch eq $ch or $nch eq $CH) {
1364              !!!cp (24);
1365              ## Stay in the state.
1366              $self->{s_kwd} .= $nch;
1367              !!!next-input-character;
1368              redo A;
1369            } else {
1370              !!!cp (25);
1371              $self->{state} = DATA_STATE;
1372              ## Reconsume.
1373              !!!emit ({type => CHARACTER_TOKEN,
1374                        data => '</' . $self->{s_kwd},
1375                        line => $self->{line_prev},
1376                        column => $self->{column_prev} - 1 - length $self->{s_kwd},
1377                       });
1378              redo A;
1379            }
1380          } else { # after "<{tag-name}"
1381            unless ($is_space->{$self->{nc}} or
1382                    {
1383                     0x003E => 1, # >
1384                     0x002F => 1, # /
1385                     -1 => 1, # EOF
1386                    }->{$self->{nc}}) {
1387              !!!cp (26);
1388              ## Reconsume.
1389              $self->{state} = DATA_STATE;
1390              !!!emit ({type => CHARACTER_TOKEN,
1391                        data => '</' . $self->{s_kwd},
1392                        line => $self->{line_prev},
1393                        column => $self->{column_prev} - 1 - length $self->{s_kwd},
1394                       });
1395              redo A;
1396            } else {
1397              !!!cp (27);
1398              $self->{ct}
1399                  = {type => END_TAG_TOKEN,
1400                     tag_name => $self->{last_stag_name},
1401                     line => $self->{line_prev},
1402                     column => $self->{column_prev} - 1 - length $self->{s_kwd}};
1403              $self->{state} = TAG_NAME_STATE;
1404              ## Reconsume.
1405              redo A;
1406            }
1407        }        }
1408      } elsif ($self->{state} == TAG_NAME_STATE) {      } elsif ($self->{state} == TAG_NAME_STATE) {
1409        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
1410          !!!cp (34);          !!!cp (34);
1411          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1412          !!!next-input-character;          !!!next-input-character;
1413          redo A;          redo A;
1414        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1415          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1416            !!!cp (35);            !!!cp (35);
1417            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1418          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1419            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1420            #if ($self->{current_token}->{attributes}) {            #if ($self->{ct}->{attributes}) {
1421            #  ## NOTE: This should never be reached.            #  ## NOTE: This should never be reached.
1422            #  !!! cp (36);            #  !!! cp (36);
1423            #  !!! parse-error (type => 'end tag attribute');            #  !!! parse-error (type => 'end tag attribute');
# Line 1254  sub _get_next_token ($) { Line 1425  sub _get_next_token ($) {
1425              !!!cp (37);              !!!cp (37);
1426            #}            #}
1427          } else {          } else {
1428            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1429          }          }
1430          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1431          !!!next-input-character;          !!!next-input-character;
1432    
1433          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1434    
1435          redo A;          redo A;
1436        } elsif (0x0041 <= $self->{next_char} and        } elsif (0x0041 <= $self->{nc} and
1437                 $self->{next_char} <= 0x005A) { # A..Z                 $self->{nc} <= 0x005A) { # A..Z
1438          !!!cp (38);          !!!cp (38);
1439          $self->{current_token}->{tag_name} .= chr ($self->{next_char} + 0x0020);          $self->{ct}->{tag_name} .= chr ($self->{nc} + 0x0020);
1440            # start tag or end tag            # start tag or end tag
1441          ## Stay in this state          ## Stay in this state
1442          !!!next-input-character;          !!!next-input-character;
1443          redo A;          redo A;
1444        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1445          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1446          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1447            !!!cp (39);            !!!cp (39);
1448            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1449          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1450            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1451            #if ($self->{current_token}->{attributes}) {            #if ($self->{ct}->{attributes}) {
1452            #  ## NOTE: This state should never be reached.            #  ## NOTE: This state should never be reached.
1453            #  !!! cp (40);            #  !!! cp (40);
1454            #  !!! parse-error (type => 'end tag attribute');            #  !!! parse-error (type => 'end tag attribute');
# Line 1285  sub _get_next_token ($) { Line 1456  sub _get_next_token ($) {
1456              !!!cp (41);              !!!cp (41);
1457            #}            #}
1458          } else {          } else {
1459            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1460          }          }
1461          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1462          # reconsume          # reconsume
1463    
1464          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1465    
1466          redo A;          redo A;
1467        } elsif ($self->{next_char} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
1468          !!!cp (42);          !!!cp (42);
1469          $self->{state} = SELF_CLOSING_START_TAG_STATE;          $self->{state} = SELF_CLOSING_START_TAG_STATE;
1470          !!!next-input-character;          !!!next-input-character;
1471          redo A;          redo A;
1472        } else {        } else {
1473          !!!cp (44);          !!!cp (44);
1474          $self->{current_token}->{tag_name} .= chr $self->{next_char};          $self->{ct}->{tag_name} .= chr $self->{nc};
1475            # start tag or end tag            # start tag or end tag
1476          ## Stay in the state          ## Stay in the state
1477          !!!next-input-character;          !!!next-input-character;
1478          redo A;          redo A;
1479        }        }
1480      } elsif ($self->{state} == BEFORE_ATTRIBUTE_NAME_STATE) {      } elsif ($self->{state} == BEFORE_ATTRIBUTE_NAME_STATE) {
1481        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
1482          !!!cp (45);          !!!cp (45);
1483          ## Stay in the state          ## Stay in the state
1484          !!!next-input-character;          !!!next-input-character;
1485          redo A;          redo A;
1486        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1487          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1488            !!!cp (46);            !!!cp (46);
1489            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1490          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1491            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1492            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1493              !!!cp (47);              !!!cp (47);
1494              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1495            } else {            } else {
1496              !!!cp (48);              !!!cp (48);
1497            }            }
1498          } else {          } else {
1499            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1500          }          }
1501          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1502          !!!next-input-character;          !!!next-input-character;
1503    
1504          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1505    
1506          redo A;          redo A;
1507        } elsif (0x0041 <= $self->{next_char} and        } elsif (0x0041 <= $self->{nc} and
1508                 $self->{next_char} <= 0x005A) { # A..Z                 $self->{nc} <= 0x005A) { # A..Z
1509          !!!cp (49);          !!!cp (49);
1510          $self->{current_attribute}          $self->{ca}
1511              = {name => chr ($self->{next_char} + 0x0020),              = {name => chr ($self->{nc} + 0x0020),
1512                 value => '',                 value => '',
1513                 line => $self->{line}, column => $self->{column}};                 line => $self->{line}, column => $self->{column}};
1514          $self->{state} = ATTRIBUTE_NAME_STATE;          $self->{state} = ATTRIBUTE_NAME_STATE;
1515          !!!next-input-character;          !!!next-input-character;
1516          redo A;          redo A;
1517        } elsif ($self->{next_char} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
1518          !!!cp (50);          !!!cp (50);
1519          $self->{state} = SELF_CLOSING_START_TAG_STATE;          $self->{state} = SELF_CLOSING_START_TAG_STATE;
1520          !!!next-input-character;          !!!next-input-character;
1521          redo A;          redo A;
1522        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1523          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1524          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1525            !!!cp (52);            !!!cp (52);
1526            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1527          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1528            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1529            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1530              !!!cp (53);              !!!cp (53);
1531              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1532            } else {            } else {
1533              !!!cp (54);              !!!cp (54);
1534            }            }
1535          } else {          } else {
1536            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1537          }          }
1538          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1539          # reconsume          # reconsume
1540    
1541          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1542    
1543          redo A;          redo A;
1544        } else {        } else {
# Line 1379  sub _get_next_token ($) { Line 1546  sub _get_next_token ($) {
1546               0x0022 => 1, # "               0x0022 => 1, # "
1547               0x0027 => 1, # '               0x0027 => 1, # '
1548               0x003D => 1, # =               0x003D => 1, # =
1549              }->{$self->{next_char}}) {              }->{$self->{nc}}) {
1550            !!!cp (55);            !!!cp (55);
1551            !!!parse-error (type => 'bad attribute name');            !!!parse-error (type => 'bad attribute name');
1552          } else {          } else {
1553            !!!cp (56);            !!!cp (56);
1554          }          }
1555          $self->{current_attribute}          $self->{ca}
1556              = {name => chr ($self->{next_char}),              = {name => chr ($self->{nc}),
1557                 value => '',                 value => '',
1558                 line => $self->{line}, column => $self->{column}};                 line => $self->{line}, column => $self->{column}};
1559          $self->{state} = ATTRIBUTE_NAME_STATE;          $self->{state} = ATTRIBUTE_NAME_STATE;
# Line 1395  sub _get_next_token ($) { Line 1562  sub _get_next_token ($) {
1562        }        }
1563      } elsif ($self->{state} == ATTRIBUTE_NAME_STATE) {      } elsif ($self->{state} == ATTRIBUTE_NAME_STATE) {
1564        my $before_leave = sub {        my $before_leave = sub {
1565          if (exists $self->{current_token}->{attributes} # start tag or end tag          if (exists $self->{ct}->{attributes} # start tag or end tag
1566              ->{$self->{current_attribute}->{name}}) { # MUST              ->{$self->{ca}->{name}}) { # MUST
1567            !!!cp (57);            !!!cp (57);
1568            !!!parse-error (type => 'duplicate attribute', text => $self->{current_attribute}->{name}, line => $self->{current_attribute}->{line}, column => $self->{current_attribute}->{column});            !!!parse-error (type => 'duplicate attribute', text => $self->{ca}->{name}, line => $self->{ca}->{line}, column => $self->{ca}->{column});
1569            ## Discard $self->{current_attribute} # MUST            ## Discard $self->{ca} # MUST
1570          } else {          } else {
1571            !!!cp (58);            !!!cp (58);
1572            $self->{current_token}->{attributes}->{$self->{current_attribute}->{name}}            $self->{ct}->{attributes}->{$self->{ca}->{name}}
1573              = $self->{current_attribute};              = $self->{ca};
1574          }          }
1575        }; # $before_leave        }; # $before_leave
1576    
1577        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
1578          !!!cp (59);          !!!cp (59);
1579          $before_leave->();          $before_leave->();
1580          $self->{state} = AFTER_ATTRIBUTE_NAME_STATE;          $self->{state} = AFTER_ATTRIBUTE_NAME_STATE;
1581          !!!next-input-character;          !!!next-input-character;
1582          redo A;          redo A;
1583        } elsif ($self->{next_char} == 0x003D) { # =        } elsif ($self->{nc} == 0x003D) { # =
1584          !!!cp (60);          !!!cp (60);
1585          $before_leave->();          $before_leave->();
1586          $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;          $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;
1587          !!!next-input-character;          !!!next-input-character;
1588          redo A;          redo A;
1589        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1590          $before_leave->();          $before_leave->();
1591          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1592            !!!cp (61);            !!!cp (61);
1593            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1594          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1595            !!!cp (62);            !!!cp (62);
1596            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1597            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1598              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1599            }            }
1600          } else {          } else {
1601            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1602          }          }
1603          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1604          !!!next-input-character;          !!!next-input-character;
1605    
1606          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1607    
1608          redo A;          redo A;
1609        } elsif (0x0041 <= $self->{next_char} and        } elsif (0x0041 <= $self->{nc} and
1610                 $self->{next_char} <= 0x005A) { # A..Z                 $self->{nc} <= 0x005A) { # A..Z
1611          !!!cp (63);          !!!cp (63);
1612          $self->{current_attribute}->{name} .= chr ($self->{next_char} + 0x0020);          $self->{ca}->{name} .= chr ($self->{nc} + 0x0020);
1613          ## Stay in the state          ## Stay in the state
1614          !!!next-input-character;          !!!next-input-character;
1615          redo A;          redo A;
1616        } elsif ($self->{next_char} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
1617          !!!cp (64);          !!!cp (64);
1618          $before_leave->();          $before_leave->();
1619          $self->{state} = SELF_CLOSING_START_TAG_STATE;          $self->{state} = SELF_CLOSING_START_TAG_STATE;
1620          !!!next-input-character;          !!!next-input-character;
1621          redo A;          redo A;
1622        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1623          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1624          $before_leave->();          $before_leave->();
1625          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1626            !!!cp (66);            !!!cp (66);
1627            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1628          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1629            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1630            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1631              !!!cp (67);              !!!cp (67);
1632              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1633            } else {            } else {
# Line 1472  sub _get_next_token ($) { Line 1635  sub _get_next_token ($) {
1635              !!!cp (68);              !!!cp (68);
1636            }            }
1637          } else {          } else {
1638            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1639          }          }
1640          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1641          # reconsume          # reconsume
1642    
1643          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1644    
1645          redo A;          redo A;
1646        } else {        } else {
1647          if ($self->{next_char} == 0x0022 or # "          if ($self->{nc} == 0x0022 or # "
1648              $self->{next_char} == 0x0027) { # '              $self->{nc} == 0x0027) { # '
1649            !!!cp (69);            !!!cp (69);
1650            !!!parse-error (type => 'bad attribute name');            !!!parse-error (type => 'bad attribute name');
1651          } else {          } else {
1652            !!!cp (70);            !!!cp (70);
1653          }          }
1654          $self->{current_attribute}->{name} .= chr ($self->{next_char});          $self->{ca}->{name} .= chr ($self->{nc});
1655          ## Stay in the state          ## Stay in the state
1656          !!!next-input-character;          !!!next-input-character;
1657          redo A;          redo A;
1658        }        }
1659      } elsif ($self->{state} == AFTER_ATTRIBUTE_NAME_STATE) {      } elsif ($self->{state} == AFTER_ATTRIBUTE_NAME_STATE) {
1660        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
1661          !!!cp (71);          !!!cp (71);
1662          ## Stay in the state          ## Stay in the state
1663          !!!next-input-character;          !!!next-input-character;
1664          redo A;          redo A;
1665        } elsif ($self->{next_char} == 0x003D) { # =        } elsif ($self->{nc} == 0x003D) { # =
1666          !!!cp (72);          !!!cp (72);
1667          $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;          $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;
1668          !!!next-input-character;          !!!next-input-character;
1669          redo A;          redo A;
1670        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1671          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1672            !!!cp (73);            !!!cp (73);
1673            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1674          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1675            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1676            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1677              !!!cp (74);              !!!cp (74);
1678              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1679            } else {            } else {
# Line 1522  sub _get_next_token ($) { Line 1681  sub _get_next_token ($) {
1681              !!!cp (75);              !!!cp (75);
1682            }            }
1683          } else {          } else {
1684            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1685          }          }
1686          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1687          !!!next-input-character;          !!!next-input-character;
1688    
1689          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1690    
1691          redo A;          redo A;
1692        } elsif (0x0041 <= $self->{next_char} and        } elsif (0x0041 <= $self->{nc} and
1693                 $self->{next_char} <= 0x005A) { # A..Z                 $self->{nc} <= 0x005A) { # A..Z
1694          !!!cp (76);          !!!cp (76);
1695          $self->{current_attribute}          $self->{ca}
1696              = {name => chr ($self->{next_char} + 0x0020),              = {name => chr ($self->{nc} + 0x0020),
1697                 value => '',                 value => '',
1698                 line => $self->{line}, column => $self->{column}};                 line => $self->{line}, column => $self->{column}};
1699          $self->{state} = ATTRIBUTE_NAME_STATE;          $self->{state} = ATTRIBUTE_NAME_STATE;
1700          !!!next-input-character;          !!!next-input-character;
1701          redo A;          redo A;
1702        } elsif ($self->{next_char} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
1703          !!!cp (77);          !!!cp (77);
1704          $self->{state} = SELF_CLOSING_START_TAG_STATE;          $self->{state} = SELF_CLOSING_START_TAG_STATE;
1705          !!!next-input-character;          !!!next-input-character;
1706          redo A;          redo A;
1707        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1708          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1709          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1710            !!!cp (79);            !!!cp (79);
1711            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1712          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1713            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1714            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1715              !!!cp (80);              !!!cp (80);
1716              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1717            } else {            } else {
# Line 1560  sub _get_next_token ($) { Line 1719  sub _get_next_token ($) {
1719              !!!cp (81);              !!!cp (81);
1720            }            }
1721          } else {          } else {
1722            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1723          }          }
1724          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1725          # reconsume          # reconsume
1726    
1727          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1728    
1729          redo A;          redo A;
1730        } else {        } else {
1731          if ($self->{next_char} == 0x0022 or # "          if ($self->{nc} == 0x0022 or # "
1732              $self->{next_char} == 0x0027) { # '              $self->{nc} == 0x0027) { # '
1733            !!!cp (78);            !!!cp (78);
1734            !!!parse-error (type => 'bad attribute name');            !!!parse-error (type => 'bad attribute name');
1735          } else {          } else {
1736            !!!cp (82);            !!!cp (82);
1737          }          }
1738          $self->{current_attribute}          $self->{ca}
1739              = {name => chr ($self->{next_char}),              = {name => chr ($self->{nc}),
1740                 value => '',                 value => '',
1741                 line => $self->{line}, column => $self->{column}};                 line => $self->{line}, column => $self->{column}};
1742          $self->{state} = ATTRIBUTE_NAME_STATE;          $self->{state} = ATTRIBUTE_NAME_STATE;
# Line 1585  sub _get_next_token ($) { Line 1744  sub _get_next_token ($) {
1744          redo A;                  redo A;        
1745        }        }
1746      } elsif ($self->{state} == BEFORE_ATTRIBUTE_VALUE_STATE) {      } elsif ($self->{state} == BEFORE_ATTRIBUTE_VALUE_STATE) {
1747        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP        
1748          !!!cp (83);          !!!cp (83);
1749          ## Stay in the state          ## Stay in the state
1750          !!!next-input-character;          !!!next-input-character;
1751          redo A;          redo A;
1752        } elsif ($self->{next_char} == 0x0022) { # "        } elsif ($self->{nc} == 0x0022) { # "
1753          !!!cp (84);          !!!cp (84);
1754          $self->{state} = ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE;          $self->{state} = ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE;
1755          !!!next-input-character;          !!!next-input-character;
1756          redo A;          redo A;
1757        } elsif ($self->{next_char} == 0x0026) { # &        } elsif ($self->{nc} == 0x0026) { # &
1758          !!!cp (85);          !!!cp (85);
1759          $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;          $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;
1760          ## reconsume          ## reconsume
1761          redo A;          redo A;
1762        } elsif ($self->{next_char} == 0x0027) { # '        } elsif ($self->{nc} == 0x0027) { # '
1763          !!!cp (86);          !!!cp (86);
1764          $self->{state} = ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE;          $self->{state} = ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE;
1765          !!!next-input-character;          !!!next-input-character;
1766          redo A;          redo A;
1767        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1768          !!!parse-error (type => 'empty unquoted attribute value');          !!!parse-error (type => 'empty unquoted attribute value');
1769          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1770            !!!cp (87);            !!!cp (87);
1771            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1772          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1773            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1774            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1775              !!!cp (88);              !!!cp (88);
1776              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1777            } else {            } else {
# Line 1624  sub _get_next_token ($) { Line 1779  sub _get_next_token ($) {
1779              !!!cp (89);              !!!cp (89);
1780            }            }
1781          } else {          } else {
1782            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1783          }          }
1784          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1785          !!!next-input-character;          !!!next-input-character;
1786    
1787          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1788    
1789          redo A;          redo A;
1790        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1791          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1792          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1793            !!!cp (90);            !!!cp (90);
1794            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1795          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1796            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1797            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1798              !!!cp (91);              !!!cp (91);
1799              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1800            } else {            } else {
# Line 1647  sub _get_next_token ($) { Line 1802  sub _get_next_token ($) {
1802              !!!cp (92);              !!!cp (92);
1803            }            }
1804          } else {          } else {
1805            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1806          }          }
1807          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1808          ## reconsume          ## reconsume
1809    
1810          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1811    
1812          redo A;          redo A;
1813        } else {        } else {
1814          if ($self->{next_char} == 0x003D) { # =          if ($self->{nc} == 0x003D) { # =
1815            !!!cp (93);            !!!cp (93);
1816            !!!parse-error (type => 'bad attribute value');            !!!parse-error (type => 'bad attribute value');
1817          } else {          } else {
1818            !!!cp (94);            !!!cp (94);
1819          }          }
1820          $self->{current_attribute}->{value} .= chr ($self->{next_char});          $self->{ca}->{value} .= chr ($self->{nc});
1821          $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;          $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;
1822          !!!next-input-character;          !!!next-input-character;
1823          redo A;          redo A;
1824        }        }
1825      } elsif ($self->{state} == ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE) {      } elsif ($self->{state} == ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE) {
1826        if ($self->{next_char} == 0x0022) { # "        if ($self->{nc} == 0x0022) { # "
1827          !!!cp (95);          !!!cp (95);
1828          $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;          $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;
1829          !!!next-input-character;          !!!next-input-character;
1830          redo A;          redo A;
1831        } elsif ($self->{next_char} == 0x0026) { # &        } elsif ($self->{nc} == 0x0026) { # &
1832          !!!cp (96);          !!!cp (96);
1833          $self->{last_attribute_value_state} = $self->{state};          ## NOTE: In the spec, the tokenizer is switched to the
1834          $self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE;          ## "entity in attribute value state".  In this implementation, the
1835            ## tokenizer is switched to the |ENTITY_STATE|, which is an
1836            ## implementation of the "consume a character reference" algorithm.
1837            $self->{prev_state} = $self->{state};
1838            $self->{entity_add} = 0x0022; # "
1839            $self->{state} = ENTITY_STATE;
1840          !!!next-input-character;          !!!next-input-character;
1841          redo A;          redo A;
1842        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1843          !!!parse-error (type => 'unclosed attribute value');          !!!parse-error (type => 'unclosed attribute value');
1844          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1845            !!!cp (97);            !!!cp (97);
1846            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1847          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1848            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1849            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1850              !!!cp (98);              !!!cp (98);
1851              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1852            } else {            } else {
# Line 1694  sub _get_next_token ($) { Line 1854  sub _get_next_token ($) {
1854              !!!cp (99);              !!!cp (99);
1855            }            }
1856          } else {          } else {
1857            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1858          }          }
1859          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1860          ## reconsume          ## reconsume
1861    
1862          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1863    
1864          redo A;          redo A;
1865        } else {        } else {
1866          !!!cp (100);          !!!cp (100);
1867          $self->{current_attribute}->{value} .= chr ($self->{next_char});          $self->{ca}->{value} .= chr ($self->{nc});
1868            $self->{read_until}->($self->{ca}->{value},
1869                                  q["&],
1870                                  length $self->{ca}->{value});
1871    
1872          ## Stay in the state          ## Stay in the state
1873          !!!next-input-character;          !!!next-input-character;
1874          redo A;          redo A;
1875        }        }
1876      } elsif ($self->{state} == ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE) {      } elsif ($self->{state} == ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE) {
1877        if ($self->{next_char} == 0x0027) { # '        if ($self->{nc} == 0x0027) { # '
1878          !!!cp (101);          !!!cp (101);
1879          $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;          $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;
1880          !!!next-input-character;          !!!next-input-character;
1881          redo A;          redo A;
1882        } elsif ($self->{next_char} == 0x0026) { # &        } elsif ($self->{nc} == 0x0026) { # &
1883          !!!cp (102);          !!!cp (102);
1884          $self->{last_attribute_value_state} = $self->{state};          ## NOTE: In the spec, the tokenizer is switched to the
1885          $self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE;          ## "entity in attribute value state".  In this implementation, the
1886            ## tokenizer is switched to the |ENTITY_STATE|, which is an
1887            ## implementation of the "consume a character reference" algorithm.
1888            $self->{entity_add} = 0x0027; # '
1889            $self->{prev_state} = $self->{state};
1890            $self->{state} = ENTITY_STATE;
1891          !!!next-input-character;          !!!next-input-character;
1892          redo A;          redo A;
1893        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1894          !!!parse-error (type => 'unclosed attribute value');          !!!parse-error (type => 'unclosed attribute value');
1895          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1896            !!!cp (103);            !!!cp (103);
1897            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1898          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1899            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1900            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1901              !!!cp (104);              !!!cp (104);
1902              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1903            } else {            } else {
# Line 1736  sub _get_next_token ($) { Line 1905  sub _get_next_token ($) {
1905              !!!cp (105);              !!!cp (105);
1906            }            }
1907          } else {          } else {
1908            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1909          }          }
1910          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1911          ## reconsume          ## reconsume
1912    
1913          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1914    
1915          redo A;          redo A;
1916        } else {        } else {
1917          !!!cp (106);          !!!cp (106);
1918          $self->{current_attribute}->{value} .= chr ($self->{next_char});          $self->{ca}->{value} .= chr ($self->{nc});
1919            $self->{read_until}->($self->{ca}->{value},
1920                                  q['&],
1921                                  length $self->{ca}->{value});
1922    
1923          ## Stay in the state          ## Stay in the state
1924          !!!next-input-character;          !!!next-input-character;
1925          redo A;          redo A;
1926        }        }
1927      } elsif ($self->{state} == ATTRIBUTE_VALUE_UNQUOTED_STATE) {      } elsif ($self->{state} == ATTRIBUTE_VALUE_UNQUOTED_STATE) {
1928        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # HT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
1929          !!!cp (107);          !!!cp (107);
1930          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1931          !!!next-input-character;          !!!next-input-character;
1932          redo A;          redo A;
1933        } elsif ($self->{next_char} == 0x0026) { # &        } elsif ($self->{nc} == 0x0026) { # &
1934          !!!cp (108);          !!!cp (108);
1935          $self->{last_attribute_value_state} = $self->{state};          ## NOTE: In the spec, the tokenizer is switched to the
1936          $self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE;          ## "entity in attribute value state".  In this implementation, the
1937            ## tokenizer is switched to the |ENTITY_STATE|, which is an
1938            ## implementation of the "consume a character reference" algorithm.
1939            $self->{entity_add} = -1;
1940            $self->{prev_state} = $self->{state};
1941            $self->{state} = ENTITY_STATE;
1942          !!!next-input-character;          !!!next-input-character;
1943          redo A;          redo A;
1944        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1945          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1946            !!!cp (109);            !!!cp (109);
1947            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1948          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1949            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1950            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1951              !!!cp (110);              !!!cp (110);
1952              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1953            } else {            } else {
# Line 1781  sub _get_next_token ($) { Line 1955  sub _get_next_token ($) {
1955              !!!cp (111);              !!!cp (111);
1956            }            }
1957          } else {          } else {
1958            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1959          }          }
1960          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1961          !!!next-input-character;          !!!next-input-character;
1962    
1963          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1964    
1965          redo A;          redo A;
1966        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1967          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1968          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1969            !!!cp (112);            !!!cp (112);
1970            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1971          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1972            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1973            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1974              !!!cp (113);              !!!cp (113);
1975              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1976            } else {            } else {
# Line 1804  sub _get_next_token ($) { Line 1978  sub _get_next_token ($) {
1978              !!!cp (114);              !!!cp (114);
1979            }            }
1980          } else {          } else {
1981            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1982          }          }
1983          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1984          ## reconsume          ## reconsume
1985    
1986          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1987    
1988          redo A;          redo A;
1989        } else {        } else {
# Line 1817  sub _get_next_token ($) { Line 1991  sub _get_next_token ($) {
1991               0x0022 => 1, # "               0x0022 => 1, # "
1992               0x0027 => 1, # '               0x0027 => 1, # '
1993               0x003D => 1, # =               0x003D => 1, # =
1994              }->{$self->{next_char}}) {              }->{$self->{nc}}) {
1995            !!!cp (115);            !!!cp (115);
1996            !!!parse-error (type => 'bad attribute value');            !!!parse-error (type => 'bad attribute value');
1997          } else {          } else {
1998            !!!cp (116);            !!!cp (116);
1999          }          }
2000          $self->{current_attribute}->{value} .= chr ($self->{next_char});          $self->{ca}->{value} .= chr ($self->{nc});
2001            $self->{read_until}->($self->{ca}->{value},
2002                                  q["'=& >],
2003                                  length $self->{ca}->{value});
2004    
2005          ## Stay in the state          ## Stay in the state
2006          !!!next-input-character;          !!!next-input-character;
2007          redo A;          redo A;
2008        }        }
     } elsif ($self->{state} == ENTITY_IN_ATTRIBUTE_VALUE_STATE) {  
       my $token = $self->_tokenize_attempt_to_consume_an_entity  
           (1,  
            $self->{last_attribute_value_state}  
              == ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE ? 0x0022 : # "  
            $self->{last_attribute_value_state}  
              == ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE ? 0x0027 : # '  
            -1);  
   
       unless (defined $token) {  
         !!!cp (117);  
         $self->{current_attribute}->{value} .= '&';  
       } else {  
         !!!cp (118);  
         $self->{current_attribute}->{value} .= $token->{data};  
         $self->{current_attribute}->{has_reference} = $token->{has_reference};  
         ## ISSUE: spec says "append the returned character token to the current attribute's value"  
       }  
   
       $self->{state} = $self->{last_attribute_value_state};  
       # next-input-character is already done  
       redo A;  
2009      } elsif ($self->{state} == AFTER_ATTRIBUTE_VALUE_QUOTED_STATE) {      } elsif ($self->{state} == AFTER_ATTRIBUTE_VALUE_QUOTED_STATE) {
2010        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
2011          !!!cp (118);          !!!cp (118);
2012          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
2013          !!!next-input-character;          !!!next-input-character;
2014          redo A;          redo A;
2015        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2016          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
2017            !!!cp (119);            !!!cp (119);
2018            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
2019          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
2020            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
2021            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
2022              !!!cp (120);              !!!cp (120);
2023              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
2024            } else {            } else {
# Line 1874  sub _get_next_token ($) { Line 2026  sub _get_next_token ($) {
2026              !!!cp (121);              !!!cp (121);
2027            }            }
2028          } else {          } else {
2029            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
2030          }          }
2031          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2032          !!!next-input-character;          !!!next-input-character;
2033    
2034          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
2035    
2036          redo A;          redo A;
2037        } elsif ($self->{next_char} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
2038          !!!cp (122);          !!!cp (122);
2039          $self->{state} = SELF_CLOSING_START_TAG_STATE;          $self->{state} = SELF_CLOSING_START_TAG_STATE;
2040          !!!next-input-character;          !!!next-input-character;
2041          redo A;          redo A;
2042        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2043          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
2044          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
2045            !!!cp (122.3);            !!!cp (122.3);
2046            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
2047          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
2048            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
2049              !!!cp (122.1);              !!!cp (122.1);
2050              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
2051            } else {            } else {
# Line 1901  sub _get_next_token ($) { Line 2053  sub _get_next_token ($) {
2053              !!!cp (122.2);              !!!cp (122.2);
2054            }            }
2055          } else {          } else {
2056            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
2057          }          }
2058          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2059          ## Reconsume.          ## Reconsume.
2060          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
2061          redo A;          redo A;
2062        } else {        } else {
2063          !!!cp ('124.1');          !!!cp ('124.1');
# Line 1915  sub _get_next_token ($) { Line 2067  sub _get_next_token ($) {
2067          redo A;          redo A;
2068        }        }
2069      } elsif ($self->{state} == SELF_CLOSING_START_TAG_STATE) {      } elsif ($self->{state} == SELF_CLOSING_START_TAG_STATE) {
2070        if ($self->{next_char} == 0x003E) { # >        if ($self->{nc} == 0x003E) { # >
2071          if ($self->{current_token}->{type} == END_TAG_TOKEN) {          if ($self->{ct}->{type} == END_TAG_TOKEN) {
2072            !!!cp ('124.2');            !!!cp ('124.2');
2073            !!!parse-error (type => 'nestc', token => $self->{current_token});            !!!parse-error (type => 'nestc', token => $self->{ct});
2074            ## TODO: Different type than slash in start tag            ## TODO: Different type than slash in start tag
2075            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
2076            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
2077              !!!cp ('124.4');              !!!cp ('124.4');
2078              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
2079            } else {            } else {
# Line 1936  sub _get_next_token ($) { Line 2088  sub _get_next_token ($) {
2088          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2089          !!!next-input-character;          !!!next-input-character;
2090    
2091          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
2092    
2093          redo A;          redo A;
2094        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2095          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
2096          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
2097            !!!cp (124.7);            !!!cp (124.7);
2098            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
2099          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
2100            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
2101              !!!cp (124.5);              !!!cp (124.5);
2102              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
2103            } else {            } else {
# Line 1953  sub _get_next_token ($) { Line 2105  sub _get_next_token ($) {
2105              !!!cp (124.6);              !!!cp (124.6);
2106            }            }
2107          } else {          } else {
2108            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
2109          }          }
2110          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2111          ## Reconsume.          ## Reconsume.
2112          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
2113          redo A;          redo A;
2114        } else {        } else {
2115          !!!cp ('124.4');          !!!cp ('124.4');
# Line 1969  sub _get_next_token ($) { Line 2121  sub _get_next_token ($) {
2121        }        }
2122      } elsif ($self->{state} == BOGUS_COMMENT_STATE) {      } elsif ($self->{state} == BOGUS_COMMENT_STATE) {
2123        ## (only happen if PCDATA state)        ## (only happen if PCDATA state)
         
       ## NOTE: Set by the previous state  
       #my $token = {type => COMMENT_TOKEN, data => ''};  
2124    
2125        BC: {        ## NOTE: Unlike spec's "bogus comment state", this implementation
2126          if ($self->{next_char} == 0x003E) { # >        ## consumes characters one-by-one basis.
2127            !!!cp (124);        
2128            $self->{state} = DATA_STATE;        if ($self->{nc} == 0x003E) { # >
2129            !!!next-input-character;          !!!cp (124);
2130            $self->{state} = DATA_STATE;
2131            !!!emit ($self->{current_token}); # comment          !!!next-input-character;
   
           redo A;  
         } elsif ($self->{next_char} == -1) {  
           !!!cp (125);  
           $self->{state} = DATA_STATE;  
           ## reconsume  
2132    
2133            !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2134            redo A;
2135          } elsif ($self->{nc} == -1) {
2136            !!!cp (125);
2137            $self->{state} = DATA_STATE;
2138            ## reconsume
2139    
2140            redo A;          !!!emit ($self->{ct}); # comment
2141          } else {          redo A;
2142            !!!cp (126);        } else {
2143            $self->{current_token}->{data} .= chr ($self->{next_char}); # comment          !!!cp (126);
2144            !!!next-input-character;          $self->{ct}->{data} .= chr ($self->{nc}); # comment
2145            redo BC;          $self->{read_until}->($self->{ct}->{data},
2146          }                                q[>],
2147        } # BC                                length $self->{ct}->{data});
2148    
2149        die "$0: _get_next_token: unexpected case [BC]";          ## Stay in the state.
2150            !!!next-input-character;
2151            redo A;
2152          }
2153      } elsif ($self->{state} == MARKUP_DECLARATION_OPEN_STATE) {      } elsif ($self->{state} == MARKUP_DECLARATION_OPEN_STATE) {
2154        ## (only happen if PCDATA state)        ## (only happen if PCDATA state)
   
       my ($l, $c) = ($self->{line_prev}, $self->{column_prev} - 1);  
   
       my @next_char;  
       push @next_char, $self->{next_char};  
2155                
2156        if ($self->{next_char} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2157            !!!cp (133);
2158            $self->{state} = MD_HYPHEN_STATE;
2159          !!!next-input-character;          !!!next-input-character;
2160          push @next_char, $self->{next_char};          redo A;
2161          if ($self->{next_char} == 0x002D) { # -        } elsif ($self->{nc} == 0x0044 or # D
2162            !!!cp (127);                 $self->{nc} == 0x0064) { # d
2163            $self->{current_token} = {type => COMMENT_TOKEN, data => '',          ## ASCII case-insensitive.
2164                                      line => $l, column => $c,          !!!cp (130);
2165                                     };          $self->{state} = MD_DOCTYPE_STATE;
2166            $self->{state} = COMMENT_START_STATE;          $self->{s_kwd} = chr $self->{nc};
           !!!next-input-character;  
           redo A;  
         } else {  
           !!!cp (128);  
         }  
       } elsif ($self->{next_char} == 0x0044 or # D  
                $self->{next_char} == 0x0064) { # d  
2167          !!!next-input-character;          !!!next-input-character;
2168          push @next_char, $self->{next_char};          redo A;
         if ($self->{next_char} == 0x004F or # O  
             $self->{next_char} == 0x006F) { # o  
           !!!next-input-character;  
           push @next_char, $self->{next_char};  
           if ($self->{next_char} == 0x0043 or # C  
               $self->{next_char} == 0x0063) { # c  
             !!!next-input-character;  
             push @next_char, $self->{next_char};  
             if ($self->{next_char} == 0x0054 or # T  
                 $self->{next_char} == 0x0074) { # t  
               !!!next-input-character;  
               push @next_char, $self->{next_char};  
               if ($self->{next_char} == 0x0059 or # Y  
                   $self->{next_char} == 0x0079) { # y  
                 !!!next-input-character;  
                 push @next_char, $self->{next_char};  
                 if ($self->{next_char} == 0x0050 or # P  
                     $self->{next_char} == 0x0070) { # p  
                   !!!next-input-character;  
                   push @next_char, $self->{next_char};  
                   if ($self->{next_char} == 0x0045 or # E  
                       $self->{next_char} == 0x0065) { # e  
                     !!!cp (129);  
                     ## TODO: What a stupid code this is!  
                     $self->{state} = DOCTYPE_STATE;  
                     $self->{current_token} = {type => DOCTYPE_TOKEN,  
                                               quirks => 1,  
                                               line => $l, column => $c,  
                                              };  
                     !!!next-input-character;  
                     redo A;  
                   } else {  
                     !!!cp (130);  
                   }  
                 } else {  
                   !!!cp (131);  
                 }  
               } else {  
                 !!!cp (132);  
               }  
             } else {  
               !!!cp (133);  
             }  
           } else {  
             !!!cp (134);  
           }  
         } else {  
           !!!cp (135);  
         }  
2169        } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM and        } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM and
2170                 $self->{open_elements}->[-1]->[1] & FOREIGN_EL and                 $self->{open_elements}->[-1]->[1] & FOREIGN_EL and
2171                 $self->{next_char} == 0x005B) { # [                 $self->{nc} == 0x005B) { # [
2172            !!!cp (135.4);                
2173            $self->{state} = MD_CDATA_STATE;
2174            $self->{s_kwd} = '[';
2175          !!!next-input-character;          !!!next-input-character;
2176          push @next_char, $self->{next_char};          redo A;
         if ($self->{next_char} == 0x0043) { # C  
           !!!next-input-character;  
           push @next_char, $self->{next_char};  
           if ($self->{next_char} == 0x0044) { # D  
             !!!next-input-character;  
             push @next_char, $self->{next_char};  
             if ($self->{next_char} == 0x0041) { # A  
               !!!next-input-character;  
               push @next_char, $self->{next_char};  
               if ($self->{next_char} == 0x0054) { # T  
                 !!!next-input-character;  
                 push @next_char, $self->{next_char};  
                 if ($self->{next_char} == 0x0041) { # A  
                   !!!next-input-character;  
                   push @next_char, $self->{next_char};  
                   if ($self->{next_char} == 0x005B) { # [  
                     !!!cp (135.1);  
                     $self->{state} = CDATA_BLOCK_STATE;  
                     !!!next-input-character;  
                     redo A;  
                   } else {  
                     !!!cp (135.2);  
                   }  
                 } else {  
                   !!!cp (135.3);  
                 }  
               } else {  
                 !!!cp (135.4);                  
               }  
             } else {  
               !!!cp (135.5);  
             }  
           } else {  
             !!!cp (135.6);  
           }  
         } else {  
           !!!cp (135.7);  
         }  
2177        } else {        } else {
2178          !!!cp (136);          !!!cp (136);
2179        }        }
2180    
2181        !!!parse-error (type => 'bogus comment');        !!!parse-error (type => 'bogus comment',
2182        $self->{next_char} = shift @next_char;                        line => $self->{line_prev},
2183        !!!back-next-input-character (@next_char);                        column => $self->{column_prev} - 1);
2184          ## Reconsume.
2185        $self->{state} = BOGUS_COMMENT_STATE;        $self->{state} = BOGUS_COMMENT_STATE;
2186        $self->{current_token} = {type => COMMENT_TOKEN, data => '',        $self->{ct} = {type => COMMENT_TOKEN, data => '',
2187                                  line => $l, column => $c,                                  line => $self->{line_prev},
2188                                    column => $self->{column_prev} - 1,
2189                                 };                                 };
2190        redo A;        redo A;
2191              } elsif ($self->{state} == MD_HYPHEN_STATE) {
2192        ## ISSUE: typos in spec: chacacters, is is a parse error        if ($self->{nc} == 0x002D) { # -
2193        ## ISSUE: spec is somewhat unclear on "is the first character that will be in the comment"; what is "that will be in the comment" is what the algorithm defines, isn't it?          !!!cp (127);
2194            $self->{ct} = {type => COMMENT_TOKEN, data => '',
2195                                      line => $self->{line_prev},
2196                                      column => $self->{column_prev} - 2,
2197                                     };
2198            $self->{state} = COMMENT_START_STATE;
2199            !!!next-input-character;
2200            redo A;
2201          } else {
2202            !!!cp (128);
2203            !!!parse-error (type => 'bogus comment',
2204                            line => $self->{line_prev},
2205                            column => $self->{column_prev} - 2);
2206            $self->{state} = BOGUS_COMMENT_STATE;
2207            ## Reconsume.
2208            $self->{ct} = {type => COMMENT_TOKEN,
2209                                      data => '-',
2210                                      line => $self->{line_prev},
2211                                      column => $self->{column_prev} - 2,
2212                                     };
2213            redo A;
2214          }
2215        } elsif ($self->{state} == MD_DOCTYPE_STATE) {
2216          ## ASCII case-insensitive.
2217          if ($self->{nc} == [
2218                undef,
2219                0x004F, # O
2220                0x0043, # C
2221                0x0054, # T
2222                0x0059, # Y
2223                0x0050, # P
2224              ]->[length $self->{s_kwd}] or
2225              $self->{nc} == [
2226                undef,
2227                0x006F, # o
2228                0x0063, # c
2229                0x0074, # t
2230                0x0079, # y
2231                0x0070, # p
2232              ]->[length $self->{s_kwd}]) {
2233            !!!cp (131);
2234            ## Stay in the state.
2235            $self->{s_kwd} .= chr $self->{nc};
2236            !!!next-input-character;
2237            redo A;
2238          } elsif ((length $self->{s_kwd}) == 6 and
2239                   ($self->{nc} == 0x0045 or # E
2240                    $self->{nc} == 0x0065)) { # e
2241            !!!cp (129);
2242            $self->{state} = DOCTYPE_STATE;
2243            $self->{ct} = {type => DOCTYPE_TOKEN,
2244                                      quirks => 1,
2245                                      line => $self->{line_prev},
2246                                      column => $self->{column_prev} - 7,
2247                                     };
2248            !!!next-input-character;
2249            redo A;
2250          } else {
2251            !!!cp (132);        
2252            !!!parse-error (type => 'bogus comment',
2253                            line => $self->{line_prev},
2254                            column => $self->{column_prev} - 1 - length $self->{s_kwd});
2255            $self->{state} = BOGUS_COMMENT_STATE;
2256            ## Reconsume.
2257            $self->{ct} = {type => COMMENT_TOKEN,
2258                                      data => $self->{s_kwd},
2259                                      line => $self->{line_prev},
2260                                      column => $self->{column_prev} - 1 - length $self->{s_kwd},
2261                                     };
2262            redo A;
2263          }
2264        } elsif ($self->{state} == MD_CDATA_STATE) {
2265          if ($self->{nc} == {
2266                '[' => 0x0043, # C
2267                '[C' => 0x0044, # D
2268                '[CD' => 0x0041, # A
2269                '[CDA' => 0x0054, # T
2270                '[CDAT' => 0x0041, # A
2271              }->{$self->{s_kwd}}) {
2272            !!!cp (135.1);
2273            ## Stay in the state.
2274            $self->{s_kwd} .= chr $self->{nc};
2275            !!!next-input-character;
2276            redo A;
2277          } elsif ($self->{s_kwd} eq '[CDATA' and
2278                   $self->{nc} == 0x005B) { # [
2279            !!!cp (135.2);
2280            $self->{ct} = {type => CHARACTER_TOKEN,
2281                                      data => '',
2282                                      line => $self->{line_prev},
2283                                      column => $self->{column_prev} - 7};
2284            $self->{state} = CDATA_SECTION_STATE;
2285            !!!next-input-character;
2286            redo A;
2287          } else {
2288            !!!cp (135.3);
2289            !!!parse-error (type => 'bogus comment',
2290                            line => $self->{line_prev},
2291                            column => $self->{column_prev} - 1 - length $self->{s_kwd});
2292            $self->{state} = BOGUS_COMMENT_STATE;
2293            ## Reconsume.
2294            $self->{ct} = {type => COMMENT_TOKEN,
2295                                      data => $self->{s_kwd},
2296                                      line => $self->{line_prev},
2297                                      column => $self->{column_prev} - 1 - length $self->{s_kwd},
2298                                     };
2299            redo A;
2300          }
2301      } elsif ($self->{state} == COMMENT_START_STATE) {      } elsif ($self->{state} == COMMENT_START_STATE) {
2302        if ($self->{next_char} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2303          !!!cp (137);          !!!cp (137);
2304          $self->{state} = COMMENT_START_DASH_STATE;          $self->{state} = COMMENT_START_DASH_STATE;
2305          !!!next-input-character;          !!!next-input-character;
2306          redo A;          redo A;
2307        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2308          !!!cp (138);          !!!cp (138);
2309          !!!parse-error (type => 'bogus comment');          !!!parse-error (type => 'bogus comment');
2310          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2311          !!!next-input-character;          !!!next-input-character;
2312    
2313          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2314    
2315          redo A;          redo A;
2316        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2317          !!!cp (139);          !!!cp (139);
2318          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2319          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2320          ## reconsume          ## reconsume
2321    
2322          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2323    
2324          redo A;          redo A;
2325        } else {        } else {
2326          !!!cp (140);          !!!cp (140);
2327          $self->{current_token}->{data} # comment          $self->{ct}->{data} # comment
2328              .= chr ($self->{next_char});              .= chr ($self->{nc});
2329          $self->{state} = COMMENT_STATE;          $self->{state} = COMMENT_STATE;
2330          !!!next-input-character;          !!!next-input-character;
2331          redo A;          redo A;
2332        }        }
2333      } elsif ($self->{state} == COMMENT_START_DASH_STATE) {      } elsif ($self->{state} == COMMENT_START_DASH_STATE) {
2334        if ($self->{next_char} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2335          !!!cp (141);          !!!cp (141);
2336          $self->{state} = COMMENT_END_STATE;          $self->{state} = COMMENT_END_STATE;
2337          !!!next-input-character;          !!!next-input-character;
2338          redo A;          redo A;
2339        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2340          !!!cp (142);          !!!cp (142);
2341          !!!parse-error (type => 'bogus comment');          !!!parse-error (type => 'bogus comment');
2342          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2343          !!!next-input-character;          !!!next-input-character;
2344    
2345          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2346    
2347          redo A;          redo A;
2348        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2349          !!!cp (143);          !!!cp (143);
2350          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2351          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2352          ## reconsume          ## reconsume
2353    
2354          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2355    
2356          redo A;          redo A;
2357        } else {        } else {
2358          !!!cp (144);          !!!cp (144);
2359          $self->{current_token}->{data} # comment          $self->{ct}->{data} # comment
2360              .= '-' . chr ($self->{next_char});              .= '-' . chr ($self->{nc});
2361          $self->{state} = COMMENT_STATE;          $self->{state} = COMMENT_STATE;
2362          !!!next-input-character;          !!!next-input-character;
2363          redo A;          redo A;
2364        }        }
2365      } elsif ($self->{state} == COMMENT_STATE) {      } elsif ($self->{state} == COMMENT_STATE) {
2366        if ($self->{next_char} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2367          !!!cp (145);          !!!cp (145);
2368          $self->{state} = COMMENT_END_DASH_STATE;          $self->{state} = COMMENT_END_DASH_STATE;
2369          !!!next-input-character;          !!!next-input-character;
2370          redo A;          redo A;
2371        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2372          !!!cp (146);          !!!cp (146);
2373          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2374          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2375          ## reconsume          ## reconsume
2376    
2377          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2378    
2379          redo A;          redo A;
2380        } else {        } else {
2381          !!!cp (147);          !!!cp (147);
2382          $self->{current_token}->{data} .= chr ($self->{next_char}); # comment          $self->{ct}->{data} .= chr ($self->{nc}); # comment
2383            $self->{read_until}->($self->{ct}->{data},
2384                                  q[-],
2385                                  length $self->{ct}->{data});
2386    
2387          ## Stay in the state          ## Stay in the state
2388          !!!next-input-character;          !!!next-input-character;
2389          redo A;          redo A;
2390        }        }
2391      } elsif ($self->{state} == COMMENT_END_DASH_STATE) {      } elsif ($self->{state} == COMMENT_END_DASH_STATE) {
2392        if ($self->{next_char} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2393          !!!cp (148);          !!!cp (148);
2394          $self->{state} = COMMENT_END_STATE;          $self->{state} = COMMENT_END_STATE;
2395          !!!next-input-character;          !!!next-input-character;
2396          redo A;          redo A;
2397        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2398          !!!cp (149);          !!!cp (149);
2399          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2400          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2401          ## reconsume          ## reconsume
2402    
2403          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2404    
2405          redo A;          redo A;
2406        } else {        } else {
2407          !!!cp (150);          !!!cp (150);
2408          $self->{current_token}->{data} .= '-' . chr ($self->{next_char}); # comment          $self->{ct}->{data} .= '-' . chr ($self->{nc}); # comment
2409          $self->{state} = COMMENT_STATE;          $self->{state} = COMMENT_STATE;
2410          !!!next-input-character;          !!!next-input-character;
2411          redo A;          redo A;
2412        }        }
2413      } elsif ($self->{state} == COMMENT_END_STATE) {      } elsif ($self->{state} == COMMENT_END_STATE) {
2414        if ($self->{next_char} == 0x003E) { # >        if ($self->{nc} == 0x003E) { # >
2415          !!!cp (151);          !!!cp (151);
2416          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2417          !!!next-input-character;          !!!next-input-character;
2418    
2419          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2420    
2421          redo A;          redo A;
2422        } elsif ($self->{next_char} == 0x002D) { # -        } elsif ($self->{nc} == 0x002D) { # -
2423          !!!cp (152);          !!!cp (152);
2424          !!!parse-error (type => 'dash in comment',          !!!parse-error (type => 'dash in comment',
2425                          line => $self->{line_prev},                          line => $self->{line_prev},
2426                          column => $self->{column_prev});                          column => $self->{column_prev});
2427          $self->{current_token}->{data} .= '-'; # comment          $self->{ct}->{data} .= '-'; # comment
2428          ## Stay in the state          ## Stay in the state
2429          !!!next-input-character;          !!!next-input-character;
2430          redo A;          redo A;
2431        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2432          !!!cp (153);          !!!cp (153);
2433          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2434          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2435          ## reconsume          ## reconsume
2436    
2437          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2438    
2439          redo A;          redo A;
2440        } else {        } else {
# Line 2272  sub _get_next_token ($) { Line 2442  sub _get_next_token ($) {
2442          !!!parse-error (type => 'dash in comment',          !!!parse-error (type => 'dash in comment',
2443                          line => $self->{line_prev},                          line => $self->{line_prev},
2444                          column => $self->{column_prev});                          column => $self->{column_prev});
2445          $self->{current_token}->{data} .= '--' . chr ($self->{next_char}); # comment          $self->{ct}->{data} .= '--' . chr ($self->{nc}); # comment
2446          $self->{state} = COMMENT_STATE;          $self->{state} = COMMENT_STATE;
2447          !!!next-input-character;          !!!next-input-character;
2448          redo A;          redo A;
2449        }        }
2450      } elsif ($self->{state} == DOCTYPE_STATE) {      } elsif ($self->{state} == DOCTYPE_STATE) {
2451        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
2452          !!!cp (155);          !!!cp (155);
2453          $self->{state} = BEFORE_DOCTYPE_NAME_STATE;          $self->{state} = BEFORE_DOCTYPE_NAME_STATE;
2454          !!!next-input-character;          !!!next-input-character;
# Line 2295  sub _get_next_token ($) { Line 2461  sub _get_next_token ($) {
2461          redo A;          redo A;
2462        }        }
2463      } elsif ($self->{state} == BEFORE_DOCTYPE_NAME_STATE) {      } elsif ($self->{state} == BEFORE_DOCTYPE_NAME_STATE) {
2464        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
2465          !!!cp (157);          !!!cp (157);
2466          ## Stay in the state          ## Stay in the state
2467          !!!next-input-character;          !!!next-input-character;
2468          redo A;          redo A;
2469        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2470          !!!cp (158);          !!!cp (158);
2471          !!!parse-error (type => 'no DOCTYPE name');          !!!parse-error (type => 'no DOCTYPE name');
2472          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2473          !!!next-input-character;          !!!next-input-character;
2474    
2475          !!!emit ($self->{current_token}); # DOCTYPE (quirks)          !!!emit ($self->{ct}); # DOCTYPE (quirks)
2476    
2477          redo A;          redo A;
2478        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2479          !!!cp (159);          !!!cp (159);
2480          !!!parse-error (type => 'no DOCTYPE name');          !!!parse-error (type => 'no DOCTYPE name');
2481          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2482          ## reconsume          ## reconsume
2483    
2484          !!!emit ($self->{current_token}); # DOCTYPE (quirks)          !!!emit ($self->{ct}); # DOCTYPE (quirks)
2485    
2486          redo A;          redo A;
2487        } else {        } else {
2488          !!!cp (160);          !!!cp (160);
2489          $self->{current_token}->{name} = chr $self->{next_char};          $self->{ct}->{name} = chr $self->{nc};
2490          delete $self->{current_token}->{quirks};          delete $self->{ct}->{quirks};
 ## ISSUE: "Set the token's name name to the" in the spec  
2491          $self->{state} = DOCTYPE_NAME_STATE;          $self->{state} = DOCTYPE_NAME_STATE;
2492          !!!next-input-character;          !!!next-input-character;
2493          redo A;          redo A;
2494        }        }
2495      } elsif ($self->{state} == DOCTYPE_NAME_STATE) {      } elsif ($self->{state} == DOCTYPE_NAME_STATE) {
2496  ## ISSUE: Redundant "First," in the spec.  ## ISSUE: Redundant "First," in the spec.
2497        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
2498          !!!cp (161);          !!!cp (161);
2499          $self->{state} = AFTER_DOCTYPE_NAME_STATE;          $self->{state} = AFTER_DOCTYPE_NAME_STATE;
2500          !!!next-input-character;          !!!next-input-character;
2501          redo A;          redo A;
2502        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2503          !!!cp (162);          !!!cp (162);
2504          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2505          !!!next-input-character;          !!!next-input-character;
2506    
2507          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2508    
2509          redo A;          redo A;
2510        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2511          !!!cp (163);          !!!cp (163);
2512          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2513          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2514          ## reconsume          ## reconsume
2515    
2516          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2517          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2518    
2519          redo A;          redo A;
2520        } else {        } else {
2521          !!!cp (164);          !!!cp (164);
2522          $self->{current_token}->{name}          $self->{ct}->{name}
2523            .= chr ($self->{next_char}); # DOCTYPE            .= chr ($self->{nc}); # DOCTYPE
2524          ## Stay in the state          ## Stay in the state
2525          !!!next-input-character;          !!!next-input-character;
2526          redo A;          redo A;
2527        }        }
2528      } elsif ($self->{state} == AFTER_DOCTYPE_NAME_STATE) {      } elsif ($self->{state} == AFTER_DOCTYPE_NAME_STATE) {
2529        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
2530          !!!cp (165);          !!!cp (165);
2531          ## Stay in the state          ## Stay in the state
2532          !!!next-input-character;          !!!next-input-character;
2533          redo A;          redo A;
2534        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2535          !!!cp (166);          !!!cp (166);
2536          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2537          !!!next-input-character;          !!!next-input-character;
2538    
2539          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2540    
2541          redo A;          redo A;
2542        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2543          !!!cp (167);          !!!cp (167);
2544          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2545          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2546          ## reconsume          ## reconsume
2547    
2548          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2549          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2550    
2551          redo A;          redo A;
2552        } elsif ($self->{next_char} == 0x0050 or # P        } elsif ($self->{nc} == 0x0050 or # P
2553                 $self->{next_char} == 0x0070) { # p                 $self->{nc} == 0x0070) { # p
2554            $self->{state} = PUBLIC_STATE;
2555            $self->{s_kwd} = chr $self->{nc};
2556          !!!next-input-character;          !!!next-input-character;
2557          if ($self->{next_char} == 0x0055 or # U          redo A;
2558              $self->{next_char} == 0x0075) { # u        } elsif ($self->{nc} == 0x0053 or # S
2559            !!!next-input-character;                 $self->{nc} == 0x0073) { # s
2560            if ($self->{next_char} == 0x0042 or # B          $self->{state} = SYSTEM_STATE;
2561                $self->{next_char} == 0x0062) { # b          $self->{s_kwd} = chr $self->{nc};
             !!!next-input-character;  
             if ($self->{next_char} == 0x004C or # L  
                 $self->{next_char} == 0x006C) { # l  
               !!!next-input-character;  
               if ($self->{next_char} == 0x0049 or # I  
                   $self->{next_char} == 0x0069) { # i  
                 !!!next-input-character;  
                 if ($self->{next_char} == 0x0043 or # C  
                     $self->{next_char} == 0x0063) { # c  
                   !!!cp (168);  
                   $self->{state} = BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE;  
                   !!!next-input-character;  
                   redo A;  
                 } else {  
                   !!!cp (169);  
                 }  
               } else {  
                 !!!cp (170);  
               }  
             } else {  
               !!!cp (171);  
             }  
           } else {  
             !!!cp (172);  
           }  
         } else {  
           !!!cp (173);  
         }  
   
         #  
       } elsif ($self->{next_char} == 0x0053 or # S  
                $self->{next_char} == 0x0073) { # s  
2562          !!!next-input-character;          !!!next-input-character;
2563          if ($self->{next_char} == 0x0059 or # Y          redo A;
             $self->{next_char} == 0x0079) { # y  
           !!!next-input-character;  
           if ($self->{next_char} == 0x0053 or # S  
               $self->{next_char} == 0x0073) { # s  
             !!!next-input-character;  
             if ($self->{next_char} == 0x0054 or # T  
                 $self->{next_char} == 0x0074) { # t  
               !!!next-input-character;  
               if ($self->{next_char} == 0x0045 or # E  
                   $self->{next_char} == 0x0065) { # e  
                 !!!next-input-character;  
                 if ($self->{next_char} == 0x004D or # M  
                     $self->{next_char} == 0x006D) { # m  
                   !!!cp (174);  
                   $self->{state} = BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE;  
                   !!!next-input-character;  
                   redo A;  
                 } else {  
                   !!!cp (175);  
                 }  
               } else {  
                 !!!cp (176);  
               }  
             } else {  
               !!!cp (177);  
             }  
           } else {  
             !!!cp (178);  
           }  
         } else {  
           !!!cp (179);  
         }  
   
         #  
2564        } else {        } else {
2565          !!!cp (180);          !!!cp (180);
2566            !!!parse-error (type => 'string after DOCTYPE name');
2567            $self->{ct}->{quirks} = 1;
2568    
2569            $self->{state} = BOGUS_DOCTYPE_STATE;
2570          !!!next-input-character;          !!!next-input-character;
2571          #          redo A;
2572        }        }
2573        } elsif ($self->{state} == PUBLIC_STATE) {
2574          ## ASCII case-insensitive
2575          if ($self->{nc} == [
2576                undef,
2577                0x0055, # U
2578                0x0042, # B
2579                0x004C, # L
2580                0x0049, # I
2581              ]->[length $self->{s_kwd}] or
2582              $self->{nc} == [
2583                undef,
2584                0x0075, # u
2585                0x0062, # b
2586                0x006C, # l
2587                0x0069, # i
2588              ]->[length $self->{s_kwd}]) {
2589            !!!cp (175);
2590            ## Stay in the state.
2591            $self->{s_kwd} .= chr $self->{nc};
2592            !!!next-input-character;
2593            redo A;
2594          } elsif ((length $self->{s_kwd}) == 5 and
2595                   ($self->{nc} == 0x0043 or # C
2596                    $self->{nc} == 0x0063)) { # c
2597            !!!cp (168);
2598            $self->{state} = BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2599            !!!next-input-character;
2600            redo A;
2601          } else {
2602            !!!cp (169);
2603            !!!parse-error (type => 'string after DOCTYPE name',
2604                            line => $self->{line_prev},
2605                            column => $self->{column_prev} + 1 - length $self->{s_kwd});
2606            $self->{ct}->{quirks} = 1;
2607    
2608        !!!parse-error (type => 'string after DOCTYPE name');          $self->{state} = BOGUS_DOCTYPE_STATE;
2609        $self->{current_token}->{quirks} = 1;          ## Reconsume.
2610            redo A;
2611          }
2612        } elsif ($self->{state} == SYSTEM_STATE) {
2613          ## ASCII case-insensitive
2614          if ($self->{nc} == [
2615                undef,
2616                0x0059, # Y
2617                0x0053, # S
2618                0x0054, # T
2619                0x0045, # E
2620              ]->[length $self->{s_kwd}] or
2621              $self->{nc} == [
2622                undef,
2623                0x0079, # y
2624                0x0073, # s
2625                0x0074, # t
2626                0x0065, # e
2627              ]->[length $self->{s_kwd}]) {
2628            !!!cp (170);
2629            ## Stay in the state.
2630            $self->{s_kwd} .= chr $self->{nc};
2631            !!!next-input-character;
2632            redo A;
2633          } elsif ((length $self->{s_kwd}) == 5 and
2634                   ($self->{nc} == 0x004D or # M
2635                    $self->{nc} == 0x006D)) { # m
2636            !!!cp (171);
2637            $self->{state} = BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2638            !!!next-input-character;
2639            redo A;
2640          } else {
2641            !!!cp (172);
2642            !!!parse-error (type => 'string after DOCTYPE name',
2643                            line => $self->{line_prev},
2644                            column => $self->{column_prev} + 1 - length $self->{s_kwd});
2645            $self->{ct}->{quirks} = 1;
2646    
2647        $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2648        # next-input-character is already done          ## Reconsume.
2649        redo A;          redo A;
2650          }
2651      } elsif ($self->{state} == BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {      } elsif ($self->{state} == BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {
2652        if ({        if ($is_space->{$self->{nc}}) {
             0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,  
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_char}}) {  
2653          !!!cp (181);          !!!cp (181);
2654          ## Stay in the state          ## Stay in the state
2655          !!!next-input-character;          !!!next-input-character;
2656          redo A;          redo A;
2657        } elsif ($self->{next_char} eq 0x0022) { # "        } elsif ($self->{nc} eq 0x0022) { # "
2658          !!!cp (182);          !!!cp (182);
2659          $self->{current_token}->{public_identifier} = ''; # DOCTYPE          $self->{ct}->{pubid} = ''; # DOCTYPE
2660          $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE;          $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE;
2661          !!!next-input-character;          !!!next-input-character;
2662          redo A;          redo A;
2663        } elsif ($self->{next_char} eq 0x0027) { # '        } elsif ($self->{nc} eq 0x0027) { # '
2664          !!!cp (183);          !!!cp (183);
2665          $self->{current_token}->{public_identifier} = ''; # DOCTYPE          $self->{ct}->{pubid} = ''; # DOCTYPE
2666          $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE;          $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE;
2667          !!!next-input-character;          !!!next-input-character;
2668          redo A;          redo A;
2669        } elsif ($self->{next_char} eq 0x003E) { # >        } elsif ($self->{nc} eq 0x003E) { # >
2670          !!!cp (184);          !!!cp (184);
2671          !!!parse-error (type => 'no PUBLIC literal');          !!!parse-error (type => 'no PUBLIC literal');
2672    
2673          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2674          !!!next-input-character;          !!!next-input-character;
2675    
2676          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2677          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2678    
2679          redo A;          redo A;
2680        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2681          !!!cp (185);          !!!cp (185);
2682          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2683    
2684          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2685          ## reconsume          ## reconsume
2686    
2687          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2688          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2689    
2690          redo A;          redo A;
2691        } else {        } else {
2692          !!!cp (186);          !!!cp (186);
2693          !!!parse-error (type => 'string after PUBLIC');          !!!parse-error (type => 'string after PUBLIC');
2694          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2695    
2696          $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2697          !!!next-input-character;          !!!next-input-character;
2698          redo A;          redo A;
2699        }        }
2700      } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE) {      } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE) {
2701        if ($self->{next_char} == 0x0022) { # "        if ($self->{nc} == 0x0022) { # "
2702          !!!cp (187);          !!!cp (187);
2703          $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;          $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2704          !!!next-input-character;          !!!next-input-character;
2705          redo A;          redo A;
2706        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2707          !!!cp (188);          !!!cp (188);
2708          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed PUBLIC literal');
2709    
2710          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2711          !!!next-input-character;          !!!next-input-character;
2712    
2713          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2714          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2715    
2716          redo A;          redo A;
2717        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2718          !!!cp (189);          !!!cp (189);
2719          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed PUBLIC literal');
2720    
2721          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2722          ## reconsume          ## reconsume
2723    
2724          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2725          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2726    
2727          redo A;          redo A;
2728        } else {        } else {
2729          !!!cp (190);          !!!cp (190);
2730          $self->{current_token}->{public_identifier} # DOCTYPE          $self->{ct}->{pubid} # DOCTYPE
2731              .= chr $self->{next_char};              .= chr $self->{nc};
2732            $self->{read_until}->($self->{ct}->{pubid}, q[">],
2733                                  length $self->{ct}->{pubid});
2734    
2735          ## Stay in the state          ## Stay in the state
2736          !!!next-input-character;          !!!next-input-character;
2737          redo A;          redo A;
2738        }        }
2739      } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE) {      } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE) {
2740        if ($self->{next_char} == 0x0027) { # '        if ($self->{nc} == 0x0027) { # '
2741          !!!cp (191);          !!!cp (191);
2742          $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;          $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2743          !!!next-input-character;          !!!next-input-character;
2744          redo A;          redo A;
2745        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2746          !!!cp (192);          !!!cp (192);
2747          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed PUBLIC literal');
2748    
2749          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2750          !!!next-input-character;          !!!next-input-character;
2751    
2752          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2753          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2754    
2755          redo A;          redo A;
2756        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2757          !!!cp (193);          !!!cp (193);
2758          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed PUBLIC literal');
2759    
2760          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2761          ## reconsume          ## reconsume
2762    
2763          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2764          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2765    
2766          redo A;          redo A;
2767        } else {        } else {
2768          !!!cp (194);          !!!cp (194);
2769          $self->{current_token}->{public_identifier} # DOCTYPE          $self->{ct}->{pubid} # DOCTYPE
2770              .= chr $self->{next_char};              .= chr $self->{nc};
2771            $self->{read_until}->($self->{ct}->{pubid}, q['>],
2772                                  length $self->{ct}->{pubid});
2773    
2774          ## Stay in the state          ## Stay in the state
2775          !!!next-input-character;          !!!next-input-character;
2776          redo A;          redo A;
2777        }        }
2778      } elsif ($self->{state} == AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {      } elsif ($self->{state} == AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {
2779        if ({        if ($is_space->{$self->{nc}}) {
             0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,  
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_char}}) {  
2780          !!!cp (195);          !!!cp (195);
2781          ## Stay in the state          ## Stay in the state
2782          !!!next-input-character;          !!!next-input-character;
2783          redo A;          redo A;
2784        } elsif ($self->{next_char} == 0x0022) { # "        } elsif ($self->{nc} == 0x0022) { # "
2785          !!!cp (196);          !!!cp (196);
2786          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          $self->{ct}->{sysid} = ''; # DOCTYPE
2787          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;
2788          !!!next-input-character;          !!!next-input-character;
2789          redo A;          redo A;
2790        } elsif ($self->{next_char} == 0x0027) { # '        } elsif ($self->{nc} == 0x0027) { # '
2791          !!!cp (197);          !!!cp (197);
2792          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          $self->{ct}->{sysid} = ''; # DOCTYPE
2793          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;
2794          !!!next-input-character;          !!!next-input-character;
2795          redo A;          redo A;
2796        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2797          !!!cp (198);          !!!cp (198);
2798          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2799          !!!next-input-character;          !!!next-input-character;
2800    
2801          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2802    
2803          redo A;          redo A;
2804        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2805          !!!cp (199);          !!!cp (199);
2806          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2807    
2808          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2809          ## reconsume          ## reconsume
2810    
2811          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2812          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2813    
2814          redo A;          redo A;
2815        } else {        } else {
2816          !!!cp (200);          !!!cp (200);
2817          !!!parse-error (type => 'string after PUBLIC literal');          !!!parse-error (type => 'string after PUBLIC literal');
2818          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2819    
2820          $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2821          !!!next-input-character;          !!!next-input-character;
2822          redo A;          redo A;
2823        }        }
2824      } elsif ($self->{state} == BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {      } elsif ($self->{state} == BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {
2825        if ({        if ($is_space->{$self->{nc}}) {
             0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,  
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_char}}) {  
2826          !!!cp (201);          !!!cp (201);
2827          ## Stay in the state          ## Stay in the state
2828          !!!next-input-character;          !!!next-input-character;
2829          redo A;          redo A;
2830        } elsif ($self->{next_char} == 0x0022) { # "        } elsif ($self->{nc} == 0x0022) { # "
2831          !!!cp (202);          !!!cp (202);
2832          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          $self->{ct}->{sysid} = ''; # DOCTYPE
2833          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;
2834          !!!next-input-character;          !!!next-input-character;
2835          redo A;          redo A;
2836        } elsif ($self->{next_char} == 0x0027) { # '        } elsif ($self->{nc} == 0x0027) { # '
2837          !!!cp (203);          !!!cp (203);
2838          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          $self->{ct}->{sysid} = ''; # DOCTYPE
2839          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;
2840          !!!next-input-character;          !!!next-input-character;
2841          redo A;          redo A;
2842        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2843          !!!cp (204);          !!!cp (204);
2844          !!!parse-error (type => 'no SYSTEM literal');          !!!parse-error (type => 'no SYSTEM literal');
2845          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2846          !!!next-input-character;          !!!next-input-character;
2847    
2848          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2849          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2850    
2851          redo A;          redo A;
2852        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2853          !!!cp (205);          !!!cp (205);
2854          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2855    
2856          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2857          ## reconsume          ## reconsume
2858    
2859          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2860          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2861    
2862          redo A;          redo A;
2863        } else {        } else {
2864          !!!cp (206);          !!!cp (206);
2865          !!!parse-error (type => 'string after SYSTEM');          !!!parse-error (type => 'string after SYSTEM');
2866          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2867    
2868          $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2869          !!!next-input-character;          !!!next-input-character;
2870          redo A;          redo A;
2871        }        }
2872      } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE) {      } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE) {
2873        if ($self->{next_char} == 0x0022) { # "        if ($self->{nc} == 0x0022) { # "
2874          !!!cp (207);          !!!cp (207);
2875          $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;          $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2876          !!!next-input-character;          !!!next-input-character;
2877          redo A;          redo A;
2878        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2879          !!!cp (208);          !!!cp (208);
2880          !!!parse-error (type => 'unclosed SYSTEM literal');          !!!parse-error (type => 'unclosed SYSTEM literal');
2881    
2882          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2883          !!!next-input-character;          !!!next-input-character;
2884    
2885          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2886          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2887    
2888          redo A;          redo A;
2889        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2890          !!!cp (209);          !!!cp (209);
2891          !!!parse-error (type => 'unclosed SYSTEM literal');          !!!parse-error (type => 'unclosed SYSTEM literal');
2892    
2893          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2894          ## reconsume          ## reconsume
2895    
2896          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2897          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2898    
2899          redo A;          redo A;
2900        } else {        } else {
2901          !!!cp (210);          !!!cp (210);
2902          $self->{current_token}->{system_identifier} # DOCTYPE          $self->{ct}->{sysid} # DOCTYPE
2903              .= chr $self->{next_char};              .= chr $self->{nc};
2904            $self->{read_until}->($self->{ct}->{sysid}, q[">],
2905                                  length $self->{ct}->{sysid});
2906    
2907          ## Stay in the state          ## Stay in the state
2908          !!!next-input-character;          !!!next-input-character;
2909          redo A;          redo A;
2910        }        }
2911      } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE) {      } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE) {
2912        if ($self->{next_char} == 0x0027) { # '        if ($self->{nc} == 0x0027) { # '
2913          !!!cp (211);          !!!cp (211);
2914          $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;          $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2915          !!!next-input-character;          !!!next-input-character;
2916          redo A;          redo A;
2917        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2918          !!!cp (212);          !!!cp (212);
2919          !!!parse-error (type => 'unclosed SYSTEM literal');          !!!parse-error (type => 'unclosed SYSTEM literal');
2920    
2921          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2922          !!!next-input-character;          !!!next-input-character;
2923    
2924          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2925          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2926    
2927          redo A;          redo A;
2928        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2929          !!!cp (213);          !!!cp (213);
2930          !!!parse-error (type => 'unclosed SYSTEM literal');          !!!parse-error (type => 'unclosed SYSTEM literal');
2931    
2932          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2933          ## reconsume          ## reconsume
2934    
2935          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2936          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2937    
2938          redo A;          redo A;
2939        } else {        } else {
2940          !!!cp (214);          !!!cp (214);
2941          $self->{current_token}->{system_identifier} # DOCTYPE          $self->{ct}->{sysid} # DOCTYPE
2942              .= chr $self->{next_char};              .= chr $self->{nc};
2943            $self->{read_until}->($self->{ct}->{sysid}, q['>],
2944                                  length $self->{ct}->{sysid});
2945    
2946          ## Stay in the state          ## Stay in the state
2947          !!!next-input-character;          !!!next-input-character;
2948          redo A;          redo A;
2949        }        }
2950      } elsif ($self->{state} == AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {      } elsif ($self->{state} == AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {
2951        if ({        if ($is_space->{$self->{nc}}) {
             0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,  
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_char}}) {  
2952          !!!cp (215);          !!!cp (215);
2953          ## Stay in the state          ## Stay in the state
2954          !!!next-input-character;          !!!next-input-character;
2955          redo A;          redo A;
2956        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2957          !!!cp (216);          !!!cp (216);
2958          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2959          !!!next-input-character;          !!!next-input-character;
2960    
2961          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2962    
2963          redo A;          redo A;
2964        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2965          !!!cp (217);          !!!cp (217);
2966          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2967          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2968          ## reconsume          ## reconsume
2969    
2970          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2971          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2972    
2973          redo A;          redo A;
2974        } else {        } else {
2975          !!!cp (218);          !!!cp (218);
2976          !!!parse-error (type => 'string after SYSTEM literal');          !!!parse-error (type => 'string after SYSTEM literal');
2977          #$self->{current_token}->{quirks} = 1;          #$self->{ct}->{quirks} = 1;
2978    
2979          $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2980          !!!next-input-character;          !!!next-input-character;
2981          redo A;          redo A;
2982        }        }
2983      } elsif ($self->{state} == BOGUS_DOCTYPE_STATE) {      } elsif ($self->{state} == BOGUS_DOCTYPE_STATE) {
2984        if ($self->{next_char} == 0x003E) { # >        if ($self->{nc} == 0x003E) { # >
2985          !!!cp (219);          !!!cp (219);
2986          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2987          !!!next-input-character;          !!!next-input-character;
2988    
2989          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2990    
2991          redo A;          redo A;
2992        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2993          !!!cp (220);          !!!cp (220);
         !!!parse-error (type => 'unclosed DOCTYPE');  
2994          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2995          ## reconsume          ## reconsume
2996    
2997          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2998    
2999          redo A;          redo A;
3000        } else {        } else {
3001          !!!cp (221);          !!!cp (221);
3002            my $s = '';
3003            $self->{read_until}->($s, q[>], 0);
3004    
3005          ## Stay in the state          ## Stay in the state
3006          !!!next-input-character;          !!!next-input-character;
3007          redo A;          redo A;
3008        }        }
3009      } elsif ($self->{state} == CDATA_BLOCK_STATE) {      } elsif ($self->{state} == CDATA_SECTION_STATE) {
3010        my $s = '';        ## NOTE: "CDATA section state" in the state is jointly implemented
3011          ## by three states, |CDATA_SECTION_STATE|, |CDATA_SECTION_MSE1_STATE|,
3012          ## and |CDATA_SECTION_MSE2_STATE|.
3013                
3014        my ($l, $c) = ($self->{line}, $self->{column});        if ($self->{nc} == 0x005D) { # ]
3015            !!!cp (221.1);
3016        CS: while ($self->{next_char} != -1) {          $self->{state} = CDATA_SECTION_MSE1_STATE;
         if ($self->{next_char} == 0x005D) { # ]  
           !!!next-input-character;  
           if ($self->{next_char} == 0x005D) { # ]  
             !!!next-input-character;  
             MDC: {  
               if ($self->{next_char} == 0x003E) { # >  
                 !!!cp (221.1);  
                 !!!next-input-character;  
                 last CS;  
               } elsif ($self->{next_char} == 0x005D) { # ]  
                 !!!cp (221.2);  
                 $s .= ']';  
                 !!!next-input-character;  
                 redo MDC;  
               } else {  
                 !!!cp (221.3);  
                 $s .= ']]';  
                 #  
               }  
             } # MDC  
           } else {  
             !!!cp (221.4);  
             $s .= ']';  
             #  
           }  
         } else {  
           !!!cp (221.5);  
           #  
         }  
         $s .= chr $self->{next_char};  
3017          !!!next-input-character;          !!!next-input-character;
3018        } # CS          redo A;
3019          } elsif ($self->{nc} == -1) {
3020            $self->{state} = DATA_STATE;
3021            !!!next-input-character;
3022            if (length $self->{ct}->{data}) { # character
3023              !!!cp (221.2);
3024              !!!emit ($self->{ct}); # character
3025            } else {
3026              !!!cp (221.3);
3027              ## No token to emit. $self->{ct} is discarded.
3028            }        
3029            redo A;
3030          } else {
3031            !!!cp (221.4);
3032            $self->{ct}->{data} .= chr $self->{nc};
3033            $self->{read_until}->($self->{ct}->{data},
3034                                  q<]>,
3035                                  length $self->{ct}->{data});
3036    
3037        $self->{state} = DATA_STATE;          ## Stay in the state.
3038        ## next-input-character done or EOF, which is reconsumed.          !!!next-input-character;
3039            redo A;
3040          }
3041    
3042        if (length $s) {        ## ISSUE: "text tokens" in spec.
3043        } elsif ($self->{state} == CDATA_SECTION_MSE1_STATE) {
3044          if ($self->{nc} == 0x005D) { # ]
3045            !!!cp (221.5);
3046            $self->{state} = CDATA_SECTION_MSE2_STATE;
3047            !!!next-input-character;
3048            redo A;
3049          } else {
3050          !!!cp (221.6);          !!!cp (221.6);
3051          !!!emit ({type => CHARACTER_TOKEN, data => $s,          $self->{ct}->{data} .= ']';
3052                    line => $l, column => $c});          $self->{state} = CDATA_SECTION_STATE;
3053            ## Reconsume.
3054            redo A;
3055          }
3056        } elsif ($self->{state} == CDATA_SECTION_MSE2_STATE) {
3057          if ($self->{nc} == 0x003E) { # >
3058            $self->{state} = DATA_STATE;
3059            !!!next-input-character;
3060            if (length $self->{ct}->{data}) { # character
3061              !!!cp (221.7);
3062              !!!emit ($self->{ct}); # character
3063            } else {
3064              !!!cp (221.8);
3065              ## No token to emit. $self->{ct} is discarded.
3066            }
3067            redo A;
3068          } elsif ($self->{nc} == 0x005D) { # ]
3069            !!!cp (221.9); # character
3070            $self->{ct}->{data} .= ']'; ## Add first "]" of "]]]".
3071            ## Stay in the state.
3072            !!!next-input-character;
3073            redo A;
3074        } else {        } else {
3075          !!!cp (221.7);          !!!cp (221.11);
3076            $self->{ct}->{data} .= ']]'; # character
3077            $self->{state} = CDATA_SECTION_STATE;
3078            ## Reconsume.
3079            redo A;
3080          }
3081        } elsif ($self->{state} == ENTITY_STATE) {
3082          if ($is_space->{$self->{nc}} or
3083              {
3084                0x003C => 1, 0x0026 => 1, -1 => 1, # <, &
3085                $self->{entity_add} => 1,
3086              }->{$self->{nc}}) {
3087            !!!cp (1001);
3088            ## Don't consume
3089            ## No error
3090            ## Return nothing.
3091            #
3092          } elsif ($self->{nc} == 0x0023) { # #
3093            !!!cp (999);
3094            $self->{state} = ENTITY_HASH_STATE;
3095            $self->{s_kwd} = '#';
3096            !!!next-input-character;
3097            redo A;
3098          } elsif ((0x0041 <= $self->{nc} and
3099                    $self->{nc} <= 0x005A) or # A..Z
3100                   (0x0061 <= $self->{nc} and
3101                    $self->{nc} <= 0x007A)) { # a..z
3102            !!!cp (998);
3103            require Whatpm::_NamedEntityList;
3104            $self->{state} = ENTITY_NAME_STATE;
3105            $self->{s_kwd} = chr $self->{nc};
3106            $self->{entity__value} = $self->{s_kwd};
3107            $self->{entity__match} = 0;
3108            !!!next-input-character;
3109            redo A;
3110          } else {
3111            !!!cp (1027);
3112            !!!parse-error (type => 'bare ero');
3113            ## Return nothing.
3114            #
3115        }        }
3116    
3117        redo A;        ## NOTE: No character is consumed by the "consume a character
3118          ## reference" algorithm.  In other word, there is an "&" character
3119        ## ISSUE: "text tokens" in spec.        ## that does not introduce a character reference, which would be
3120        ## TODO: Streaming support        ## appended to the parent element or the attribute value in later
3121      } else {        ## process of the tokenizer.
3122        die "$0: $self->{state}: Unknown state";  
3123      }        if ($self->{prev_state} == DATA_STATE) {
3124    } # A            !!!cp (997);
3125            $self->{state} = $self->{prev_state};
3126    die "$0: _get_next_token: unexpected case";          ## Reconsume.
3127  } # _get_next_token          !!!emit ({type => CHARACTER_TOKEN, data => '&',
3128                      line => $self->{line_prev},
3129  sub _tokenize_attempt_to_consume_an_entity ($$$) {                    column => $self->{column_prev},
3130    my ($self, $in_attr, $additional) = @_;                   });
3131            redo A;
3132    my ($l, $c) = ($self->{line_prev}, $self->{column_prev});        } else {
3133            !!!cp (996);
3134            $self->{ca}->{value} .= '&';
3135            $self->{state} = $self->{prev_state};
3136            ## Reconsume.
3137            redo A;
3138          }
3139        } elsif ($self->{state} == ENTITY_HASH_STATE) {
3140          if ($self->{nc} == 0x0078 or # x
3141              $self->{nc} == 0x0058) { # X
3142            !!!cp (995);
3143            $self->{state} = HEXREF_X_STATE;
3144            $self->{s_kwd} .= chr $self->{nc};
3145            !!!next-input-character;
3146            redo A;
3147          } elsif (0x0030 <= $self->{nc} and
3148                   $self->{nc} <= 0x0039) { # 0..9
3149            !!!cp (994);
3150            $self->{state} = NCR_NUM_STATE;
3151            $self->{s_kwd} = $self->{nc} - 0x0030;
3152            !!!next-input-character;
3153            redo A;
3154          } else {
3155            !!!parse-error (type => 'bare nero',
3156                            line => $self->{line_prev},
3157                            column => $self->{column_prev} - 1);
3158    
3159    if ({          ## NOTE: According to the spec algorithm, nothing is returned,
3160         0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, # HT, LF, VT, FF,          ## and then "&#" is appended to the parent element or the attribute
3161         0x0020 => 1, 0x003C => 1, 0x0026 => 1, -1 => 1, # SP, <, & # 0x000D # CR          ## value in the later processing.
3162         $additional => 1,  
3163        }->{$self->{next_char}}) {          if ($self->{prev_state} == DATA_STATE) {
3164      !!!cp (1001);            !!!cp (1019);
3165      ## Don't consume            $self->{state} = $self->{prev_state};
3166      ## No error            ## Reconsume.
3167      return undef;            !!!emit ({type => CHARACTER_TOKEN,
3168    } elsif ($self->{next_char} == 0x0023) { # #                      data => '&#',
3169      !!!next-input-character;                      line => $self->{line_prev},
3170      if ($self->{next_char} == 0x0078 or # x                      column => $self->{column_prev} - 1,
3171          $self->{next_char} == 0x0058) { # X                     });
3172        my $code;            redo A;
       X: {  
         my $x_char = $self->{next_char};  
         !!!next-input-character;  
         if (0x0030 <= $self->{next_char} and  
             $self->{next_char} <= 0x0039) { # 0..9  
           !!!cp (1002);  
           $code ||= 0;  
           $code *= 0x10;  
           $code += $self->{next_char} - 0x0030;  
           redo X;  
         } elsif (0x0061 <= $self->{next_char} and  
                  $self->{next_char} <= 0x0066) { # a..f  
           !!!cp (1003);  
           $code ||= 0;  
           $code *= 0x10;  
           $code += $self->{next_char} - 0x0060 + 9;  
           redo X;  
         } elsif (0x0041 <= $self->{next_char} and  
                  $self->{next_char} <= 0x0046) { # A..F  
           !!!cp (1004);  
           $code ||= 0;  
           $code *= 0x10;  
           $code += $self->{next_char} - 0x0040 + 9;  
           redo X;  
         } elsif (not defined $code) { # no hexadecimal digit  
           !!!cp (1005);  
           !!!parse-error (type => 'bare hcro', line => $l, column => $c);  
           !!!back-next-input-character ($x_char, $self->{next_char});  
           $self->{next_char} = 0x0023; # #  
           return undef;  
         } elsif ($self->{next_char} == 0x003B) { # ;  
           !!!cp (1006);  
           !!!next-input-character;  
3173          } else {          } else {
3174            !!!cp (1007);            !!!cp (993);
3175            !!!parse-error (type => 'no refc', line => $l, column => $c);            $self->{ca}->{value} .= '&#';
3176              $self->{state} = $self->{prev_state};
3177              ## Reconsume.
3178              redo A;
3179          }          }
3180          }
3181          if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) {      } elsif ($self->{state} == NCR_NUM_STATE) {
3182            !!!cp (1008);        if (0x0030 <= $self->{nc} and
3183            !!!parse-error (type => 'invalid character reference',            $self->{nc} <= 0x0039) { # 0..9
                           text => (sprintf 'U+%04X', $code),  
                           line => $l, column => $c);  
           $code = 0xFFFD;  
         } elsif ($code > 0x10FFFF) {  
           !!!cp (1009);  
           !!!parse-error (type => 'invalid character reference',  
                           text => (sprintf 'U-%08X', $code),  
                           line => $l, column => $c);  
           $code = 0xFFFD;  
         } elsif ($code == 0x000D) {  
           !!!cp (1010);  
           !!!parse-error (type => 'CR character reference', line => $l, column => $c);  
           $code = 0x000A;  
         } elsif (0x80 <= $code and $code <= 0x9F) {  
           !!!cp (1011);  
           !!!parse-error (type => 'C1 character reference', text => (sprintf 'U+%04X', $code), line => $l, column => $c);  
           $code = $c1_entity_char->{$code};  
         }  
   
         return {type => CHARACTER_TOKEN, data => chr $code,  
                 has_reference => 1,  
                 line => $l, column => $c,  
                };  
       } # X  
     } elsif (0x0030 <= $self->{next_char} and  
              $self->{next_char} <= 0x0039) { # 0..9  
       my $code = $self->{next_char} - 0x0030;  
       !!!next-input-character;  
         
       while (0x0030 <= $self->{next_char} and  
                 $self->{next_char} <= 0x0039) { # 0..9  
3184          !!!cp (1012);          !!!cp (1012);
3185          $code *= 10;          $self->{s_kwd} *= 10;
3186          $code += $self->{next_char} - 0x0030;          $self->{s_kwd} += $self->{nc} - 0x0030;
3187                    
3188            ## Stay in the state.
3189          !!!next-input-character;          !!!next-input-character;
3190        }          redo A;
3191          } elsif ($self->{nc} == 0x003B) { # ;
       if ($self->{next_char} == 0x003B) { # ;  
3192          !!!cp (1013);          !!!cp (1013);
3193          !!!next-input-character;          !!!next-input-character;
3194            #
3195        } else {        } else {
3196          !!!cp (1014);          !!!cp (1014);
3197          !!!parse-error (type => 'no refc', line => $l, column => $c);          !!!parse-error (type => 'no refc');
3198            ## Reconsume.
3199            #
3200        }        }
3201    
3202        if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) {        my $code = $self->{s_kwd};
3203          my $l = $self->{line_prev};
3204          my $c = $self->{column_prev};
3205          if ($charref_map->{$code}) {
3206          !!!cp (1015);          !!!cp (1015);
3207          !!!parse-error (type => 'invalid character reference',          !!!parse-error (type => 'invalid character reference',
3208                          text => (sprintf 'U+%04X', $code),                          text => (sprintf 'U+%04X', $code),
3209                          line => $l, column => $c);                          line => $l, column => $c);
3210          $code = 0xFFFD;          $code = $charref_map->{$code};
3211        } elsif ($code > 0x10FFFF) {        } elsif ($code > 0x10FFFF) {
3212          !!!cp (1016);          !!!cp (1016);
3213          !!!parse-error (type => 'invalid character reference',          !!!parse-error (type => 'invalid character reference',
3214                          text => (sprintf 'U-%08X', $code),                          text => (sprintf 'U-%08X', $code),
3215                          line => $l, column => $c);                          line => $l, column => $c);
3216          $code = 0xFFFD;          $code = 0xFFFD;
3217        } elsif ($code == 0x000D) {        }
3218          !!!cp (1017);  
3219          !!!parse-error (type => 'CR character reference',        if ($self->{prev_state} == DATA_STATE) {
3220                          line => $l, column => $c);          !!!cp (992);
3221          $code = 0x000A;          $self->{state} = $self->{prev_state};
3222        } elsif (0x80 <= $code and $code <= 0x9F) {          ## Reconsume.
3223          !!!cp (1018);          !!!emit ({type => CHARACTER_TOKEN, data => chr $code,
3224          !!!parse-error (type => 'C1 character reference',                    line => $l, column => $c,
3225                     });
3226            redo A;
3227          } else {
3228            !!!cp (991);
3229            $self->{ca}->{value} .= chr $code;
3230            $self->{ca}->{has_reference} = 1;
3231            $self->{state} = $self->{prev_state};
3232            ## Reconsume.
3233            redo A;
3234          }
3235        } elsif ($self->{state} == HEXREF_X_STATE) {
3236          if ((0x0030 <= $self->{nc} and $self->{nc} <= 0x0039) or
3237              (0x0041 <= $self->{nc} and $self->{nc} <= 0x0046) or
3238              (0x0061 <= $self->{nc} and $self->{nc} <= 0x0066)) {
3239            # 0..9, A..F, a..f
3240            !!!cp (990);
3241            $self->{state} = HEXREF_HEX_STATE;
3242            $self->{s_kwd} = 0;
3243            ## Reconsume.
3244            redo A;
3245          } else {
3246            !!!parse-error (type => 'bare hcro',
3247                            line => $self->{line_prev},
3248                            column => $self->{column_prev} - 2);
3249    
3250            ## NOTE: According to the spec algorithm, nothing is returned,
3251            ## and then "&#" followed by "X" or "x" is appended to the parent
3252            ## element or the attribute value in the later processing.
3253    
3254            if ($self->{prev_state} == DATA_STATE) {
3255              !!!cp (1005);
3256              $self->{state} = $self->{prev_state};
3257              ## Reconsume.
3258              !!!emit ({type => CHARACTER_TOKEN,
3259                        data => '&' . $self->{s_kwd},
3260                        line => $self->{line_prev},
3261                        column => $self->{column_prev} - length $self->{s_kwd},
3262                       });
3263              redo A;
3264            } else {
3265              !!!cp (989);
3266              $self->{ca}->{value} .= '&' . $self->{s_kwd};
3267              $self->{state} = $self->{prev_state};
3268              ## Reconsume.
3269              redo A;
3270            }
3271          }
3272        } elsif ($self->{state} == HEXREF_HEX_STATE) {
3273          if (0x0030 <= $self->{nc} and $self->{nc} <= 0x0039) {
3274            # 0..9
3275            !!!cp (1002);
3276            $self->{s_kwd} *= 0x10;
3277            $self->{s_kwd} += $self->{nc} - 0x0030;
3278            ## Stay in the state.
3279            !!!next-input-character;
3280            redo A;
3281          } elsif (0x0061 <= $self->{nc} and
3282                   $self->{nc} <= 0x0066) { # a..f
3283            !!!cp (1003);
3284            $self->{s_kwd} *= 0x10;
3285            $self->{s_kwd} += $self->{nc} - 0x0060 + 9;
3286            ## Stay in the state.
3287            !!!next-input-character;
3288            redo A;
3289          } elsif (0x0041 <= $self->{nc} and
3290                   $self->{nc} <= 0x0046) { # A..F
3291            !!!cp (1004);
3292            $self->{s_kwd} *= 0x10;
3293            $self->{s_kwd} += $self->{nc} - 0x0040 + 9;
3294            ## Stay in the state.
3295            !!!next-input-character;
3296            redo A;
3297          } elsif ($self->{nc} == 0x003B) { # ;
3298            !!!cp (1006);
3299            !!!next-input-character;
3300            #
3301          } else {
3302            !!!cp (1007);
3303            !!!parse-error (type => 'no refc',
3304                            line => $self->{line},
3305                            column => $self->{column});
3306            ## Reconsume.
3307            #
3308          }
3309    
3310          my $code = $self->{s_kwd};
3311          my $l = $self->{line_prev};
3312          my $c = $self->{column_prev};
3313          if ($charref_map->{$code}) {
3314            !!!cp (1008);
3315            !!!parse-error (type => 'invalid character reference',
3316                          text => (sprintf 'U+%04X', $code),                          text => (sprintf 'U+%04X', $code),
3317                          line => $l, column => $c);                          line => $l, column => $c);
3318          $code = $c1_entity_char->{$code};          $code = $charref_map->{$code};
3319          } elsif ($code > 0x10FFFF) {
3320            !!!cp (1009);
3321            !!!parse-error (type => 'invalid character reference',
3322                            text => (sprintf 'U-%08X', $code),
3323                            line => $l, column => $c);
3324            $code = 0xFFFD;
3325        }        }
3326          
3327        return {type => CHARACTER_TOKEN, data => chr $code, has_reference => 1,        if ($self->{prev_state} == DATA_STATE) {
3328                line => $l, column => $c,          !!!cp (988);
3329               };          $self->{state} = $self->{prev_state};
3330      } else {          ## Reconsume.
3331        !!!cp (1019);          !!!emit ({type => CHARACTER_TOKEN, data => chr $code,
3332        !!!parse-error (type => 'bare nero', line => $l, column => $c);                    line => $l, column => $c,
3333        !!!back-next-input-character ($self->{next_char});                   });
3334        $self->{next_char} = 0x0023; # #          redo A;
3335        return undef;        } else {
3336      }          !!!cp (987);
3337    } elsif ((0x0041 <= $self->{next_char} and          $self->{ca}->{value} .= chr $code;
3338              $self->{next_char} <= 0x005A) or          $self->{ca}->{has_reference} = 1;
3339             (0x0061 <= $self->{next_char} and          $self->{state} = $self->{prev_state};
3340              $self->{next_char} <= 0x007A)) {          ## Reconsume.
3341      my $entity_name = chr $self->{next_char};          redo A;
3342      !!!next-input-character;        }
3343        } elsif ($self->{state} == ENTITY_NAME_STATE) {
3344      my $value = $entity_name;        if (length $self->{s_kwd} < 30 and
3345      my $match = 0;            ## NOTE: Some number greater than the maximum length of entity name
3346      require Whatpm::_NamedEntityList;            ((0x0041 <= $self->{nc} and # a
3347      our $EntityChar;              $self->{nc} <= 0x005A) or # x
3348               (0x0061 <= $self->{nc} and # a
3349      while (length $entity_name < 30 and              $self->{nc} <= 0x007A) or # z
3350             ## NOTE: Some number greater than the maximum length of entity name             (0x0030 <= $self->{nc} and # 0
3351             ((0x0041 <= $self->{next_char} and # a              $self->{nc} <= 0x0039) or # 9
3352               $self->{next_char} <= 0x005A) or # x             $self->{nc} == 0x003B)) { # ;
3353              (0x0061 <= $self->{next_char} and # a          our $EntityChar;
3354               $self->{next_char} <= 0x007A) or # z          $self->{s_kwd} .= chr $self->{nc};
3355              (0x0030 <= $self->{next_char} and # 0          if (defined $EntityChar->{$self->{s_kwd}}) {
3356               $self->{next_char} <= 0x0039) or # 9            if ($self->{nc} == 0x003B) { # ;
3357              $self->{next_char} == 0x003B)) { # ;              !!!cp (1020);
3358        $entity_name .= chr $self->{next_char};              $self->{entity__value} = $EntityChar->{$self->{s_kwd}};
3359        if (defined $EntityChar->{$entity_name}) {              $self->{entity__match} = 1;
3360          if ($self->{next_char} == 0x003B) { # ;              !!!next-input-character;
3361            !!!cp (1020);              #
3362            $value = $EntityChar->{$entity_name};            } else {
3363            $match = 1;              !!!cp (1021);
3364            !!!next-input-character;              $self->{entity__value} = $EntityChar->{$self->{s_kwd}};
3365            last;              $self->{entity__match} = -1;
3366                ## Stay in the state.
3367                !!!next-input-character;
3368                redo A;
3369              }
3370          } else {          } else {
3371            !!!cp (1021);            !!!cp (1022);
3372            $value = $EntityChar->{$entity_name};            $self->{entity__value} .= chr $self->{nc};
3373            $match = -1;            $self->{entity__match} *= 2;
3374              ## Stay in the state.
3375            !!!next-input-character;            !!!next-input-character;
3376              redo A;
3377            }
3378          }
3379    
3380          my $data;
3381          my $has_ref;
3382          if ($self->{entity__match} > 0) {
3383            !!!cp (1023);
3384            $data = $self->{entity__value};
3385            $has_ref = 1;
3386            #
3387          } elsif ($self->{entity__match} < 0) {
3388            !!!parse-error (type => 'no refc');
3389            if ($self->{prev_state} != DATA_STATE and # in attribute
3390                $self->{entity__match} < -1) {
3391              !!!cp (1024);
3392              $data = '&' . $self->{s_kwd};
3393              #
3394            } else {
3395              !!!cp (1025);
3396              $data = $self->{entity__value};
3397              $has_ref = 1;
3398              #
3399          }          }
3400        } else {        } else {
3401          !!!cp (1022);          !!!cp (1026);
3402          $value .= chr $self->{next_char};          !!!parse-error (type => 'bare ero',
3403          $match *= 2;                          line => $self->{line_prev},
3404          !!!next-input-character;                          column => $self->{column_prev} - length $self->{s_kwd});
3405            $data = '&' . $self->{s_kwd};
3406            #
3407        }        }
3408      }    
3409              ## NOTE: In these cases, when a character reference is found,
3410      if ($match > 0) {        ## it is consumed and a character token is returned, or, otherwise,
3411        !!!cp (1023);        ## nothing is consumed and returned, according to the spec algorithm.
3412        return {type => CHARACTER_TOKEN, data => $value, has_reference => 1,        ## In this implementation, anything that has been examined by the
3413                line => $l, column => $c,        ## tokenizer is appended to the parent element or the attribute value
3414               };        ## as string, either literal string when no character reference or
3415      } elsif ($match < 0) {        ## entity-replaced string otherwise, in this stage, since any characters
3416        !!!parse-error (type => 'no refc', line => $l, column => $c);        ## that would not be consumed are appended in the data state or in an
3417        if ($in_attr and $match < -1) {        ## appropriate attribute value state anyway.
3418          !!!cp (1024);  
3419          return {type => CHARACTER_TOKEN, data => '&'.$entity_name,        if ($self->{prev_state} == DATA_STATE) {
3420                  line => $l, column => $c,          !!!cp (986);
3421                 };          $self->{state} = $self->{prev_state};
3422        } else {          ## Reconsume.
3423          !!!cp (1025);          !!!emit ({type => CHARACTER_TOKEN,
3424          return {type => CHARACTER_TOKEN, data => $value, has_reference => 1,                    data => $data,
3425                  line => $l, column => $c,                    line => $self->{line_prev},
3426                 };                    column => $self->{column_prev} + 1 - length $self->{s_kwd},
3427                     });
3428            redo A;
3429          } else {
3430            !!!cp (985);
3431            $self->{ca}->{value} .= $data;
3432            $self->{ca}->{has_reference} = 1 if $has_ref;
3433            $self->{state} = $self->{prev_state};
3434            ## Reconsume.
3435            redo A;
3436        }        }
3437      } else {      } else {
3438        !!!cp (1026);        die "$0: $self->{state}: Unknown state";
       !!!parse-error (type => 'bare ero', line => $l, column => $c);  
       ## NOTE: "No characters are consumed" in the spec.  
       return {type => CHARACTER_TOKEN, data => '&'.$value,  
               line => $l, column => $c,  
              };  
3439      }      }
3440    } else {    } # A  
3441      !!!cp (1027);  
3442      ## no characters are consumed    die "$0: _get_next_token: unexpected case";
3443      !!!parse-error (type => 'bare ero', line => $l, column => $c);  } # _get_next_token
     return undef;  
   }  
 } # _tokenize_attempt_to_consume_an_entity  
3444    
3445  sub _initialize_tree_constructor ($) {  sub _initialize_tree_constructor ($) {
3446    my $self = shift;    my $self = shift;
# Line 3148  sub _construct_tree ($) { Line 3470  sub _construct_tree ($) {
3470    ## When an interactive UA render the $self->{document} available    ## When an interactive UA render the $self->{document} available
3471    ## to the user, or when it begin accepting user input, are    ## to the user, or when it begin accepting user input, are
3472    ## not defined.    ## not defined.
   
   ## Append a character: collect it and all subsequent consecutive  
   ## characters and insert one Text node whose data is concatenation  
   ## of all those characters. # MUST  
3473        
3474    !!!next-token;    !!!next-token;
3475    
3476    undef $self->{form_element};    undef $self->{form_element};
3477    undef $self->{head_element};    undef $self->{head_element};
3478      undef $self->{head_element_inserted};
3479    $self->{open_elements} = [];    $self->{open_elements} = [];
3480    undef $self->{inner_html_node};    undef $self->{inner_html_node};
3481    
# Line 3185  sub _tree_construction_initial ($) { Line 3504  sub _tree_construction_initial ($) {
3504        $doctype_name = '' unless defined $doctype_name;        $doctype_name = '' unless defined $doctype_name;
3505        $doctype_name =~ tr/a-z/A-Z/; # ASCII case-insensitive        $doctype_name =~ tr/a-z/A-Z/; # ASCII case-insensitive
3506        if (not defined $token->{name} or # <!DOCTYPE>        if (not defined $token->{name} or # <!DOCTYPE>
3507            defined $token->{system_identifier}) {            defined $token->{sysid}) {
3508          !!!cp ('t1');          !!!cp ('t1');
3509          !!!parse-error (type => 'not HTML5', token => $token);          !!!parse-error (type => 'not HTML5', token => $token);
3510        } elsif ($doctype_name ne 'HTML') {        } elsif ($doctype_name ne 'HTML') {
3511          !!!cp ('t2');          !!!cp ('t2');
3512          !!!parse-error (type => 'not HTML5', token => $token);          !!!parse-error (type => 'not HTML5', token => $token);
3513        } elsif (defined $token->{public_identifier}) {        } elsif (defined $token->{pubid}) {
3514          if ($token->{public_identifier} eq 'XSLT-compat') {          if ($token->{pubid} eq 'XSLT-compat') {
3515            !!!cp ('t1.2');            !!!cp ('t1.2');
3516            !!!parse-error (type => 'XSLT-compat', token => $token,            !!!parse-error (type => 'XSLT-compat', token => $token,
3517                            level => $self->{level}->{should});                            level => $self->{level}->{should});
# Line 3208  sub _tree_construction_initial ($) { Line 3527  sub _tree_construction_initial ($) {
3527          ($token->{name}); ## ISSUE: If name is missing (e.g. <!DOCTYPE>)?          ($token->{name}); ## ISSUE: If name is missing (e.g. <!DOCTYPE>)?
3528        ## NOTE: Default value for both |public_id| and |system_id| attributes        ## NOTE: Default value for both |public_id| and |system_id| attributes
3529        ## are empty strings, so that we don't set any value in missing cases.        ## are empty strings, so that we don't set any value in missing cases.
3530        $doctype->public_id ($token->{public_identifier})        $doctype->public_id ($token->{pubid}) if defined $token->{pubid};
3531            if defined $token->{public_identifier};        $doctype->system_id ($token->{sysid}) if defined $token->{sysid};
       $doctype->system_id ($token->{system_identifier})  
           if defined $token->{system_identifier};  
3532        ## NOTE: Other DocumentType attributes are null or empty lists.        ## NOTE: Other DocumentType attributes are null or empty lists.
3533        ## ISSUE: internalSubset = null??        ## ISSUE: internalSubset = null??
3534        $self->{document}->append_child ($doctype);        $self->{document}->append_child ($doctype);
# Line 3219  sub _tree_construction_initial ($) { Line 3536  sub _tree_construction_initial ($) {
3536        if ($token->{quirks} or $doctype_name ne 'HTML') {        if ($token->{quirks} or $doctype_name ne 'HTML') {
3537          !!!cp ('t4');          !!!cp ('t4');
3538          $self->{document}->manakai_compat_mode ('quirks');          $self->{document}->manakai_compat_mode ('quirks');
3539        } elsif (defined $token->{public_identifier}) {        } elsif (defined $token->{pubid}) {
3540          my $pubid = $token->{public_identifier};          my $pubid = $token->{pubid};
3541          $pubid =~ tr/a-z/A-z/;          $pubid =~ tr/a-z/A-z/;
3542          my $prefix = [          my $prefix = [
3543            "+//SILMARIL//DTD HTML PRO V0R11 19970101//",            "+//SILMARIL//DTD HTML PRO V0R11 19970101//",
# Line 3294  sub _tree_construction_initial ($) { Line 3611  sub _tree_construction_initial ($) {
3611            $self->{document}->manakai_compat_mode ('quirks');            $self->{document}->manakai_compat_mode ('quirks');
3612          } elsif ($pubid =~ m[^-//W3C//DTD HTML 4.01 FRAMESET//] or          } elsif ($pubid =~ m[^-//W3C//DTD HTML 4.01 FRAMESET//] or
3613                   $pubid =~ m[^-//W3C//DTD HTML 4.01 TRANSITIONAL//]) {                   $pubid =~ m[^-//W3C//DTD HTML 4.01 TRANSITIONAL//]) {
3614            if (defined $token->{system_identifier}) {            if (defined $token->{sysid}) {
3615              !!!cp ('t6');              !!!cp ('t6');
3616              $self->{document}->manakai_compat_mode ('quirks');              $self->{document}->manakai_compat_mode ('quirks');
3617            } else {            } else {
# Line 3311  sub _tree_construction_initial ($) { Line 3628  sub _tree_construction_initial ($) {
3628        } else {        } else {
3629          !!!cp ('t10');          !!!cp ('t10');
3630        }        }
3631        if (defined $token->{system_identifier}) {        if (defined $token->{sysid}) {
3632          my $sysid = $token->{system_identifier};          my $sysid = $token->{sysid};
3633          $sysid =~ tr/A-Z/a-z/;          $sysid =~ tr/A-Z/a-z/;
3634          if ($sysid eq "http://www.ibm.com/data/dtd/v11/ibmxhtml1-transitional.dtd") {          if ($sysid eq "http://www.ibm.com/data/dtd/v11/ibmxhtml1-transitional.dtd") {
3635            ## NOTE: Ensure that |PUBLIC "(limited quirks)" "(quirks)"| is            ## NOTE: Ensure that |PUBLIC "(limited quirks)" "(quirks)"| is
# Line 3342  sub _tree_construction_initial ($) { Line 3659  sub _tree_construction_initial ($) {
3659        !!!ack-later;        !!!ack-later;
3660        return;        return;
3661      } elsif ($token->{type} == CHARACTER_TOKEN) {      } elsif ($token->{type} == CHARACTER_TOKEN) {
3662        if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D        if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
3663          ## Ignore the token          ## Ignore the token
3664    
3665          unless (length $token->{data}) {          unless (length $token->{data}) {
# Line 3399  sub _tree_construction_root_element ($) Line 3716  sub _tree_construction_root_element ($)
3716          !!!next-token;          !!!next-token;
3717          redo B;          redo B;
3718        } elsif ($token->{type} == CHARACTER_TOKEN) {        } elsif ($token->{type} == CHARACTER_TOKEN) {
3719          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D          if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
3720            ## Ignore the token.            ## Ignore the token.
3721    
3722            unless (length $token->{data}) {            unless (length $token->{data}) {
# Line 3466  sub _tree_construction_root_element ($) Line 3783  sub _tree_construction_root_element ($)
3783      ## NOTE: Reprocess the token.      ## NOTE: Reprocess the token.
3784      !!!ack-later;      !!!ack-later;
3785      return; ## Go to the "before head" insertion mode.      return; ## Go to the "before head" insertion mode.
   
     ## ISSUE: There is an issue in the spec  
3786    } # B    } # B
3787    
3788    die "$0: _tree_construction_root_element: This should never be reached";    die "$0: _tree_construction_root_element: This should never be reached";
# Line 3664  sub _tree_construction_main ($) { Line 3979  sub _tree_construction_main ($) {
3979    
3980      ## Step 1      ## Step 1
3981      my $start_tag_name = $token->{tag_name};      my $start_tag_name = $token->{tag_name};
3982      my $el;      !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
     !!!create-element ($el, $HTML_NS, $start_tag_name, $token->{attributes}, $token);  
3983    
3984      ## Step 2      ## Step 2
     $insert->($el);  
   
     ## Step 3  
3985      $self->{content_model} = $content_model_flag; # CDATA or RCDATA      $self->{content_model} = $content_model_flag; # CDATA or RCDATA
3986      delete $self->{escape}; # MUST      delete $self->{escape}; # MUST
3987    
3988      ## Step 4      ## Step 3, 4
3989      my $text = '';      $self->{insertion_mode} |= IN_CDATA_RCDATA_IM;
     !!!nack ('t40.1');  
     !!!next-token;  
     while ($token->{type} == CHARACTER_TOKEN) { # or until stop tokenizing  
       !!!cp ('t40');  
       $text .= $token->{data};  
       !!!next-token;  
     }  
   
     ## Step 5  
     if (length $text) {  
       !!!cp ('t41');  
       my $text = $self->{document}->create_text_node ($text);  
       $el->append_child ($text);  
     }  
   
     ## Step 6  
     $self->{content_model} = PCDATA_CONTENT_MODEL;  
3990    
3991      ## Step 7      !!!nack ('t40.1');
     if ($token->{type} == END_TAG_TOKEN and  
         $token->{tag_name} eq $start_tag_name) {  
       !!!cp ('t42');  
       ## Ignore the token  
     } else {  
       ## NOTE: An end-of-file token.  
       if ($content_model_flag == CDATA_CONTENT_MODEL) {  
         !!!cp ('t43');  
         !!!parse-error (type => 'in CDATA:#eof', token => $token);  
       } elsif ($content_model_flag == RCDATA_CONTENT_MODEL) {  
         !!!cp ('t44');  
         !!!parse-error (type => 'in RCDATA:#eof', token => $token);  
       } else {  
         die "$0: $content_model_flag in parse_rcdata";  
       }  
     }  
3992      !!!next-token;      !!!next-token;
3993    }; # $parse_rcdata    }; # $parse_rcdata
3994    
3995    my $script_start_tag = sub () {    my $script_start_tag = sub () {
3996        ## Step 1
3997      my $script_el;      my $script_el;
3998      !!!create-element ($script_el, $HTML_NS, 'script', $token->{attributes}, $token);      !!!create-element ($script_el, $HTML_NS, 'script', $token->{attributes}, $token);
3999    
4000        ## Step 2
4001      ## TODO: mark as "parser-inserted"      ## TODO: mark as "parser-inserted"
4002    
4003        ## Step 3
4004        ## TODO: Mark as "already executed", if ...
4005    
4006        ## Step 4
4007        $insert->($script_el);
4008    
4009        ## ISSUE: $script_el is not put into the stack
4010        push @{$self->{open_elements}}, [$script_el, $el_category->{script}];
4011    
4012        ## Step 5
4013      $self->{content_model} = CDATA_CONTENT_MODEL;      $self->{content_model} = CDATA_CONTENT_MODEL;
4014      delete $self->{escape}; # MUST      delete $self->{escape}; # MUST
       
     my $text = '';  
     !!!nack ('t45.1');  
     !!!next-token;  
     while ($token->{type} == CHARACTER_TOKEN) {  
       !!!cp ('t45');  
       $text .= $token->{data};  
       !!!next-token;  
     } # stop if non-character token or tokenizer stops tokenising  
     if (length $text) {  
       !!!cp ('t46');  
       $script_el->manakai_append_text ($text);  
     }  
                 
     $self->{content_model} = PCDATA_CONTENT_MODEL;  
4015    
4016      if ($token->{type} == END_TAG_TOKEN and      ## Step 6-7
4017          $token->{tag_name} eq 'script') {      $self->{insertion_mode} |= IN_CDATA_RCDATA_IM;
       !!!cp ('t47');  
       ## Ignore the token  
     } else {  
       !!!cp ('t48');  
       !!!parse-error (type => 'in CDATA:#eof', token => $token);  
       ## ISSUE: And ignore?  
       ## TODO: mark as "already executed"  
     }  
       
     if (defined $self->{inner_html_node}) {  
       !!!cp ('t49');  
       ## TODO: mark as "already executed"  
     } else {  
       !!!cp ('t50');  
       ## TODO: $old_insertion_point = current insertion point  
       ## TODO: insertion point = just before the next input character  
4018    
4019        $insert->($script_el);      !!!nack ('t40.2');
         
       ## TODO: insertion point = $old_insertion_point (might be "undefined")  
         
       ## TODO: if there is a script that will execute as soon as the parser resume, then...  
     }  
       
4020      !!!next-token;      !!!next-token;
4021    }; # $script_start_tag    }; # $script_start_tag
4022    
4023    ## NOTE: $open_tables->[-1]->[0] is the "current table" element node.    ## NOTE: $open_tables->[-1]->[0] is the "current table" element node.
4024    ## NOTE: $open_tables->[-1]->[1] is the "tainted" flag.    ## NOTE: $open_tables->[-1]->[1] is the "tainted" flag.
4025      ## NOTE: $open_tables->[-1]->[2] is set false when non-Text node inserted.
4026    my $open_tables = [[$self->{open_elements}->[0]->[0]]];    my $open_tables = [[$self->{open_elements}->[0]->[0]]];
4027    
4028    my $formatting_end_tag = sub {    my $formatting_end_tag = sub {
# Line 3852  sub _tree_construction_main ($) { Line 4107  sub _tree_construction_main ($) {
4107            !!!cp ('t59');            !!!cp ('t59');
4108            $furthest_block = $node;            $furthest_block = $node;
4109            $furthest_block_i_in_open = $_;            $furthest_block_i_in_open = $_;
4110              ## NOTE: The topmost (eldest) node.
4111          } elsif ($node->[0] eq $formatting_element->[0]) {          } elsif ($node->[0] eq $formatting_element->[0]) {
4112            !!!cp ('t60');            !!!cp ('t60');
4113            last OE;            last OE;
# Line 3998  sub _tree_construction_main ($) { Line 4254  sub _tree_construction_main ($) {
4254            $i = $_;            $i = $_;
4255          }          }
4256        } # OE        } # OE
4257        splice @{$self->{open_elements}}, $i + 1, 1, $clone;        splice @{$self->{open_elements}}, $i + 1, 0, $clone;
4258                
4259        ## Step 14        ## Step 14
4260        redo FET;        redo FET;
# Line 4041  sub _tree_construction_main ($) { Line 4297  sub _tree_construction_main ($) {
4297      }      }
4298    }; # $insert_to_foster    }; # $insert_to_foster
4299    
4300      ## NOTE: Insert a character (MUST): When a character is inserted, if
4301      ## the last node that was inserted by the parser is a Text node and
4302      ## the character has to be inserted after that node, then the
4303      ## character is appended to the Text node.  However, if any other
4304      ## node is inserted by the parser, then a new Text node is created
4305      ## and the character is appended as that Text node.  If I'm not
4306      ## wrong, for a parser with scripting disabled, there are only two
4307      ## cases where this occurs.  One is the case where an element node
4308      ## is inserted to the |head| element.  This is covered by using the
4309      ## |$self->{head_element_inserted}| flag.  Another is the case where
4310      ## an element or comment is inserted into the |table| subtree while
4311      ## foster parenting happens.  This is covered by using the [2] flag
4312      ## of the |$open_tables| structure.  All other cases are handled
4313      ## simply by calling |manakai_append_text| method.
4314    
4315      ## TODO: |<body><script>document.write("a<br>");
4316      ## document.body.removeChild (document.body.lastChild);
4317      ## document.write ("b")</script>|
4318    
4319    B: while (1) {    B: while (1) {
4320      if ($token->{type} == DOCTYPE_TOKEN) {      if ($token->{type} == DOCTYPE_TOKEN) {
4321        !!!cp ('t73');        !!!cp ('t73');
# Line 4088  sub _tree_construction_main ($) { Line 4363  sub _tree_construction_main ($) {
4363        } else {        } else {
4364          !!!cp ('t87');          !!!cp ('t87');
4365          $self->{open_elements}->[-1]->[0]->append_child ($comment);          $self->{open_elements}->[-1]->[0]->append_child ($comment);
4366            $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
4367        }        }
4368        !!!next-token;        !!!next-token;
4369        next B;        next B;
4370        } elsif ($self->{insertion_mode} & IN_CDATA_RCDATA_IM) {
4371          if ($token->{type} == CHARACTER_TOKEN) {
4372            $token->{data} =~ s/^\x0A// if $self->{ignore_newline};
4373            delete $self->{ignore_newline};
4374    
4375            if (length $token->{data}) {
4376              !!!cp ('t43');
4377              $self->{open_elements}->[-1]->[0]->manakai_append_text
4378                  ($token->{data});
4379            } else {
4380              !!!cp ('t43.1');
4381            }
4382            !!!next-token;
4383            next B;
4384          } elsif ($token->{type} == END_TAG_TOKEN) {
4385            delete $self->{ignore_newline};
4386    
4387            if ($token->{tag_name} eq 'script') {
4388              !!!cp ('t50');
4389              
4390              ## Para 1-2
4391              my $script = pop @{$self->{open_elements}};
4392              
4393              ## Para 3
4394              $self->{insertion_mode} &= ~ IN_CDATA_RCDATA_IM;
4395    
4396              ## Para 4
4397              ## TODO: $old_insertion_point = $current_insertion_point;
4398              ## TODO: $current_insertion_point = just before $self->{nc};
4399    
4400              ## Para 5
4401              ## TODO: Run the $script->[0].
4402    
4403              ## Para 6
4404              ## TODO: $current_insertion_point = $old_insertion_point;
4405    
4406              ## Para 7
4407              ## TODO: if ($pending_external_script) {
4408                ## TODO: ...
4409              ## TODO: }
4410    
4411              !!!next-token;
4412              next B;
4413            } else {
4414              !!!cp ('t42');
4415    
4416              pop @{$self->{open_elements}};
4417    
4418              $self->{insertion_mode} &= ~ IN_CDATA_RCDATA_IM;
4419              !!!next-token;
4420              next B;
4421            }
4422          } elsif ($token->{type} == END_OF_FILE_TOKEN) {
4423            delete $self->{ignore_newline};
4424    
4425            !!!cp ('t44');
4426            !!!parse-error (type => 'not closed',
4427                            text => $self->{open_elements}->[-1]->[0]
4428                                ->manakai_local_name,
4429                            token => $token);
4430    
4431            #if ($self->{open_elements}->[-1]->[1] & SCRIPT_EL) {
4432            #  ## TODO: Mark as "already executed"
4433            #}
4434    
4435            pop @{$self->{open_elements}};
4436    
4437            $self->{insertion_mode} &= ~ IN_CDATA_RCDATA_IM;
4438            ## Reprocess.
4439            next B;
4440          } else {
4441            die "$0: $token->{type}: In CDATA/RCDATA: Unknown token type";        
4442          }
4443      } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM) {      } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM) {
4444        if ($token->{type} == CHARACTER_TOKEN) {        if ($token->{type} == CHARACTER_TOKEN) {
4445          !!!cp ('t87.1');          !!!cp ('t87.1');
# Line 4203  sub _tree_construction_main ($) { Line 4552  sub _tree_construction_main ($) {
4552          pop @{$self->{open_elements}}          pop @{$self->{open_elements}}
4553              while $self->{open_elements}->[-1]->[1] & FOREIGN_EL;              while $self->{open_elements}->[-1]->[1] & FOREIGN_EL;
4554    
4555            ## NOTE: |<span><svg>| ... two parse errors, |<svg>| ... a parse error.
4556    
4557          $self->{insertion_mode} &= ~ IN_FOREIGN_CONTENT_IM;          $self->{insertion_mode} &= ~ IN_FOREIGN_CONTENT_IM;
4558          ## Reprocess.          ## Reprocess.
4559          next B;          next B;
# Line 4213  sub _tree_construction_main ($) { Line 4564  sub _tree_construction_main ($) {
4564    
4565      if ($self->{insertion_mode} & HEAD_IMS) {      if ($self->{insertion_mode} & HEAD_IMS) {
4566        if ($token->{type} == CHARACTER_TOKEN) {        if ($token->{type} == CHARACTER_TOKEN) {
4567          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {          if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
4568            unless ($self->{insertion_mode} == BEFORE_HEAD_IM) {            unless ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4569              !!!cp ('t88.2');              if ($self->{head_element_inserted}) {
4570              $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);                !!!cp ('t88.3');
4571                  $self->{open_elements}->[-1]->[0]->append_child
4572                    ($self->{document}->create_text_node ($1));
4573                  delete $self->{head_element_inserted};
4574                  ## NOTE: |</head> <link> |
4575                  #
4576                } else {
4577                  !!!cp ('t88.2');
4578                  $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
4579                  ## NOTE: |</head> &#x20;|
4580                  #
4581                }
4582            } else {            } else {
4583              !!!cp ('t88.1');              !!!cp ('t88.1');
4584              ## Ignore the token.              ## Ignore the token.
4585              !!!next-token;              #
             next B;  
4586            }            }
4587            unless (length $token->{data}) {            unless (length $token->{data}) {
4588              !!!cp ('t88');              !!!cp ('t88');
4589              !!!next-token;              !!!next-token;
4590              next B;              next B;
4591            }            }
4592    ## TODO: set $token->{column} appropriately
4593          }          }
4594    
4595          if ($self->{insertion_mode} == BEFORE_HEAD_IM) {          if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
# Line 4312  sub _tree_construction_main ($) { Line 4674  sub _tree_construction_main ($) {
4674            !!!cp ('t97');            !!!cp ('t97');
4675          }          }
4676    
4677              if ($token->{tag_name} eq 'base') {          if ($token->{tag_name} eq 'base') {
4678                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {            if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4679                  !!!cp ('t98');              !!!cp ('t98');
4680                  ## As if </noscript>              ## As if </noscript>
4681                  pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
4682                  !!!parse-error (type => 'in noscript', text => 'base',              !!!parse-error (type => 'in noscript', text => 'base',
4683                                  token => $token);                              token => $token);
4684                            
4685                  $self->{insertion_mode} = IN_HEAD_IM;              $self->{insertion_mode} = IN_HEAD_IM;
4686                  ## Reprocess in the "in head" insertion mode...              ## Reprocess in the "in head" insertion mode...
4687                } else {            } else {
4688                  !!!cp ('t99');              !!!cp ('t99');
4689                }            }
4690    
4691                ## NOTE: There is a "as if in head" code clone.            ## NOTE: There is a "as if in head" code clone.
4692                if ($self->{insertion_mode} == AFTER_HEAD_IM) {            if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4693                  !!!cp ('t100');              !!!cp ('t100');
4694                  !!!parse-error (type => 'after head',              !!!parse-error (type => 'after head',
4695                                  text => $token->{tag_name}, token => $token);                              text => $token->{tag_name}, token => $token);
4696                  push @{$self->{open_elements}},              push @{$self->{open_elements}},
4697                      [$self->{head_element}, $el_category->{head}];                  [$self->{head_element}, $el_category->{head}];
4698                } else {              $self->{head_element_inserted} = 1;
4699                  !!!cp ('t101');            } else {
4700                }              !!!cp ('t101');
4701                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);            }
4702                pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.            !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4703                pop @{$self->{open_elements}} # <head>            pop @{$self->{open_elements}};
4704                    if $self->{insertion_mode} == AFTER_HEAD_IM;            pop @{$self->{open_elements}} # <head>
4705                !!!nack ('t101.1');                if $self->{insertion_mode} == AFTER_HEAD_IM;
4706                !!!next-token;            !!!nack ('t101.1');
4707                next B;            !!!next-token;
4708              } elsif ($token->{tag_name} eq 'link') {            next B;
4709                ## NOTE: There is a "as if in head" code clone.          } elsif ($token->{tag_name} eq 'link') {
4710                if ($self->{insertion_mode} == AFTER_HEAD_IM) {            ## NOTE: There is a "as if in head" code clone.
4711                  !!!cp ('t102');            if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4712                  !!!parse-error (type => 'after head',              !!!cp ('t102');
4713                                  text => $token->{tag_name}, token => $token);              !!!parse-error (type => 'after head',
4714                  push @{$self->{open_elements}},                              text => $token->{tag_name}, token => $token);
4715                      [$self->{head_element}, $el_category->{head}];              push @{$self->{open_elements}},
4716                } else {                  [$self->{head_element}, $el_category->{head}];
4717                  !!!cp ('t103');              $self->{head_element_inserted} = 1;
4718                }            } else {
4719                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);              !!!cp ('t103');
4720                pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.            }
4721                pop @{$self->{open_elements}} # <head>            !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4722                    if $self->{insertion_mode} == AFTER_HEAD_IM;            pop @{$self->{open_elements}};
4723                !!!ack ('t103.1');            pop @{$self->{open_elements}} # <head>
4724                !!!next-token;                if $self->{insertion_mode} == AFTER_HEAD_IM;
4725                next B;            !!!ack ('t103.1');
4726              } elsif ($token->{tag_name} eq 'meta') {            !!!next-token;
4727                ## NOTE: There is a "as if in head" code clone.            next B;
4728                if ($self->{insertion_mode} == AFTER_HEAD_IM) {          } elsif ($token->{tag_name} eq 'command' or
4729                  !!!cp ('t104');                   $token->{tag_name} eq 'eventsource') {
4730                  !!!parse-error (type => 'after head',            if ($self->{insertion_mode} == IN_HEAD_IM) {
4731                                  text => $token->{tag_name}, token => $token);              ## NOTE: If the insertion mode at the time of the emission
4732                  push @{$self->{open_elements}},              ## of the token was "before head", $self->{insertion_mode}
4733                      [$self->{head_element}, $el_category->{head}];              ## is already changed to |IN_HEAD_IM|.
4734                } else {  
4735                  !!!cp ('t105');              ## NOTE: There is a "as if in head" code clone.
4736                }              !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4737                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);              pop @{$self->{open_elements}};
4738                my $meta_el = pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.              pop @{$self->{open_elements}} # <head>
4739                    if $self->{insertion_mode} == AFTER_HEAD_IM;
4740                !!!ack ('t103.2');
4741                !!!next-token;
4742                next B;
4743              } else {
4744                ## NOTE: "in head noscript" or "after head" insertion mode
4745                ## - in these cases, these tags are treated as same as
4746                ## normal in-body tags.
4747                !!!cp ('t103.3');
4748                #
4749              }
4750            } elsif ($token->{tag_name} eq 'meta') {
4751              ## NOTE: There is a "as if in head" code clone.
4752              if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4753                !!!cp ('t104');
4754                !!!parse-error (type => 'after head',
4755                                text => $token->{tag_name}, token => $token);
4756                push @{$self->{open_elements}},
4757                    [$self->{head_element}, $el_category->{head}];
4758                $self->{head_element_inserted} = 1;
4759              } else {
4760                !!!cp ('t105');
4761              }
4762              !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4763              my $meta_el = pop @{$self->{open_elements}};
4764    
4765                unless ($self->{confident}) {                unless ($self->{confident}) {
4766                  if ($token->{attributes}->{charset}) {                  if ($token->{attributes}->{charset}) {
# Line 4391  sub _tree_construction_main ($) { Line 4778  sub _tree_construction_main ($) {
4778                  } elsif ($token->{attributes}->{content}) {                  } elsif ($token->{attributes}->{content}) {
4779                    if ($token->{attributes}->{content}->{value}                    if ($token->{attributes}->{content}->{value}
4780                        =~ /[Cc][Hh][Aa][Rr][Ss][Ee][Tt]                        =~ /[Cc][Hh][Aa][Rr][Ss][Ee][Tt]
4781                            [\x09-\x0D\x20]*=                            [\x09\x0A\x0C\x0D\x20]*=
4782                            [\x09-\x0D\x20]*(?>"([^"]*)"|'([^']*)'|                            [\x09\x0A\x0C\x0D\x20]*(?>"([^"]*)"|'([^']*)'|
4783                            ([^"'\x09-\x0D\x20][^\x09-\x0D\x20\x3B]*))/x) {                            ([^"'\x09\x0A\x0C\x0D\x20]
4784                               [^\x09\x0A\x0C\x0D\x20\x3B]*))/x) {
4785                      !!!cp ('t107');                      !!!cp ('t107');
4786                      ## NOTE: Whether the encoding is supported or not is handled                      ## NOTE: Whether the encoding is supported or not is handled
4787                      ## in the {change_encoding} callback.                      ## in the {change_encoding} callback.
# Line 4430  sub _tree_construction_main ($) { Line 4818  sub _tree_construction_main ($) {
4818                !!!ack ('t110.1');                !!!ack ('t110.1');
4819                !!!next-token;                !!!next-token;
4820                next B;                next B;
4821              } elsif ($token->{tag_name} eq 'title') {          } elsif ($token->{tag_name} eq 'title') {
4822                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {            if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4823                  !!!cp ('t111');              !!!cp ('t111');
4824                  ## As if </noscript>              ## As if </noscript>
4825                  pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
4826                  !!!parse-error (type => 'in noscript', text => 'title',              !!!parse-error (type => 'in noscript', text => 'title',
4827                                  token => $token);                              token => $token);
4828                            
4829                  $self->{insertion_mode} = IN_HEAD_IM;              $self->{insertion_mode} = IN_HEAD_IM;
4830                  ## Reprocess in the "in head" insertion mode...              ## Reprocess in the "in head" insertion mode...
4831                } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {            } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4832                  !!!cp ('t112');              !!!cp ('t112');
4833                  !!!parse-error (type => 'after head',              !!!parse-error (type => 'after head',
4834                                  text => $token->{tag_name}, token => $token);                              text => $token->{tag_name}, token => $token);
4835                  push @{$self->{open_elements}},              push @{$self->{open_elements}},
4836                      [$self->{head_element}, $el_category->{head}];                  [$self->{head_element}, $el_category->{head}];
4837                } else {              $self->{head_element_inserted} = 1;
4838                  !!!cp ('t113');            } else {
4839                }              !!!cp ('t113');
4840              }
4841    
4842                ## NOTE: There is a "as if in head" code clone.            ## NOTE: There is a "as if in head" code clone.
4843                my $parent = defined $self->{head_element} ? $self->{head_element}            $parse_rcdata->(RCDATA_CONTENT_MODEL);
4844                    : $self->{open_elements}->[-1]->[0];            ## ISSUE: A spec bug [Bug 6038]
4845                $parse_rcdata->(RCDATA_CONTENT_MODEL);            splice @{$self->{open_elements}}, -2, 1, () # <head>
4846                pop @{$self->{open_elements}} # <head>                if ($self->{insertion_mode} & AFTER_HEAD_IM) == AFTER_HEAD_IM;
4847                    if $self->{insertion_mode} == AFTER_HEAD_IM;            next B;
4848                next B;          } elsif ($token->{tag_name} eq 'style' or
4849              } elsif ($token->{tag_name} eq 'style' or                   $token->{tag_name} eq 'noframes') {
4850                       $token->{tag_name} eq 'noframes') {            ## NOTE: Or (scripting is enabled and tag_name eq 'noscript' and
4851                ## NOTE: Or (scripting is enabled and tag_name eq 'noscript' and            ## insertion mode IN_HEAD_IM)
4852                ## insertion mode IN_HEAD_IM)            ## NOTE: There is a "as if in head" code clone.
4853                ## NOTE: There is a "as if in head" code clone.            if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4854                if ($self->{insertion_mode} == AFTER_HEAD_IM) {              !!!cp ('t114');
4855                  !!!cp ('t114');              !!!parse-error (type => 'after head',
4856                  !!!parse-error (type => 'after head',                              text => $token->{tag_name}, token => $token);
4857                                  text => $token->{tag_name}, token => $token);              push @{$self->{open_elements}},
4858                  push @{$self->{open_elements}},                  [$self->{head_element}, $el_category->{head}];
4859                      [$self->{head_element}, $el_category->{head}];              $self->{head_element_inserted} = 1;
4860                } else {            } else {
4861                  !!!cp ('t115');              !!!cp ('t115');
4862                }            }
4863                $parse_rcdata->(CDATA_CONTENT_MODEL);            $parse_rcdata->(CDATA_CONTENT_MODEL);
4864                pop @{$self->{open_elements}} # <head>            ## ISSUE: A spec bug [Bug 6038]
4865                    if $self->{insertion_mode} == AFTER_HEAD_IM;            splice @{$self->{open_elements}}, -2, 1, () # <head>
4866                next B;                if ($self->{insertion_mode} & AFTER_HEAD_IM) == AFTER_HEAD_IM;
4867              } elsif ($token->{tag_name} eq 'noscript') {            next B;
4868            } elsif ($token->{tag_name} eq 'noscript') {
4869                if ($self->{insertion_mode} == IN_HEAD_IM) {                if ($self->{insertion_mode} == IN_HEAD_IM) {
4870                  !!!cp ('t116');                  !!!cp ('t116');
4871                  ## NOTE: and scripting is disalbed                  ## NOTE: and scripting is disalbed
# Line 4496  sub _tree_construction_main ($) { Line 4886  sub _tree_construction_main ($) {
4886                  !!!cp ('t118');                  !!!cp ('t118');
4887                  #                  #
4888                }                }
4889              } elsif ($token->{tag_name} eq 'script') {          } elsif ($token->{tag_name} eq 'script') {
4890                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {            if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4891                  !!!cp ('t119');              !!!cp ('t119');
4892                  ## As if </noscript>              ## As if </noscript>
4893                  pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
4894                  !!!parse-error (type => 'in noscript', text => 'script',              !!!parse-error (type => 'in noscript', text => 'script',
4895                                  token => $token);                              token => $token);
4896                            
4897                  $self->{insertion_mode} = IN_HEAD_IM;              $self->{insertion_mode} = IN_HEAD_IM;
4898                  ## Reprocess in the "in head" insertion mode...              ## Reprocess in the "in head" insertion mode...
4899                } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {            } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4900                  !!!cp ('t120');              !!!cp ('t120');
4901                  !!!parse-error (type => 'after head',              !!!parse-error (type => 'after head',
4902                                  text => $token->{tag_name}, token => $token);                              text => $token->{tag_name}, token => $token);
4903                  push @{$self->{open_elements}},              push @{$self->{open_elements}},
4904                      [$self->{head_element}, $el_category->{head}];                  [$self->{head_element}, $el_category->{head}];
4905                } else {              $self->{head_element_inserted} = 1;
4906                  !!!cp ('t121');            } else {
4907                }              !!!cp ('t121');
4908              }
4909    
4910                ## NOTE: There is a "as if in head" code clone.            ## NOTE: There is a "as if in head" code clone.
4911                $script_start_tag->();            $script_start_tag->();
4912                pop @{$self->{open_elements}} # <head>            ## ISSUE: A spec bug  [Bug 6038]
4913                    if $self->{insertion_mode} == AFTER_HEAD_IM;            splice @{$self->{open_elements}}, -2, 1 # <head>
4914                next B;                if ($self->{insertion_mode} & AFTER_HEAD_IM) == AFTER_HEAD_IM;
4915              } elsif ($token->{tag_name} eq 'body' or            next B;
4916                       $token->{tag_name} eq 'frameset') {          } elsif ($token->{tag_name} eq 'body' or
4917                     $token->{tag_name} eq 'frameset') {
4918                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4919                  !!!cp ('t122');                  !!!cp ('t122');
4920                  ## As if </noscript>                  ## As if </noscript>
# Line 4657  sub _tree_construction_main ($) { Line 5049  sub _tree_construction_main ($) {
5049              } elsif ({              } elsif ({
5050                        body => 1, html => 1,                        body => 1, html => 1,
5051                       }->{$token->{tag_name}}) {                       }->{$token->{tag_name}}) {
5052                if ($self->{insertion_mode} == BEFORE_HEAD_IM or                ## TODO: This branch is entirely redundant.
5053                  if ($self->{insertion_mode} == BEFORE_HEAD_IM or
5054                    $self->{insertion_mode} == IN_HEAD_IM or                    $self->{insertion_mode} == IN_HEAD_IM or
5055                    $self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {                    $self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
5056                  !!!cp ('t140');                  !!!cp ('t140');
# Line 4829  sub _tree_construction_main ($) { Line 5222  sub _tree_construction_main ($) {
5222        } else {        } else {
5223          die "$0: $token->{type}: Unknown token type";          die "$0: $token->{type}: Unknown token type";
5224        }        }
   
           ## ISSUE: An issue in the spec.  
5225      } elsif ($self->{insertion_mode} & BODY_IMS) {      } elsif ($self->{insertion_mode} & BODY_IMS) {
5226            if ($token->{type} == CHARACTER_TOKEN) {            if ($token->{type} == CHARACTER_TOKEN) {
5227              !!!cp ('t150');              !!!cp ('t150');
# Line 5202  sub _tree_construction_main ($) { Line 5593  sub _tree_construction_main ($) {
5593      } elsif ($self->{insertion_mode} & TABLE_IMS) {      } elsif ($self->{insertion_mode} & TABLE_IMS) {
5594        if ($token->{type} == CHARACTER_TOKEN) {        if ($token->{type} == CHARACTER_TOKEN) {
5595          if (not $open_tables->[-1]->[1] and # tainted          if (not $open_tables->[-1]->[1] and # tainted
5596              $token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {              $token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
5597            $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);            $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
5598                                
5599            unless (length $token->{data}) {            unless (length $token->{data}) {
# Line 5216  sub _tree_construction_main ($) { Line 5607  sub _tree_construction_main ($) {
5607    
5608          !!!parse-error (type => 'in table:#text', token => $token);          !!!parse-error (type => 'in table:#text', token => $token);
5609    
5610              ## As if in body, but insert into foster parent element          ## NOTE: As if in body, but insert into the foster parent element.
5611              ## ISSUE: Spec says that "whenever a node would be inserted          $reconstruct_active_formatting_elements->($insert_to_foster);
             ## into the current node" while characters might not be  
             ## result in a new Text node.  
             $reconstruct_active_formatting_elements->($insert_to_foster);  
5612                            
5613              if ($self->{open_elements}->[-1]->[1] & TABLE_ROWS_EL) {          if ($self->{open_elements}->[-1]->[1] & TABLE_ROWS_EL) {
5614                # MUST            # MUST
5615                my $foster_parent_element;            my $foster_parent_element;
5616                my $next_sibling;            my $next_sibling;
5617                my $prev_sibling;            my $prev_sibling;
5618                OE: for (reverse 0..$#{$self->{open_elements}}) {            OE: for (reverse 0..$#{$self->{open_elements}}) {
5619                  if ($self->{open_elements}->[$_]->[1] & TABLE_EL) {              if ($self->{open_elements}->[$_]->[1] & TABLE_EL) {
5620                    my $parent = $self->{open_elements}->[$_]->[0]->parent_node;                my $parent = $self->{open_elements}->[$_]->[0]->parent_node;
5621                    if (defined $parent and $parent->node_type == 1) {                if (defined $parent and $parent->node_type == 1) {
5622                      !!!cp ('t196');                  $foster_parent_element = $parent;
5623                      $foster_parent_element = $parent;                  !!!cp ('t196');
5624                      $next_sibling = $self->{open_elements}->[$_]->[0];                  $next_sibling = $self->{open_elements}->[$_]->[0];
5625                      $prev_sibling = $next_sibling->previous_sibling;                  $prev_sibling = $next_sibling->previous_sibling;
5626                    } else {                  #
                     !!!cp ('t197');  
                     $foster_parent_element = $self->{open_elements}->[$_ - 1]->[0];  
                     $prev_sibling = $foster_parent_element->last_child;  
                   }  
                   last OE;  
                 }  
               } # OE  
               $foster_parent_element = $self->{open_elements}->[0]->[0] and  
               $prev_sibling = $foster_parent_element->last_child  
                 unless defined $foster_parent_element;  
               if (defined $prev_sibling and  
                   $prev_sibling->node_type == 3) {  
                 !!!cp ('t198');  
                 $prev_sibling->manakai_append_text ($token->{data});  
5627                } else {                } else {
5628                  !!!cp ('t199');                  !!!cp ('t197');
5629                  $foster_parent_element->insert_before                  $foster_parent_element = $self->{open_elements}->[$_ - 1]->[0];
5630                    ($self->{document}->create_text_node ($token->{data}),                  $prev_sibling = $foster_parent_element->last_child;
5631                     $next_sibling);                  #
5632                }                }
5633                  last OE;
5634                }
5635              } # OE
5636              $foster_parent_element = $self->{open_elements}->[0]->[0] and
5637              $prev_sibling = $foster_parent_element->last_child
5638                  unless defined $foster_parent_element;
5639              undef $prev_sibling unless $open_tables->[-1]->[2]; # ~node inserted
5640              if (defined $prev_sibling and
5641                  $prev_sibling->node_type == 3) {
5642                !!!cp ('t198');
5643                $prev_sibling->manakai_append_text ($token->{data});
5644              } else {
5645                !!!cp ('t199');
5646                $foster_parent_element->insert_before
5647                    ($self->{document}->create_text_node ($token->{data}),
5648                     $next_sibling);
5649              }
5650            $open_tables->[-1]->[1] = 1; # tainted            $open_tables->[-1]->[1] = 1; # tainted
5651              $open_tables->[-1]->[2] = 1; # ~node inserted
5652          } else {          } else {
5653              ## NOTE: Fragment case or in a foster parent'ed element
5654              ## (e.g. |<table><span>a|).  In fragment case, whether the
5655              ## character is appended to existing node or a new node is
5656              ## created is irrelevant, since the foster parent'ed nodes
5657              ## are discarded and fragment parsing does not invoke any
5658              ## script.
5659            !!!cp ('t200');            !!!cp ('t200');
5660            $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});            $self->{open_elements}->[-1]->[0]->manakai_append_text
5661                  ($token->{data});
5662          }          }
5663                            
5664          !!!next-token;          !!!next-token;
# Line 5296  sub _tree_construction_main ($) { Line 5695  sub _tree_construction_main ($) {
5695                pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
5696              }              }
5697                                    
5698                  $self->{insertion_mode} = IN_ROW_IM;              $self->{insertion_mode} = IN_ROW_IM;
5699                  if ($token->{tag_name} eq 'tr') {              if ($token->{tag_name} eq 'tr') {
5700                    !!!cp ('t204');                !!!cp ('t204');
5701                    !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5702                    !!!nack ('t204');                $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5703                    !!!next-token;                !!!nack ('t204');
5704                    next B;                !!!next-token;
5705                  } else {                next B;
5706                    !!!cp ('t205');              } else {
5707                    !!!insert-element ('tr',, $token);                !!!cp ('t205');
5708                    ## reprocess in the "in row" insertion mode                !!!insert-element ('tr',, $token);
5709                  }                ## reprocess in the "in row" insertion mode
5710                } else {              }
5711                  !!!cp ('t206');            } else {
5712                }              !!!cp ('t206');
5713              }
5714    
5715                ## Clear back to table row context                ## Clear back to table row context
5716                while (not ($self->{open_elements}->[-1]->[1]                while (not ($self->{open_elements}->[-1]->[1]
# Line 5319  sub _tree_construction_main ($) { Line 5719  sub _tree_construction_main ($) {
5719                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
5720                }                }
5721                                
5722                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);            !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5723                $self->{insertion_mode} = IN_CELL_IM;            $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5724              $self->{insertion_mode} = IN_CELL_IM;
5725    
5726                push @$active_formatting_elements, ['#marker', ''];            push @$active_formatting_elements, ['#marker', ''];
5727                                
5728                !!!nack ('t207.1');            !!!nack ('t207.1');
5729              !!!next-token;
5730              next B;
5731            } elsif ({
5732                      caption => 1, col => 1, colgroup => 1,
5733                      tbody => 1, tfoot => 1, thead => 1,
5734                      tr => 1, # $self->{insertion_mode} == IN_ROW_IM
5735                     }->{$token->{tag_name}}) {
5736              if ($self->{insertion_mode} == IN_ROW_IM) {
5737                ## As if </tr>
5738                ## have an element in table scope
5739                my $i;
5740                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5741                  my $node = $self->{open_elements}->[$_];
5742                  if ($node->[1] & TABLE_ROW_EL) {
5743                    !!!cp ('t208');
5744                    $i = $_;
5745                    last INSCOPE;
5746                  } elsif ($node->[1] & TABLE_SCOPING_EL) {
5747                    !!!cp ('t209');
5748                    last INSCOPE;
5749                  }
5750                } # INSCOPE
5751                unless (defined $i) {
5752                  !!!cp ('t210');
5753                  ## TODO: This type is wrong.
5754                  !!!parse-error (type => 'unmacthed end tag',
5755                                  text => $token->{tag_name}, token => $token);
5756                  ## Ignore the token
5757                  !!!nack ('t210.1');
5758                !!!next-token;                !!!next-token;
5759                next B;                next B;
5760              } elsif ({              }
                       caption => 1, col => 1, colgroup => 1,  
                       tbody => 1, tfoot => 1, thead => 1,  
                       tr => 1, # $self->{insertion_mode} == IN_ROW_IM  
                      }->{$token->{tag_name}}) {  
               if ($self->{insertion_mode} == IN_ROW_IM) {  
                 ## As if </tr>  
                 ## have an element in table scope  
                 my $i;  
                 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
                   my $node = $self->{open_elements}->[$_];  
                   if ($node->[1] & TABLE_ROW_EL) {  
                     !!!cp ('t208');  
                     $i = $_;  
                     last INSCOPE;  
                   } elsif ($node->[1] & TABLE_SCOPING_EL) {  
                     !!!cp ('t209');  
                     last INSCOPE;  
                   }  
                 } # INSCOPE  
                 unless (defined $i) {  
                   !!!cp ('t210');  
 ## TODO: This type is wrong.  
                   !!!parse-error (type => 'unmacthed end tag',  
                                   text => $token->{tag_name}, token => $token);  
                   ## Ignore the token  
                   !!!nack ('t210.1');  
                   !!!next-token;  
                   next B;  
                 }  
5761                                    
5762                  ## Clear back to table row context                  ## Clear back to table row context
5763                  while (not ($self->{open_elements}->[-1]->[1]                  while (not ($self->{open_elements}->[-1]->[1]
# Line 5426  sub _tree_construction_main ($) { Line 5827  sub _tree_construction_main ($) {
5827                  !!!cp ('t218');                  !!!cp ('t218');
5828                }                }
5829    
5830                if ($token->{tag_name} eq 'col') {            if ($token->{tag_name} eq 'col') {
5831                  ## Clear back to table context              ## Clear back to table context
5832                  while (not ($self->{open_elements}->[-1]->[1]              while (not ($self->{open_elements}->[-1]->[1]
5833                                  & TABLE_SCOPING_EL)) {                              & TABLE_SCOPING_EL)) {
5834                    !!!cp ('t219');                !!!cp ('t219');
5835                    ## ISSUE: Can this state be reached?                ## ISSUE: Can this state be reached?
5836                    pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
5837                  }              }
5838                                
5839                  !!!insert-element ('colgroup',, $token);              !!!insert-element ('colgroup',, $token);
5840                  $self->{insertion_mode} = IN_COLUMN_GROUP_IM;              $self->{insertion_mode} = IN_COLUMN_GROUP_IM;
5841                  ## reprocess              ## reprocess
5842                  !!!ack-later;              $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5843                  next B;              !!!ack-later;
5844                } elsif ({              next B;
5845                          caption => 1,            } elsif ({
5846                          colgroup => 1,                      caption => 1,
5847                          tbody => 1, tfoot => 1, thead => 1,                      colgroup => 1,
5848                         }->{$token->{tag_name}}) {                      tbody => 1, tfoot => 1, thead => 1,
5849                  ## Clear back to table context                     }->{$token->{tag_name}}) {
5850                ## Clear back to table context
5851                  while (not ($self->{open_elements}->[-1]->[1]                  while (not ($self->{open_elements}->[-1]->[1]
5852                                  & TABLE_SCOPING_EL)) {                                  & TABLE_SCOPING_EL)) {
5853                    !!!cp ('t220');                    !!!cp ('t220');
# Line 5453  sub _tree_construction_main ($) { Line 5855  sub _tree_construction_main ($) {
5855                    pop @{$self->{open_elements}};                    pop @{$self->{open_elements}};
5856                  }                  }
5857                                    
5858                  push @$active_formatting_elements, ['#marker', '']              push @$active_formatting_elements, ['#marker', '']
5859                      if $token->{tag_name} eq 'caption';                  if $token->{tag_name} eq 'caption';
5860                                    
5861                  !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);              !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5862                  $self->{insertion_mode} = {              $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5863                                             caption => IN_CAPTION_IM,              $self->{insertion_mode} = {
5864                                             colgroup => IN_COLUMN_GROUP_IM,                                         caption => IN_CAPTION_IM,
5865                                             tbody => IN_TABLE_BODY_IM,                                         colgroup => IN_COLUMN_GROUP_IM,
5866                                             tfoot => IN_TABLE_BODY_IM,                                         tbody => IN_TABLE_BODY_IM,
5867                                             thead => IN_TABLE_BODY_IM,                                         tfoot => IN_TABLE_BODY_IM,
5868                                            }->{$token->{tag_name}};                                         thead => IN_TABLE_BODY_IM,
5869                  !!!next-token;                                        }->{$token->{tag_name}};
5870                  !!!nack ('t220.1');              !!!next-token;
5871                  next B;              !!!nack ('t220.1');
5872                } else {              next B;
5873                  die "$0: in table: <>: $token->{tag_name}";            } else {
5874                }              die "$0: in table: <>: $token->{tag_name}";
5875              }
5876              } elsif ($token->{tag_name} eq 'table') {              } elsif ($token->{tag_name} eq 'table') {
5877                !!!parse-error (type => 'not closed',                !!!parse-error (type => 'not closed',
5878                                text => $self->{open_elements}->[-1]->[0]                                text => $self->{open_elements}->[-1]->[0]
# Line 5532  sub _tree_construction_main ($) { Line 5935  sub _tree_construction_main ($) {
5935              !!!cp ('t227.8');              !!!cp ('t227.8');
5936              ## NOTE: This is a "as if in head" code clone.              ## NOTE: This is a "as if in head" code clone.
5937              $parse_rcdata->(CDATA_CONTENT_MODEL);              $parse_rcdata->(CDATA_CONTENT_MODEL);
5938                $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5939              next B;              next B;
5940            } else {            } else {
5941              !!!cp ('t227.7');              !!!cp ('t227.7');
# Line 5542  sub _tree_construction_main ($) { Line 5946  sub _tree_construction_main ($) {
5946              !!!cp ('t227.6');              !!!cp ('t227.6');
5947              ## NOTE: This is a "as if in head" code clone.              ## NOTE: This is a "as if in head" code clone.
5948              $script_start_tag->();              $script_start_tag->();
5949                $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5950              next B;              next B;
5951            } else {            } else {
5952              !!!cp ('t227.5');              !!!cp ('t227.5');
# Line 5557  sub _tree_construction_main ($) { Line 5962  sub _tree_construction_main ($) {
5962                                  text => $token->{tag_name}, token => $token);                                  text => $token->{tag_name}, token => $token);
5963    
5964                  !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);                  !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5965                    $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5966    
5967                  ## TODO: form element pointer                  ## TODO: form element pointer
5968    
# Line 5886  sub _tree_construction_main ($) { Line 6292  sub _tree_construction_main ($) {
6292        }        }
6293      } elsif ($self->{insertion_mode} == IN_COLUMN_GROUP_IM) {      } elsif ($self->{insertion_mode} == IN_COLUMN_GROUP_IM) {
6294            if ($token->{type} == CHARACTER_TOKEN) {            if ($token->{type} == CHARACTER_TOKEN) {
6295              if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {              if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
6296                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
6297                unless (length $token->{data}) {                unless (length $token->{data}) {
6298                  !!!cp ('t260');                  !!!cp ('t260');
# Line 6227  sub _tree_construction_main ($) { Line 6633  sub _tree_construction_main ($) {
6633        }        }
6634      } elsif ($self->{insertion_mode} & BODY_AFTER_IMS) {      } elsif ($self->{insertion_mode} & BODY_AFTER_IMS) {
6635        if ($token->{type} == CHARACTER_TOKEN) {        if ($token->{type} == CHARACTER_TOKEN) {
6636          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {          if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
6637            my $data = $1;            my $data = $1;
6638            ## As if in body            ## As if in body
6639            $reconstruct_active_formatting_elements->($insert_to_current);            $reconstruct_active_formatting_elements->($insert_to_current);
# Line 6244  sub _tree_construction_main ($) { Line 6650  sub _tree_construction_main ($) {
6650          if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {          if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
6651            !!!cp ('t301');            !!!cp ('t301');
6652            !!!parse-error (type => 'after html:#text', token => $token);            !!!parse-error (type => 'after html:#text', token => $token);
6653              #
           ## Reprocess in the "after body" insertion mode.  
6654          } else {          } else {
6655            !!!cp ('t302');            !!!cp ('t302');
6656              ## "after body" insertion mode
6657              !!!parse-error (type => 'after body:#text', token => $token);
6658              #
6659          }          }
           
         ## "after body" insertion mode  
         !!!parse-error (type => 'after body:#text', token => $token);  
6660    
6661          $self->{insertion_mode} = IN_BODY_IM;          $self->{insertion_mode} = IN_BODY_IM;
6662          ## reprocess          ## reprocess
# Line 6261  sub _tree_construction_main ($) { Line 6666  sub _tree_construction_main ($) {
6666            !!!cp ('t303');            !!!cp ('t303');
6667            !!!parse-error (type => 'after html',            !!!parse-error (type => 'after html',
6668                            text => $token->{tag_name}, token => $token);                            text => $token->{tag_name}, token => $token);
6669                        #
           ## Reprocess in the "after body" insertion mode.  
6670          } else {          } else {
6671            !!!cp ('t304');            !!!cp ('t304');
6672              ## "after body" insertion mode
6673              !!!parse-error (type => 'after body',
6674                              text => $token->{tag_name}, token => $token);
6675              #
6676          }          }
6677    
         ## "after body" insertion mode  
         !!!parse-error (type => 'after body',  
                         text => $token->{tag_name}, token => $token);  
   
6678          $self->{insertion_mode} = IN_BODY_IM;          $self->{insertion_mode} = IN_BODY_IM;
6679          !!!ack-later;          !!!ack-later;
6680          ## reprocess          ## reprocess
# Line 6281  sub _tree_construction_main ($) { Line 6685  sub _tree_construction_main ($) {
6685            !!!parse-error (type => 'after html:/',            !!!parse-error (type => 'after html:/',
6686                            text => $token->{tag_name}, token => $token);                            text => $token->{tag_name}, token => $token);
6687                        
6688            $self->{insertion_mode} = AFTER_BODY_IM;            $self->{insertion_mode} = IN_BODY_IM;
6689            ## Reprocess in the "after body" insertion mode.            ## Reprocess.
6690              next B;
6691          } else {          } else {
6692            !!!cp ('t306');            !!!cp ('t306');
6693          }          }
# Line 6320  sub _tree_construction_main ($) { Line 6725  sub _tree_construction_main ($) {
6725        }        }
6726      } elsif ($self->{insertion_mode} & FRAME_IMS) {      } elsif ($self->{insertion_mode} & FRAME_IMS) {
6727        if ($token->{type} == CHARACTER_TOKEN) {        if ($token->{type} == CHARACTER_TOKEN) {
6728          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {          if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
6729            $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);            $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
6730                        
6731            unless (length $token->{data}) {            unless (length $token->{data}) {
# Line 6330  sub _tree_construction_main ($) { Line 6735  sub _tree_construction_main ($) {
6735            }            }
6736          }          }
6737                    
6738          if ($token->{data} =~ s/^[^\x09\x0A\x0B\x0C\x20]+//) {          if ($token->{data} =~ s/^[^\x09\x0A\x0C\x20]+//) {
6739            if ($self->{insertion_mode} == IN_FRAMESET_IM) {            if ($self->{insertion_mode} == IN_FRAMESET_IM) {
6740              !!!cp ('t311');              !!!cp ('t311');
6741              !!!parse-error (type => 'in frameset:#text', token => $token);              !!!parse-error (type => 'in frameset:#text', token => $token);
# Line 6459  sub _tree_construction_main ($) { Line 6864  sub _tree_construction_main ($) {
6864        } else {        } else {
6865          die "$0: $token->{type}: Unknown token type";          die "$0: $token->{type}: Unknown token type";
6866        }        }
   
       ## ISSUE: An issue in spec here  
6867      } else {      } else {
6868        die "$0: $self->{insertion_mode}: Unknown insertion mode";        die "$0: $self->{insertion_mode}: Unknown insertion mode";
6869      }      }
# Line 6478  sub _tree_construction_main ($) { Line 6881  sub _tree_construction_main ($) {
6881          $parse_rcdata->(CDATA_CONTENT_MODEL);          $parse_rcdata->(CDATA_CONTENT_MODEL);
6882          next B;          next B;
6883        } elsif ({        } elsif ({
6884                  base => 1, link => 1,                  base => 1, command => 1, eventsource => 1, link => 1,
6885                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
6886          !!!cp ('t334');          !!!cp ('t334');
6887          ## NOTE: This is an "as if in head" code clone, only "-t" differs          ## NOTE: This is an "as if in head" code clone, only "-t" differs
6888          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
6889          pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.          pop @{$self->{open_elements}};
6890          !!!ack ('t334.1');          !!!ack ('t334.1');
6891          !!!next-token;          !!!next-token;
6892          next B;          next B;
6893        } elsif ($token->{tag_name} eq 'meta') {        } elsif ($token->{tag_name} eq 'meta') {
6894          ## NOTE: This is an "as if in head" code clone, only "-t" differs          ## NOTE: This is an "as if in head" code clone, only "-t" differs
6895          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
6896          my $meta_el = pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.          my $meta_el = pop @{$self->{open_elements}};
6897    
6898          unless ($self->{confident}) {          unless ($self->{confident}) {
6899            if ($token->{attributes}->{charset}) {            if ($token->{attributes}->{charset}) {
# Line 6507  sub _tree_construction_main ($) { Line 6910  sub _tree_construction_main ($) {
6910            } elsif ($token->{attributes}->{content}) {            } elsif ($token->{attributes}->{content}) {
6911              if ($token->{attributes}->{content}->{value}              if ($token->{attributes}->{content}->{value}
6912                  =~ /[Cc][Hh][Aa][Rr][Ss][Ee][Tt]                  =~ /[Cc][Hh][Aa][Rr][Ss][Ee][Tt]
6913                      [\x09-\x0D\x20]*=                      [\x09\x0A\x0C\x0D\x20]*=
6914                      [\x09-\x0D\x20]*(?>"([^"]*)"|'([^']*)'|                      [\x09\x0A\x0C\x0D\x20]*(?>"([^"]*)"|'([^']*)'|
6915                      ([^"'\x09-\x0D\x20][^\x09-\x0D\x20\x3B]*))/x) {                      ([^"'\x09\x0A\x0C\x0D\x20][^\x09\x0A\x0C\x0D\x20\x3B]*))
6916                       /x) {
6917                !!!cp ('t336');                !!!cp ('t336');
6918                ## NOTE: Whether the encoding is supported or not is handled                ## NOTE: Whether the encoding is supported or not is handled
6919                ## in the {change_encoding} callback.                ## in the {change_encoding} callback.
# Line 6568  sub _tree_construction_main ($) { Line 6972  sub _tree_construction_main ($) {
6972          !!!next-token;          !!!next-token;
6973          next B;          next B;
6974        } elsif ({        } elsif ({
6975                  address => 1, blockquote => 1, center => 1, dir => 1,                  ## NOTE: Start tags for non-phrasing flow content elements
6976                  div => 1, dl => 1, fieldset => 1,  
6977                  h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,                  ## NOTE: The normal one
6978                  menu => 1, ol => 1, p => 1, ul => 1,                  address => 1, article => 1, aside => 1, blockquote => 1,
6979                    center => 1, datagrid => 1, details => 1, dialog => 1,
6980                    dir => 1, div => 1, dl => 1, fieldset => 1, figure => 1,
6981                    footer => 1, h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1,
6982                    h6 => 1, header => 1, menu => 1, nav => 1, ol => 1, p => 1,
6983                    section => 1, ul => 1,
6984                    ## NOTE: As normal, but drops leading newline
6985                  pre => 1, listing => 1,                  pre => 1, listing => 1,
6986                    ## NOTE: As normal, but interacts with the form element pointer
6987                  form => 1,                  form => 1,
6988                    
6989                  table => 1,                  table => 1,
6990                  hr => 1,                  hr => 1,
6991                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
# Line 6640  sub _tree_construction_main ($) { Line 7052  sub _tree_construction_main ($) {
7052            !!!next-token;            !!!next-token;
7053          }          }
7054          next B;          next B;
7055        } elsif ({li => 1, dt => 1, dd => 1}->{$token->{tag_name}}) {        } elsif ($token->{tag_name} eq 'li') {
7056          ## has a p element in scope          ## NOTE: As normal, but imply </li> when there's another <li> ...
7057    
7058            ## NOTE: Special, Scope (<li><foo><li> == <li><foo><li/></foo></li>)
7059              ## Interpreted as <li><foo/></li><li/> (non-conforming)
7060              ## blockquote (O9.27), center (O), dd (Fx3, O, S3.1.2, IE7),
7061              ## dt (Fx, O, S, IE), dl (O), fieldset (O, S, IE), form (Fx, O, S),
7062              ## hn (O), pre (O), applet (O, S), button (O, S), marquee (Fx, O, S),
7063              ## object (Fx)
7064              ## Generate non-tree (non-conforming)
7065              ## basefont (IE7 (where basefont is non-void)), center (IE),
7066              ## form (IE), hn (IE)
7067            ## address, div, p (<li><foo><li> == <li><foo/></li><li/>)
7068              ## Interpreted as <li><foo><li/></foo></li> (non-conforming)
7069              ## div (Fx, S)
7070    
7071            my $non_optional;
7072            my $i = -1;
7073    
7074            ## 1.
7075            for my $node (reverse @{$self->{open_elements}}) {
7076              if ($node->[1] & LI_EL) {
7077                ## 2. (a) As if </li>
7078                {
7079                  ## If no </li> - not applied
7080                  #
7081    
7082                  ## Otherwise
7083    
7084                  ## 1. generate implied end tags, except for </li>
7085                  #
7086    
7087                  ## 2. If current node != "li", parse error
7088                  if ($non_optional) {
7089                    !!!parse-error (type => 'not closed',
7090                                    text => $non_optional->[0]->manakai_local_name,
7091                                    token => $token);
7092                    !!!cp ('t355');
7093                  } else {
7094                    !!!cp ('t356');
7095                  }
7096    
7097                  ## 3. Pop
7098                  splice @{$self->{open_elements}}, $i;
7099                }
7100    
7101                last; ## 2. (b) goto 5.
7102              } elsif (
7103                       ## NOTE: not "formatting" and not "phrasing"
7104                       ($node->[1] & SPECIAL_EL or
7105                        $node->[1] & SCOPING_EL) and
7106                       ## NOTE: "li", "dt", and "dd" are in |SPECIAL_EL|.
7107    
7108                       (not $node->[1] & ADDRESS_EL) &
7109                       (not $node->[1] & DIV_EL) &
7110                       (not $node->[1] & P_EL)) {
7111                ## 3.
7112                !!!cp ('t357');
7113                last; ## goto 5.
7114              } elsif ($node->[1] & END_TAG_OPTIONAL_EL) {
7115                !!!cp ('t358');
7116                #
7117              } else {
7118                !!!cp ('t359');
7119                $non_optional ||= $node;
7120                #
7121              }
7122              ## 4.
7123              ## goto 2.
7124              $i--;
7125            }
7126    
7127            ## 5. (a) has a |p| element in scope
7128          INSCOPE: for (reverse @{$self->{open_elements}}) {          INSCOPE: for (reverse @{$self->{open_elements}}) {
7129            if ($_->[1] & P_EL) {            if ($_->[1] & P_EL) {
7130              !!!cp ('t353');              !!!cp ('t353');
7131    
7132                ## NOTE: |<p><li>|, for example.
7133    
7134              !!!back-token; # <x>              !!!back-token; # <x>
7135              $token = {type => END_TAG_TOKEN, tag_name => 'p',              $token = {type => END_TAG_TOKEN, tag_name => 'p',
7136                        line => $token->{line}, column => $token->{column}};                        line => $token->{line}, column => $token->{column}};
# Line 6654  sub _tree_construction_main ($) { Line 7140  sub _tree_construction_main ($) {
7140              last INSCOPE;              last INSCOPE;
7141            }            }
7142          } # INSCOPE          } # INSCOPE
7143              
7144          ## Step 1          ## 5. (b) insert
7145            !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7146            !!!nack ('t359.1');
7147            !!!next-token;
7148            next B;
7149          } elsif ($token->{tag_name} eq 'dt' or
7150                   $token->{tag_name} eq 'dd') {
7151            ## NOTE: As normal, but imply </dt> or </dd> when ...
7152    
7153            my $non_optional;
7154          my $i = -1;          my $i = -1;
7155          my $node = $self->{open_elements}->[$i];  
7156          my $li_or_dtdd = {li => {li => 1},          ## 1.
7157                            dt => {dt => 1, dd => 1},          for my $node (reverse @{$self->{open_elements}}) {
7158                            dd => {dt => 1, dd => 1}}->{$token->{tag_name}};            if ($node->[1] & DT_EL or $node->[1] & DD_EL) {
7159          LI: {              ## 2. (a) As if </li>
7160            ## Step 2              {
7161            if ($li_or_dtdd->{$node->[0]->manakai_local_name}) {                ## If no </li> - not applied
7162              if ($i != -1) {                #
7163                !!!cp ('t355');  
7164                !!!parse-error (type => 'not closed',                ## Otherwise
7165                                text => $self->{open_elements}->[-1]->[0]  
7166                                    ->manakai_local_name,                ## 1. generate implied end tags, except for </dt> or </dd>
7167                                token => $token);                #
7168              } else {  
7169                !!!cp ('t356');                ## 2. If current node != "dt"|"dd", parse error
7170                  if ($non_optional) {
7171                    !!!parse-error (type => 'not closed',
7172                                    text => $non_optional->[0]->manakai_local_name,
7173                                    token => $token);
7174                    !!!cp ('t355.1');
7175                  } else {
7176                    !!!cp ('t356.1');
7177                  }
7178    
7179                  ## 3. Pop
7180                  splice @{$self->{open_elements}}, $i;
7181              }              }
7182              splice @{$self->{open_elements}}, $i;  
7183              last LI;              last; ## 2. (b) goto 5.
7184              } elsif (
7185                       ## NOTE: not "formatting" and not "phrasing"
7186                       ($node->[1] & SPECIAL_EL or
7187                        $node->[1] & SCOPING_EL) and
7188                       ## NOTE: "li", "dt", and "dd" are in |SPECIAL_EL|.
7189    
7190                       (not $node->[1] & ADDRESS_EL) &
7191                       (not $node->[1] & DIV_EL) &
7192                       (not $node->[1] & P_EL)) {
7193                ## 3.
7194                !!!cp ('t357.1');
7195                last; ## goto 5.
7196              } elsif ($node->[1] & END_TAG_OPTIONAL_EL) {
7197                !!!cp ('t358.1');
7198                #
7199            } else {            } else {
7200              !!!cp ('t357');              !!!cp ('t359.1');
7201            }              $non_optional ||= $node;
7202                          #
           ## Step 3  
           if (not ($node->[1] & FORMATTING_EL) and  
               #not $phrasing_category->{$node->[1]} and  
               ($node->[1] & SPECIAL_EL or  
                $node->[1] & SCOPING_EL) and  
               not ($node->[1] & ADDRESS_EL) and  
               not ($node->[1] & DIV_EL)) {  
             !!!cp ('t358');  
             last LI;  
7203            }            }
7204                        ## 4.
7205            !!!cp ('t359');            ## goto 2.
           ## Step 4  
7206            $i--;            $i--;
7207            $node = $self->{open_elements}->[$i];          }
7208            redo LI;  
7209          } # LI          ## 5. (a) has a |p| element in scope
7210                      INSCOPE: for (reverse @{$self->{open_elements}}) {
7211              if ($_->[1] & P_EL) {
7212                !!!cp ('t353.1');
7213                !!!back-token; # <x>
7214                $token = {type => END_TAG_TOKEN, tag_name => 'p',
7215                          line => $token->{line}, column => $token->{column}};
7216                next B;
7217              } elsif ($_->[1] & SCOPING_EL) {
7218                !!!cp ('t354.1');
7219                last INSCOPE;
7220              }
7221            } # INSCOPE
7222    
7223            ## 5. (b) insert
7224          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7225          !!!nack ('t359.1');          !!!nack ('t359.2');
7226          !!!next-token;          !!!next-token;
7227          next B;          next B;
7228        } elsif ($token->{tag_name} eq 'plaintext') {        } elsif ($token->{tag_name} eq 'plaintext') {
7229            ## NOTE: As normal, but effectively ends parsing
7230    
7231          ## has a p element in scope          ## has a p element in scope
7232          INSCOPE: for (reverse @{$self->{open_elements}}) {          INSCOPE: for (reverse @{$self->{open_elements}}) {
7233            if ($_->[1] & P_EL) {            if ($_->[1] & P_EL) {
# Line 6893  sub _tree_construction_main ($) { Line 7419  sub _tree_construction_main ($) {
7419            next B;            next B;
7420          }          }
7421        } elsif ($token->{tag_name} eq 'textarea') {        } elsif ($token->{tag_name} eq 'textarea') {
7422          my $tag_name = $token->{tag_name};          ## Step 1
7423          my $el;          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
         !!!create-element ($el, $HTML_NS, $token->{tag_name}, $token->{attributes}, $token);  
7424                    
7425            ## Step 2
7426          ## TODO: $self->{form_element} if defined          ## TODO: $self->{form_element} if defined
7427    
7428            ## Step 3
7429            $self->{ignore_newline} = 1;
7430    
7431            ## Step 4
7432            ## ISSUE: This step is wrong. (r2302 enbugged)
7433    
7434            ## Step 5
7435          $self->{content_model} = RCDATA_CONTENT_MODEL;          $self->{content_model} = RCDATA_CONTENT_MODEL;
7436          delete $self->{escape}; # MUST          delete $self->{escape}; # MUST
7437            
7438          $insert->($el);          ## Step 6-7
7439                    $self->{insertion_mode} |= IN_CDATA_RCDATA_IM;
7440          my $text = '';  
7441          !!!nack ('t392.1');          !!!nack ('t392.1');
7442          !!!next-token;          !!!next-token;
7443          if ($token->{type} == CHARACTER_TOKEN) {          next B;
7444            $token->{data} =~ s/^\x0A//;        } elsif ($token->{tag_name} eq 'optgroup' or
7445            unless (length $token->{data}) {                 $token->{tag_name} eq 'option') {
7446              !!!cp ('t392');          ## has an |option| element in scope
7447              !!!next-token;          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7448            } else {            my $node = $self->{open_elements}->[$_];
7449              !!!cp ('t393');            if ($node->[1] & OPTION_EL) {
7450                !!!cp ('t397.1');
7451                ## NOTE: As if </option>
7452                !!!back-token; # <option> or <optgroup>
7453                $token = {type => END_TAG_TOKEN, tag_name => 'option',
7454                          line => $token->{line}, column => $token->{column}};
7455                next B;
7456              } elsif ($node->[1] & SCOPING_EL) {
7457                !!!cp ('t397.2');
7458                last INSCOPE;
7459            }            }
7460          } else {          } # INSCOPE
7461            !!!cp ('t394');  
7462          }          $reconstruct_active_formatting_elements->($insert_to_current);
7463          while ($token->{type} == CHARACTER_TOKEN) {  
7464            !!!cp ('t395');          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7465            $text .= $token->{data};  
7466            !!!next-token;          !!!nack ('t397.3');
         }  
         if (length $text) {  
           !!!cp ('t396');  
           $el->manakai_append_text ($text);  
         }  
           
         $self->{content_model} = PCDATA_CONTENT_MODEL;  
           
         if ($token->{type} == END_TAG_TOKEN and  
             $token->{tag_name} eq $tag_name) {  
           !!!cp ('t397');  
           ## Ignore the token  
         } else {  
           !!!cp ('t398');  
           !!!parse-error (type => 'in RCDATA:#eof', token => $token);  
         }  
7467          !!!next-token;          !!!next-token;
7468          next B;          redo B;
7469        } elsif ($token->{tag_name} eq 'rt' or        } elsif ($token->{tag_name} eq 'rt' or
7470                 $token->{tag_name} eq 'rp') {                 $token->{tag_name} eq 'rp') {
7471          ## has a |ruby| element in scope          ## has a |ruby| element in scope
# Line 6986  sub _tree_construction_main ($) { Line 7513  sub _tree_construction_main ($) {
7513                    
7514          if ($self->{self_closing}) {          if ($self->{self_closing}) {
7515            pop @{$self->{open_elements}};            pop @{$self->{open_elements}};
7516            !!!ack ('t398.1');            !!!ack ('t398.6');
7517          } else {          } else {
7518            !!!cp ('t398.2');            !!!cp ('t398.7');
7519            $self->{insertion_mode} |= IN_FOREIGN_CONTENT_IM;            $self->{insertion_mode} |= IN_FOREIGN_CONTENT_IM;
7520            ## NOTE: |<body><math><mi><svg>| -> "in foreign content" insertion            ## NOTE: |<body><math><mi><svg>| -> "in foreign content" insertion
7521            ## mode, "in body" (not "in foreign content") secondary insertion            ## mode, "in body" (not "in foreign content") secondary insertion
# Line 6999  sub _tree_construction_main ($) { Line 7526  sub _tree_construction_main ($) {
7526          next B;          next B;
7527        } elsif ({        } elsif ({
7528                  caption => 1, col => 1, colgroup => 1, frame => 1,                  caption => 1, col => 1, colgroup => 1, frame => 1,
7529                  frameset => 1, head => 1, option => 1, optgroup => 1,                  frameset => 1, head => 1,
7530                  tbody => 1, td => 1, tfoot => 1, th => 1,                  tbody => 1, td => 1, tfoot => 1, th => 1,
7531                  thead => 1, tr => 1,                  thead => 1, tr => 1,
7532                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
# Line 7010  sub _tree_construction_main ($) { Line 7537  sub _tree_construction_main ($) {
7537          !!!nack ('t401.1'); ## NOTE: |<col/>| or |<frame/>| here is an error.          !!!nack ('t401.1'); ## NOTE: |<col/>| or |<frame/>| here is an error.
7538          !!!next-token;          !!!next-token;
7539          next B;          next B;
7540                  } elsif ($token->{tag_name} eq 'param' or
7541          ## ISSUE: An issue on HTML5 new elements in the spec.                 $token->{tag_name} eq 'source') {
7542            !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7543            pop @{$self->{open_elements}};
7544    
7545            !!!ack ('t398.5');
7546            !!!next-token;
7547            redo B;
7548        } else {        } else {
7549          if ($token->{tag_name} eq 'image') {          if ($token->{tag_name} eq 'image') {
7550            !!!cp ('t384');            !!!cp ('t384');
# Line 7034  sub _tree_construction_main ($) { Line 7567  sub _tree_construction_main ($) {
7567            !!!nack ('t380.1');            !!!nack ('t380.1');
7568          } elsif ({          } elsif ({
7569                    b => 1, big => 1, em => 1, font => 1, i => 1,                    b => 1, big => 1, em => 1, font => 1, i => 1,
7570                    s => 1, small => 1, strile => 1,                    s => 1, small => 1, strike => 1,
7571                    strong => 1, tt => 1, u => 1,                    strong => 1, tt => 1, u => 1,
7572                   }->{$token->{tag_name}}) {                   }->{$token->{tag_name}}) {
7573            !!!cp ('t375');            !!!cp ('t375');
# Line 7047  sub _tree_construction_main ($) { Line 7580  sub _tree_construction_main ($) {
7580            !!!ack ('t388.2');            !!!ack ('t388.2');
7581          } elsif ({          } elsif ({
7582                    area => 1, basefont => 1, bgsound => 1, br => 1,                    area => 1, basefont => 1, bgsound => 1, br => 1,
7583                    embed => 1, img => 1, param => 1, spacer => 1, wbr => 1,                    embed => 1, img => 1, spacer => 1, wbr => 1,
                   #image => 1,  
7584                   }->{$token->{tag_name}}) {                   }->{$token->{tag_name}}) {
7585            !!!cp ('t388.1');            !!!cp ('t388.1');
7586            pop @{$self->{open_elements}};            pop @{$self->{open_elements}};
# Line 7089  sub _tree_construction_main ($) { Line 7621  sub _tree_construction_main ($) {
7621              }              }
7622            }            }
7623    
7624            !!!parse-error (type => 'start tag not allowed',            ## NOTE: |<marquee></body>|, |<svg><foreignobject></body>|
7625    
7626              !!!parse-error (type => 'unmatched end tag',
7627                            text => $token->{tag_name}, token => $token);                            text => $token->{tag_name}, token => $token);
7628            ## NOTE: Ignore the token.            ## NOTE: Ignore the token.
7629            !!!next-token;            !!!next-token;
# Line 7116  sub _tree_construction_main ($) { Line 7650  sub _tree_construction_main ($) {
7650          ## up-to-date, though it has same effect as speced.          ## up-to-date, though it has same effect as speced.
7651          if (@{$self->{open_elements}} > 1 and          if (@{$self->{open_elements}} > 1 and
7652              $self->{open_elements}->[1]->[1] & BODY_EL) {              $self->{open_elements}->[1]->[1] & BODY_EL) {
           ## ISSUE: There is an issue in the spec.  
7653            unless ($self->{open_elements}->[-1]->[1] & BODY_EL) {            unless ($self->{open_elements}->[-1]->[1] & BODY_EL) {
7654              !!!cp ('t406');              !!!cp ('t406');
7655              !!!parse-error (type => 'not closed',              !!!parse-error (type => 'not closed',
# Line 7138  sub _tree_construction_main ($) { Line 7671  sub _tree_construction_main ($) {
7671            next B;            next B;
7672          }          }
7673        } elsif ({        } elsif ({
7674                  address => 1, blockquote => 1, center => 1, dir => 1,                  ## NOTE: End tags for non-phrasing flow content elements
7675                  div => 1, dl => 1, fieldset => 1, listing => 1,  
7676                  menu => 1, ol => 1, pre => 1, ul => 1,                  ## NOTE: The normal ones
7677                    address => 1, article => 1, aside => 1, blockquote => 1,
7678                    center => 1, datagrid => 1, details => 1, dialog => 1,
7679                    dir => 1, div => 1, dl => 1, fieldset => 1, figure => 1,
7680                    footer => 1, header => 1, listing => 1, menu => 1, nav => 1,
7681                    ol => 1, pre => 1, section => 1, ul => 1,
7682    
7683                    ## NOTE: As normal, but ... optional tags
7684                  dd => 1, dt => 1, li => 1,                  dd => 1, dt => 1, li => 1,
7685    
7686                  applet => 1, button => 1, marquee => 1, object => 1,                  applet => 1, button => 1, marquee => 1, object => 1,
7687                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
7688            ## NOTE: Code for <li> start tags includes "as if </li>" code.
7689            ## Code for <dt> or <dd> start tags includes "as if </dt> or
7690            ## </dd>" code.
7691    
7692          ## has an element in scope          ## has an element in scope
7693          my $i;          my $i;
7694          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
# Line 7170  sub _tree_construction_main ($) { Line 7715  sub _tree_construction_main ($) {
7715                    dd => ($token->{tag_name} ne 'dd'),                    dd => ($token->{tag_name} ne 'dd'),
7716                    dt => ($token->{tag_name} ne 'dt'),                    dt => ($token->{tag_name} ne 'dt'),
7717                    li => ($token->{tag_name} ne 'li'),                    li => ($token->{tag_name} ne 'li'),
7718                      option => 1,
7719                      optgroup => 1,
7720                    p => 1,                    p => 1,
7721                    rt => 1,                    rt => 1,
7722                    rp => 1,                    rp => 1,
# Line 7202  sub _tree_construction_main ($) { Line 7749  sub _tree_construction_main ($) {
7749          !!!next-token;          !!!next-token;
7750          next B;          next B;
7751        } elsif ($token->{tag_name} eq 'form') {        } elsif ($token->{tag_name} eq 'form') {
7752            ## NOTE: As normal, but interacts with the form element pointer
7753    
7754          undef $self->{form_element};          undef $self->{form_element};
7755    
7756          ## has an element in scope          ## has an element in scope
# Line 7249  sub _tree_construction_main ($) { Line 7798  sub _tree_construction_main ($) {
7798          !!!next-token;          !!!next-token;
7799          next B;          next B;
7800        } elsif ({        } elsif ({
7801                    ## NOTE: As normal, except acts as a closer for any ...
7802                  h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,                  h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,
7803                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
7804          ## has an element in scope          ## has an element in scope
# Line 7294  sub _tree_construction_main ($) { Line 7844  sub _tree_construction_main ($) {
7844          !!!next-token;          !!!next-token;
7845          next B;          next B;
7846        } elsif ($token->{tag_name} eq 'p') {        } elsif ($token->{tag_name} eq 'p') {
7847            ## NOTE: As normal, except </p> implies <p> and ...
7848    
7849          ## has an element in scope          ## has an element in scope
7850            my $non_optional;
7851          my $i;          my $i;
7852          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7853            my $node = $self->{open_elements}->[$_];            my $node = $self->{open_elements}->[$_];
# Line 7305  sub _tree_construction_main ($) { Line 7858  sub _tree_construction_main ($) {
7858            } elsif ($node->[1] & SCOPING_EL) {            } elsif ($node->[1] & SCOPING_EL) {
7859              !!!cp ('t411.1');              !!!cp ('t411.1');
7860              last INSCOPE;              last INSCOPE;
7861              } elsif ($node->[1] & END_TAG_OPTIONAL_EL) {
7862                ## NOTE: |END_TAG_OPTIONAL_EL| includes "p"
7863                !!!cp ('t411.2');
7864                #
7865              } else {
7866                !!!cp ('t411.3');
7867                $non_optional ||= $node;
7868                #
7869            }            }
7870          } # INSCOPE          } # INSCOPE
7871    
7872          if (defined $i) {          if (defined $i) {
7873            if ($self->{open_elements}->[-1]->[0]->manakai_local_name            ## 1. Generate implied end tags
7874                    ne $token->{tag_name}) {            #
7875    
7876              ## 2. If current node != "p", parse error
7877              if ($non_optional) {
7878              !!!cp ('t412.1');              !!!cp ('t412.1');
7879              !!!parse-error (type => 'not closed',              !!!parse-error (type => 'not closed',
7880                              text => $self->{open_elements}->[-1]->[0]                              text => $non_optional->[0]->manakai_local_name,
                                 ->manakai_local_name,  
7881                              token => $token);                              token => $token);
7882            } else {            } else {
7883              !!!cp ('t414.1');              !!!cp ('t414.1');
7884            }            }
7885    
7886              ## 3. Pop
7887            splice @{$self->{open_elements}}, $i;            splice @{$self->{open_elements}}, $i;
7888          } else {          } else {
7889            !!!cp ('t413.1');            !!!cp ('t413.1');
# Line 7339  sub _tree_construction_main ($) { Line 7903  sub _tree_construction_main ($) {
7903        } elsif ({        } elsif ({
7904                  a => 1,                  a => 1,
7905                  b => 1, big => 1, em => 1, font => 1, i => 1,                  b => 1, big => 1, em => 1, font => 1, i => 1,
7906                  nobr => 1, s => 1, small => 1, strile => 1,                  nobr => 1, s => 1, small => 1, strike => 1,
7907                  strong => 1, tt => 1, u => 1,                  strong => 1, tt => 1, u => 1,
7908                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
7909          !!!cp ('t427');          !!!cp ('t427');
# Line 7360  sub _tree_construction_main ($) { Line 7924  sub _tree_construction_main ($) {
7924          ## Ignore the token.          ## Ignore the token.
7925          !!!next-token;          !!!next-token;
7926          next B;          next B;
       } elsif ({  
                 caption => 1, col => 1, colgroup => 1, frame => 1,  
                 frameset => 1, head => 1, option => 1, optgroup => 1,  
                 tbody => 1, td => 1, tfoot => 1, th => 1,  
                 thead => 1, tr => 1,  
                 area => 1, basefont => 1, bgsound => 1,  
                 embed => 1, hr => 1, iframe => 1, image => 1,  
                 img => 1, input => 1, isindex => 1, noembed => 1,  
                 noframes => 1, param => 1, select => 1, spacer => 1,  
                 table => 1, textarea => 1, wbr => 1,  
                 noscript => 0, ## TODO: if scripting is enabled  
                }->{$token->{tag_name}}) {  
         !!!cp ('t429');  
         !!!parse-error (type => 'unmatched end tag',  
                         text => $token->{tag_name}, token => $token);  
         ## Ignore the token  
         !!!next-token;  
         next B;  
           
         ## ISSUE: Issue on HTML5 new elements in spec  
           
7927        } else {        } else {
7928            if ($token->{tag_name} eq 'sarcasm') {
7929              sleep 0.001; # take a deep breath
7930            }
7931    
7932          ## Step 1          ## Step 1
7933          my $node_i = -1;          my $node_i = -1;
7934          my $node = $self->{open_elements}->[$node_i];          my $node = $self->{open_elements}->[$node_i];
7935    
7936          ## Step 2          ## Step 2
7937          S2: {          S2: {
7938            if ($node->[0]->manakai_local_name eq $token->{tag_name}) {            my $node_tag_name = $node->[0]->manakai_local_name;
7939              $node_tag_name =~ tr/A-Z/a-z/; # for SVG camelCase tag names
7940              if ($node_tag_name eq $token->{tag_name}) {
7941              ## Step 1              ## Step 1
7942              ## generate implied end tags              ## generate implied end tags
7943              while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {              while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
# Line 7401  sub _tree_construction_main ($) { Line 7950  sub _tree_construction_main ($) {
7950              }              }
7951                    
7952              ## Step 2              ## Step 2
7953              if ($self->{open_elements}->[-1]->[0]->manakai_local_name              my $current_tag_name
7954                      ne $token->{tag_name}) {                  = $self->{open_elements}->[-1]->[0]->manakai_local_name;
7955                $current_tag_name =~ tr/A-Z/a-z/;
7956                if ($current_tag_name ne $token->{tag_name}) {
7957                !!!cp ('t431');                !!!cp ('t431');
7958                ## NOTE: <x><y></x>                ## NOTE: <x><y></x>
7959                !!!parse-error (type => 'not closed',                !!!parse-error (type => 'not closed',
# Line 7430  sub _tree_construction_main ($) { Line 7981  sub _tree_construction_main ($) {
7981                ## Ignore the token                ## Ignore the token
7982                !!!next-token;                !!!next-token;
7983                last S2;                last S2;
             }  
7984    
7985                  ## NOTE: |<span><dd></span>a|: In Safari 3.1.2 and Opera
7986                  ## 9.27, "a" is a child of <dd> (conforming).  In
7987                  ## Firefox 3.0.2, "a" is a child of <body>.  In WinIE 7,
7988                  ## "a" is a child of both <body> and <dd>.
7989                }
7990                
7991              !!!cp ('t434');              !!!cp ('t434');
7992            }            }
7993                        
# Line 7472  sub _tree_construction_main ($) { Line 8028  sub _tree_construction_main ($) {
8028    ## TODO: script stuffs    ## TODO: script stuffs
8029  } # _tree_construct_main  } # _tree_construct_main
8030    
8031  sub set_inner_html ($$$;$) {  sub set_inner_html ($$$$;$) {
8032    my $class = shift;    my $class = shift;
8033    my $node = shift;    my $node = shift;
8034    my $s = \$_[0];    #my $s = \$_[0];
8035    my $onerror = $_[1];    my $onerror = $_[1];
8036    my $get_wrapper = $_[2] || sub ($) { return $_[0] };    my $get_wrapper = $_[2] || sub ($) { return $_[0] };
8037    
# Line 7496  sub set_inner_html ($$$;$) { Line 8052  sub set_inner_html ($$$;$) {
8052      }      }
8053    
8054      ## Step 3, 4, 5 # MUST      ## Step 3, 4, 5 # MUST
8055      $class->parse_char_string ($$s => $node, $onerror, $get_wrapper);      $class->parse_char_string ($_[0] => $node, $onerror, $get_wrapper);
8056    } elsif ($nt == 1) {    } elsif ($nt == 1) {
8057      ## TODO: If non-html element      ## TODO: If non-html element
8058    
# Line 7515  sub set_inner_html ($$$;$) { Line 8071  sub set_inner_html ($$$;$) {
8071      my $i = 0;      my $i = 0;
8072      $p->{line_prev} = $p->{line} = 1;      $p->{line_prev} = $p->{line} = 1;
8073      $p->{column_prev} = $p->{column} = 0;      $p->{column_prev} = $p->{column} = 0;
8074      $p->{set_next_char} = sub {      require Whatpm::Charset::DecodeHandle;
8075        my $input = Whatpm::Charset::DecodeHandle::CharString->new (\($_[0]));
8076        $input = $get_wrapper->($input);
8077        $p->{set_nc} = sub {
8078        my $self = shift;        my $self = shift;
8079    
8080        pop @{$self->{prev_char}};        my $char = '';
8081        unshift @{$self->{prev_char}}, $self->{next_char};        if (defined $self->{next_nc}) {
8082            $char = $self->{next_nc};
8083        $self->{next_char} = -1 and return if $i >= length $$s;          delete $self->{next_nc};
8084        $self->{next_char} = ord substr $$s, $i++, 1;          $self->{nc} = ord $char;
8085          } else {
8086            $self->{char_buffer} = '';
8087            $self->{char_buffer_pos} = 0;
8088            
8089            my $count = $input->manakai_read_until
8090                ($self->{char_buffer}, qr/[^\x00\x0A\x0D]/,
8091                 $self->{char_buffer_pos});
8092            if ($count) {
8093              $self->{line_prev} = $self->{line};
8094              $self->{column_prev} = $self->{column};
8095              $self->{column}++;
8096              $self->{nc}
8097                  = ord substr ($self->{char_buffer},
8098                                $self->{char_buffer_pos}++, 1);
8099              return;
8100            }
8101            
8102            if ($input->read ($char, 1)) {
8103              $self->{nc} = ord $char;
8104            } else {
8105              $self->{nc} = -1;
8106              return;
8107            }
8108          }
8109    
8110        ($p->{line_prev}, $p->{column_prev}) = ($p->{line}, $p->{column});        ($p->{line_prev}, $p->{column_prev}) = ($p->{line}, $p->{column});
8111        $p->{column}++;        $p->{column}++;
8112    
8113        if ($self->{next_char} == 0x000A) { # LF        if ($self->{nc} == 0x000A) { # LF
8114          $p->{line}++;          $p->{line}++;
8115          $p->{column} = 0;          $p->{column} = 0;
8116          !!!cp ('i1');          !!!cp ('i1');
8117        } elsif ($self->{next_char} == 0x000D) { # CR        } elsif ($self->{nc} == 0x000D) { # CR
8118          $i++ if substr ($$s, $i, 1) eq "\x0A";  ## TODO: support for abort/streaming
8119          $self->{next_char} = 0x000A; # LF # MUST          my $next = '';
8120            if ($input->read ($next, 1) and $next ne "\x0A") {
8121              $self->{next_nc} = $next;
8122            }
8123            $self->{nc} = 0x000A; # LF # MUST
8124          $p->{line}++;          $p->{line}++;
8125          $p->{column} = 0;          $p->{column} = 0;
8126          !!!cp ('i2');          !!!cp ('i2');
8127        } elsif ($self->{next_char} > 0x10FFFF) {        } elsif ($self->{nc} == 0x0000) { # NULL
         $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST  
         !!!cp ('i3');  
       } elsif ($self->{next_char} == 0x0000) { # NULL  
8128          !!!cp ('i4');          !!!cp ('i4');
8129          !!!parse-error (type => 'NULL');          !!!parse-error (type => 'NULL');
8130          $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST          $self->{nc} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
       } elsif ($self->{next_char} <= 0x0008 or  
                (0x000E <= $self->{next_char} and  
                 $self->{next_char} <= 0x001F) or  
                (0x007F <= $self->{next_char} and  
                 $self->{next_char} <= 0x009F) or  
                (0xD800 <= $self->{next_char} and  
                 $self->{next_char} <= 0xDFFF) or  
                (0xFDD0 <= $self->{next_char} and  
                 $self->{next_char} <= 0xFDDF) or  
                {  
                 0xFFFE => 1, 0xFFFF => 1, 0x1FFFE => 1, 0x1FFFF => 1,  
                 0x2FFFE => 1, 0x2FFFF => 1, 0x3FFFE => 1, 0x3FFFF => 1,  
                 0x4FFFE => 1, 0x4FFFF => 1, 0x5FFFE => 1, 0x5FFFF => 1,  
                 0x6FFFE => 1, 0x6FFFF => 1, 0x7FFFE => 1, 0x7FFFF => 1,  
                 0x8FFFE => 1, 0x8FFFF => 1, 0x9FFFE => 1, 0x9FFFF => 1,  
                 0xAFFFE => 1, 0xAFFFF => 1, 0xBFFFE => 1, 0xBFFFF => 1,  
                 0xCFFFE => 1, 0xCFFFF => 1, 0xDFFFE => 1, 0xDFFFF => 1,  
                 0xEFFFE => 1, 0xEFFFF => 1, 0xFFFFE => 1, 0xFFFFF => 1,  
                 0x10FFFE => 1, 0x10FFFF => 1,  
                }->{$self->{next_char}}) {  
         !!!cp ('i4.1');  
         if ($self->{next_char} < 0x10000) {  
           !!!parse-error (type => 'control char',  
                           text => (sprintf 'U+%04X', $self->{next_char}));  
         } else {  
           !!!parse-error (type => 'control char',  
                           text => (sprintf 'U-%08X', $self->{next_char}));  
         }  
8131        }        }
8132      };      };
8133      $p->{prev_char} = [-1, -1, -1];  
8134      $p->{next_char} = -1;      $p->{read_until} = sub {
8135              #my ($scalar, $specials_range, $offset) = @_;
8136          return 0 if defined $p->{next_nc};
8137    
8138          my $pattern = qr/[^$_[1]\x00\x0A\x0D]/;
8139          my $offset = $_[2] || 0;
8140          
8141          if ($p->{char_buffer_pos} < length $p->{char_buffer}) {
8142            pos ($p->{char_buffer}) = $p->{char_buffer_pos};
8143            if ($p->{char_buffer} =~ /\G(?>$pattern)+/) {
8144              substr ($_[0], $offset)
8145                  = substr ($p->{char_buffer}, $-[0], $+[0] - $-[0]);
8146              my $count = $+[0] - $-[0];
8147              if ($count) {
8148                $p->{column} += $count;
8149                $p->{char_buffer_pos} += $count;
8150                $p->{line_prev} = $p->{line};
8151                $p->{column_prev} = $p->{column} - 1;
8152                $p->{nc} = -1;
8153              }
8154              return $count;
8155            } else {
8156              return 0;
8157            }
8158          } else {
8159            my $count = $input->manakai_read_until ($_[0], $pattern, $_[2]);
8160            if ($count) {
8161              $p->{column} += $count;
8162              $p->{column_prev} += $count;
8163              $p->{nc} = -1;
8164            }
8165            return $count;
8166          }
8167        }; # $p->{read_until}
8168    
8169      my $ponerror = $onerror || sub {      my $ponerror = $onerror || sub {
8170        my (%opt) = @_;        my (%opt) = @_;
8171        my $line = $opt{line};        my $line = $opt{line};
# Line 7591  sub set_inner_html ($$$;$) { Line 8180  sub set_inner_html ($$$;$) {
8180        $ponerror->(line => $p->{line}, column => $p->{column}, @_);        $ponerror->(line => $p->{line}, column => $p->{column}, @_);
8181      };      };
8182            
8183        my $char_onerror = sub {
8184          my (undef, $type, %opt) = @_;
8185          $ponerror->(layer => 'encode',
8186                      line => $p->{line}, column => $p->{column} + 1,
8187                      %opt, type => $type);
8188        }; # $char_onerror
8189        $input->onerror ($char_onerror);
8190    
8191      $p->_initialize_tokenizer;      $p->_initialize_tokenizer;
8192      $p->_initialize_tree_constructor;      $p->_initialize_tree_constructor;
8193    
# Line 7626  sub set_inner_html ($$$;$) { Line 8223  sub set_inner_html ($$$;$) {
8223      push @{$p->{open_elements}}, [$root, $el_category->{html}];      push @{$p->{open_elements}}, [$root, $el_category->{html}];
8224    
8225      undef $p->{head_element};      undef $p->{head_element};
8226        undef $p->{head_element_inserted};
8227    
8228      ## Step 6 # MUST      ## Step 6 # MUST
8229      $p->_reset_insertion_mode;      $p->_reset_insertion_mode;

Legend:
Removed from v.1.162  
changed lines
  Added in v.1.205

[email protected]
ViewVC Help
Powered by ViewVC 1.1.24