/[suikacvs]/markup/html/whatpm/Whatpm/HTML.pm.src
Suika

Diff of /markup/html/whatpm/Whatpm/HTML.pm.src

Parent Directory Parent Directory | Revision Log Revision Log | View Patch Patch

revision 1.141 by wakaba, Sat May 24 10:18:26 2008 UTC revision 1.192 by wakaba, Thu Oct 2 10:59:04 2008 UTC
# Line 3  use strict; Line 3  use strict;
3  our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r};  our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r};
4  use Error qw(:try);  use Error qw(:try);
5    
6    ## NOTE: This module don't check all HTML5 parse errors; character
7    ## encoding related parse errors are expected to be handled by relevant
8    ## modules.
9    ## Parse errors for control characters that are not allowed in HTML5
10    ## documents, for surrogate code points, and for noncharacter code
11    ## points, as well as U+FFFD substitions for characters whose code points
12    ## is higher than U+10FFFF may be detected by combining the parser with
13    ## the checker implemented by Whatpm::Charset::UnicodeChecker (for its
14    ## usage example, see |t/HTML-tree.t| in the Whatpm package or the
15    ## WebHACC::Language::HTML module in the WebHACC package).
16    
17  ## ISSUE:  ## ISSUE:
18  ## var doc = implementation.createDocument (null, null, null);  ## var doc = implementation.createDocument (null, null, null);
19  ## doc.write ('');  ## doc.write ('');
20  ## alert (doc.compatMode);  ## alert (doc.compatMode);
21    
 ## TODO: 1252 parse error (revision 1264)  
 ## TODO: 8859-11 = 874 (revision 1271)  
   
22  require IO::Handle;  require IO::Handle;
23    
24  my $HTML_NS = q<http://www.w3.org/1999/xhtml>;  my $HTML_NS = q<http://www.w3.org/1999/xhtml>;
# Line 48  sub MISC_SPECIAL_EL () { 0b1000000000000 Line 56  sub MISC_SPECIAL_EL () { 0b1000000000000
56  sub FOREIGN_EL () { 0b10000000000000000000000000 }  sub FOREIGN_EL () { 0b10000000000000000000000000 }
57  sub FOREIGN_FLOW_CONTENT_EL () { 0b100000000000000000000000000 }  sub FOREIGN_FLOW_CONTENT_EL () { 0b100000000000000000000000000 }
58  sub MML_AXML_EL () { 0b1000000000000000000000000000 }  sub MML_AXML_EL () { 0b1000000000000000000000000000 }
59    sub RUBY_EL () { 0b10000000000000000000000000000 }
60    sub RUBY_COMPONENT_EL () { 0b100000000000000000000000000000 }
61    
62  sub TABLE_ROWS_EL () {  sub TABLE_ROWS_EL () {
63    TABLE_EL |    TABLE_EL |
# Line 55  sub TABLE_ROWS_EL () { Line 65  sub TABLE_ROWS_EL () {
65    TABLE_ROW_GROUP_EL    TABLE_ROW_GROUP_EL
66  }  }
67    
68    ## NOTE: Used in "generate implied end tags" algorithm.
69    ## NOTE: There is a code where a modified version of END_TAG_OPTIONAL_EL
70    ## is used in "generate implied end tags" implementation (search for the
71    ## function mae).
72  sub END_TAG_OPTIONAL_EL () {  sub END_TAG_OPTIONAL_EL () {
73    DD_EL |    DD_EL |
74    DT_EL |    DT_EL |
75    LI_EL |    LI_EL |
76    P_EL    P_EL |
77      RUBY_COMPONENT_EL
78  }  }
79    
80    ## NOTE: Used in </body> and EOF algorithms.
81  sub ALL_END_TAG_OPTIONAL_EL () {  sub ALL_END_TAG_OPTIONAL_EL () {
82    END_TAG_OPTIONAL_EL |    DD_EL |
83      DT_EL |
84      LI_EL |
85      P_EL |
86    
87    BODY_EL |    BODY_EL |
88    HTML_EL |    HTML_EL |
89    TABLE_CELL_EL |    TABLE_CELL_EL |
# Line 99  sub SPECIAL_EL () { Line 119  sub SPECIAL_EL () {
119    ADDRESS_EL |    ADDRESS_EL |
120    BODY_EL |    BODY_EL |
121    DIV_EL |    DIV_EL |
122    END_TAG_OPTIONAL_EL |  
123      DD_EL |
124      DT_EL |
125      LI_EL |
126      P_EL |
127    
128    FORM_EL |    FORM_EL |
129    FRAMESET_EL |    FRAMESET_EL |
130    HEADING_EL |    HEADING_EL |
# Line 173  my $el_category = { Line 198  my $el_category = {
198    param => MISC_SPECIAL_EL,    param => MISC_SPECIAL_EL,
199    plaintext => MISC_SPECIAL_EL,    plaintext => MISC_SPECIAL_EL,
200    pre => MISC_SPECIAL_EL,    pre => MISC_SPECIAL_EL,
201      rp => RUBY_COMPONENT_EL,
202      rt => RUBY_COMPONENT_EL,
203      ruby => RUBY_EL,
204    s => FORMATTING_EL,    s => FORMATTING_EL,
205    script => MISC_SPECIAL_EL,    script => MISC_SPECIAL_EL,
206    select => SELECT_EL,    select => SELECT_EL,
# Line 214  my $el_category_f = { Line 242  my $el_category_f = {
242  };  };
243    
244  my $svg_attr_name = {  my $svg_attr_name = {
245      attributename => 'attributeName',
246    attributetype => 'attributeType',    attributetype => 'attributeType',
247    basefrequency => 'baseFrequency',    basefrequency => 'baseFrequency',
248    baseprofile => 'baseProfile',    baseprofile => 'baseProfile',
# Line 224  my $svg_attr_name = { Line 253  my $svg_attr_name = {
253    diffuseconstant => 'diffuseConstant',    diffuseconstant => 'diffuseConstant',
254    edgemode => 'edgeMode',    edgemode => 'edgeMode',
255    externalresourcesrequired => 'externalResourcesRequired',    externalresourcesrequired => 'externalResourcesRequired',
   fecolormatrix => 'feColorMatrix',  
   fecomposite => 'feComposite',  
   fegaussianblur => 'feGaussianBlur',  
   femorphology => 'feMorphology',  
   fetile => 'feTile',  
256    filterres => 'filterRes',    filterres => 'filterRes',
257    filterunits => 'filterUnits',    filterunits => 'filterUnits',
258    glyphref => 'glyphRef',    glyphref => 'glyphRef',
# Line 262  my $svg_attr_name = { Line 286  my $svg_attr_name = {
286    repeatcount => 'repeatCount',    repeatcount => 'repeatCount',
287    repeatdur => 'repeatDur',    repeatdur => 'repeatDur',
288    requiredextensions => 'requiredExtensions',    requiredextensions => 'requiredExtensions',
289      requiredfeatures => 'requiredFeatures',
290    specularconstant => 'specularConstant',    specularconstant => 'specularConstant',
291    specularexponent => 'specularExponent',    specularexponent => 'specularExponent',
292    spreadmethod => 'spreadMethod',    spreadmethod => 'spreadMethod',
# Line 298  my $foreign_attr_xname = { Line 323  my $foreign_attr_xname = {
323    
324  ## ISSUE: xmlns:xlink="non-xlink-ns" is not an error.  ## ISSUE: xmlns:xlink="non-xlink-ns" is not an error.
325    
326  my $c1_entity_char = {  my $charref_map = {
327      0x0D => 0x000A,
328    0x80 => 0x20AC,    0x80 => 0x20AC,
329    0x81 => 0xFFFD,    0x81 => 0xFFFD,
330    0x82 => 0x201A,    0x82 => 0x201A,
# Line 331  my $c1_entity_char = { Line 357  my $c1_entity_char = {
357    0x9D => 0xFFFD,    0x9D => 0xFFFD,
358    0x9E => 0x017E,    0x9E => 0x017E,
359    0x9F => 0x0178,    0x9F => 0x0178,
360  }; # $c1_entity_char  }; # $charref_map
361    $charref_map->{$_} = 0xFFFD
362        for 0x0000..0x0008, 0x000B, 0x000E..0x001F, 0x007F,
363            0xD800..0xDFFF, 0xFDD0..0xFDDF, ## ISSUE: 0xFDEF
364            0xFFFE, 0xFFFF, 0x1FFFE, 0x1FFFF, 0x2FFFE, 0x2FFFF, 0x3FFFE, 0x3FFFF,
365            0x4FFFE, 0x4FFFF, 0x5FFFE, 0x5FFFF, 0x6FFFE, 0x6FFFF, 0x7FFFE,
366            0x7FFFF, 0x8FFFE, 0x8FFFF, 0x9FFFE, 0x9FFFF, 0xAFFFE, 0xAFFFF,
367            0xBFFFE, 0xBFFFF, 0xCFFFE, 0xCFFFF, 0xDFFFE, 0xDFFFF, 0xEFFFE,
368            0xEFFFF, 0xFFFFE, 0xFFFFF, 0x10FFFE, 0x10FFFF;
369    
370    ## TODO: Invoke the reset algorithm when a resettable element is
371    ## created (cf. HTML5 revision 2259).
372    
373  sub parse_byte_string ($$$$;$) {  sub parse_byte_string ($$$$;$) {
374    my $self = shift;    my $self = shift;
# Line 340  sub parse_byte_string ($$$$;$) { Line 377  sub parse_byte_string ($$$$;$) {
377    return $self->parse_byte_stream ($charset_name, $input, @_[1..$#_]);    return $self->parse_byte_stream ($charset_name, $input, @_[1..$#_]);
378  } # parse_byte_string  } # parse_byte_string
379    
380  sub parse_byte_stream ($$$$;$) {  sub parse_byte_stream ($$$$;$$) {
381      # my ($self, $charset_name, $byte_stream, $doc, $onerror, $get_wrapper) = @_;
382    my $self = ref $_[0] ? shift : shift->new;    my $self = ref $_[0] ? shift : shift->new;
383    my $charset_name = shift;    my $charset_name = shift;
384    my $byte_stream = $_[0];    my $byte_stream = $_[0];
# Line 351  sub parse_byte_stream ($$$$;$) { Line 389  sub parse_byte_stream ($$$$;$) {
389    };    };
390    $self->{parse_error} = $onerror; # updated later by parse_char_string    $self->{parse_error} = $onerror; # updated later by parse_char_string
391    
392      my $get_wrapper = $_[3] || sub ($) {
393        return $_[0]; # $_[0] = byte stream handle, returned = arg to char handle
394      };
395    
396    ## HTML5 encoding sniffing algorithm    ## HTML5 encoding sniffing algorithm
397    require Message::Charset::Info;    require Message::Charset::Info;
398    my $charset;    my $charset;
# Line 358  sub parse_byte_stream ($$$$;$) { Line 400  sub parse_byte_stream ($$$$;$) {
400    my ($char_stream, $e_status);    my ($char_stream, $e_status);
401    
402    SNIFFING: {    SNIFFING: {
403        ## NOTE: By setting |allow_fallback| option true when the
404        ## |get_decode_handle| method is invoked, we ignore what the HTML5
405        ## spec requires, i.e. unsupported encoding should be ignored.
406          ## TODO: We should not do this unless the parser is invoked
407          ## in the conformance checking mode, in which this behavior
408          ## would be useful.
409    
410      ## Step 1      ## Step 1
411      if (defined $charset_name) {      if (defined $charset_name) {
412        $charset = Message::Charset::Info->get_by_iana_name ($charset_name);        $charset = Message::Charset::Info->get_by_html_name ($charset_name);
413              ## TODO: Is this ok?  Transfer protocol's parameter should be
414              ## interpreted in its semantics?
415    
       ## ISSUE: Unsupported encoding is not ignored according to the spec.  
416        ($char_stream, $e_status) = $charset->get_decode_handle        ($char_stream, $e_status) = $charset->get_decode_handle
417            ($byte_stream, allow_error_reporting => 1,            ($byte_stream, allow_error_reporting => 1,
418             allow_fallback => 1);             allow_fallback => 1);
# Line 371  sub parse_byte_stream ($$$$;$) { Line 420  sub parse_byte_stream ($$$$;$) {
420          $self->{confident} = 1;          $self->{confident} = 1;
421          last SNIFFING;          last SNIFFING;
422        } else {        } else {
423          ## TODO: unsupported error          !!!parse-error (type => 'charset:not supported',
424                            layer => 'encode',
425                            line => 1, column => 1,
426                            value => $charset_name,
427                            level => $self->{level}->{uncertain});
428        }        }
429      }      }
430    
# Line 385  sub parse_byte_stream ($$$$;$) { Line 438  sub parse_byte_stream ($$$$;$) {
438    
439      ## Step 3      ## Step 3
440      if ($byte_buffer =~ /^\xFE\xFF/) {      if ($byte_buffer =~ /^\xFE\xFF/) {
441        $charset = Message::Charset::Info->get_by_iana_name ('utf-16be');        $charset = Message::Charset::Info->get_by_html_name ('utf-16be');
442        ($char_stream, $e_status) = $charset->get_decode_handle        ($char_stream, $e_status) = $charset->get_decode_handle
443            ($byte_stream, allow_error_reporting => 1,            ($byte_stream, allow_error_reporting => 1,
444             allow_fallback => 1, byte_buffer => \$byte_buffer);             allow_fallback => 1, byte_buffer => \$byte_buffer);
445        $self->{confident} = 1;        $self->{confident} = 1;
446        last SNIFFING;        last SNIFFING;
447      } elsif ($byte_buffer =~ /^\xFF\xFE/) {      } elsif ($byte_buffer =~ /^\xFF\xFE/) {
448        $charset = Message::Charset::Info->get_by_iana_name ('utf-16le');        $charset = Message::Charset::Info->get_by_html_name ('utf-16le');
449        ($char_stream, $e_status) = $charset->get_decode_handle        ($char_stream, $e_status) = $charset->get_decode_handle
450            ($byte_stream, allow_error_reporting => 1,            ($byte_stream, allow_error_reporting => 1,
451             allow_fallback => 1, byte_buffer => \$byte_buffer);             allow_fallback => 1, byte_buffer => \$byte_buffer);
452        $self->{confident} = 1;        $self->{confident} = 1;
453        last SNIFFING;        last SNIFFING;
454      } elsif ($byte_buffer =~ /^\xEF\xBB\xBF/) {      } elsif ($byte_buffer =~ /^\xEF\xBB\xBF/) {
455        $charset = Message::Charset::Info->get_by_iana_name ('utf-8');        $charset = Message::Charset::Info->get_by_html_name ('utf-8');
456        ($char_stream, $e_status) = $charset->get_decode_handle        ($char_stream, $e_status) = $charset->get_decode_handle
457            ($byte_stream, allow_error_reporting => 1,            ($byte_stream, allow_error_reporting => 1,
458             allow_fallback => 1, byte_buffer => \$byte_buffer);             allow_fallback => 1, byte_buffer => \$byte_buffer);
# Line 418  sub parse_byte_stream ($$$$;$) { Line 471  sub parse_byte_stream ($$$$;$) {
471      $charset_name = Whatpm::Charset::UniversalCharDet->detect_byte_string      $charset_name = Whatpm::Charset::UniversalCharDet->detect_byte_string
472          ($byte_buffer);          ($byte_buffer);
473      if (defined $charset_name) {      if (defined $charset_name) {
474        $charset = Message::Charset::Info->get_by_iana_name ($charset_name);        $charset = Message::Charset::Info->get_by_html_name ($charset_name);
475    
476        ## ISSUE: Unsupported encoding is not ignored according to the spec.        ## ISSUE: Unsupported encoding is not ignored according to the spec.
477        require Whatpm::Charset::DecodeHandle;        require Whatpm::Charset::DecodeHandle;
# Line 429  sub parse_byte_stream ($$$$;$) { Line 482  sub parse_byte_stream ($$$$;$) {
482             allow_fallback => 1, byte_buffer => \$byte_buffer);             allow_fallback => 1, byte_buffer => \$byte_buffer);
483        if ($char_stream) {        if ($char_stream) {
484          $buffer->{buffer} = $byte_buffer;          $buffer->{buffer} = $byte_buffer;
485          !!!parse-error (type => 'sniffing:chardet', ## TODO: type name          !!!parse-error (type => 'sniffing:chardet',
486                          value => $charset_name,                          text => $charset_name,
487                          level => $self->{info_level},                          level => $self->{level}->{info},
488                            layer => 'encode',
489                          line => 1, column => 1);                          line => 1, column => 1);
490          $self->{confident} = 0;          $self->{confident} = 0;
491          last SNIFFING;          last SNIFFING;
# Line 440  sub parse_byte_stream ($$$$;$) { Line 494  sub parse_byte_stream ($$$$;$) {
494    
495      ## Step 7: default      ## Step 7: default
496      ## TODO: Make this configurable.      ## TODO: Make this configurable.
497      $charset = Message::Charset::Info->get_by_iana_name ('windows-1252');      $charset = Message::Charset::Info->get_by_html_name ('windows-1252');
498          ## NOTE: We choose |windows-1252| here, since |utf-8| should be          ## NOTE: We choose |windows-1252| here, since |utf-8| should be
499          ## detectable in the step 6.          ## detectable in the step 6.
500      require Whatpm::Charset::DecodeHandle;      require Whatpm::Charset::DecodeHandle;
# Line 452  sub parse_byte_stream ($$$$;$) { Line 506  sub parse_byte_stream ($$$$;$) {
506                                         allow_fallback => 1,                                         allow_fallback => 1,
507                                         byte_buffer => \$byte_buffer);                                         byte_buffer => \$byte_buffer);
508      $buffer->{buffer} = $byte_buffer;      $buffer->{buffer} = $byte_buffer;
509      !!!parse-error (type => 'sniffing:default', ## TODO: type name      !!!parse-error (type => 'sniffing:default',
510                      value => 'windows-1252',                      text => 'windows-1252',
511                      level => $self->{info_level},                      level => $self->{level}->{info},
512                      line => 1, column => 1);                      line => 1, column => 1,
513                        layer => 'encode');
514      $self->{confident} = 0;      $self->{confident} = 0;
515    } # SNIFFING    } # SNIFFING
516    
   $self->{input_encoding} = $charset->get_iana_name;  
517    if ($e_status & Message::Charset::Info::FALLBACK_ENCODING_IMPL ()) {    if ($e_status & Message::Charset::Info::FALLBACK_ENCODING_IMPL ()) {
518      !!!parse-error (type => 'chardecode:fallback', ## TODO: type name      $self->{input_encoding} = $charset->get_iana_name; ## TODO: Should we set actual charset decoder's encoding name?
519                      value => $self->{input_encoding},      !!!parse-error (type => 'chardecode:fallback',
520                      level => $self->{unsupported_level},                      #text => $self->{input_encoding},
521                      line => 1, column => 1);                      level => $self->{level}->{uncertain},
522                        line => 1, column => 1,
523                        layer => 'encode');
524    } elsif (not ($e_status &    } elsif (not ($e_status &
525                  Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL())) {                  Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL ())) {
526      !!!parse-error (type => 'chardecode:no error', ## TODO: type name      $self->{input_encoding} = $charset->get_iana_name;
527                      value => $self->{input_encoding},      !!!parse-error (type => 'chardecode:no error',
528                      level => $self->{unsupported_level},                      text => $self->{input_encoding},
529                      line => 1, column => 1);                      level => $self->{level}->{uncertain},
530                        line => 1, column => 1,
531                        layer => 'encode');
532      } else {
533        $self->{input_encoding} = $charset->get_iana_name;
534    }    }
535    
536    $self->{change_encoding} = sub {    $self->{change_encoding} = sub {
# Line 478  sub parse_byte_stream ($$$$;$) { Line 538  sub parse_byte_stream ($$$$;$) {
538      $charset_name = shift;      $charset_name = shift;
539      my $token = shift;      my $token = shift;
540    
541      $charset = Message::Charset::Info->get_by_iana_name ($charset_name);      $charset = Message::Charset::Info->get_by_html_name ($charset_name);
542      ($char_stream, $e_status) = $charset->get_decode_handle      ($char_stream, $e_status) = $charset->get_decode_handle
543          ($byte_stream, allow_error_reporting => 1, allow_fallback => 1,          ($byte_stream, allow_error_reporting => 1, allow_fallback => 1,
544           byte_buffer => \ $buffer->{buffer});           byte_buffer => \ $buffer->{buffer});
# Line 487  sub parse_byte_stream ($$$$;$) { Line 547  sub parse_byte_stream ($$$$;$) {
547        ## "Change the encoding" algorithm:        ## "Change the encoding" algorithm:
548    
549        ## Step 1            ## Step 1    
550        if ($charset->{iana_names}->{'utf-16'}) { ## ISSUE: UTF-16BE -> UTF-8? UTF-16LE -> UTF-8?        if ($charset->{category} &
551          $charset = Message::Charset::Info->get_by_iana_name ('utf-8');            Message::Charset::Info::CHARSET_CATEGORY_UTF16 ()) {
552            $charset = Message::Charset::Info->get_by_html_name ('utf-8');
553          ($char_stream, $e_status) = $charset->get_decode_handle          ($char_stream, $e_status) = $charset->get_decode_handle
554              ($byte_stream,              ($byte_stream,
555               byte_buffer => \ $buffer->{buffer});               byte_buffer => \ $buffer->{buffer});
# Line 498  sub parse_byte_stream ($$$$;$) { Line 559  sub parse_byte_stream ($$$$;$) {
559        ## Step 2        ## Step 2
560        if (defined $self->{input_encoding} and        if (defined $self->{input_encoding} and
561            $self->{input_encoding} eq $charset_name) {            $self->{input_encoding} eq $charset_name) {
562          !!!parse-error (type => 'charset label:matching', ## TODO: type          !!!parse-error (type => 'charset label:matching',
563                          value => $charset_name,                          text => $charset_name,
564                          level => $self->{info_level});                          level => $self->{level}->{info});
565          $self->{confident} = 1;          $self->{confident} = 1;
566          return;          return;
567        }        }
568    
569        !!!parse-error (type => 'charset label detected:'.$self->{input_encoding}.        !!!parse-error (type => 'charset label detected',
570            ':'.$charset_name, level => 'w', token => $token);                        text => $self->{input_encoding},
571                          value => $charset_name,
572                          level => $self->{level}->{warn},
573                          token => $token);
574                
575        ## Step 3        ## Step 3
576        # if (can) {        # if (can) {
# Line 522  sub parse_byte_stream ($$$$;$) { Line 586  sub parse_byte_stream ($$$$;$) {
586    
587    my $char_onerror = sub {    my $char_onerror = sub {
588      my (undef, $type, %opt) = @_;      my (undef, $type, %opt) = @_;
589      !!!parse-error (%opt, type => $type,      !!!parse-error (layer => 'encode',
590                      line => $self->{line}, column => $self->{column} + 1);                      line => $self->{line}, column => $self->{column} + 1,
591                        %opt, type => $type);
592      if ($opt{octets}) {      if ($opt{octets}) {
593        ${$opt{octets}} = "\x{FFFD}"; # relacement character        ${$opt{octets}} = "\x{FFFD}"; # relacement character
594      }      }
595    };    };
   $char_stream->onerror ($char_onerror);  
596    
597    my @args = @_; shift @args; # $s    my $wrapped_char_stream = $get_wrapper->($char_stream);
598      $wrapped_char_stream->onerror ($char_onerror);
599    
600      my @args = ($_[1], $_[2]); # $doc, $onerror - $get_wrapper = undef;
601    my $return;    my $return;
602    try {    try {
603      $return = $self->parse_char_stream ($char_stream, @args);        $return = $self->parse_char_stream ($wrapped_char_stream, @args);  
604    } catch Whatpm::HTML::RestartParser with {    } catch Whatpm::HTML::RestartParser with {
605      ## NOTE: Invoked after {change_encoding}.      ## NOTE: Invoked after {change_encoding}.
606    
     $self->{input_encoding} = $charset->get_iana_name;  
607      if ($e_status & Message::Charset::Info::FALLBACK_ENCODING_IMPL ()) {      if ($e_status & Message::Charset::Info::FALLBACK_ENCODING_IMPL ()) {
608        !!!parse-error (type => 'chardecode:fallback', ## TODO: type name        $self->{input_encoding} = $charset->get_iana_name; ## TODO: Should we set actual charset decoder's encoding name?
609                        value => $self->{input_encoding},        !!!parse-error (type => 'chardecode:fallback',
610                        level => $self->{unsupported_level},                        level => $self->{level}->{uncertain},
611                        line => 1, column => 1);                        #text => $self->{input_encoding},
612                          line => 1, column => 1,
613                          layer => 'encode');
614      } elsif (not ($e_status &      } elsif (not ($e_status &
615                    Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL())) {                    Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL ())) {
616        !!!parse-error (type => 'chardecode:no error', ## TODO: type name        $self->{input_encoding} = $charset->get_iana_name;
617                        value => $self->{input_encoding},        !!!parse-error (type => 'chardecode:no error',
618                        level => $self->{unsupported_level},                        text => $self->{input_encoding},
619                        line => 1, column => 1);                        level => $self->{level}->{uncertain},
620                          line => 1, column => 1,
621                          layer => 'encode');
622        } else {
623          $self->{input_encoding} = $charset->get_iana_name;
624      }      }
625      $self->{confident} = 1;      $self->{confident} = 1;
626      $char_stream->onerror ($char_onerror);  
627      $return = $self->parse_char_stream ($char_stream, @args);      $wrapped_char_stream = $get_wrapper->($char_stream);
628        $wrapped_char_stream->onerror ($char_onerror);
629    
630        $return = $self->parse_char_stream ($wrapped_char_stream, @args);
631    };    };
632    return $return;    return $return;
633  } # parse_byte_stream  } # parse_byte_stream
# Line 566  sub parse_byte_stream ($$$$;$) { Line 641  sub parse_byte_stream ($$$$;$) {
641  ## such as |parse_byte_string| in this module, must ensure that it does  ## such as |parse_byte_string| in this module, must ensure that it does
642  ## strip the BOM and never strip any ZWNBSP.  ## strip the BOM and never strip any ZWNBSP.
643    
644  sub parse_char_string ($$$;$) {  sub parse_char_string ($$$;$$) {
645      #my ($self, $s, $doc, $onerror, $get_wrapper) = @_;
646    my $self = shift;    my $self = shift;
   require utf8;  
647    my $s = ref $_[0] ? $_[0] : \($_[0]);    my $s = ref $_[0] ? $_[0] : \($_[0]);
648    open my $input, '<' . (utf8::is_utf8 ($$s) ? ':utf8' : ''), $s;    require Whatpm::Charset::DecodeHandle;
649      my $input = Whatpm::Charset::DecodeHandle::CharString->new ($s);
650    return $self->parse_char_stream ($input, @_[1..$#_]);    return $self->parse_char_stream ($input, @_[1..$#_]);
651  } # parse_char_string  } # parse_char_string
652  *parse_string = \&parse_char_string;  *parse_string = \&parse_char_string; ## NOTE: Alias for backward compatibility.
653    
654  sub parse_char_stream ($$$;$) {  sub parse_char_stream ($$$;$$) {
655    my $self = ref $_[0] ? shift : shift->new;    my $self = ref $_[0] ? shift : shift->new;
656    my $input = $_[0];    my $input = $_[0];
657    $self->{document} = $_[1];    $self->{document} = $_[1];
# Line 586  sub parse_char_stream ($$$;$) { Line 662  sub parse_char_stream ($$$;$) {
662    $self->{confident} = 1 unless exists $self->{confident};    $self->{confident} = 1 unless exists $self->{confident};
663    $self->{document}->input_encoding ($self->{input_encoding})    $self->{document}->input_encoding ($self->{input_encoding})
664        if defined $self->{input_encoding};        if defined $self->{input_encoding};
665    ## TODO: |{input_encoding}| is needless?
666    
   my $i = 0;  
667    $self->{line_prev} = $self->{line} = 1;    $self->{line_prev} = $self->{line} = 1;
668    $self->{column_prev} = $self->{column} = 0;    $self->{column_prev} = -1;
669    $self->{set_next_char} = sub {    $self->{column} = 0;
670      $self->{set_nc} = sub {
671      my $self = shift;      my $self = shift;
672    
673      pop @{$self->{prev_char}};      my $char = '';
674      unshift @{$self->{prev_char}}, $self->{next_char};      if (defined $self->{next_nc}) {
675          $char = $self->{next_nc};
676      my $char;        delete $self->{next_nc};
677      if (defined $self->{next_next_char}) {        $self->{nc} = ord $char;
       $char = $self->{next_next_char};  
       delete $self->{next_next_char};  
678      } else {      } else {
679        $char = $input->getc;        $self->{char_buffer} = '';
680          $self->{char_buffer_pos} = 0;
681    
682          my $count = $input->manakai_read_until
683             ($self->{char_buffer}, qr/[^\x00\x0A\x0D]/, $self->{char_buffer_pos});
684          if ($count) {
685            $self->{line_prev} = $self->{line};
686            $self->{column_prev} = $self->{column};
687            $self->{column}++;
688            $self->{nc}
689                = ord substr ($self->{char_buffer}, $self->{char_buffer_pos}++, 1);
690            return;
691          }
692    
693          if ($input->read ($char, 1)) {
694            $self->{nc} = ord $char;
695          } else {
696            $self->{nc} = -1;
697            return;
698          }
699      }      }
     $self->{next_char} = -1 and return unless defined $char;  
     $self->{next_char} = ord $char;  
700    
701      ($self->{line_prev}, $self->{column_prev})      ($self->{line_prev}, $self->{column_prev})
702          = ($self->{line}, $self->{column});          = ($self->{line}, $self->{column});
703      $self->{column}++;      $self->{column}++;
704            
705      if ($self->{next_char} == 0x000A) { # LF      if ($self->{nc} == 0x000A) { # LF
706        !!!cp ('j1');        !!!cp ('j1');
707        $self->{line}++;        $self->{line}++;
708        $self->{column} = 0;        $self->{column} = 0;
709      } elsif ($self->{next_char} == 0x000D) { # CR      } elsif ($self->{nc} == 0x000D) { # CR
710        !!!cp ('j2');        !!!cp ('j2');
711        my $next = $input->getc;  ## TODO: support for abort/streaming
712        if (defined $next and $next ne "\x0A") {        my $next = '';
713          $self->{next_next_char} = $next;        if ($input->read ($next, 1) and $next ne "\x0A") {
714            $self->{next_nc} = $next;
715        }        }
716        $self->{next_char} = 0x000A; # LF # MUST        $self->{nc} = 0x000A; # LF # MUST
717        $self->{line}++;        $self->{line}++;
718        $self->{column} = 0;        $self->{column} = 0;
719      } elsif ($self->{next_char} > 0x10FFFF) {      } elsif ($self->{nc} == 0x0000) { # NULL
       !!!cp ('j3');  
       $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST  
     } elsif ($self->{next_char} == 0x0000) { # NULL  
720        !!!cp ('j4');        !!!cp ('j4');
721        !!!parse-error (type => 'NULL');        !!!parse-error (type => 'NULL');
722        $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST        $self->{nc} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
     } elsif ($self->{next_char} <= 0x0008 or  
              (0x000E <= $self->{next_char} and $self->{next_char} <= 0x001F) or  
              (0x007F <= $self->{next_char} and $self->{next_char} <= 0x009F) or  
              (0xD800 <= $self->{next_char} and $self->{next_char} <= 0xDFFF) or  
              (0xFDD0 <= $self->{next_char} and $self->{next_char} <= 0xFDDF) or  
              {  
               0xFFFE => 1, 0xFFFF => 1, 0x1FFFE => 1, 0x1FFFF => 1,  
               0x2FFFE => 1, 0x2FFFF => 1, 0x3FFFE => 1, 0x3FFFF => 1,  
               0x4FFFE => 1, 0x4FFFF => 1, 0x5FFFE => 1, 0x5FFFF => 1,  
               0x6FFFE => 1, 0x6FFFF => 1, 0x7FFFE => 1, 0x7FFFF => 1,  
               0x8FFFE => 1, 0x8FFFF => 1, 0x9FFFE => 1, 0x9FFFF => 1,  
               0xAFFFE => 1, 0xAFFFF => 1, 0xBFFFE => 1, 0xBFFFF => 1,  
               0xCFFFE => 1, 0xCFFFF => 1, 0xDFFFE => 1, 0xDFFFF => 1,  
               0xEFFFE => 1, 0xEFFFF => 1, 0xFFFFE => 1, 0xFFFFF => 1,  
               0x10FFFE => 1, 0x10FFFF => 1,  
              }->{$self->{next_char}}) {  
       !!!cp ('j5');  
       !!!parse-error (type => 'control char', level => $self->{must_level});  
 ## TODO: error type documentation  
723      }      }
724    };    };
725    $self->{prev_char} = [-1, -1, -1];  
726    $self->{next_char} = -1;    $self->{read_until} = sub {
727        #my ($scalar, $specials_range, $offset) = @_;
728        return 0 if defined $self->{next_nc};
729    
730        my $pattern = qr/[^$_[1]\x00\x0A\x0D]/;
731        my $offset = $_[2] || 0;
732    
733        if ($self->{char_buffer_pos} < length $self->{char_buffer}) {
734          pos ($self->{char_buffer}) = $self->{char_buffer_pos};
735          if ($self->{char_buffer} =~ /\G(?>$pattern)+/) {
736            substr ($_[0], $offset)
737                = substr ($self->{char_buffer}, $-[0], $+[0] - $-[0]);
738            my $count = $+[0] - $-[0];
739            if ($count) {
740              $self->{column} += $count;
741              $self->{char_buffer_pos} += $count;
742              $self->{line_prev} = $self->{line};
743              $self->{column_prev} = $self->{column} - 1;
744              $self->{nc} = -1;
745            }
746            return $count;
747          } else {
748            return 0;
749          }
750        } else {
751          my $count = $input->manakai_read_until ($_[0], $pattern, $_[2]);
752          if ($count) {
753            $self->{column} += $count;
754            $self->{line_prev} = $self->{line};
755            $self->{column_prev} = $self->{column} - 1;
756            $self->{nc} = -1;
757          }
758          return $count;
759        }
760      }; # $self->{read_until}
761    
762    my $onerror = $_[2] || sub {    my $onerror = $_[2] || sub {
763      my (%opt) = @_;      my (%opt) = @_;
# Line 664  sub parse_char_stream ($$$;$) { Line 769  sub parse_char_stream ($$$;$) {
769      $onerror->(line => $self->{line}, column => $self->{column}, @_);      $onerror->(line => $self->{line}, column => $self->{column}, @_);
770    };    };
771    
772      my $char_onerror = sub {
773        my (undef, $type, %opt) = @_;
774        !!!parse-error (layer => 'encode',
775                        line => $self->{line}, column => $self->{column} + 1,
776                        %opt, type => $type);
777      }; # $char_onerror
778    
779      if ($_[3]) {
780        $input = $_[3]->($input);
781        $input->onerror ($char_onerror);
782      } else {
783        $input->onerror ($char_onerror) unless defined $input->onerror;
784      }
785    
786    $self->_initialize_tokenizer;    $self->_initialize_tokenizer;
787    $self->_initialize_tree_constructor;    $self->_initialize_tree_constructor;
788    $self->_construct_tree;    $self->_construct_tree;
# Line 677  sub parse_char_stream ($$$;$) { Line 796  sub parse_char_stream ($$$;$) {
796  sub new ($) {  sub new ($) {
797    my $class = shift;    my $class = shift;
798    my $self = bless {    my $self = bless {
799      must_level => 'm',      level => {must => 'm',
800      should_level => 's',                should => 's',
801      good_level => 'w',                warn => 'w',
802      warn_level => 'w',                info => 'i',
803      info_level => 'i',                uncertain => 'u'},
     unsupported_level => 'u',  
804    }, $class;    }, $class;
805    $self->{set_next_char} = sub {    $self->{set_nc} = sub {
806      $self->{next_char} = -1;      $self->{nc} = -1;
807    };    };
808    $self->{parse_error} = sub {    $self->{parse_error} = sub {
809      #      #
# Line 712  sub RCDATA_CONTENT_MODEL () { CM_ENTITY Line 830  sub RCDATA_CONTENT_MODEL () { CM_ENTITY
830  sub PCDATA_CONTENT_MODEL () { CM_ENTITY | CM_FULL_MARKUP }  sub PCDATA_CONTENT_MODEL () { CM_ENTITY | CM_FULL_MARKUP }
831    
832  sub DATA_STATE () { 0 }  sub DATA_STATE () { 0 }
833  sub ENTITY_DATA_STATE () { 1 }  #sub ENTITY_DATA_STATE () { 1 }
834  sub TAG_OPEN_STATE () { 2 }  sub TAG_OPEN_STATE () { 2 }
835  sub CLOSE_TAG_OPEN_STATE () { 3 }  sub CLOSE_TAG_OPEN_STATE () { 3 }
836  sub TAG_NAME_STATE () { 4 }  sub TAG_NAME_STATE () { 4 }
# Line 723  sub BEFORE_ATTRIBUTE_VALUE_STATE () { 8 Line 841  sub BEFORE_ATTRIBUTE_VALUE_STATE () { 8
841  sub ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE () { 9 }  sub ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE () { 9 }
842  sub ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE () { 10 }  sub ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE () { 10 }
843  sub ATTRIBUTE_VALUE_UNQUOTED_STATE () { 11 }  sub ATTRIBUTE_VALUE_UNQUOTED_STATE () { 11 }
844  sub ENTITY_IN_ATTRIBUTE_VALUE_STATE () { 12 }  #sub ENTITY_IN_ATTRIBUTE_VALUE_STATE () { 12 }
845  sub MARKUP_DECLARATION_OPEN_STATE () { 13 }  sub MARKUP_DECLARATION_OPEN_STATE () { 13 }
846  sub COMMENT_START_STATE () { 14 }  sub COMMENT_START_STATE () { 14 }
847  sub COMMENT_START_DASH_STATE () { 15 }  sub COMMENT_START_DASH_STATE () { 15 }
# Line 746  sub AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STAT Line 864  sub AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STAT
864  sub BOGUS_DOCTYPE_STATE () { 32 }  sub BOGUS_DOCTYPE_STATE () { 32 }
865  sub AFTER_ATTRIBUTE_VALUE_QUOTED_STATE () { 33 }  sub AFTER_ATTRIBUTE_VALUE_QUOTED_STATE () { 33 }
866  sub SELF_CLOSING_START_TAG_STATE () { 34 }  sub SELF_CLOSING_START_TAG_STATE () { 34 }
867  sub CDATA_BLOCK_STATE () { 35 }  sub CDATA_SECTION_STATE () { 35 }
868    sub MD_HYPHEN_STATE () { 36 } # "markup declaration open state" in the spec
869    sub MD_DOCTYPE_STATE () { 37 } # "markup declaration open state" in the spec
870    sub MD_CDATA_STATE () { 38 } # "markup declaration open state" in the spec
871    sub CDATA_RCDATA_CLOSE_TAG_STATE () { 39 } # "close tag open state" in the spec
872    sub CDATA_SECTION_MSE1_STATE () { 40 } # "CDATA section state" in the spec
873    sub CDATA_SECTION_MSE2_STATE () { 41 } # "CDATA section state" in the spec
874    sub PUBLIC_STATE () { 42 } # "after DOCTYPE name state" in the spec
875    sub SYSTEM_STATE () { 43 } # "after DOCTYPE name state" in the spec
876    ## NOTE: "Entity data state", "entity in attribute value state", and
877    ## "consume a character reference" algorithm are jointly implemented
878    ## using the following six states:
879    sub ENTITY_STATE () { 44 }
880    sub ENTITY_HASH_STATE () { 45 }
881    sub NCR_NUM_STATE () { 46 }
882    sub HEXREF_X_STATE () { 47 }
883    sub HEXREF_HEX_STATE () { 48 }
884    sub ENTITY_NAME_STATE () { 49 }
885    sub PCDATA_STATE () { 50 } # "data state" in the spec
886    
887  sub DOCTYPE_TOKEN () { 1 }  sub DOCTYPE_TOKEN () { 1 }
888  sub COMMENT_TOKEN () { 2 }  sub COMMENT_TOKEN () { 2 }
# Line 799  sub IN_COLUMN_GROUP_IM () { 0b10 } Line 935  sub IN_COLUMN_GROUP_IM () { 0b10 }
935  sub _initialize_tokenizer ($) {  sub _initialize_tokenizer ($) {
936    my $self = shift;    my $self = shift;
937    $self->{state} = DATA_STATE; # MUST    $self->{state} = DATA_STATE; # MUST
938      #$self->{s_kwd}; # state keyword - initialized when used
939      #$self->{entity__value}; # initialized when used
940      #$self->{entity__match}; # initialized when used
941    $self->{content_model} = PCDATA_CONTENT_MODEL; # be    $self->{content_model} = PCDATA_CONTENT_MODEL; # be
942    undef $self->{current_token}; # start tag, end tag, comment, or DOCTYPE    undef $self->{ct}; # current token
943    undef $self->{current_attribute};    undef $self->{ca}; # current attribute
944    undef $self->{last_emitted_start_tag_name};    undef $self->{last_stag_name}; # last emitted start tag name
945    undef $self->{last_attribute_value_state};    #$self->{prev_state}; # initialized when used
946    delete $self->{self_closing};    delete $self->{self_closing};
947    $self->{char} = [];    $self->{char_buffer} = '';
948    # $self->{next_char}    $self->{char_buffer_pos} = 0;
949      $self->{nc} = -1; # next input character
950      #$self->{next_nc}
951    !!!next-input-character;    !!!next-input-character;
952    $self->{token} = [];    $self->{token} = [];
953    # $self->{escape}    # $self->{escape}
# Line 817  sub _initialize_tokenizer ($) { Line 958  sub _initialize_tokenizer ($) {
958  ##       CHARACTER_TOKEN, or END_OF_FILE_TOKEN  ##       CHARACTER_TOKEN, or END_OF_FILE_TOKEN
959  ##   ->{name} (DOCTYPE_TOKEN)  ##   ->{name} (DOCTYPE_TOKEN)
960  ##   ->{tag_name} (START_TAG_TOKEN, END_TAG_TOKEN)  ##   ->{tag_name} (START_TAG_TOKEN, END_TAG_TOKEN)
961  ##   ->{public_identifier} (DOCTYPE_TOKEN)  ##   ->{pubid} (DOCTYPE_TOKEN)
962  ##   ->{system_identifier} (DOCTYPE_TOKEN)  ##   ->{sysid} (DOCTYPE_TOKEN)
963  ##   ->{quirks} == 1 or 0 (DOCTYPE_TOKEN): "force-quirks" flag  ##   ->{quirks} == 1 or 0 (DOCTYPE_TOKEN): "force-quirks" flag
964  ##   ->{attributes} isa HASH (START_TAG_TOKEN, END_TAG_TOKEN)  ##   ->{attributes} isa HASH (START_TAG_TOKEN, END_TAG_TOKEN)
965  ##        ->{name}  ##        ->{name}
# Line 829  sub _initialize_tokenizer ($) { Line 970  sub _initialize_tokenizer ($) {
970  ##     |->{self_closing}| is used to save the value of |$self->{self_closing}|  ##     |->{self_closing}| is used to save the value of |$self->{self_closing}|
971  ##     while the token is pushed back to the stack.  ##     while the token is pushed back to the stack.
972    
 ## ISSUE: "When a DOCTYPE token is created, its  
 ## <i>self-closing flag</i> must be unset (its other state is that it  
 ## be set), and its attributes list must be empty.": Wrong subject?  
   
973  ## Emitted token MUST immediately be handled by the tree construction state.  ## Emitted token MUST immediately be handled by the tree construction state.
974    
975  ## Before each step, UA MAY check to see if either one of the scripts in  ## Before each step, UA MAY check to see if either one of the scripts in
# Line 841  sub _initialize_tokenizer ($) { Line 978  sub _initialize_tokenizer ($) {
978  ## has completed loading.  If one has, then it MUST be executed  ## has completed loading.  If one has, then it MUST be executed
979  ## and removed from the list.  ## and removed from the list.
980    
981  ## NOTE: HTML5 "Writing HTML documents" section, applied to  ## TODO: Polytheistic slash SHOULD NOT be used. (Applied only to atheists.)
982  ## documents and not to user agents and conformance checkers,  ## (This requirement was dropped from HTML5 spec, unfortunately.)
983  ## contains some requirements that are not detected by the  
984  ## parsing algorithm:  my $is_space = {
985  ## - Some requirements on character encoding declarations. ## TODO    0x0009 => 1, # CHARACTER TABULATION (HT)
986  ## - "Elements MUST NOT contain content that their content model disallows."    0x000A => 1, # LINE FEED (LF)
987  ##   ... Some are parse error, some are not (will be reported by c.c.).    #0x000B => 0, # LINE TABULATION (VT)
988  ## - Polytheistic slash SHOULD NOT be used. (Applied only to atheists.) ## TODO    0x000C => 1, # FORM FEED (FF)
989  ## - Text (in elements, attributes, and comments) SHOULD NOT contain    #0x000D => 1, # CARRIAGE RETURN (CR)
990  ##   control characters other than space characters. ## TODO: (what is control character? C0, C1 and DEL?  Unicode control character?)    0x0020 => 1, # SPACE (SP)
991    };
 ## TODO: HTML5 poses authors two SHOULD-level requirements that cannot  
 ## be detected by the HTML5 parsing algorithm:  
 ## - Text,  
992    
993  sub _get_next_token ($) {  sub _get_next_token ($) {
994    my $self = shift;    my $self = shift;
995    
996    if ($self->{self_closing}) {    if ($self->{self_closing}) {
997      !!!parse-error (type => 'nestc', token => $self->{current_token});      !!!parse-error (type => 'nestc', token => $self->{ct});
998      ## NOTE: The |self_closing| flag is only set by start tag token.      ## NOTE: The |self_closing| flag is only set by start tag token.
999      ## In addition, when a start tag token is emitted, it is always set to      ## In addition, when a start tag token is emitted, it is always set to
1000      ## |current_token|.      ## |ct|.
1001      delete $self->{self_closing};      delete $self->{self_closing};
1002    }    }
1003    
# Line 873  sub _get_next_token ($) { Line 1007  sub _get_next_token ($) {
1007    }    }
1008    
1009    A: {    A: {
1010      if ($self->{state} == DATA_STATE) {      if ($self->{state} == PCDATA_STATE) {
1011        if ($self->{next_char} == 0x0026) { # &        ## NOTE: Same as |DATA_STATE|, but only for |PCDATA| content model.
1012    
1013          if ($self->{nc} == 0x0026) { # &
1014            !!!cp (0.1);
1015            ## NOTE: In the spec, the tokenizer is switched to the
1016            ## "entity data state".  In this implementation, the tokenizer
1017            ## is switched to the |ENTITY_STATE|, which is an implementation
1018            ## of the "consume a character reference" algorithm.
1019            $self->{entity_add} = -1;
1020            $self->{prev_state} = DATA_STATE;
1021            $self->{state} = ENTITY_STATE;
1022            !!!next-input-character;
1023            redo A;
1024          } elsif ($self->{nc} == 0x003C) { # <
1025            !!!cp (0.2);
1026            $self->{state} = TAG_OPEN_STATE;
1027            !!!next-input-character;
1028            redo A;
1029          } elsif ($self->{nc} == -1) {
1030            !!!cp (0.3);
1031            !!!emit ({type => END_OF_FILE_TOKEN,
1032                      line => $self->{line}, column => $self->{column}});
1033            last A; ## TODO: ok?
1034          } else {
1035            !!!cp (0.4);
1036            #
1037          }
1038    
1039          # Anything else
1040          my $token = {type => CHARACTER_TOKEN,
1041                       data => chr $self->{nc},
1042                       line => $self->{line}, column => $self->{column},
1043                      };
1044          $self->{read_until}->($token->{data}, q[<&], length $token->{data});
1045    
1046          ## Stay in the state.
1047          !!!next-input-character;
1048          !!!emit ($token);
1049          redo A;
1050        } elsif ($self->{state} == DATA_STATE) {
1051          $self->{s_kwd} = '' unless defined $self->{s_kwd};
1052          if ($self->{nc} == 0x0026) { # &
1053            $self->{s_kwd} = '';
1054          if ($self->{content_model} & CM_ENTITY and # PCDATA | RCDATA          if ($self->{content_model} & CM_ENTITY and # PCDATA | RCDATA
1055              not $self->{escape}) {              not $self->{escape}) {
1056            !!!cp (1);            !!!cp (1);
1057            $self->{state} = ENTITY_DATA_STATE;            ## NOTE: In the spec, the tokenizer is switched to the
1058              ## "entity data state".  In this implementation, the tokenizer
1059              ## is switched to the |ENTITY_STATE|, which is an implementation
1060              ## of the "consume a character reference" algorithm.
1061              $self->{entity_add} = -1;
1062              $self->{prev_state} = DATA_STATE;
1063              $self->{state} = ENTITY_STATE;
1064            !!!next-input-character;            !!!next-input-character;
1065            redo A;            redo A;
1066          } else {          } else {
1067            !!!cp (2);            !!!cp (2);
1068            #            #
1069          }          }
1070        } elsif ($self->{next_char} == 0x002D) { # -        } elsif ($self->{nc} == 0x002D) { # -
1071          if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA          if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
1072            unless ($self->{escape}) {            $self->{s_kwd} .= '-';
1073              if ($self->{prev_char}->[0] == 0x002D and # -            
1074                  $self->{prev_char}->[1] == 0x0021 and # !            if ($self->{s_kwd} eq '<!--') {
1075                  $self->{prev_char}->[2] == 0x003C) { # <              !!!cp (3);
1076                !!!cp (3);              $self->{escape} = 1; # unless $self->{escape};
1077                $self->{escape} = 1;              $self->{s_kwd} = '--';
1078              } else {              #
1079                !!!cp (4);            } elsif ($self->{s_kwd} eq '---') {
1080              }              !!!cp (4);
1081                $self->{s_kwd} = '--';
1082                #
1083            } else {            } else {
1084              !!!cp (5);              !!!cp (5);
1085                #
1086            }            }
1087          }          }
1088                    
1089          #          #
1090        } elsif ($self->{next_char} == 0x003C) { # <        } elsif ($self->{nc} == 0x0021) { # !
1091            if (length $self->{s_kwd}) {
1092              !!!cp (5.1);
1093              $self->{s_kwd} .= '!';
1094              #
1095            } else {
1096              !!!cp (5.2);
1097              #$self->{s_kwd} = '';
1098              #
1099            }
1100            #
1101          } elsif ($self->{nc} == 0x003C) { # <
1102          if ($self->{content_model} & CM_FULL_MARKUP or # PCDATA          if ($self->{content_model} & CM_FULL_MARKUP or # PCDATA
1103              (($self->{content_model} & CM_LIMITED_MARKUP) and # CDATA | RCDATA              (($self->{content_model} & CM_LIMITED_MARKUP) and # CDATA | RCDATA
1104               not $self->{escape})) {               not $self->{escape})) {
# Line 912  sub _get_next_token ($) { Line 1108  sub _get_next_token ($) {
1108            redo A;            redo A;
1109          } else {          } else {
1110            !!!cp (7);            !!!cp (7);
1111              $self->{s_kwd} = '';
1112            #            #
1113          }          }
1114        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1115          if ($self->{escape} and          if ($self->{escape} and
1116              ($self->{content_model} & CM_LIMITED_MARKUP)) { # RCDATA | CDATA              ($self->{content_model} & CM_LIMITED_MARKUP)) { # RCDATA | CDATA
1117            if ($self->{prev_char}->[0] == 0x002D and # -            if ($self->{s_kwd} eq '--') {
               $self->{prev_char}->[1] == 0x002D) { # -  
1118              !!!cp (8);              !!!cp (8);
1119              delete $self->{escape};              delete $self->{escape};
1120            } else {            } else {
# Line 928  sub _get_next_token ($) { Line 1124  sub _get_next_token ($) {
1124            !!!cp (10);            !!!cp (10);
1125          }          }
1126                    
1127            $self->{s_kwd} = '';
1128          #          #
1129        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1130          !!!cp (11);          !!!cp (11);
1131            $self->{s_kwd} = '';
1132          !!!emit ({type => END_OF_FILE_TOKEN,          !!!emit ({type => END_OF_FILE_TOKEN,
1133                    line => $self->{line}, column => $self->{column}});                    line => $self->{line}, column => $self->{column}});
1134          last A; ## TODO: ok?          last A; ## TODO: ok?
1135        } else {        } else {
1136          !!!cp (12);          !!!cp (12);
1137            $self->{s_kwd} = '';
1138            #
1139        }        }
1140    
1141        # Anything else        # Anything else
1142        my $token = {type => CHARACTER_TOKEN,        my $token = {type => CHARACTER_TOKEN,
1143                     data => chr $self->{next_char},                     data => chr $self->{nc},
1144                     line => $self->{line}, column => $self->{column},                     line => $self->{line}, column => $self->{column},
1145                    };                    };
1146        ## Stay in the data state        if ($self->{read_until}->($token->{data}, q[-!<>&],
1147        !!!next-input-character;                                  length $token->{data})) {
1148            $self->{s_kwd} = '';
1149        !!!emit ($token);        }
   
       redo A;  
     } elsif ($self->{state} == ENTITY_DATA_STATE) {  
       ## (cannot happen in CDATA state)  
   
       my ($l, $c) = ($self->{line_prev}, $self->{column_prev});  
         
       my $token = $self->_tokenize_attempt_to_consume_an_entity (0, -1);  
   
       $self->{state} = DATA_STATE;  
       # next-input-character is already done  
1150    
1151        unless (defined $token) {        ## Stay in the data state.
1152          if ($self->{content_model} == PCDATA_CONTENT_MODEL) {
1153          !!!cp (13);          !!!cp (13);
1154          !!!emit ({type => CHARACTER_TOKEN, data => '&',          $self->{state} = PCDATA_STATE;
                   line => $l, column => $c,  
                  });  
1155        } else {        } else {
1156          !!!cp (14);          !!!cp (14);
1157          !!!emit ($token);          ## Stay in the state.
1158        }        }
1159          !!!next-input-character;
1160          !!!emit ($token);
1161        redo A;        redo A;
1162      } elsif ($self->{state} == TAG_OPEN_STATE) {      } elsif ($self->{state} == TAG_OPEN_STATE) {
1163        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
1164          if ($self->{next_char} == 0x002F) { # /          if ($self->{nc} == 0x002F) { # /
1165            !!!cp (15);            !!!cp (15);
1166            !!!next-input-character;            !!!next-input-character;
1167            $self->{state} = CLOSE_TAG_OPEN_STATE;            $self->{state} = CLOSE_TAG_OPEN_STATE;
1168            redo A;            redo A;
1169            } elsif ($self->{nc} == 0x0021) { # !
1170              !!!cp (15.1);
1171              $self->{s_kwd} = '<' unless $self->{escape};
1172              #
1173          } else {          } else {
1174            !!!cp (16);            !!!cp (16);
1175            ## reconsume            #
           $self->{state} = DATA_STATE;  
   
           !!!emit ({type => CHARACTER_TOKEN, data => '<',  
                     line => $self->{line_prev},  
                     column => $self->{column_prev},  
                    });  
   
           redo A;  
1176          }          }
1177    
1178            ## reconsume
1179            $self->{state} = DATA_STATE;
1180            !!!emit ({type => CHARACTER_TOKEN, data => '<',
1181                      line => $self->{line_prev},
1182                      column => $self->{column_prev},
1183                     });
1184            redo A;
1185        } elsif ($self->{content_model} & CM_FULL_MARKUP) { # PCDATA        } elsif ($self->{content_model} & CM_FULL_MARKUP) { # PCDATA
1186          if ($self->{next_char} == 0x0021) { # !          if ($self->{nc} == 0x0021) { # !
1187            !!!cp (17);            !!!cp (17);
1188            $self->{state} = MARKUP_DECLARATION_OPEN_STATE;            $self->{state} = MARKUP_DECLARATION_OPEN_STATE;
1189            !!!next-input-character;            !!!next-input-character;
1190            redo A;            redo A;
1191          } elsif ($self->{next_char} == 0x002F) { # /          } elsif ($self->{nc} == 0x002F) { # /
1192            !!!cp (18);            !!!cp (18);
1193            $self->{state} = CLOSE_TAG_OPEN_STATE;            $self->{state} = CLOSE_TAG_OPEN_STATE;
1194            !!!next-input-character;            !!!next-input-character;
1195            redo A;            redo A;
1196          } elsif (0x0041 <= $self->{next_char} and          } elsif (0x0041 <= $self->{nc} and
1197                   $self->{next_char} <= 0x005A) { # A..Z                   $self->{nc} <= 0x005A) { # A..Z
1198            !!!cp (19);            !!!cp (19);
1199            $self->{current_token}            $self->{ct}
1200              = {type => START_TAG_TOKEN,              = {type => START_TAG_TOKEN,
1201                 tag_name => chr ($self->{next_char} + 0x0020),                 tag_name => chr ($self->{nc} + 0x0020),
1202                 line => $self->{line_prev},                 line => $self->{line_prev},
1203                 column => $self->{column_prev}};                 column => $self->{column_prev}};
1204            $self->{state} = TAG_NAME_STATE;            $self->{state} = TAG_NAME_STATE;
1205            !!!next-input-character;            !!!next-input-character;
1206            redo A;            redo A;
1207          } elsif (0x0061 <= $self->{next_char} and          } elsif (0x0061 <= $self->{nc} and
1208                   $self->{next_char} <= 0x007A) { # a..z                   $self->{nc} <= 0x007A) { # a..z
1209            !!!cp (20);            !!!cp (20);
1210            $self->{current_token} = {type => START_TAG_TOKEN,            $self->{ct} = {type => START_TAG_TOKEN,
1211                                      tag_name => chr ($self->{next_char}),                                      tag_name => chr ($self->{nc}),
1212                                      line => $self->{line_prev},                                      line => $self->{line_prev},
1213                                      column => $self->{column_prev}};                                      column => $self->{column_prev}};
1214            $self->{state} = TAG_NAME_STATE;            $self->{state} = TAG_NAME_STATE;
1215            !!!next-input-character;            !!!next-input-character;
1216            redo A;            redo A;
1217          } elsif ($self->{next_char} == 0x003E) { # >          } elsif ($self->{nc} == 0x003E) { # >
1218            !!!cp (21);            !!!cp (21);
1219            !!!parse-error (type => 'empty start tag',            !!!parse-error (type => 'empty start tag',
1220                            line => $self->{line_prev},                            line => $self->{line_prev},
# Line 1034  sub _get_next_token ($) { Line 1228  sub _get_next_token ($) {
1228                     });                     });
1229    
1230            redo A;            redo A;
1231          } elsif ($self->{next_char} == 0x003F) { # ?          } elsif ($self->{nc} == 0x003F) { # ?
1232            !!!cp (22);            !!!cp (22);
1233            !!!parse-error (type => 'pio',            !!!parse-error (type => 'pio',
1234                            line => $self->{line_prev},                            line => $self->{line_prev},
1235                            column => $self->{column_prev});                            column => $self->{column_prev});
1236            $self->{state} = BOGUS_COMMENT_STATE;            $self->{state} = BOGUS_COMMENT_STATE;
1237            $self->{current_token} = {type => COMMENT_TOKEN, data => '',            $self->{ct} = {type => COMMENT_TOKEN, data => '',
1238                                      line => $self->{line_prev},                                      line => $self->{line_prev},
1239                                      column => $self->{column_prev},                                      column => $self->{column_prev},
1240                                     };                                     };
1241            ## $self->{next_char} is intentionally left as is            ## $self->{nc} is intentionally left as is
1242            redo A;            redo A;
1243          } else {          } else {
1244            !!!cp (23);            !!!cp (23);
# Line 1065  sub _get_next_token ($) { Line 1259  sub _get_next_token ($) {
1259          die "$0: $self->{content_model} in tag open";          die "$0: $self->{content_model} in tag open";
1260        }        }
1261      } elsif ($self->{state} == CLOSE_TAG_OPEN_STATE) {      } elsif ($self->{state} == CLOSE_TAG_OPEN_STATE) {
1262          ## NOTE: The "close tag open state" in the spec is implemented as
1263          ## |CLOSE_TAG_OPEN_STATE| and |CDATA_RCDATA_CLOSE_TAG_STATE|.
1264    
1265        my ($l, $c) = ($self->{line_prev}, $self->{column_prev} - 1); # "<"of"</"        my ($l, $c) = ($self->{line_prev}, $self->{column_prev} - 1); # "<"of"</"
1266        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
1267          if (defined $self->{last_emitted_start_tag_name}) {          if (defined $self->{last_stag_name}) {
1268              $self->{state} = CDATA_RCDATA_CLOSE_TAG_STATE;
1269            ## NOTE: <http://krijnhoetmer.nl/irc-logs/whatwg/20070626#l-564>            $self->{s_kwd} = '';
1270            my @next_char;            ## Reconsume.
1271            TAGNAME: for (my $i = 0; $i < length $self->{last_emitted_start_tag_name}; $i++) {            redo A;
             push @next_char, $self->{next_char};  
             my $c = ord substr ($self->{last_emitted_start_tag_name}, $i, 1);  
             my $C = 0x0061 <= $c && $c <= 0x007A ? $c - 0x0020 : $c;  
             if ($self->{next_char} == $c or $self->{next_char} == $C) {  
               !!!cp (24);  
               !!!next-input-character;  
               next TAGNAME;  
             } else {  
               !!!cp (25);  
               $self->{next_char} = shift @next_char; # reconsume  
               !!!back-next-input-character (@next_char);  
               $self->{state} = DATA_STATE;  
   
               !!!emit ({type => CHARACTER_TOKEN, data => '</',  
                         line => $l, column => $c,  
                        });  
     
               redo A;  
             }  
           }  
           push @next_char, $self->{next_char};  
         
           unless ($self->{next_char} == 0x0009 or # HT  
                   $self->{next_char} == 0x000A or # LF  
                   $self->{next_char} == 0x000B or # VT  
                   $self->{next_char} == 0x000C or # FF  
                   $self->{next_char} == 0x0020 or # SP  
                   $self->{next_char} == 0x003E or # >  
                   $self->{next_char} == 0x002F or # /  
                   $self->{next_char} == -1) {  
             !!!cp (26);  
             $self->{next_char} = shift @next_char; # reconsume  
             !!!back-next-input-character (@next_char);  
             $self->{state} = DATA_STATE;  
             !!!emit ({type => CHARACTER_TOKEN, data => '</',  
                       line => $l, column => $c,  
                      });  
             redo A;  
           } else {  
             !!!cp (27);  
             $self->{next_char} = shift @next_char;  
             !!!back-next-input-character (@next_char);  
             # and consume...  
           }  
1272          } else {          } else {
1273            ## No start tag token has ever been emitted            ## No start tag token has ever been emitted
1274              ## NOTE: See <http://krijnhoetmer.nl/irc-logs/whatwg/20070626#l-564>.
1275            !!!cp (28);            !!!cp (28);
           # next-input-character is already done  
1276            $self->{state} = DATA_STATE;            $self->{state} = DATA_STATE;
1277              ## Reconsume.
1278            !!!emit ({type => CHARACTER_TOKEN, data => '</',            !!!emit ({type => CHARACTER_TOKEN, data => '</',
1279                      line => $l, column => $c,                      line => $l, column => $c,
1280                     });                     });
1281            redo A;            redo A;
1282          }          }
1283        }        }
1284          
1285        if (0x0041 <= $self->{next_char} and        if (0x0041 <= $self->{nc} and
1286            $self->{next_char} <= 0x005A) { # A..Z            $self->{nc} <= 0x005A) { # A..Z
1287          !!!cp (29);          !!!cp (29);
1288          $self->{current_token}          $self->{ct}
1289              = {type => END_TAG_TOKEN,              = {type => END_TAG_TOKEN,
1290                 tag_name => chr ($self->{next_char} + 0x0020),                 tag_name => chr ($self->{nc} + 0x0020),
1291                 line => $l, column => $c};                 line => $l, column => $c};
1292          $self->{state} = TAG_NAME_STATE;          $self->{state} = TAG_NAME_STATE;
1293          !!!next-input-character;          !!!next-input-character;
1294          redo A;          redo A;
1295        } elsif (0x0061 <= $self->{next_char} and        } elsif (0x0061 <= $self->{nc} and
1296                 $self->{next_char} <= 0x007A) { # a..z                 $self->{nc} <= 0x007A) { # a..z
1297          !!!cp (30);          !!!cp (30);
1298          $self->{current_token} = {type => END_TAG_TOKEN,          $self->{ct} = {type => END_TAG_TOKEN,
1299                                    tag_name => chr ($self->{next_char}),                                    tag_name => chr ($self->{nc}),
1300                                    line => $l, column => $c};                                    line => $l, column => $c};
1301          $self->{state} = TAG_NAME_STATE;          $self->{state} = TAG_NAME_STATE;
1302          !!!next-input-character;          !!!next-input-character;
1303          redo A;          redo A;
1304        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1305          !!!cp (31);          !!!cp (31);
1306          !!!parse-error (type => 'empty end tag',          !!!parse-error (type => 'empty end tag',
1307                          line => $self->{line_prev}, ## "<" in "</>"                          line => $self->{line_prev}, ## "<" in "</>"
# Line 1155  sub _get_next_token ($) { Line 1309  sub _get_next_token ($) {
1309          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1310          !!!next-input-character;          !!!next-input-character;
1311          redo A;          redo A;
1312        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1313          !!!cp (32);          !!!cp (32);
1314          !!!parse-error (type => 'bare etago');          !!!parse-error (type => 'bare etago');
1315          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
# Line 1170  sub _get_next_token ($) { Line 1324  sub _get_next_token ($) {
1324          !!!cp (33);          !!!cp (33);
1325          !!!parse-error (type => 'bogus end tag');          !!!parse-error (type => 'bogus end tag');
1326          $self->{state} = BOGUS_COMMENT_STATE;          $self->{state} = BOGUS_COMMENT_STATE;
1327          $self->{current_token} = {type => COMMENT_TOKEN, data => '',          $self->{ct} = {type => COMMENT_TOKEN, data => '',
1328                                    line => $self->{line_prev}, # "<" of "</"                                    line => $self->{line_prev}, # "<" of "</"
1329                                    column => $self->{column_prev} - 1,                                    column => $self->{column_prev} - 1,
1330                                   };                                   };
1331          ## $self->{next_char} is intentionally left as is          ## NOTE: $self->{nc} is intentionally left as is.
1332          redo A;          ## Although the "anything else" case of the spec not explicitly
1333            ## states that the next input character is to be reconsumed,
1334            ## it will be included to the |data| of the comment token
1335            ## generated from the bogus end tag, as defined in the
1336            ## "bogus comment state" entry.
1337            redo A;
1338          }
1339        } elsif ($self->{state} == CDATA_RCDATA_CLOSE_TAG_STATE) {
1340          my $ch = substr $self->{last_stag_name}, length $self->{s_kwd}, 1;
1341          if (length $ch) {
1342            my $CH = $ch;
1343            $ch =~ tr/a-z/A-Z/;
1344            my $nch = chr $self->{nc};
1345            if ($nch eq $ch or $nch eq $CH) {
1346              !!!cp (24);
1347              ## Stay in the state.
1348              $self->{s_kwd} .= $nch;
1349              !!!next-input-character;
1350              redo A;
1351            } else {
1352              !!!cp (25);
1353              $self->{state} = DATA_STATE;
1354              ## Reconsume.
1355              !!!emit ({type => CHARACTER_TOKEN,
1356                        data => '</' . $self->{s_kwd},
1357                        line => $self->{line_prev},
1358                        column => $self->{column_prev} - 1 - length $self->{s_kwd},
1359                       });
1360              redo A;
1361            }
1362          } else { # after "<{tag-name}"
1363            unless ($is_space->{$self->{nc}} or
1364                    {
1365                     0x003E => 1, # >
1366                     0x002F => 1, # /
1367                     -1 => 1, # EOF
1368                    }->{$self->{nc}}) {
1369              !!!cp (26);
1370              ## Reconsume.
1371              $self->{state} = DATA_STATE;
1372              !!!emit ({type => CHARACTER_TOKEN,
1373                        data => '</' . $self->{s_kwd},
1374                        line => $self->{line_prev},
1375                        column => $self->{column_prev} - 1 - length $self->{s_kwd},
1376                       });
1377              redo A;
1378            } else {
1379              !!!cp (27);
1380              $self->{ct}
1381                  = {type => END_TAG_TOKEN,
1382                     tag_name => $self->{last_stag_name},
1383                     line => $self->{line_prev},
1384                     column => $self->{column_prev} - 1 - length $self->{s_kwd}};
1385              $self->{state} = TAG_NAME_STATE;
1386              ## Reconsume.
1387              redo A;
1388            }
1389        }        }
1390      } elsif ($self->{state} == TAG_NAME_STATE) {      } elsif ($self->{state} == TAG_NAME_STATE) {
1391        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
1392          !!!cp (34);          !!!cp (34);
1393          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1394          !!!next-input-character;          !!!next-input-character;
1395          redo A;          redo A;
1396        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1397          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1398            !!!cp (35);            !!!cp (35);
1399            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1400          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1401            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1402            #if ($self->{current_token}->{attributes}) {            #if ($self->{ct}->{attributes}) {
1403            #  ## NOTE: This should never be reached.            #  ## NOTE: This should never be reached.
1404            #  !!! cp (36);            #  !!! cp (36);
1405            #  !!! parse-error (type => 'end tag attribute');            #  !!! parse-error (type => 'end tag attribute');
# Line 1201  sub _get_next_token ($) { Line 1407  sub _get_next_token ($) {
1407              !!!cp (37);              !!!cp (37);
1408            #}            #}
1409          } else {          } else {
1410            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1411          }          }
1412          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1413          !!!next-input-character;          !!!next-input-character;
1414    
1415          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1416    
1417          redo A;          redo A;
1418        } elsif (0x0041 <= $self->{next_char} and        } elsif (0x0041 <= $self->{nc} and
1419                 $self->{next_char} <= 0x005A) { # A..Z                 $self->{nc} <= 0x005A) { # A..Z
1420          !!!cp (38);          !!!cp (38);
1421          $self->{current_token}->{tag_name} .= chr ($self->{next_char} + 0x0020);          $self->{ct}->{tag_name} .= chr ($self->{nc} + 0x0020);
1422            # start tag or end tag            # start tag or end tag
1423          ## Stay in this state          ## Stay in this state
1424          !!!next-input-character;          !!!next-input-character;
1425          redo A;          redo A;
1426        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1427          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1428          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1429            !!!cp (39);            !!!cp (39);
1430            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1431          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1432            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1433            #if ($self->{current_token}->{attributes}) {            #if ($self->{ct}->{attributes}) {
1434            #  ## NOTE: This state should never be reached.            #  ## NOTE: This state should never be reached.
1435            #  !!! cp (40);            #  !!! cp (40);
1436            #  !!! parse-error (type => 'end tag attribute');            #  !!! parse-error (type => 'end tag attribute');
# Line 1232  sub _get_next_token ($) { Line 1438  sub _get_next_token ($) {
1438              !!!cp (41);              !!!cp (41);
1439            #}            #}
1440          } else {          } else {
1441            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1442          }          }
1443          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1444          # reconsume          # reconsume
1445    
1446          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1447    
1448          redo A;          redo A;
1449        } elsif ($self->{next_char} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
1450          !!!cp (42);          !!!cp (42);
1451          $self->{state} = SELF_CLOSING_START_TAG_STATE;          $self->{state} = SELF_CLOSING_START_TAG_STATE;
1452          !!!next-input-character;          !!!next-input-character;
1453          redo A;          redo A;
1454        } else {        } else {
1455          !!!cp (44);          !!!cp (44);
1456          $self->{current_token}->{tag_name} .= chr $self->{next_char};          $self->{ct}->{tag_name} .= chr $self->{nc};
1457            # start tag or end tag            # start tag or end tag
1458          ## Stay in the state          ## Stay in the state
1459          !!!next-input-character;          !!!next-input-character;
1460          redo A;          redo A;
1461        }        }
1462      } elsif ($self->{state} == BEFORE_ATTRIBUTE_NAME_STATE) {      } elsif ($self->{state} == BEFORE_ATTRIBUTE_NAME_STATE) {
1463        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
1464          !!!cp (45);          !!!cp (45);
1465          ## Stay in the state          ## Stay in the state
1466          !!!next-input-character;          !!!next-input-character;
1467          redo A;          redo A;
1468        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1469          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1470            !!!cp (46);            !!!cp (46);
1471            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1472          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1473            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1474            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1475              !!!cp (47);              !!!cp (47);
1476              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1477            } else {            } else {
1478              !!!cp (48);              !!!cp (48);
1479            }            }
1480          } else {          } else {
1481            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1482          }          }
1483          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1484          !!!next-input-character;          !!!next-input-character;
1485    
1486          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1487    
1488          redo A;          redo A;
1489        } elsif (0x0041 <= $self->{next_char} and        } elsif (0x0041 <= $self->{nc} and
1490                 $self->{next_char} <= 0x005A) { # A..Z                 $self->{nc} <= 0x005A) { # A..Z
1491          !!!cp (49);          !!!cp (49);
1492          $self->{current_attribute}          $self->{ca}
1493              = {name => chr ($self->{next_char} + 0x0020),              = {name => chr ($self->{nc} + 0x0020),
1494                 value => '',                 value => '',
1495                 line => $self->{line}, column => $self->{column}};                 line => $self->{line}, column => $self->{column}};
1496          $self->{state} = ATTRIBUTE_NAME_STATE;          $self->{state} = ATTRIBUTE_NAME_STATE;
1497          !!!next-input-character;          !!!next-input-character;
1498          redo A;          redo A;
1499        } elsif ($self->{next_char} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
1500          !!!cp (50);          !!!cp (50);
1501          $self->{state} = SELF_CLOSING_START_TAG_STATE;          $self->{state} = SELF_CLOSING_START_TAG_STATE;
1502          !!!next-input-character;          !!!next-input-character;
1503          redo A;          redo A;
1504        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1505          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1506          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1507            !!!cp (52);            !!!cp (52);
1508            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1509          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1510            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1511            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1512              !!!cp (53);              !!!cp (53);
1513              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1514            } else {            } else {
1515              !!!cp (54);              !!!cp (54);
1516            }            }
1517          } else {          } else {
1518            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1519          }          }
1520          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1521          # reconsume          # reconsume
1522    
1523          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1524    
1525          redo A;          redo A;
1526        } else {        } else {
# Line 1326  sub _get_next_token ($) { Line 1528  sub _get_next_token ($) {
1528               0x0022 => 1, # "               0x0022 => 1, # "
1529               0x0027 => 1, # '               0x0027 => 1, # '
1530               0x003D => 1, # =               0x003D => 1, # =
1531              }->{$self->{next_char}}) {              }->{$self->{nc}}) {
1532            !!!cp (55);            !!!cp (55);
1533            !!!parse-error (type => 'bad attribute name');            !!!parse-error (type => 'bad attribute name');
1534          } else {          } else {
1535            !!!cp (56);            !!!cp (56);
1536          }          }
1537          $self->{current_attribute}          $self->{ca}
1538              = {name => chr ($self->{next_char}),              = {name => chr ($self->{nc}),
1539                 value => '',                 value => '',
1540                 line => $self->{line}, column => $self->{column}};                 line => $self->{line}, column => $self->{column}};
1541          $self->{state} = ATTRIBUTE_NAME_STATE;          $self->{state} = ATTRIBUTE_NAME_STATE;
# Line 1342  sub _get_next_token ($) { Line 1544  sub _get_next_token ($) {
1544        }        }
1545      } elsif ($self->{state} == ATTRIBUTE_NAME_STATE) {      } elsif ($self->{state} == ATTRIBUTE_NAME_STATE) {
1546        my $before_leave = sub {        my $before_leave = sub {
1547          if (exists $self->{current_token}->{attributes} # start tag or end tag          if (exists $self->{ct}->{attributes} # start tag or end tag
1548              ->{$self->{current_attribute}->{name}}) { # MUST              ->{$self->{ca}->{name}}) { # MUST
1549            !!!cp (57);            !!!cp (57);
1550            !!!parse-error (type => 'duplicate attribute:'.$self->{current_attribute}->{name}, line => $self->{current_attribute}->{line}, column => $self->{current_attribute}->{column});            !!!parse-error (type => 'duplicate attribute', text => $self->{ca}->{name}, line => $self->{ca}->{line}, column => $self->{ca}->{column});
1551            ## Discard $self->{current_attribute} # MUST            ## Discard $self->{ca} # MUST
1552          } else {          } else {
1553            !!!cp (58);            !!!cp (58);
1554            $self->{current_token}->{attributes}->{$self->{current_attribute}->{name}}            $self->{ct}->{attributes}->{$self->{ca}->{name}}
1555              = $self->{current_attribute};              = $self->{ca};
1556          }          }
1557        }; # $before_leave        }; # $before_leave
1558    
1559        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
1560          !!!cp (59);          !!!cp (59);
1561          $before_leave->();          $before_leave->();
1562          $self->{state} = AFTER_ATTRIBUTE_NAME_STATE;          $self->{state} = AFTER_ATTRIBUTE_NAME_STATE;
1563          !!!next-input-character;          !!!next-input-character;
1564          redo A;          redo A;
1565        } elsif ($self->{next_char} == 0x003D) { # =        } elsif ($self->{nc} == 0x003D) { # =
1566          !!!cp (60);          !!!cp (60);
1567          $before_leave->();          $before_leave->();
1568          $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;          $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;
1569          !!!next-input-character;          !!!next-input-character;
1570          redo A;          redo A;
1571        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1572          $before_leave->();          $before_leave->();
1573          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1574            !!!cp (61);            !!!cp (61);
1575            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1576          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1577            !!!cp (62);            !!!cp (62);
1578            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1579            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1580              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1581            }            }
1582          } else {          } else {
1583            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1584          }          }
1585          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1586          !!!next-input-character;          !!!next-input-character;
1587    
1588          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1589    
1590          redo A;          redo A;
1591        } elsif (0x0041 <= $self->{next_char} and        } elsif (0x0041 <= $self->{nc} and
1592                 $self->{next_char} <= 0x005A) { # A..Z                 $self->{nc} <= 0x005A) { # A..Z
1593          !!!cp (63);          !!!cp (63);
1594          $self->{current_attribute}->{name} .= chr ($self->{next_char} + 0x0020);          $self->{ca}->{name} .= chr ($self->{nc} + 0x0020);
1595          ## Stay in the state          ## Stay in the state
1596          !!!next-input-character;          !!!next-input-character;
1597          redo A;          redo A;
1598        } elsif ($self->{next_char} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
1599          !!!cp (64);          !!!cp (64);
1600          $before_leave->();          $before_leave->();
1601          $self->{state} = SELF_CLOSING_START_TAG_STATE;          $self->{state} = SELF_CLOSING_START_TAG_STATE;
1602          !!!next-input-character;          !!!next-input-character;
1603          redo A;          redo A;
1604        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1605          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1606          $before_leave->();          $before_leave->();
1607          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1608            !!!cp (66);            !!!cp (66);
1609            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1610          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1611            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1612            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1613              !!!cp (67);              !!!cp (67);
1614              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1615            } else {            } else {
# Line 1419  sub _get_next_token ($) { Line 1617  sub _get_next_token ($) {
1617              !!!cp (68);              !!!cp (68);
1618            }            }
1619          } else {          } else {
1620            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1621          }          }
1622          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1623          # reconsume          # reconsume
1624    
1625          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1626    
1627          redo A;          redo A;
1628        } else {        } else {
1629          if ($self->{next_char} == 0x0022 or # "          if ($self->{nc} == 0x0022 or # "
1630              $self->{next_char} == 0x0027) { # '              $self->{nc} == 0x0027) { # '
1631            !!!cp (69);            !!!cp (69);
1632            !!!parse-error (type => 'bad attribute name');            !!!parse-error (type => 'bad attribute name');
1633          } else {          } else {
1634            !!!cp (70);            !!!cp (70);
1635          }          }
1636          $self->{current_attribute}->{name} .= chr ($self->{next_char});          $self->{ca}->{name} .= chr ($self->{nc});
1637          ## Stay in the state          ## Stay in the state
1638          !!!next-input-character;          !!!next-input-character;
1639          redo A;          redo A;
1640        }        }
1641      } elsif ($self->{state} == AFTER_ATTRIBUTE_NAME_STATE) {      } elsif ($self->{state} == AFTER_ATTRIBUTE_NAME_STATE) {
1642        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
1643          !!!cp (71);          !!!cp (71);
1644          ## Stay in the state          ## Stay in the state
1645          !!!next-input-character;          !!!next-input-character;
1646          redo A;          redo A;
1647        } elsif ($self->{next_char} == 0x003D) { # =        } elsif ($self->{nc} == 0x003D) { # =
1648          !!!cp (72);          !!!cp (72);
1649          $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;          $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;
1650          !!!next-input-character;          !!!next-input-character;
1651          redo A;          redo A;
1652        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1653          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1654            !!!cp (73);            !!!cp (73);
1655            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1656          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1657            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1658            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1659              !!!cp (74);              !!!cp (74);
1660              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1661            } else {            } else {
# Line 1469  sub _get_next_token ($) { Line 1663  sub _get_next_token ($) {
1663              !!!cp (75);              !!!cp (75);
1664            }            }
1665          } else {          } else {
1666            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1667          }          }
1668          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1669          !!!next-input-character;          !!!next-input-character;
1670    
1671          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1672    
1673          redo A;          redo A;
1674        } elsif (0x0041 <= $self->{next_char} and        } elsif (0x0041 <= $self->{nc} and
1675                 $self->{next_char} <= 0x005A) { # A..Z                 $self->{nc} <= 0x005A) { # A..Z
1676          !!!cp (76);          !!!cp (76);
1677          $self->{current_attribute}          $self->{ca}
1678              = {name => chr ($self->{next_char} + 0x0020),              = {name => chr ($self->{nc} + 0x0020),
1679                 value => '',                 value => '',
1680                 line => $self->{line}, column => $self->{column}};                 line => $self->{line}, column => $self->{column}};
1681          $self->{state} = ATTRIBUTE_NAME_STATE;          $self->{state} = ATTRIBUTE_NAME_STATE;
1682          !!!next-input-character;          !!!next-input-character;
1683          redo A;          redo A;
1684        } elsif ($self->{next_char} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
1685          !!!cp (77);          !!!cp (77);
1686          $self->{state} = SELF_CLOSING_START_TAG_STATE;          $self->{state} = SELF_CLOSING_START_TAG_STATE;
1687          !!!next-input-character;          !!!next-input-character;
1688          redo A;          redo A;
1689        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1690          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1691          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1692            !!!cp (79);            !!!cp (79);
1693            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1694          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1695            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1696            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1697              !!!cp (80);              !!!cp (80);
1698              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1699            } else {            } else {
# Line 1507  sub _get_next_token ($) { Line 1701  sub _get_next_token ($) {
1701              !!!cp (81);              !!!cp (81);
1702            }            }
1703          } else {          } else {
1704            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1705          }          }
1706          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1707          # reconsume          # reconsume
1708    
1709          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1710    
1711          redo A;          redo A;
1712        } else {        } else {
1713          !!!cp (82);          if ($self->{nc} == 0x0022 or # "
1714          $self->{current_attribute}              $self->{nc} == 0x0027) { # '
1715              = {name => chr ($self->{next_char}),            !!!cp (78);
1716              !!!parse-error (type => 'bad attribute name');
1717            } else {
1718              !!!cp (82);
1719            }
1720            $self->{ca}
1721                = {name => chr ($self->{nc}),
1722                 value => '',                 value => '',
1723                 line => $self->{line}, column => $self->{column}};                 line => $self->{line}, column => $self->{column}};
1724          $self->{state} = ATTRIBUTE_NAME_STATE;          $self->{state} = ATTRIBUTE_NAME_STATE;
# Line 1526  sub _get_next_token ($) { Line 1726  sub _get_next_token ($) {
1726          redo A;                  redo A;        
1727        }        }
1728      } elsif ($self->{state} == BEFORE_ATTRIBUTE_VALUE_STATE) {      } elsif ($self->{state} == BEFORE_ATTRIBUTE_VALUE_STATE) {
1729        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP        
1730          !!!cp (83);          !!!cp (83);
1731          ## Stay in the state          ## Stay in the state
1732          !!!next-input-character;          !!!next-input-character;
1733          redo A;          redo A;
1734        } elsif ($self->{next_char} == 0x0022) { # "        } elsif ($self->{nc} == 0x0022) { # "
1735          !!!cp (84);          !!!cp (84);
1736          $self->{state} = ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE;          $self->{state} = ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE;
1737          !!!next-input-character;          !!!next-input-character;
1738          redo A;          redo A;
1739        } elsif ($self->{next_char} == 0x0026) { # &        } elsif ($self->{nc} == 0x0026) { # &
1740          !!!cp (85);          !!!cp (85);
1741          $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;          $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;
1742          ## reconsume          ## reconsume
1743          redo A;          redo A;
1744        } elsif ($self->{next_char} == 0x0027) { # '        } elsif ($self->{nc} == 0x0027) { # '
1745          !!!cp (86);          !!!cp (86);
1746          $self->{state} = ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE;          $self->{state} = ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE;
1747          !!!next-input-character;          !!!next-input-character;
1748          redo A;          redo A;
1749        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1750          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          !!!parse-error (type => 'empty unquoted attribute value');
1751            if ($self->{ct}->{type} == START_TAG_TOKEN) {
1752            !!!cp (87);            !!!cp (87);
1753            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1754          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1755            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1756            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1757              !!!cp (88);              !!!cp (88);
1758              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1759            } else {            } else {
# Line 1564  sub _get_next_token ($) { Line 1761  sub _get_next_token ($) {
1761              !!!cp (89);              !!!cp (89);
1762            }            }
1763          } else {          } else {
1764            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1765          }          }
1766          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1767          !!!next-input-character;          !!!next-input-character;
1768    
1769          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1770    
1771          redo A;          redo A;
1772        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1773          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1774          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1775            !!!cp (90);            !!!cp (90);
1776            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1777          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1778            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1779            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1780              !!!cp (91);              !!!cp (91);
1781              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1782            } else {            } else {
# Line 1587  sub _get_next_token ($) { Line 1784  sub _get_next_token ($) {
1784              !!!cp (92);              !!!cp (92);
1785            }            }
1786          } else {          } else {
1787            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1788          }          }
1789          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1790          ## reconsume          ## reconsume
1791    
1792          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1793    
1794          redo A;          redo A;
1795        } else {        } else {
1796          if ($self->{next_char} == 0x003D) { # =          if ($self->{nc} == 0x003D) { # =
1797            !!!cp (93);            !!!cp (93);
1798            !!!parse-error (type => 'bad attribute value');            !!!parse-error (type => 'bad attribute value');
1799          } else {          } else {
1800            !!!cp (94);            !!!cp (94);
1801          }          }
1802          $self->{current_attribute}->{value} .= chr ($self->{next_char});          $self->{ca}->{value} .= chr ($self->{nc});
1803          $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;          $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;
1804          !!!next-input-character;          !!!next-input-character;
1805          redo A;          redo A;
1806        }        }
1807      } elsif ($self->{state} == ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE) {      } elsif ($self->{state} == ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE) {
1808        if ($self->{next_char} == 0x0022) { # "        if ($self->{nc} == 0x0022) { # "
1809          !!!cp (95);          !!!cp (95);
1810          $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;          $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;
1811          !!!next-input-character;          !!!next-input-character;
1812          redo A;          redo A;
1813        } elsif ($self->{next_char} == 0x0026) { # &        } elsif ($self->{nc} == 0x0026) { # &
1814          !!!cp (96);          !!!cp (96);
1815          $self->{last_attribute_value_state} = $self->{state};          ## NOTE: In the spec, the tokenizer is switched to the
1816          $self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE;          ## "entity in attribute value state".  In this implementation, the
1817            ## tokenizer is switched to the |ENTITY_STATE|, which is an
1818            ## implementation of the "consume a character reference" algorithm.
1819            $self->{prev_state} = $self->{state};
1820            $self->{entity_add} = 0x0022; # "
1821            $self->{state} = ENTITY_STATE;
1822          !!!next-input-character;          !!!next-input-character;
1823          redo A;          redo A;
1824        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1825          !!!parse-error (type => 'unclosed attribute value');          !!!parse-error (type => 'unclosed attribute value');
1826          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1827            !!!cp (97);            !!!cp (97);
1828            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1829          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1830            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1831            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1832              !!!cp (98);              !!!cp (98);
1833              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1834            } else {            } else {
# Line 1634  sub _get_next_token ($) { Line 1836  sub _get_next_token ($) {
1836              !!!cp (99);              !!!cp (99);
1837            }            }
1838          } else {          } else {
1839            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1840          }          }
1841          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1842          ## reconsume          ## reconsume
1843    
1844          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1845    
1846          redo A;          redo A;
1847        } else {        } else {
1848          !!!cp (100);          !!!cp (100);
1849          $self->{current_attribute}->{value} .= chr ($self->{next_char});          $self->{ca}->{value} .= chr ($self->{nc});
1850            $self->{read_until}->($self->{ca}->{value},
1851                                  q["&],
1852                                  length $self->{ca}->{value});
1853    
1854          ## Stay in the state          ## Stay in the state
1855          !!!next-input-character;          !!!next-input-character;
1856          redo A;          redo A;
1857        }        }
1858      } elsif ($self->{state} == ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE) {      } elsif ($self->{state} == ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE) {
1859        if ($self->{next_char} == 0x0027) { # '        if ($self->{nc} == 0x0027) { # '
1860          !!!cp (101);          !!!cp (101);
1861          $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;          $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;
1862          !!!next-input-character;          !!!next-input-character;
1863          redo A;          redo A;
1864        } elsif ($self->{next_char} == 0x0026) { # &        } elsif ($self->{nc} == 0x0026) { # &
1865          !!!cp (102);          !!!cp (102);
1866          $self->{last_attribute_value_state} = $self->{state};          ## NOTE: In the spec, the tokenizer is switched to the
1867          $self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE;          ## "entity in attribute value state".  In this implementation, the
1868            ## tokenizer is switched to the |ENTITY_STATE|, which is an
1869            ## implementation of the "consume a character reference" algorithm.
1870            $self->{entity_add} = 0x0027; # '
1871            $self->{prev_state} = $self->{state};
1872            $self->{state} = ENTITY_STATE;
1873          !!!next-input-character;          !!!next-input-character;
1874          redo A;          redo A;
1875        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1876          !!!parse-error (type => 'unclosed attribute value');          !!!parse-error (type => 'unclosed attribute value');
1877          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1878            !!!cp (103);            !!!cp (103);
1879            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1880          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1881            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1882            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1883              !!!cp (104);              !!!cp (104);
1884              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1885            } else {            } else {
# Line 1676  sub _get_next_token ($) { Line 1887  sub _get_next_token ($) {
1887              !!!cp (105);              !!!cp (105);
1888            }            }
1889          } else {          } else {
1890            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1891          }          }
1892          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1893          ## reconsume          ## reconsume
1894    
1895          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1896    
1897          redo A;          redo A;
1898        } else {        } else {
1899          !!!cp (106);          !!!cp (106);
1900          $self->{current_attribute}->{value} .= chr ($self->{next_char});          $self->{ca}->{value} .= chr ($self->{nc});
1901            $self->{read_until}->($self->{ca}->{value},
1902                                  q['&],
1903                                  length $self->{ca}->{value});
1904    
1905          ## Stay in the state          ## Stay in the state
1906          !!!next-input-character;          !!!next-input-character;
1907          redo A;          redo A;
1908        }        }
1909      } elsif ($self->{state} == ATTRIBUTE_VALUE_UNQUOTED_STATE) {      } elsif ($self->{state} == ATTRIBUTE_VALUE_UNQUOTED_STATE) {
1910        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # HT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
1911          !!!cp (107);          !!!cp (107);
1912          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1913          !!!next-input-character;          !!!next-input-character;
1914          redo A;          redo A;
1915        } elsif ($self->{next_char} == 0x0026) { # &        } elsif ($self->{nc} == 0x0026) { # &
1916          !!!cp (108);          !!!cp (108);
1917          $self->{last_attribute_value_state} = $self->{state};          ## NOTE: In the spec, the tokenizer is switched to the
1918          $self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE;          ## "entity in attribute value state".  In this implementation, the
1919            ## tokenizer is switched to the |ENTITY_STATE|, which is an
1920            ## implementation of the "consume a character reference" algorithm.
1921            $self->{entity_add} = -1;
1922            $self->{prev_state} = $self->{state};
1923            $self->{state} = ENTITY_STATE;
1924          !!!next-input-character;          !!!next-input-character;
1925          redo A;          redo A;
1926        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1927          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1928            !!!cp (109);            !!!cp (109);
1929            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1930          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1931            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1932            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1933              !!!cp (110);              !!!cp (110);
1934              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1935            } else {            } else {
# Line 1721  sub _get_next_token ($) { Line 1937  sub _get_next_token ($) {
1937              !!!cp (111);              !!!cp (111);
1938            }            }
1939          } else {          } else {
1940            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1941          }          }
1942          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1943          !!!next-input-character;          !!!next-input-character;
1944    
1945          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1946    
1947          redo A;          redo A;
1948        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1949          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1950          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1951            !!!cp (112);            !!!cp (112);
1952            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1953          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1954            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1955            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1956              !!!cp (113);              !!!cp (113);
1957              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1958            } else {            } else {
# Line 1744  sub _get_next_token ($) { Line 1960  sub _get_next_token ($) {
1960              !!!cp (114);              !!!cp (114);
1961            }            }
1962          } else {          } else {
1963            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1964          }          }
1965          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1966          ## reconsume          ## reconsume
1967    
1968          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1969    
1970          redo A;          redo A;
1971        } else {        } else {
# Line 1757  sub _get_next_token ($) { Line 1973  sub _get_next_token ($) {
1973               0x0022 => 1, # "               0x0022 => 1, # "
1974               0x0027 => 1, # '               0x0027 => 1, # '
1975               0x003D => 1, # =               0x003D => 1, # =
1976              }->{$self->{next_char}}) {              }->{$self->{nc}}) {
1977            !!!cp (115);            !!!cp (115);
1978            !!!parse-error (type => 'bad attribute value');            !!!parse-error (type => 'bad attribute value');
1979          } else {          } else {
1980            !!!cp (116);            !!!cp (116);
1981          }          }
1982          $self->{current_attribute}->{value} .= chr ($self->{next_char});          $self->{ca}->{value} .= chr ($self->{nc});
1983            $self->{read_until}->($self->{ca}->{value},
1984                                  q["'=& >],
1985                                  length $self->{ca}->{value});
1986    
1987          ## Stay in the state          ## Stay in the state
1988          !!!next-input-character;          !!!next-input-character;
1989          redo A;          redo A;
1990        }        }
     } elsif ($self->{state} == ENTITY_IN_ATTRIBUTE_VALUE_STATE) {  
       my $token = $self->_tokenize_attempt_to_consume_an_entity  
           (1,  
            $self->{last_attribute_value_state}  
              == ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE ? 0x0022 : # "  
            $self->{last_attribute_value_state}  
              == ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE ? 0x0027 : # '  
            -1);  
   
       unless (defined $token) {  
         !!!cp (117);  
         $self->{current_attribute}->{value} .= '&';  
       } else {  
         !!!cp (118);  
         $self->{current_attribute}->{value} .= $token->{data};  
         $self->{current_attribute}->{has_reference} = $token->{has_reference};  
         ## ISSUE: spec says "append the returned character token to the current attribute's value"  
       }  
   
       $self->{state} = $self->{last_attribute_value_state};  
       # next-input-character is already done  
       redo A;  
1991      } elsif ($self->{state} == AFTER_ATTRIBUTE_VALUE_QUOTED_STATE) {      } elsif ($self->{state} == AFTER_ATTRIBUTE_VALUE_QUOTED_STATE) {
1992        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
1993          !!!cp (118);          !!!cp (118);
1994          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1995          !!!next-input-character;          !!!next-input-character;
1996          redo A;          redo A;
1997        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1998          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1999            !!!cp (119);            !!!cp (119);
2000            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
2001          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
2002            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
2003            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
2004              !!!cp (120);              !!!cp (120);
2005              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
2006            } else {            } else {
# Line 1814  sub _get_next_token ($) { Line 2008  sub _get_next_token ($) {
2008              !!!cp (121);              !!!cp (121);
2009            }            }
2010          } else {          } else {
2011            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
2012          }          }
2013          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2014          !!!next-input-character;          !!!next-input-character;
2015    
2016          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
2017    
2018          redo A;          redo A;
2019        } elsif ($self->{next_char} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
2020          !!!cp (122);          !!!cp (122);
2021          $self->{state} = SELF_CLOSING_START_TAG_STATE;          $self->{state} = SELF_CLOSING_START_TAG_STATE;
2022          !!!next-input-character;          !!!next-input-character;
2023          redo A;          redo A;
2024        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2025          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
2026          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
2027            !!!cp (122.3);            !!!cp (122.3);
2028            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
2029          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
2030            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
2031              !!!cp (122.1);              !!!cp (122.1);
2032              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
2033            } else {            } else {
# Line 1841  sub _get_next_token ($) { Line 2035  sub _get_next_token ($) {
2035              !!!cp (122.2);              !!!cp (122.2);
2036            }            }
2037          } else {          } else {
2038            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
2039          }          }
2040          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2041          ## Reconsume.          ## Reconsume.
2042          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
2043          redo A;          redo A;
2044        } else {        } else {
2045          !!!cp ('124.1');          !!!cp ('124.1');
# Line 1855  sub _get_next_token ($) { Line 2049  sub _get_next_token ($) {
2049          redo A;          redo A;
2050        }        }
2051      } elsif ($self->{state} == SELF_CLOSING_START_TAG_STATE) {      } elsif ($self->{state} == SELF_CLOSING_START_TAG_STATE) {
2052        if ($self->{next_char} == 0x003E) { # >        if ($self->{nc} == 0x003E) { # >
2053          if ($self->{current_token}->{type} == END_TAG_TOKEN) {          if ($self->{ct}->{type} == END_TAG_TOKEN) {
2054            !!!cp ('124.2');            !!!cp ('124.2');
2055            !!!parse-error (type => 'nestc', token => $self->{current_token});            !!!parse-error (type => 'nestc', token => $self->{ct});
2056            ## TODO: Different type than slash in start tag            ## TODO: Different type than slash in start tag
2057            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
2058            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
2059              !!!cp ('124.4');              !!!cp ('124.4');
2060              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
2061            } else {            } else {
# Line 1876  sub _get_next_token ($) { Line 2070  sub _get_next_token ($) {
2070          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2071          !!!next-input-character;          !!!next-input-character;
2072    
2073          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
2074    
2075          redo A;          redo A;
2076        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2077          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
2078          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
2079            !!!cp (124.7);            !!!cp (124.7);
2080            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
2081          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
2082            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
2083              !!!cp (124.5);              !!!cp (124.5);
2084              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
2085            } else {            } else {
# Line 1893  sub _get_next_token ($) { Line 2087  sub _get_next_token ($) {
2087              !!!cp (124.6);              !!!cp (124.6);
2088            }            }
2089          } else {          } else {
2090            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
2091          }          }
2092          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2093          ## Reconsume.          ## Reconsume.
2094          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
2095          redo A;          redo A;
2096        } else {        } else {
2097          !!!cp ('124.4');          !!!cp ('124.4');
# Line 1909  sub _get_next_token ($) { Line 2103  sub _get_next_token ($) {
2103        }        }
2104      } elsif ($self->{state} == BOGUS_COMMENT_STATE) {      } elsif ($self->{state} == BOGUS_COMMENT_STATE) {
2105        ## (only happen if PCDATA state)        ## (only happen if PCDATA state)
         
       ## NOTE: Set by the previous state  
       #my $token = {type => COMMENT_TOKEN, data => ''};  
   
       BC: {  
         if ($self->{next_char} == 0x003E) { # >  
           !!!cp (124);  
           $self->{state} = DATA_STATE;  
           !!!next-input-character;  
2106    
2107            !!!emit ($self->{current_token}); # comment        ## NOTE: Unlike spec's "bogus comment state", this implementation
2108          ## consumes characters one-by-one basis.
2109            redo A;        
2110          } elsif ($self->{next_char} == -1) {        if ($self->{nc} == 0x003E) { # >
2111            !!!cp (125);          !!!cp (124);
2112            $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2113            ## reconsume          !!!next-input-character;
2114    
2115            !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2116            redo A;
2117          } elsif ($self->{nc} == -1) {
2118            !!!cp (125);
2119            $self->{state} = DATA_STATE;
2120            ## reconsume
2121    
2122            redo A;          !!!emit ($self->{ct}); # comment
2123          } else {          redo A;
2124            !!!cp (126);        } else {
2125            $self->{current_token}->{data} .= chr ($self->{next_char}); # comment          !!!cp (126);
2126            !!!next-input-character;          $self->{ct}->{data} .= chr ($self->{nc}); # comment
2127            redo BC;          $self->{read_until}->($self->{ct}->{data},
2128          }                                q[>],
2129        } # BC                                length $self->{ct}->{data});
2130    
2131        die "$0: _get_next_token: unexpected case [BC]";          ## Stay in the state.
2132            !!!next-input-character;
2133            redo A;
2134          }
2135      } elsif ($self->{state} == MARKUP_DECLARATION_OPEN_STATE) {      } elsif ($self->{state} == MARKUP_DECLARATION_OPEN_STATE) {
2136        ## (only happen if PCDATA state)        ## (only happen if PCDATA state)
   
       my ($l, $c) = ($self->{line_prev}, $self->{column_prev} - 1);  
   
       my @next_char;  
       push @next_char, $self->{next_char};  
2137                
2138        if ($self->{next_char} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2139            !!!cp (133);
2140            $self->{state} = MD_HYPHEN_STATE;
2141          !!!next-input-character;          !!!next-input-character;
2142          push @next_char, $self->{next_char};          redo A;
2143          if ($self->{next_char} == 0x002D) { # -        } elsif ($self->{nc} == 0x0044 or # D
2144            !!!cp (127);                 $self->{nc} == 0x0064) { # d
2145            $self->{current_token} = {type => COMMENT_TOKEN, data => '',          ## ASCII case-insensitive.
2146                                      line => $l, column => $c,          !!!cp (130);
2147                                     };          $self->{state} = MD_DOCTYPE_STATE;
2148            $self->{state} = COMMENT_START_STATE;          $self->{s_kwd} = chr $self->{nc};
           !!!next-input-character;  
           redo A;  
         } else {  
           !!!cp (128);  
         }  
       } elsif ($self->{next_char} == 0x0044 or # D  
                $self->{next_char} == 0x0064) { # d  
2149          !!!next-input-character;          !!!next-input-character;
2150          push @next_char, $self->{next_char};          redo A;
         if ($self->{next_char} == 0x004F or # O  
             $self->{next_char} == 0x006F) { # o  
           !!!next-input-character;  
           push @next_char, $self->{next_char};  
           if ($self->{next_char} == 0x0043 or # C  
               $self->{next_char} == 0x0063) { # c  
             !!!next-input-character;  
             push @next_char, $self->{next_char};  
             if ($self->{next_char} == 0x0054 or # T  
                 $self->{next_char} == 0x0074) { # t  
               !!!next-input-character;  
               push @next_char, $self->{next_char};  
               if ($self->{next_char} == 0x0059 or # Y  
                   $self->{next_char} == 0x0079) { # y  
                 !!!next-input-character;  
                 push @next_char, $self->{next_char};  
                 if ($self->{next_char} == 0x0050 or # P  
                     $self->{next_char} == 0x0070) { # p  
                   !!!next-input-character;  
                   push @next_char, $self->{next_char};  
                   if ($self->{next_char} == 0x0045 or # E  
                       $self->{next_char} == 0x0065) { # e  
                     !!!cp (129);  
                     ## TODO: What a stupid code this is!  
                     $self->{state} = DOCTYPE_STATE;  
                     $self->{current_token} = {type => DOCTYPE_TOKEN,  
                                               quirks => 1,  
                                               line => $l, column => $c,  
                                              };  
                     !!!next-input-character;  
                     redo A;  
                   } else {  
                     !!!cp (130);  
                   }  
                 } else {  
                   !!!cp (131);  
                 }  
               } else {  
                 !!!cp (132);  
               }  
             } else {  
               !!!cp (133);  
             }  
           } else {  
             !!!cp (134);  
           }  
         } else {  
           !!!cp (135);  
         }  
2151        } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM and        } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM and
2152                 $self->{open_elements}->[-1]->[1] & FOREIGN_EL and                 $self->{open_elements}->[-1]->[1] & FOREIGN_EL and
2153                 $self->{next_char} == 0x005B) { # [                 $self->{nc} == 0x005B) { # [
2154            !!!cp (135.4);                
2155            $self->{state} = MD_CDATA_STATE;
2156            $self->{s_kwd} = '[';
2157          !!!next-input-character;          !!!next-input-character;
2158          push @next_char, $self->{next_char};          redo A;
         if ($self->{next_char} == 0x0043) { # C  
           !!!next-input-character;  
           push @next_char, $self->{next_char};  
           if ($self->{next_char} == 0x0044) { # D  
             !!!next-input-character;  
             push @next_char, $self->{next_char};  
             if ($self->{next_char} == 0x0041) { # A  
               !!!next-input-character;  
               push @next_char, $self->{next_char};  
               if ($self->{next_char} == 0x0054) { # T  
                 !!!next-input-character;  
                 push @next_char, $self->{next_char};  
                 if ($self->{next_char} == 0x0041) { # A  
                   !!!next-input-character;  
                   push @next_char, $self->{next_char};  
                   if ($self->{next_char} == 0x005B) { # [  
                     !!!cp (135.1);  
                     $self->{state} = CDATA_BLOCK_STATE;  
                     !!!next-input-character;  
                     redo A;  
                   } else {  
                     !!!cp (135.2);  
                   }  
                 } else {  
                   !!!cp (135.3);  
                 }  
               } else {  
                 !!!cp (135.4);                  
               }  
             } else {  
               !!!cp (135.5);  
             }  
           } else {  
             !!!cp (135.6);  
           }  
         } else {  
           !!!cp (135.7);  
         }  
2159        } else {        } else {
2160          !!!cp (136);          !!!cp (136);
2161        }        }
2162    
2163        !!!parse-error (type => 'bogus comment');        !!!parse-error (type => 'bogus comment',
2164        $self->{next_char} = shift @next_char;                        line => $self->{line_prev},
2165        !!!back-next-input-character (@next_char);                        column => $self->{column_prev} - 1);
2166          ## Reconsume.
2167        $self->{state} = BOGUS_COMMENT_STATE;        $self->{state} = BOGUS_COMMENT_STATE;
2168        $self->{current_token} = {type => COMMENT_TOKEN, data => '',        $self->{ct} = {type => COMMENT_TOKEN, data => '',
2169                                  line => $l, column => $c,                                  line => $self->{line_prev},
2170                                    column => $self->{column_prev} - 1,
2171                                 };                                 };
2172        redo A;        redo A;
2173              } elsif ($self->{state} == MD_HYPHEN_STATE) {
2174        ## ISSUE: typos in spec: chacacters, is is a parse error        if ($self->{nc} == 0x002D) { # -
2175        ## ISSUE: spec is somewhat unclear on "is the first character that will be in the comment"; what is "that will be in the comment" is what the algorithm defines, isn't it?          !!!cp (127);
2176            $self->{ct} = {type => COMMENT_TOKEN, data => '',
2177                                      line => $self->{line_prev},
2178                                      column => $self->{column_prev} - 2,
2179                                     };
2180            $self->{state} = COMMENT_START_STATE;
2181            !!!next-input-character;
2182            redo A;
2183          } else {
2184            !!!cp (128);
2185            !!!parse-error (type => 'bogus comment',
2186                            line => $self->{line_prev},
2187                            column => $self->{column_prev} - 2);
2188            $self->{state} = BOGUS_COMMENT_STATE;
2189            ## Reconsume.
2190            $self->{ct} = {type => COMMENT_TOKEN,
2191                                      data => '-',
2192                                      line => $self->{line_prev},
2193                                      column => $self->{column_prev} - 2,
2194                                     };
2195            redo A;
2196          }
2197        } elsif ($self->{state} == MD_DOCTYPE_STATE) {
2198          ## ASCII case-insensitive.
2199          if ($self->{nc} == [
2200                undef,
2201                0x004F, # O
2202                0x0043, # C
2203                0x0054, # T
2204                0x0059, # Y
2205                0x0050, # P
2206              ]->[length $self->{s_kwd}] or
2207              $self->{nc} == [
2208                undef,
2209                0x006F, # o
2210                0x0063, # c
2211                0x0074, # t
2212                0x0079, # y
2213                0x0070, # p
2214              ]->[length $self->{s_kwd}]) {
2215            !!!cp (131);
2216            ## Stay in the state.
2217            $self->{s_kwd} .= chr $self->{nc};
2218            !!!next-input-character;
2219            redo A;
2220          } elsif ((length $self->{s_kwd}) == 6 and
2221                   ($self->{nc} == 0x0045 or # E
2222                    $self->{nc} == 0x0065)) { # e
2223            !!!cp (129);
2224            $self->{state} = DOCTYPE_STATE;
2225            $self->{ct} = {type => DOCTYPE_TOKEN,
2226                                      quirks => 1,
2227                                      line => $self->{line_prev},
2228                                      column => $self->{column_prev} - 7,
2229                                     };
2230            !!!next-input-character;
2231            redo A;
2232          } else {
2233            !!!cp (132);        
2234            !!!parse-error (type => 'bogus comment',
2235                            line => $self->{line_prev},
2236                            column => $self->{column_prev} - 1 - length $self->{s_kwd});
2237            $self->{state} = BOGUS_COMMENT_STATE;
2238            ## Reconsume.
2239            $self->{ct} = {type => COMMENT_TOKEN,
2240                                      data => $self->{s_kwd},
2241                                      line => $self->{line_prev},
2242                                      column => $self->{column_prev} - 1 - length $self->{s_kwd},
2243                                     };
2244            redo A;
2245          }
2246        } elsif ($self->{state} == MD_CDATA_STATE) {
2247          if ($self->{nc} == {
2248                '[' => 0x0043, # C
2249                '[C' => 0x0044, # D
2250                '[CD' => 0x0041, # A
2251                '[CDA' => 0x0054, # T
2252                '[CDAT' => 0x0041, # A
2253              }->{$self->{s_kwd}}) {
2254            !!!cp (135.1);
2255            ## Stay in the state.
2256            $self->{s_kwd} .= chr $self->{nc};
2257            !!!next-input-character;
2258            redo A;
2259          } elsif ($self->{s_kwd} eq '[CDATA' and
2260                   $self->{nc} == 0x005B) { # [
2261            !!!cp (135.2);
2262            $self->{ct} = {type => CHARACTER_TOKEN,
2263                                      data => '',
2264                                      line => $self->{line_prev},
2265                                      column => $self->{column_prev} - 7};
2266            $self->{state} = CDATA_SECTION_STATE;
2267            !!!next-input-character;
2268            redo A;
2269          } else {
2270            !!!cp (135.3);
2271            !!!parse-error (type => 'bogus comment',
2272                            line => $self->{line_prev},
2273                            column => $self->{column_prev} - 1 - length $self->{s_kwd});
2274            $self->{state} = BOGUS_COMMENT_STATE;
2275            ## Reconsume.
2276            $self->{ct} = {type => COMMENT_TOKEN,
2277                                      data => $self->{s_kwd},
2278                                      line => $self->{line_prev},
2279                                      column => $self->{column_prev} - 1 - length $self->{s_kwd},
2280                                     };
2281            redo A;
2282          }
2283      } elsif ($self->{state} == COMMENT_START_STATE) {      } elsif ($self->{state} == COMMENT_START_STATE) {
2284        if ($self->{next_char} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2285          !!!cp (137);          !!!cp (137);
2286          $self->{state} = COMMENT_START_DASH_STATE;          $self->{state} = COMMENT_START_DASH_STATE;
2287          !!!next-input-character;          !!!next-input-character;
2288          redo A;          redo A;
2289        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2290          !!!cp (138);          !!!cp (138);
2291          !!!parse-error (type => 'bogus comment');          !!!parse-error (type => 'bogus comment');
2292          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2293          !!!next-input-character;          !!!next-input-character;
2294    
2295          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2296    
2297          redo A;          redo A;
2298        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2299          !!!cp (139);          !!!cp (139);
2300          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2301          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2302          ## reconsume          ## reconsume
2303    
2304          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2305    
2306          redo A;          redo A;
2307        } else {        } else {
2308          !!!cp (140);          !!!cp (140);
2309          $self->{current_token}->{data} # comment          $self->{ct}->{data} # comment
2310              .= chr ($self->{next_char});              .= chr ($self->{nc});
2311          $self->{state} = COMMENT_STATE;          $self->{state} = COMMENT_STATE;
2312          !!!next-input-character;          !!!next-input-character;
2313          redo A;          redo A;
2314        }        }
2315      } elsif ($self->{state} == COMMENT_START_DASH_STATE) {      } elsif ($self->{state} == COMMENT_START_DASH_STATE) {
2316        if ($self->{next_char} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2317          !!!cp (141);          !!!cp (141);
2318          $self->{state} = COMMENT_END_STATE;          $self->{state} = COMMENT_END_STATE;
2319          !!!next-input-character;          !!!next-input-character;
2320          redo A;          redo A;
2321        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2322          !!!cp (142);          !!!cp (142);
2323          !!!parse-error (type => 'bogus comment');          !!!parse-error (type => 'bogus comment');
2324          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2325          !!!next-input-character;          !!!next-input-character;
2326    
2327          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2328    
2329          redo A;          redo A;
2330        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2331          !!!cp (143);          !!!cp (143);
2332          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2333          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2334          ## reconsume          ## reconsume
2335    
2336          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2337    
2338          redo A;          redo A;
2339        } else {        } else {
2340          !!!cp (144);          !!!cp (144);
2341          $self->{current_token}->{data} # comment          $self->{ct}->{data} # comment
2342              .= '-' . chr ($self->{next_char});              .= '-' . chr ($self->{nc});
2343          $self->{state} = COMMENT_STATE;          $self->{state} = COMMENT_STATE;
2344          !!!next-input-character;          !!!next-input-character;
2345          redo A;          redo A;
2346        }        }
2347      } elsif ($self->{state} == COMMENT_STATE) {      } elsif ($self->{state} == COMMENT_STATE) {
2348        if ($self->{next_char} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2349          !!!cp (145);          !!!cp (145);
2350          $self->{state} = COMMENT_END_DASH_STATE;          $self->{state} = COMMENT_END_DASH_STATE;
2351          !!!next-input-character;          !!!next-input-character;
2352          redo A;          redo A;
2353        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2354          !!!cp (146);          !!!cp (146);
2355          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2356          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2357          ## reconsume          ## reconsume
2358    
2359          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2360    
2361          redo A;          redo A;
2362        } else {        } else {
2363          !!!cp (147);          !!!cp (147);
2364          $self->{current_token}->{data} .= chr ($self->{next_char}); # comment          $self->{ct}->{data} .= chr ($self->{nc}); # comment
2365            $self->{read_until}->($self->{ct}->{data},
2366                                  q[-],
2367                                  length $self->{ct}->{data});
2368    
2369          ## Stay in the state          ## Stay in the state
2370          !!!next-input-character;          !!!next-input-character;
2371          redo A;          redo A;
2372        }        }
2373      } elsif ($self->{state} == COMMENT_END_DASH_STATE) {      } elsif ($self->{state} == COMMENT_END_DASH_STATE) {
2374        if ($self->{next_char} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2375          !!!cp (148);          !!!cp (148);
2376          $self->{state} = COMMENT_END_STATE;          $self->{state} = COMMENT_END_STATE;
2377          !!!next-input-character;          !!!next-input-character;
2378          redo A;          redo A;
2379        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2380          !!!cp (149);          !!!cp (149);
2381          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2382          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2383          ## reconsume          ## reconsume
2384    
2385          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2386    
2387          redo A;          redo A;
2388        } else {        } else {
2389          !!!cp (150);          !!!cp (150);
2390          $self->{current_token}->{data} .= '-' . chr ($self->{next_char}); # comment          $self->{ct}->{data} .= '-' . chr ($self->{nc}); # comment
2391          $self->{state} = COMMENT_STATE;          $self->{state} = COMMENT_STATE;
2392          !!!next-input-character;          !!!next-input-character;
2393          redo A;          redo A;
2394        }        }
2395      } elsif ($self->{state} == COMMENT_END_STATE) {      } elsif ($self->{state} == COMMENT_END_STATE) {
2396        if ($self->{next_char} == 0x003E) { # >        if ($self->{nc} == 0x003E) { # >
2397          !!!cp (151);          !!!cp (151);
2398          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2399          !!!next-input-character;          !!!next-input-character;
2400    
2401          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2402    
2403          redo A;          redo A;
2404        } elsif ($self->{next_char} == 0x002D) { # -        } elsif ($self->{nc} == 0x002D) { # -
2405          !!!cp (152);          !!!cp (152);
2406          !!!parse-error (type => 'dash in comment',          !!!parse-error (type => 'dash in comment',
2407                          line => $self->{line_prev},                          line => $self->{line_prev},
2408                          column => $self->{column_prev});                          column => $self->{column_prev});
2409          $self->{current_token}->{data} .= '-'; # comment          $self->{ct}->{data} .= '-'; # comment
2410          ## Stay in the state          ## Stay in the state
2411          !!!next-input-character;          !!!next-input-character;
2412          redo A;          redo A;
2413        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2414          !!!cp (153);          !!!cp (153);
2415          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2416          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2417          ## reconsume          ## reconsume
2418    
2419          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2420    
2421          redo A;          redo A;
2422        } else {        } else {
# Line 2212  sub _get_next_token ($) { Line 2424  sub _get_next_token ($) {
2424          !!!parse-error (type => 'dash in comment',          !!!parse-error (type => 'dash in comment',
2425                          line => $self->{line_prev},                          line => $self->{line_prev},
2426                          column => $self->{column_prev});                          column => $self->{column_prev});
2427          $self->{current_token}->{data} .= '--' . chr ($self->{next_char}); # comment          $self->{ct}->{data} .= '--' . chr ($self->{nc}); # comment
2428          $self->{state} = COMMENT_STATE;          $self->{state} = COMMENT_STATE;
2429          !!!next-input-character;          !!!next-input-character;
2430          redo A;          redo A;
2431        }        }
2432      } elsif ($self->{state} == DOCTYPE_STATE) {      } elsif ($self->{state} == DOCTYPE_STATE) {
2433        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
2434          !!!cp (155);          !!!cp (155);
2435          $self->{state} = BEFORE_DOCTYPE_NAME_STATE;          $self->{state} = BEFORE_DOCTYPE_NAME_STATE;
2436          !!!next-input-character;          !!!next-input-character;
# Line 2235  sub _get_next_token ($) { Line 2443  sub _get_next_token ($) {
2443          redo A;          redo A;
2444        }        }
2445      } elsif ($self->{state} == BEFORE_DOCTYPE_NAME_STATE) {      } elsif ($self->{state} == BEFORE_DOCTYPE_NAME_STATE) {
2446        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
2447          !!!cp (157);          !!!cp (157);
2448          ## Stay in the state          ## Stay in the state
2449          !!!next-input-character;          !!!next-input-character;
2450          redo A;          redo A;
2451        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2452          !!!cp (158);          !!!cp (158);
2453          !!!parse-error (type => 'no DOCTYPE name');          !!!parse-error (type => 'no DOCTYPE name');
2454          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2455          !!!next-input-character;          !!!next-input-character;
2456    
2457          !!!emit ($self->{current_token}); # DOCTYPE (quirks)          !!!emit ($self->{ct}); # DOCTYPE (quirks)
2458    
2459          redo A;          redo A;
2460        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2461          !!!cp (159);          !!!cp (159);
2462          !!!parse-error (type => 'no DOCTYPE name');          !!!parse-error (type => 'no DOCTYPE name');
2463          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2464          ## reconsume          ## reconsume
2465    
2466          !!!emit ($self->{current_token}); # DOCTYPE (quirks)          !!!emit ($self->{ct}); # DOCTYPE (quirks)
2467    
2468          redo A;          redo A;
2469        } else {        } else {
2470          !!!cp (160);          !!!cp (160);
2471          $self->{current_token}->{name} = chr $self->{next_char};          $self->{ct}->{name} = chr $self->{nc};
2472          delete $self->{current_token}->{quirks};          delete $self->{ct}->{quirks};
2473  ## ISSUE: "Set the token's name name to the" in the spec  ## ISSUE: "Set the token's name name to the" in the spec
2474          $self->{state} = DOCTYPE_NAME_STATE;          $self->{state} = DOCTYPE_NAME_STATE;
2475          !!!next-input-character;          !!!next-input-character;
# Line 2273  sub _get_next_token ($) { Line 2477  sub _get_next_token ($) {
2477        }        }
2478      } elsif ($self->{state} == DOCTYPE_NAME_STATE) {      } elsif ($self->{state} == DOCTYPE_NAME_STATE) {
2479  ## ISSUE: Redundant "First," in the spec.  ## ISSUE: Redundant "First," in the spec.
2480        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
2481          !!!cp (161);          !!!cp (161);
2482          $self->{state} = AFTER_DOCTYPE_NAME_STATE;          $self->{state} = AFTER_DOCTYPE_NAME_STATE;
2483          !!!next-input-character;          !!!next-input-character;
2484          redo A;          redo A;
2485        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2486          !!!cp (162);          !!!cp (162);
2487          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2488          !!!next-input-character;          !!!next-input-character;
2489    
2490          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2491    
2492          redo A;          redo A;
2493        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2494          !!!cp (163);          !!!cp (163);
2495          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2496          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2497          ## reconsume          ## reconsume
2498    
2499          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2500          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2501    
2502          redo A;          redo A;
2503        } else {        } else {
2504          !!!cp (164);          !!!cp (164);
2505          $self->{current_token}->{name}          $self->{ct}->{name}
2506            .= chr ($self->{next_char}); # DOCTYPE            .= chr ($self->{nc}); # DOCTYPE
2507          ## Stay in the state          ## Stay in the state
2508          !!!next-input-character;          !!!next-input-character;
2509          redo A;          redo A;
2510        }        }
2511      } elsif ($self->{state} == AFTER_DOCTYPE_NAME_STATE) {      } elsif ($self->{state} == AFTER_DOCTYPE_NAME_STATE) {
2512        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
2513          !!!cp (165);          !!!cp (165);
2514          ## Stay in the state          ## Stay in the state
2515          !!!next-input-character;          !!!next-input-character;
2516          redo A;          redo A;
2517        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2518          !!!cp (166);          !!!cp (166);
2519          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2520          !!!next-input-character;          !!!next-input-character;
2521    
2522          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2523    
2524          redo A;          redo A;
2525        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2526          !!!cp (167);          !!!cp (167);
2527          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2528          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2529          ## reconsume          ## reconsume
2530    
2531          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2532          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2533    
2534          redo A;          redo A;
2535        } elsif ($self->{next_char} == 0x0050 or # P        } elsif ($self->{nc} == 0x0050 or # P
2536                 $self->{next_char} == 0x0070) { # p                 $self->{nc} == 0x0070) { # p
2537            $self->{state} = PUBLIC_STATE;
2538            $self->{s_kwd} = chr $self->{nc};
2539          !!!next-input-character;          !!!next-input-character;
2540          if ($self->{next_char} == 0x0055 or # U          redo A;
2541              $self->{next_char} == 0x0075) { # u        } elsif ($self->{nc} == 0x0053 or # S
2542            !!!next-input-character;                 $self->{nc} == 0x0073) { # s
2543            if ($self->{next_char} == 0x0042 or # B          $self->{state} = SYSTEM_STATE;
2544                $self->{next_char} == 0x0062) { # b          $self->{s_kwd} = chr $self->{nc};
             !!!next-input-character;  
             if ($self->{next_char} == 0x004C or # L  
                 $self->{next_char} == 0x006C) { # l  
               !!!next-input-character;  
               if ($self->{next_char} == 0x0049 or # I  
                   $self->{next_char} == 0x0069) { # i  
                 !!!next-input-character;  
                 if ($self->{next_char} == 0x0043 or # C  
                     $self->{next_char} == 0x0063) { # c  
                   !!!cp (168);  
                   $self->{state} = BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE;  
                   !!!next-input-character;  
                   redo A;  
                 } else {  
                   !!!cp (169);  
                 }  
               } else {  
                 !!!cp (170);  
               }  
             } else {  
               !!!cp (171);  
             }  
           } else {  
             !!!cp (172);  
           }  
         } else {  
           !!!cp (173);  
         }  
   
         #  
       } elsif ($self->{next_char} == 0x0053 or # S  
                $self->{next_char} == 0x0073) { # s  
2545          !!!next-input-character;          !!!next-input-character;
2546          if ($self->{next_char} == 0x0059 or # Y          redo A;
             $self->{next_char} == 0x0079) { # y  
           !!!next-input-character;  
           if ($self->{next_char} == 0x0053 or # S  
               $self->{next_char} == 0x0073) { # s  
             !!!next-input-character;  
             if ($self->{next_char} == 0x0054 or # T  
                 $self->{next_char} == 0x0074) { # t  
               !!!next-input-character;  
               if ($self->{next_char} == 0x0045 or # E  
                   $self->{next_char} == 0x0065) { # e  
                 !!!next-input-character;  
                 if ($self->{next_char} == 0x004D or # M  
                     $self->{next_char} == 0x006D) { # m  
                   !!!cp (174);  
                   $self->{state} = BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE;  
                   !!!next-input-character;  
                   redo A;  
                 } else {  
                   !!!cp (175);  
                 }  
               } else {  
                 !!!cp (176);  
               }  
             } else {  
               !!!cp (177);  
             }  
           } else {  
             !!!cp (178);  
           }  
         } else {  
           !!!cp (179);  
         }  
   
         #  
2547        } else {        } else {
2548          !!!cp (180);          !!!cp (180);
2549            !!!parse-error (type => 'string after DOCTYPE name');
2550            $self->{ct}->{quirks} = 1;
2551    
2552            $self->{state} = BOGUS_DOCTYPE_STATE;
2553          !!!next-input-character;          !!!next-input-character;
2554          #          redo A;
2555        }        }
2556        } elsif ($self->{state} == PUBLIC_STATE) {
2557          ## ASCII case-insensitive
2558          if ($self->{nc} == [
2559                undef,
2560                0x0055, # U
2561                0x0042, # B
2562                0x004C, # L
2563                0x0049, # I
2564              ]->[length $self->{s_kwd}] or
2565              $self->{nc} == [
2566                undef,
2567                0x0075, # u
2568                0x0062, # b
2569                0x006C, # l
2570                0x0069, # i
2571              ]->[length $self->{s_kwd}]) {
2572            !!!cp (175);
2573            ## Stay in the state.
2574            $self->{s_kwd} .= chr $self->{nc};
2575            !!!next-input-character;
2576            redo A;
2577          } elsif ((length $self->{s_kwd}) == 5 and
2578                   ($self->{nc} == 0x0043 or # C
2579                    $self->{nc} == 0x0063)) { # c
2580            !!!cp (168);
2581            $self->{state} = BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2582            !!!next-input-character;
2583            redo A;
2584          } else {
2585            !!!cp (169);
2586            !!!parse-error (type => 'string after DOCTYPE name',
2587                            line => $self->{line_prev},
2588                            column => $self->{column_prev} + 1 - length $self->{s_kwd});
2589            $self->{ct}->{quirks} = 1;
2590    
2591        !!!parse-error (type => 'string after DOCTYPE name');          $self->{state} = BOGUS_DOCTYPE_STATE;
2592        $self->{current_token}->{quirks} = 1;          ## Reconsume.
2593            redo A;
2594          }
2595        } elsif ($self->{state} == SYSTEM_STATE) {
2596          ## ASCII case-insensitive
2597          if ($self->{nc} == [
2598                undef,
2599                0x0059, # Y
2600                0x0053, # S
2601                0x0054, # T
2602                0x0045, # E
2603              ]->[length $self->{s_kwd}] or
2604              $self->{nc} == [
2605                undef,
2606                0x0079, # y
2607                0x0073, # s
2608                0x0074, # t
2609                0x0065, # e
2610              ]->[length $self->{s_kwd}]) {
2611            !!!cp (170);
2612            ## Stay in the state.
2613            $self->{s_kwd} .= chr $self->{nc};
2614            !!!next-input-character;
2615            redo A;
2616          } elsif ((length $self->{s_kwd}) == 5 and
2617                   ($self->{nc} == 0x004D or # M
2618                    $self->{nc} == 0x006D)) { # m
2619            !!!cp (171);
2620            $self->{state} = BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2621            !!!next-input-character;
2622            redo A;
2623          } else {
2624            !!!cp (172);
2625            !!!parse-error (type => 'string after DOCTYPE name',
2626                            line => $self->{line_prev},
2627                            column => $self->{column_prev} + 1 - length $self->{s_kwd});
2628            $self->{ct}->{quirks} = 1;
2629    
2630        $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2631        # next-input-character is already done          ## Reconsume.
2632        redo A;          redo A;
2633          }
2634      } elsif ($self->{state} == BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {      } elsif ($self->{state} == BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {
2635        if ({        if ($is_space->{$self->{nc}}) {
             0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,  
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_char}}) {  
2636          !!!cp (181);          !!!cp (181);
2637          ## Stay in the state          ## Stay in the state
2638          !!!next-input-character;          !!!next-input-character;
2639          redo A;          redo A;
2640        } elsif ($self->{next_char} eq 0x0022) { # "        } elsif ($self->{nc} eq 0x0022) { # "
2641          !!!cp (182);          !!!cp (182);
2642          $self->{current_token}->{public_identifier} = ''; # DOCTYPE          $self->{ct}->{pubid} = ''; # DOCTYPE
2643          $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE;          $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE;
2644          !!!next-input-character;          !!!next-input-character;
2645          redo A;          redo A;
2646        } elsif ($self->{next_char} eq 0x0027) { # '        } elsif ($self->{nc} eq 0x0027) { # '
2647          !!!cp (183);          !!!cp (183);
2648          $self->{current_token}->{public_identifier} = ''; # DOCTYPE          $self->{ct}->{pubid} = ''; # DOCTYPE
2649          $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE;          $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE;
2650          !!!next-input-character;          !!!next-input-character;
2651          redo A;          redo A;
2652        } elsif ($self->{next_char} eq 0x003E) { # >        } elsif ($self->{nc} eq 0x003E) { # >
2653          !!!cp (184);          !!!cp (184);
2654          !!!parse-error (type => 'no PUBLIC literal');          !!!parse-error (type => 'no PUBLIC literal');
2655    
2656          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2657          !!!next-input-character;          !!!next-input-character;
2658    
2659          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2660          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2661    
2662          redo A;          redo A;
2663        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2664          !!!cp (185);          !!!cp (185);
2665          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2666    
2667          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2668          ## reconsume          ## reconsume
2669    
2670          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2671          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2672    
2673          redo A;          redo A;
2674        } else {        } else {
2675          !!!cp (186);          !!!cp (186);
2676          !!!parse-error (type => 'string after PUBLIC');          !!!parse-error (type => 'string after PUBLIC');
2677          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2678    
2679          $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2680          !!!next-input-character;          !!!next-input-character;
2681          redo A;          redo A;
2682        }        }
2683      } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE) {      } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE) {
2684        if ($self->{next_char} == 0x0022) { # "        if ($self->{nc} == 0x0022) { # "
2685          !!!cp (187);          !!!cp (187);
2686          $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;          $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2687          !!!next-input-character;          !!!next-input-character;
2688          redo A;          redo A;
2689        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2690          !!!cp (188);          !!!cp (188);
2691          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed PUBLIC literal');
2692    
2693          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2694          !!!next-input-character;          !!!next-input-character;
2695    
2696          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2697          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2698    
2699          redo A;          redo A;
2700        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2701          !!!cp (189);          !!!cp (189);
2702          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed PUBLIC literal');
2703    
2704          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2705          ## reconsume          ## reconsume
2706    
2707          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2708          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2709    
2710          redo A;          redo A;
2711        } else {        } else {
2712          !!!cp (190);          !!!cp (190);
2713          $self->{current_token}->{public_identifier} # DOCTYPE          $self->{ct}->{pubid} # DOCTYPE
2714              .= chr $self->{next_char};              .= chr $self->{nc};
2715            $self->{read_until}->($self->{ct}->{pubid}, q[">],
2716                                  length $self->{ct}->{pubid});
2717    
2718          ## Stay in the state          ## Stay in the state
2719          !!!next-input-character;          !!!next-input-character;
2720          redo A;          redo A;
2721        }        }
2722      } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE) {      } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE) {
2723        if ($self->{next_char} == 0x0027) { # '        if ($self->{nc} == 0x0027) { # '
2724          !!!cp (191);          !!!cp (191);
2725          $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;          $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2726          !!!next-input-character;          !!!next-input-character;
2727          redo A;          redo A;
2728        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2729          !!!cp (192);          !!!cp (192);
2730          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed PUBLIC literal');
2731    
2732          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2733          !!!next-input-character;          !!!next-input-character;
2734    
2735          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2736          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2737    
2738          redo A;          redo A;
2739        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2740          !!!cp (193);          !!!cp (193);
2741          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed PUBLIC literal');
2742    
2743          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2744          ## reconsume          ## reconsume
2745    
2746          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2747          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2748    
2749          redo A;          redo A;
2750        } else {        } else {
2751          !!!cp (194);          !!!cp (194);
2752          $self->{current_token}->{public_identifier} # DOCTYPE          $self->{ct}->{pubid} # DOCTYPE
2753              .= chr $self->{next_char};              .= chr $self->{nc};
2754            $self->{read_until}->($self->{ct}->{pubid}, q['>],
2755                                  length $self->{ct}->{pubid});
2756    
2757          ## Stay in the state          ## Stay in the state
2758          !!!next-input-character;          !!!next-input-character;
2759          redo A;          redo A;
2760        }        }
2761      } elsif ($self->{state} == AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {      } elsif ($self->{state} == AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {
2762        if ({        if ($is_space->{$self->{nc}}) {
             0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,  
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_char}}) {  
2763          !!!cp (195);          !!!cp (195);
2764          ## Stay in the state          ## Stay in the state
2765          !!!next-input-character;          !!!next-input-character;
2766          redo A;          redo A;
2767        } elsif ($self->{next_char} == 0x0022) { # "        } elsif ($self->{nc} == 0x0022) { # "
2768          !!!cp (196);          !!!cp (196);
2769          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          $self->{ct}->{sysid} = ''; # DOCTYPE
2770          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;
2771          !!!next-input-character;          !!!next-input-character;
2772          redo A;          redo A;
2773        } elsif ($self->{next_char} == 0x0027) { # '        } elsif ($self->{nc} == 0x0027) { # '
2774          !!!cp (197);          !!!cp (197);
2775          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          $self->{ct}->{sysid} = ''; # DOCTYPE
2776          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;
2777          !!!next-input-character;          !!!next-input-character;
2778          redo A;          redo A;
2779        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2780          !!!cp (198);          !!!cp (198);
2781          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2782          !!!next-input-character;          !!!next-input-character;
2783    
2784          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2785    
2786          redo A;          redo A;
2787        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2788          !!!cp (199);          !!!cp (199);
2789          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2790    
2791          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2792          ## reconsume          ## reconsume
2793    
2794          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2795          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2796    
2797          redo A;          redo A;
2798        } else {        } else {
2799          !!!cp (200);          !!!cp (200);
2800          !!!parse-error (type => 'string after PUBLIC literal');          !!!parse-error (type => 'string after PUBLIC literal');
2801          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2802    
2803          $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2804          !!!next-input-character;          !!!next-input-character;
2805          redo A;          redo A;
2806        }        }
2807      } elsif ($self->{state} == BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {      } elsif ($self->{state} == BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {
2808        if ({        if ($is_space->{$self->{nc}}) {
             0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,  
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_char}}) {  
2809          !!!cp (201);          !!!cp (201);
2810          ## Stay in the state          ## Stay in the state
2811          !!!next-input-character;          !!!next-input-character;
2812          redo A;          redo A;
2813        } elsif ($self->{next_char} == 0x0022) { # "        } elsif ($self->{nc} == 0x0022) { # "
2814          !!!cp (202);          !!!cp (202);
2815          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          $self->{ct}->{sysid} = ''; # DOCTYPE
2816          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;
2817          !!!next-input-character;          !!!next-input-character;
2818          redo A;          redo A;
2819        } elsif ($self->{next_char} == 0x0027) { # '        } elsif ($self->{nc} == 0x0027) { # '
2820          !!!cp (203);          !!!cp (203);
2821          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          $self->{ct}->{sysid} = ''; # DOCTYPE
2822          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;
2823          !!!next-input-character;          !!!next-input-character;
2824          redo A;          redo A;
2825        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2826          !!!cp (204);          !!!cp (204);
2827          !!!parse-error (type => 'no SYSTEM literal');          !!!parse-error (type => 'no SYSTEM literal');
2828          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2829          !!!next-input-character;          !!!next-input-character;
2830    
2831          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2832          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2833    
2834          redo A;          redo A;
2835        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2836          !!!cp (205);          !!!cp (205);
2837          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2838    
2839          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2840          ## reconsume          ## reconsume
2841    
2842          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2843          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2844    
2845          redo A;          redo A;
2846        } else {        } else {
2847          !!!cp (206);          !!!cp (206);
2848          !!!parse-error (type => 'string after SYSTEM');          !!!parse-error (type => 'string after SYSTEM');
2849          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2850    
2851          $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2852          !!!next-input-character;          !!!next-input-character;
2853          redo A;          redo A;
2854        }        }
2855      } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE) {      } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE) {
2856        if ($self->{next_char} == 0x0022) { # "        if ($self->{nc} == 0x0022) { # "
2857          !!!cp (207);          !!!cp (207);
2858          $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;          $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2859          !!!next-input-character;          !!!next-input-character;
2860          redo A;          redo A;
2861        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2862          !!!cp (208);          !!!cp (208);
2863          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed SYSTEM literal');
2864    
2865          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2866          !!!next-input-character;          !!!next-input-character;
2867    
2868          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2869          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2870    
2871          redo A;          redo A;
2872        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2873          !!!cp (209);          !!!cp (209);
2874          !!!parse-error (type => 'unclosed SYSTEM literal');          !!!parse-error (type => 'unclosed SYSTEM literal');
2875    
2876          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2877          ## reconsume          ## reconsume
2878    
2879          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2880          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2881    
2882          redo A;          redo A;
2883        } else {        } else {
2884          !!!cp (210);          !!!cp (210);
2885          $self->{current_token}->{system_identifier} # DOCTYPE          $self->{ct}->{sysid} # DOCTYPE
2886              .= chr $self->{next_char};              .= chr $self->{nc};
2887            $self->{read_until}->($self->{ct}->{sysid}, q[">],
2888                                  length $self->{ct}->{sysid});
2889    
2890          ## Stay in the state          ## Stay in the state
2891          !!!next-input-character;          !!!next-input-character;
2892          redo A;          redo A;
2893        }        }
2894      } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE) {      } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE) {
2895        if ($self->{next_char} == 0x0027) { # '        if ($self->{nc} == 0x0027) { # '
2896          !!!cp (211);          !!!cp (211);
2897          $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;          $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2898          !!!next-input-character;          !!!next-input-character;
2899          redo A;          redo A;
2900        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2901          !!!cp (212);          !!!cp (212);
2902          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed SYSTEM literal');
2903    
2904          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2905          !!!next-input-character;          !!!next-input-character;
2906    
2907          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2908          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2909    
2910          redo A;          redo A;
2911        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2912          !!!cp (213);          !!!cp (213);
2913          !!!parse-error (type => 'unclosed SYSTEM literal');          !!!parse-error (type => 'unclosed SYSTEM literal');
2914    
2915          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2916          ## reconsume          ## reconsume
2917    
2918          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2919          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2920    
2921          redo A;          redo A;
2922        } else {        } else {
2923          !!!cp (214);          !!!cp (214);
2924          $self->{current_token}->{system_identifier} # DOCTYPE          $self->{ct}->{sysid} # DOCTYPE
2925              .= chr $self->{next_char};              .= chr $self->{nc};
2926            $self->{read_until}->($self->{ct}->{sysid}, q['>],
2927                                  length $self->{ct}->{sysid});
2928    
2929          ## Stay in the state          ## Stay in the state
2930          !!!next-input-character;          !!!next-input-character;
2931          redo A;          redo A;
2932        }        }
2933      } elsif ($self->{state} == AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {      } elsif ($self->{state} == AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {
2934        if ({        if ($is_space->{$self->{nc}}) {
             0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,  
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_char}}) {  
2935          !!!cp (215);          !!!cp (215);
2936          ## Stay in the state          ## Stay in the state
2937          !!!next-input-character;          !!!next-input-character;
2938          redo A;          redo A;
2939        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2940          !!!cp (216);          !!!cp (216);
2941          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2942          !!!next-input-character;          !!!next-input-character;
2943    
2944          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2945    
2946          redo A;          redo A;
2947        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2948          !!!cp (217);          !!!cp (217);
2949          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
   
2950          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2951          ## reconsume          ## reconsume
2952    
2953          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2954          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2955    
2956          redo A;          redo A;
2957        } else {        } else {
2958          !!!cp (218);          !!!cp (218);
2959          !!!parse-error (type => 'string after SYSTEM literal');          !!!parse-error (type => 'string after SYSTEM literal');
2960          #$self->{current_token}->{quirks} = 1;          #$self->{ct}->{quirks} = 1;
2961    
2962          $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2963          !!!next-input-character;          !!!next-input-character;
2964          redo A;          redo A;
2965        }        }
2966      } elsif ($self->{state} == BOGUS_DOCTYPE_STATE) {      } elsif ($self->{state} == BOGUS_DOCTYPE_STATE) {
2967        if ($self->{next_char} == 0x003E) { # >        if ($self->{nc} == 0x003E) { # >
2968          !!!cp (219);          !!!cp (219);
2969          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2970          !!!next-input-character;          !!!next-input-character;
2971    
2972          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2973    
2974          redo A;          redo A;
2975        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2976          !!!cp (220);          !!!cp (220);
2977          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2978          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2979          ## reconsume          ## reconsume
2980    
2981          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2982    
2983          redo A;          redo A;
2984        } else {        } else {
2985          !!!cp (221);          !!!cp (221);
2986            my $s = '';
2987            $self->{read_until}->($s, q[>], 0);
2988    
2989          ## Stay in the state          ## Stay in the state
2990          !!!next-input-character;          !!!next-input-character;
2991          redo A;          redo A;
2992        }        }
2993      } elsif ($self->{state} == CDATA_BLOCK_STATE) {      } elsif ($self->{state} == CDATA_SECTION_STATE) {
2994        my $s = '';        ## NOTE: "CDATA section state" in the state is jointly implemented
2995          ## by three states, |CDATA_SECTION_STATE|, |CDATA_SECTION_MSE1_STATE|,
2996          ## and |CDATA_SECTION_MSE2_STATE|.
2997                
2998        my ($l, $c) = ($self->{line}, $self->{column});        if ($self->{nc} == 0x005D) { # ]
2999            !!!cp (221.1);
3000        CS: while ($self->{next_char} != -1) {          $self->{state} = CDATA_SECTION_MSE1_STATE;
         if ($self->{next_char} == 0x005D) { # ]  
           !!!next-input-character;  
           if ($self->{next_char} == 0x005D) { # ]  
             !!!next-input-character;  
             MDC: {  
               if ($self->{next_char} == 0x003E) { # >  
                 !!!cp (221.1);  
                 !!!next-input-character;  
                 last CS;  
               } elsif ($self->{next_char} == 0x005D) { # ]  
                 !!!cp (221.2);  
                 $s .= ']';  
                 !!!next-input-character;  
                 redo MDC;  
               } else {  
                 !!!cp (221.3);  
                 $s .= ']]';  
                 #  
               }  
             } # MDC  
           } else {  
             !!!cp (221.4);  
             $s .= ']';  
             #  
           }  
         } else {  
           !!!cp (221.5);  
           #  
         }  
         $s .= chr $self->{next_char};  
3001          !!!next-input-character;          !!!next-input-character;
3002        } # CS          redo A;
3003          } elsif ($self->{nc} == -1) {
3004            $self->{state} = DATA_STATE;
3005            !!!next-input-character;
3006            if (length $self->{ct}->{data}) { # character
3007              !!!cp (221.2);
3008              !!!emit ($self->{ct}); # character
3009            } else {
3010              !!!cp (221.3);
3011              ## No token to emit. $self->{ct} is discarded.
3012            }        
3013            redo A;
3014          } else {
3015            !!!cp (221.4);
3016            $self->{ct}->{data} .= chr $self->{nc};
3017            $self->{read_until}->($self->{ct}->{data},
3018                                  q<]>,
3019                                  length $self->{ct}->{data});
3020    
3021        $self->{state} = DATA_STATE;          ## Stay in the state.
3022        ## next-input-character done or EOF, which is reconsumed.          !!!next-input-character;
3023            redo A;
3024          }
3025    
3026        if (length $s) {        ## ISSUE: "text tokens" in spec.
3027        } elsif ($self->{state} == CDATA_SECTION_MSE1_STATE) {
3028          if ($self->{nc} == 0x005D) { # ]
3029            !!!cp (221.5);
3030            $self->{state} = CDATA_SECTION_MSE2_STATE;
3031            !!!next-input-character;
3032            redo A;
3033          } else {
3034          !!!cp (221.6);          !!!cp (221.6);
3035          !!!emit ({type => CHARACTER_TOKEN, data => $s,          $self->{ct}->{data} .= ']';
3036                    line => $l, column => $c});          $self->{state} = CDATA_SECTION_STATE;
3037            ## Reconsume.
3038            redo A;
3039          }
3040        } elsif ($self->{state} == CDATA_SECTION_MSE2_STATE) {
3041          if ($self->{nc} == 0x003E) { # >
3042            $self->{state} = DATA_STATE;
3043            !!!next-input-character;
3044            if (length $self->{ct}->{data}) { # character
3045              !!!cp (221.7);
3046              !!!emit ($self->{ct}); # character
3047            } else {
3048              !!!cp (221.8);
3049              ## No token to emit. $self->{ct} is discarded.
3050            }
3051            redo A;
3052          } elsif ($self->{nc} == 0x005D) { # ]
3053            !!!cp (221.9); # character
3054            $self->{ct}->{data} .= ']'; ## Add first "]" of "]]]".
3055            ## Stay in the state.
3056            !!!next-input-character;
3057            redo A;
3058        } else {        } else {
3059          !!!cp (221.7);          !!!cp (221.11);
3060            $self->{ct}->{data} .= ']]'; # character
3061            $self->{state} = CDATA_SECTION_STATE;
3062            ## Reconsume.
3063            redo A;
3064          }
3065        } elsif ($self->{state} == ENTITY_STATE) {
3066          if ($is_space->{$self->{nc}} or
3067              {
3068                0x003C => 1, 0x0026 => 1, -1 => 1, # <, &
3069                $self->{entity_add} => 1,
3070              }->{$self->{nc}}) {
3071            !!!cp (1001);
3072            ## Don't consume
3073            ## No error
3074            ## Return nothing.
3075            #
3076          } elsif ($self->{nc} == 0x0023) { # #
3077            !!!cp (999);
3078            $self->{state} = ENTITY_HASH_STATE;
3079            $self->{s_kwd} = '#';
3080            !!!next-input-character;
3081            redo A;
3082          } elsif ((0x0041 <= $self->{nc} and
3083                    $self->{nc} <= 0x005A) or # A..Z
3084                   (0x0061 <= $self->{nc} and
3085                    $self->{nc} <= 0x007A)) { # a..z
3086            !!!cp (998);
3087            require Whatpm::_NamedEntityList;
3088            $self->{state} = ENTITY_NAME_STATE;
3089            $self->{s_kwd} = chr $self->{nc};
3090            $self->{entity__value} = $self->{s_kwd};
3091            $self->{entity__match} = 0;
3092            !!!next-input-character;
3093            redo A;
3094          } else {
3095            !!!cp (1027);
3096            !!!parse-error (type => 'bare ero');
3097            ## Return nothing.
3098            #
3099        }        }
3100    
3101        redo A;        ## NOTE: No character is consumed by the "consume a character
3102          ## reference" algorithm.  In other word, there is an "&" character
3103        ## ISSUE: "text tokens" in spec.        ## that does not introduce a character reference, which would be
3104        ## TODO: Streaming support        ## appended to the parent element or the attribute value in later
3105      } else {        ## process of the tokenizer.
3106        die "$0: $self->{state}: Unknown state";  
3107      }        if ($self->{prev_state} == DATA_STATE) {
3108    } # A            !!!cp (997);
3109            $self->{state} = $self->{prev_state};
3110    die "$0: _get_next_token: unexpected case";          ## Reconsume.
3111  } # _get_next_token          !!!emit ({type => CHARACTER_TOKEN, data => '&',
3112                      line => $self->{line_prev},
3113  sub _tokenize_attempt_to_consume_an_entity ($$$) {                    column => $self->{column_prev},
3114    my ($self, $in_attr, $additional) = @_;                   });
3115            redo A;
3116    my ($l, $c) = ($self->{line_prev}, $self->{column_prev});        } else {
3117            !!!cp (996);
3118            $self->{ca}->{value} .= '&';
3119            $self->{state} = $self->{prev_state};
3120            ## Reconsume.
3121            redo A;
3122          }
3123        } elsif ($self->{state} == ENTITY_HASH_STATE) {
3124          if ($self->{nc} == 0x0078 or # x
3125              $self->{nc} == 0x0058) { # X
3126            !!!cp (995);
3127            $self->{state} = HEXREF_X_STATE;
3128            $self->{s_kwd} .= chr $self->{nc};
3129            !!!next-input-character;
3130            redo A;
3131          } elsif (0x0030 <= $self->{nc} and
3132                   $self->{nc} <= 0x0039) { # 0..9
3133            !!!cp (994);
3134            $self->{state} = NCR_NUM_STATE;
3135            $self->{s_kwd} = $self->{nc} - 0x0030;
3136            !!!next-input-character;
3137            redo A;
3138          } else {
3139            !!!parse-error (type => 'bare nero',
3140                            line => $self->{line_prev},
3141                            column => $self->{column_prev} - 1);
3142    
3143    if ({          ## NOTE: According to the spec algorithm, nothing is returned,
3144         0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, # HT, LF, VT, FF,          ## and then "&#" is appended to the parent element or the attribute
3145         0x0020 => 1, 0x003C => 1, 0x0026 => 1, -1 => 1, # SP, <, & # 0x000D # CR          ## value in the later processing.
3146         $additional => 1,  
3147        }->{$self->{next_char}}) {          if ($self->{prev_state} == DATA_STATE) {
3148      !!!cp (1001);            !!!cp (1019);
3149      ## Don't consume            $self->{state} = $self->{prev_state};
3150      ## No error            ## Reconsume.
3151      return undef;            !!!emit ({type => CHARACTER_TOKEN,
3152    } elsif ($self->{next_char} == 0x0023) { # #                      data => '&#',
3153      !!!next-input-character;                      line => $self->{line_prev},
3154      if ($self->{next_char} == 0x0078 or # x                      column => $self->{column_prev} - 1,
3155          $self->{next_char} == 0x0058) { # X                     });
3156        my $code;            redo A;
       X: {  
         my $x_char = $self->{next_char};  
         !!!next-input-character;  
         if (0x0030 <= $self->{next_char} and  
             $self->{next_char} <= 0x0039) { # 0..9  
           !!!cp (1002);  
           $code ||= 0;  
           $code *= 0x10;  
           $code += $self->{next_char} - 0x0030;  
           redo X;  
         } elsif (0x0061 <= $self->{next_char} and  
                  $self->{next_char} <= 0x0066) { # a..f  
           !!!cp (1003);  
           $code ||= 0;  
           $code *= 0x10;  
           $code += $self->{next_char} - 0x0060 + 9;  
           redo X;  
         } elsif (0x0041 <= $self->{next_char} and  
                  $self->{next_char} <= 0x0046) { # A..F  
           !!!cp (1004);  
           $code ||= 0;  
           $code *= 0x10;  
           $code += $self->{next_char} - 0x0040 + 9;  
           redo X;  
         } elsif (not defined $code) { # no hexadecimal digit  
           !!!cp (1005);  
           !!!parse-error (type => 'bare hcro', line => $l, column => $c);  
           !!!back-next-input-character ($x_char, $self->{next_char});  
           $self->{next_char} = 0x0023; # #  
           return undef;  
         } elsif ($self->{next_char} == 0x003B) { # ;  
           !!!cp (1006);  
           !!!next-input-character;  
3157          } else {          } else {
3158            !!!cp (1007);            !!!cp (993);
3159            !!!parse-error (type => 'no refc', line => $l, column => $c);            $self->{ca}->{value} .= '&#';
3160              $self->{state} = $self->{prev_state};
3161              ## Reconsume.
3162              redo A;
3163          }          }
3164          }
3165          if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) {      } elsif ($self->{state} == NCR_NUM_STATE) {
3166            !!!cp (1008);        if (0x0030 <= $self->{nc} and
3167            !!!parse-error (type => (sprintf 'invalid character reference:U+%04X', $code), line => $l, column => $c);            $self->{nc} <= 0x0039) { # 0..9
           $code = 0xFFFD;  
         } elsif ($code > 0x10FFFF) {  
           !!!cp (1009);  
           !!!parse-error (type => (sprintf 'invalid character reference:U-%08X', $code), line => $l, column => $c);  
           $code = 0xFFFD;  
         } elsif ($code == 0x000D) {  
           !!!cp (1010);  
           !!!parse-error (type => 'CR character reference', line => $l, column => $c);  
           $code = 0x000A;  
         } elsif (0x80 <= $code and $code <= 0x9F) {  
           !!!cp (1011);  
           !!!parse-error (type => (sprintf 'C1 character reference:U+%04X', $code), line => $l, column => $c);  
           $code = $c1_entity_char->{$code};  
         }  
   
         return {type => CHARACTER_TOKEN, data => chr $code,  
                 has_reference => 1,  
                 line => $l, column => $c,  
                };  
       } # X  
     } elsif (0x0030 <= $self->{next_char} and  
              $self->{next_char} <= 0x0039) { # 0..9  
       my $code = $self->{next_char} - 0x0030;  
       !!!next-input-character;  
         
       while (0x0030 <= $self->{next_char} and  
                 $self->{next_char} <= 0x0039) { # 0..9  
3168          !!!cp (1012);          !!!cp (1012);
3169          $code *= 10;          $self->{s_kwd} *= 10;
3170          $code += $self->{next_char} - 0x0030;          $self->{s_kwd} += $self->{nc} - 0x0030;
3171                    
3172            ## Stay in the state.
3173          !!!next-input-character;          !!!next-input-character;
3174        }          redo A;
3175          } elsif ($self->{nc} == 0x003B) { # ;
       if ($self->{next_char} == 0x003B) { # ;  
3176          !!!cp (1013);          !!!cp (1013);
3177          !!!next-input-character;          !!!next-input-character;
3178            #
3179        } else {        } else {
3180          !!!cp (1014);          !!!cp (1014);
3181          !!!parse-error (type => 'no refc', line => $l, column => $c);          !!!parse-error (type => 'no refc');
3182            ## Reconsume.
3183            #
3184        }        }
3185    
3186        if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) {        my $code = $self->{s_kwd};
3187          my $l = $self->{line_prev};
3188          my $c = $self->{column_prev};
3189          if ($charref_map->{$code}) {
3190          !!!cp (1015);          !!!cp (1015);
3191          !!!parse-error (type => (sprintf 'invalid character reference:U+%04X', $code), line => $l, column => $c);          !!!parse-error (type => 'invalid character reference',
3192          $code = 0xFFFD;                          text => (sprintf 'U+%04X', $code),
3193                            line => $l, column => $c);
3194            $code = $charref_map->{$code};
3195        } elsif ($code > 0x10FFFF) {        } elsif ($code > 0x10FFFF) {
3196          !!!cp (1016);          !!!cp (1016);
3197          !!!parse-error (type => (sprintf 'invalid character reference:U-%08X', $code), line => $l, column => $c);          !!!parse-error (type => 'invalid character reference',
3198                            text => (sprintf 'U-%08X', $code),
3199                            line => $l, column => $c);
3200          $code = 0xFFFD;          $code = 0xFFFD;
       } elsif ($code == 0x000D) {  
         !!!cp (1017);  
         !!!parse-error (type => 'CR character reference', line => $l, column => $c);  
         $code = 0x000A;  
       } elsif (0x80 <= $code and $code <= 0x9F) {  
         !!!cp (1018);  
         !!!parse-error (type => (sprintf 'C1 character reference:U+%04X', $code), line => $l, column => $c);  
         $code = $c1_entity_char->{$code};  
3201        }        }
3202          
3203        return {type => CHARACTER_TOKEN, data => chr $code, has_reference => 1,        if ($self->{prev_state} == DATA_STATE) {
3204                line => $l, column => $c,          !!!cp (992);
3205               };          $self->{state} = $self->{prev_state};
3206      } else {          ## Reconsume.
3207        !!!cp (1019);          !!!emit ({type => CHARACTER_TOKEN, data => chr $code,
3208        !!!parse-error (type => 'bare nero', line => $l, column => $c);                    line => $l, column => $c,
3209        !!!back-next-input-character ($self->{next_char});                   });
3210        $self->{next_char} = 0x0023; # #          redo A;
3211        return undef;        } else {
3212      }          !!!cp (991);
3213    } elsif ((0x0041 <= $self->{next_char} and          $self->{ca}->{value} .= chr $code;
3214              $self->{next_char} <= 0x005A) or          $self->{ca}->{has_reference} = 1;
3215             (0x0061 <= $self->{next_char} and          $self->{state} = $self->{prev_state};
3216              $self->{next_char} <= 0x007A)) {          ## Reconsume.
3217      my $entity_name = chr $self->{next_char};          redo A;
3218      !!!next-input-character;        }
3219        } elsif ($self->{state} == HEXREF_X_STATE) {
3220      my $value = $entity_name;        if ((0x0030 <= $self->{nc} and $self->{nc} <= 0x0039) or
3221      my $match = 0;            (0x0041 <= $self->{nc} and $self->{nc} <= 0x0046) or
3222      require Whatpm::_NamedEntityList;            (0x0061 <= $self->{nc} and $self->{nc} <= 0x0066)) {
3223      our $EntityChar;          # 0..9, A..F, a..f
3224            !!!cp (990);
3225      while (length $entity_name < 30 and          $self->{state} = HEXREF_HEX_STATE;
3226             ## NOTE: Some number greater than the maximum length of entity name          $self->{s_kwd} = 0;
3227             ((0x0041 <= $self->{next_char} and # a          ## Reconsume.
3228               $self->{next_char} <= 0x005A) or # x          redo A;
3229              (0x0061 <= $self->{next_char} and # a        } else {
3230               $self->{next_char} <= 0x007A) or # z          !!!parse-error (type => 'bare hcro',
3231              (0x0030 <= $self->{next_char} and # 0                          line => $self->{line_prev},
3232               $self->{next_char} <= 0x0039) or # 9                          column => $self->{column_prev} - 2);
3233              $self->{next_char} == 0x003B)) { # ;  
3234        $entity_name .= chr $self->{next_char};          ## NOTE: According to the spec algorithm, nothing is returned,
3235        if (defined $EntityChar->{$entity_name}) {          ## and then "&#" followed by "X" or "x" is appended to the parent
3236          if ($self->{next_char} == 0x003B) { # ;          ## element or the attribute value in the later processing.
3237            !!!cp (1020);  
3238            $value = $EntityChar->{$entity_name};          if ($self->{prev_state} == DATA_STATE) {
3239            $match = 1;            !!!cp (1005);
3240            !!!next-input-character;            $self->{state} = $self->{prev_state};
3241            last;            ## Reconsume.
3242              !!!emit ({type => CHARACTER_TOKEN,
3243                        data => '&' . $self->{s_kwd},
3244                        line => $self->{line_prev},
3245                        column => $self->{column_prev} - length $self->{s_kwd},
3246                       });
3247              redo A;
3248            } else {
3249              !!!cp (989);
3250              $self->{ca}->{value} .= '&' . $self->{s_kwd};
3251              $self->{state} = $self->{prev_state};
3252              ## Reconsume.
3253              redo A;
3254            }
3255          }
3256        } elsif ($self->{state} == HEXREF_HEX_STATE) {
3257          if (0x0030 <= $self->{nc} and $self->{nc} <= 0x0039) {
3258            # 0..9
3259            !!!cp (1002);
3260            $self->{s_kwd} *= 0x10;
3261            $self->{s_kwd} += $self->{nc} - 0x0030;
3262            ## Stay in the state.
3263            !!!next-input-character;
3264            redo A;
3265          } elsif (0x0061 <= $self->{nc} and
3266                   $self->{nc} <= 0x0066) { # a..f
3267            !!!cp (1003);
3268            $self->{s_kwd} *= 0x10;
3269            $self->{s_kwd} += $self->{nc} - 0x0060 + 9;
3270            ## Stay in the state.
3271            !!!next-input-character;
3272            redo A;
3273          } elsif (0x0041 <= $self->{nc} and
3274                   $self->{nc} <= 0x0046) { # A..F
3275            !!!cp (1004);
3276            $self->{s_kwd} *= 0x10;
3277            $self->{s_kwd} += $self->{nc} - 0x0040 + 9;
3278            ## Stay in the state.
3279            !!!next-input-character;
3280            redo A;
3281          } elsif ($self->{nc} == 0x003B) { # ;
3282            !!!cp (1006);
3283            !!!next-input-character;
3284            #
3285          } else {
3286            !!!cp (1007);
3287            !!!parse-error (type => 'no refc',
3288                            line => $self->{line},
3289                            column => $self->{column});
3290            ## Reconsume.
3291            #
3292          }
3293    
3294          my $code = $self->{s_kwd};
3295          my $l = $self->{line_prev};
3296          my $c = $self->{column_prev};
3297          if ($charref_map->{$code}) {
3298            !!!cp (1008);
3299            !!!parse-error (type => 'invalid character reference',
3300                            text => (sprintf 'U+%04X', $code),
3301                            line => $l, column => $c);
3302            $code = $charref_map->{$code};
3303          } elsif ($code > 0x10FFFF) {
3304            !!!cp (1009);
3305            !!!parse-error (type => 'invalid character reference',
3306                            text => (sprintf 'U-%08X', $code),
3307                            line => $l, column => $c);
3308            $code = 0xFFFD;
3309          }
3310    
3311          if ($self->{prev_state} == DATA_STATE) {
3312            !!!cp (988);
3313            $self->{state} = $self->{prev_state};
3314            ## Reconsume.
3315            !!!emit ({type => CHARACTER_TOKEN, data => chr $code,
3316                      line => $l, column => $c,
3317                     });
3318            redo A;
3319          } else {
3320            !!!cp (987);
3321            $self->{ca}->{value} .= chr $code;
3322            $self->{ca}->{has_reference} = 1;
3323            $self->{state} = $self->{prev_state};
3324            ## Reconsume.
3325            redo A;
3326          }
3327        } elsif ($self->{state} == ENTITY_NAME_STATE) {
3328          if (length $self->{s_kwd} < 30 and
3329              ## NOTE: Some number greater than the maximum length of entity name
3330              ((0x0041 <= $self->{nc} and # a
3331                $self->{nc} <= 0x005A) or # x
3332               (0x0061 <= $self->{nc} and # a
3333                $self->{nc} <= 0x007A) or # z
3334               (0x0030 <= $self->{nc} and # 0
3335                $self->{nc} <= 0x0039) or # 9
3336               $self->{nc} == 0x003B)) { # ;
3337            our $EntityChar;
3338            $self->{s_kwd} .= chr $self->{nc};
3339            if (defined $EntityChar->{$self->{s_kwd}}) {
3340              if ($self->{nc} == 0x003B) { # ;
3341                !!!cp (1020);
3342                $self->{entity__value} = $EntityChar->{$self->{s_kwd}};
3343                $self->{entity__match} = 1;
3344                !!!next-input-character;
3345                #
3346              } else {
3347                !!!cp (1021);
3348                $self->{entity__value} = $EntityChar->{$self->{s_kwd}};
3349                $self->{entity__match} = -1;
3350                ## Stay in the state.
3351                !!!next-input-character;
3352                redo A;
3353              }
3354          } else {          } else {
3355            !!!cp (1021);            !!!cp (1022);
3356            $value = $EntityChar->{$entity_name};            $self->{entity__value} .= chr $self->{nc};
3357            $match = -1;            $self->{entity__match} *= 2;
3358              ## Stay in the state.
3359            !!!next-input-character;            !!!next-input-character;
3360              redo A;
3361            }
3362          }
3363    
3364          my $data;
3365          my $has_ref;
3366          if ($self->{entity__match} > 0) {
3367            !!!cp (1023);
3368            $data = $self->{entity__value};
3369            $has_ref = 1;
3370            #
3371          } elsif ($self->{entity__match} < 0) {
3372            !!!parse-error (type => 'no refc');
3373            if ($self->{prev_state} != DATA_STATE and # in attribute
3374                $self->{entity__match} < -1) {
3375              !!!cp (1024);
3376              $data = '&' . $self->{s_kwd};
3377              #
3378            } else {
3379              !!!cp (1025);
3380              $data = $self->{entity__value};
3381              $has_ref = 1;
3382              #
3383          }          }
3384        } else {        } else {
3385          !!!cp (1022);          !!!cp (1026);
3386          $value .= chr $self->{next_char};          !!!parse-error (type => 'bare ero',
3387          $match *= 2;                          line => $self->{line_prev},
3388          !!!next-input-character;                          column => $self->{column_prev} - length $self->{s_kwd});
3389            $data = '&' . $self->{s_kwd};
3390            #
3391        }        }
3392      }    
3393              ## NOTE: In these cases, when a character reference is found,
3394      if ($match > 0) {        ## it is consumed and a character token is returned, or, otherwise,
3395        !!!cp (1023);        ## nothing is consumed and returned, according to the spec algorithm.
3396        return {type => CHARACTER_TOKEN, data => $value, has_reference => 1,        ## In this implementation, anything that has been examined by the
3397                line => $l, column => $c,        ## tokenizer is appended to the parent element or the attribute value
3398               };        ## as string, either literal string when no character reference or
3399      } elsif ($match < 0) {        ## entity-replaced string otherwise, in this stage, since any characters
3400        !!!parse-error (type => 'no refc', line => $l, column => $c);        ## that would not be consumed are appended in the data state or in an
3401        if ($in_attr and $match < -1) {        ## appropriate attribute value state anyway.
3402          !!!cp (1024);  
3403          return {type => CHARACTER_TOKEN, data => '&'.$entity_name,        if ($self->{prev_state} == DATA_STATE) {
3404                  line => $l, column => $c,          !!!cp (986);
3405                 };          $self->{state} = $self->{prev_state};
3406        } else {          ## Reconsume.
3407          !!!cp (1025);          !!!emit ({type => CHARACTER_TOKEN,
3408          return {type => CHARACTER_TOKEN, data => $value, has_reference => 1,                    data => $data,
3409                  line => $l, column => $c,                    line => $self->{line_prev},
3410                 };                    column => $self->{column_prev} + 1 - length $self->{s_kwd},
3411                     });
3412            redo A;
3413          } else {
3414            !!!cp (985);
3415            $self->{ca}->{value} .= $data;
3416            $self->{ca}->{has_reference} = 1 if $has_ref;
3417            $self->{state} = $self->{prev_state};
3418            ## Reconsume.
3419            redo A;
3420        }        }
3421      } else {      } else {
3422        !!!cp (1026);        die "$0: $self->{state}: Unknown state";
       !!!parse-error (type => 'bare ero', line => $l, column => $c);  
       ## NOTE: "No characters are consumed" in the spec.  
       return {type => CHARACTER_TOKEN, data => '&'.$value,  
               line => $l, column => $c,  
              };  
3423      }      }
3424    } else {    } # A  
3425      !!!cp (1027);  
3426      ## no characters are consumed    die "$0: _get_next_token: unexpected case";
3427      !!!parse-error (type => 'bare ero', line => $l, column => $c);  } # _get_next_token
     return undef;  
   }  
 } # _tokenize_attempt_to_consume_an_entity  
3428    
3429  sub _initialize_tree_constructor ($) {  sub _initialize_tree_constructor ($) {
3430    my $self = shift;    my $self = shift;
# Line 3057  sub _initialize_tree_constructor ($) { Line 3433  sub _initialize_tree_constructor ($) {
3433    ## TODO: Turn mutation events off # MUST    ## TODO: Turn mutation events off # MUST
3434    ## TODO: Turn loose Document option (manakai extension) on    ## TODO: Turn loose Document option (manakai extension) on
3435    $self->{document}->manakai_is_html (1); # MUST    $self->{document}->manakai_is_html (1); # MUST
3436      $self->{document}->set_user_data (manakai_source_line => 1);
3437      $self->{document}->set_user_data (manakai_source_column => 1);
3438  } # _initialize_tree_constructor  } # _initialize_tree_constructor
3439    
3440  sub _terminate_tree_constructor ($) {  sub _terminate_tree_constructor ($) {
# Line 3111  sub _tree_construction_initial ($) { Line 3489  sub _tree_construction_initial ($) {
3489        ## language.        ## language.
3490        my $doctype_name = $token->{name};        my $doctype_name = $token->{name};
3491        $doctype_name = '' unless defined $doctype_name;        $doctype_name = '' unless defined $doctype_name;
3492        $doctype_name =~ tr/a-z/A-Z/;        $doctype_name =~ tr/a-z/A-Z/; # ASCII case-insensitive
3493        if (not defined $token->{name} or # <!DOCTYPE>        if (not defined $token->{name} or # <!DOCTYPE>
3494            defined $token->{public_identifier} or            defined $token->{sysid}) {
           defined $token->{system_identifier}) {  
3495          !!!cp ('t1');          !!!cp ('t1');
3496          !!!parse-error (type => 'not HTML5', token => $token);          !!!parse-error (type => 'not HTML5', token => $token);
3497        } elsif ($doctype_name ne 'HTML') {        } elsif ($doctype_name ne 'HTML') {
3498          !!!cp ('t2');          !!!cp ('t2');
         ## ISSUE: ASCII case-insensitive? (in fact it does not matter)  
3499          !!!parse-error (type => 'not HTML5', token => $token);          !!!parse-error (type => 'not HTML5', token => $token);
3500          } elsif (defined $token->{pubid}) {
3501            if ($token->{pubid} eq 'XSLT-compat') {
3502              !!!cp ('t1.2');
3503              !!!parse-error (type => 'XSLT-compat', token => $token,
3504                              level => $self->{level}->{should});
3505            } else {
3506              !!!parse-error (type => 'not HTML5', token => $token);
3507            }
3508        } else {        } else {
3509          !!!cp ('t3');          !!!cp ('t3');
3510            #
3511        }        }
3512                
3513        my $doctype = $self->{document}->create_document_type_definition        my $doctype = $self->{document}->create_document_type_definition
3514          ($token->{name}); ## ISSUE: If name is missing (e.g. <!DOCTYPE>)?          ($token->{name}); ## ISSUE: If name is missing (e.g. <!DOCTYPE>)?
3515        ## NOTE: Default value for both |public_id| and |system_id| attributes        ## NOTE: Default value for both |public_id| and |system_id| attributes
3516        ## are empty strings, so that we don't set any value in missing cases.        ## are empty strings, so that we don't set any value in missing cases.
3517        $doctype->public_id ($token->{public_identifier})        $doctype->public_id ($token->{pubid}) if defined $token->{pubid};
3518            if defined $token->{public_identifier};        $doctype->system_id ($token->{sysid}) if defined $token->{sysid};
       $doctype->system_id ($token->{system_identifier})  
           if defined $token->{system_identifier};  
3519        ## NOTE: Other DocumentType attributes are null or empty lists.        ## NOTE: Other DocumentType attributes are null or empty lists.
3520        ## ISSUE: internalSubset = null??        ## ISSUE: internalSubset = null??
3521        $self->{document}->append_child ($doctype);        $self->{document}->append_child ($doctype);
# Line 3140  sub _tree_construction_initial ($) { Line 3523  sub _tree_construction_initial ($) {
3523        if ($token->{quirks} or $doctype_name ne 'HTML') {        if ($token->{quirks} or $doctype_name ne 'HTML') {
3524          !!!cp ('t4');          !!!cp ('t4');
3525          $self->{document}->manakai_compat_mode ('quirks');          $self->{document}->manakai_compat_mode ('quirks');
3526        } elsif (defined $token->{public_identifier}) {        } elsif (defined $token->{pubid}) {
3527          my $pubid = $token->{public_identifier};          my $pubid = $token->{pubid};
3528          $pubid =~ tr/a-z/A-z/;          $pubid =~ tr/a-z/A-z/;
3529          if ({          my $prefix = [
3530            "+//SILMARIL//DTD HTML PRO V0R11 19970101//EN" => 1,            "+//SILMARIL//DTD HTML PRO V0R11 19970101//",
3531            "-//ADVASOFT LTD//DTD HTML 3.0 ASWEDIT + EXTENSIONS//EN" => 1,            "-//ADVASOFT LTD//DTD HTML 3.0 ASWEDIT + EXTENSIONS//",
3532            "-//AS//DTD HTML 3.0 ASWEDIT + EXTENSIONS//EN" => 1,            "-//AS//DTD HTML 3.0 ASWEDIT + EXTENSIONS//",
3533            "-//IETF//DTD HTML 2.0 LEVEL 1//EN" => 1,            "-//IETF//DTD HTML 2.0 LEVEL 1//",
3534            "-//IETF//DTD HTML 2.0 LEVEL 2//EN" => 1,            "-//IETF//DTD HTML 2.0 LEVEL 2//",
3535            "-//IETF//DTD HTML 2.0 STRICT LEVEL 1//EN" => 1,            "-//IETF//DTD HTML 2.0 STRICT LEVEL 1//",
3536            "-//IETF//DTD HTML 2.0 STRICT LEVEL 2//EN" => 1,            "-//IETF//DTD HTML 2.0 STRICT LEVEL 2//",
3537            "-//IETF//DTD HTML 2.0 STRICT//EN" => 1,            "-//IETF//DTD HTML 2.0 STRICT//",
3538            "-//IETF//DTD HTML 2.0//EN" => 1,            "-//IETF//DTD HTML 2.0//",
3539            "-//IETF//DTD HTML 2.1E//EN" => 1,            "-//IETF//DTD HTML 2.1E//",
3540            "-//IETF//DTD HTML 3.0//EN" => 1,            "-//IETF//DTD HTML 3.0//",
3541            "-//IETF//DTD HTML 3.0//EN//" => 1,            "-//IETF//DTD HTML 3.2 FINAL//",
3542            "-//IETF//DTD HTML 3.2 FINAL//EN" => 1,            "-//IETF//DTD HTML 3.2//",
3543            "-//IETF//DTD HTML 3.2//EN" => 1,            "-//IETF//DTD HTML 3//",
3544            "-//IETF//DTD HTML 3//EN" => 1,            "-//IETF//DTD HTML LEVEL 0//",
3545            "-//IETF//DTD HTML LEVEL 0//EN" => 1,            "-//IETF//DTD HTML LEVEL 1//",
3546            "-//IETF//DTD HTML LEVEL 0//EN//2.0" => 1,            "-//IETF//DTD HTML LEVEL 2//",
3547            "-//IETF//DTD HTML LEVEL 1//EN" => 1,            "-//IETF//DTD HTML LEVEL 3//",
3548            "-//IETF//DTD HTML LEVEL 1//EN//2.0" => 1,            "-//IETF//DTD HTML STRICT LEVEL 0//",
3549            "-//IETF//DTD HTML LEVEL 2//EN" => 1,            "-//IETF//DTD HTML STRICT LEVEL 1//",
3550            "-//IETF//DTD HTML LEVEL 2//EN//2.0" => 1,            "-//IETF//DTD HTML STRICT LEVEL 2//",
3551            "-//IETF//DTD HTML LEVEL 3//EN" => 1,            "-//IETF//DTD HTML STRICT LEVEL 3//",
3552            "-//IETF//DTD HTML LEVEL 3//EN//3.0" => 1,            "-//IETF//DTD HTML STRICT//",
3553            "-//IETF//DTD HTML STRICT LEVEL 0//EN" => 1,            "-//IETF//DTD HTML//",
3554            "-//IETF//DTD HTML STRICT LEVEL 0//EN//2.0" => 1,            "-//METRIUS//DTD METRIUS PRESENTATIONAL//",
3555            "-//IETF//DTD HTML STRICT LEVEL 1//EN" => 1,            "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 HTML STRICT//",
3556            "-//IETF//DTD HTML STRICT LEVEL 1//EN//2.0" => 1,            "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 HTML//",
3557            "-//IETF//DTD HTML STRICT LEVEL 2//EN" => 1,            "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 TABLES//",
3558            "-//IETF//DTD HTML STRICT LEVEL 2//EN//2.0" => 1,            "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 HTML STRICT//",
3559            "-//IETF//DTD HTML STRICT LEVEL 3//EN" => 1,            "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 HTML//",
3560            "-//IETF//DTD HTML STRICT LEVEL 3//EN//3.0" => 1,            "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 TABLES//",
3561            "-//IETF//DTD HTML STRICT//EN" => 1,            "-//NETSCAPE COMM. CORP.//DTD HTML//",
3562            "-//IETF//DTD HTML STRICT//EN//2.0" => 1,            "-//NETSCAPE COMM. CORP.//DTD STRICT HTML//",
3563            "-//IETF//DTD HTML STRICT//EN//3.0" => 1,            "-//O'REILLY AND ASSOCIATES//DTD HTML 2.0//",
3564            "-//IETF//DTD HTML//EN" => 1,            "-//O'REILLY AND ASSOCIATES//DTD HTML EXTENDED 1.0//",
3565            "-//IETF//DTD HTML//EN//2.0" => 1,            "-//O'REILLY AND ASSOCIATES//DTD HTML EXTENDED RELAXED 1.0//",
3566            "-//IETF//DTD HTML//EN//3.0" => 1,            "-//SOFTQUAD SOFTWARE//DTD HOTMETAL PRO 6.0::19990601::EXTENSIONS TO HTML 4.0//",
3567            "-//METRIUS//DTD METRIUS PRESENTATIONAL//EN" => 1,            "-//SOFTQUAD//DTD HOTMETAL PRO 4.0::19971010::EXTENSIONS TO HTML 4.0//",
3568            "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 HTML STRICT//EN" => 1,            "-//SPYGLASS//DTD HTML 2.0 EXTENDED//",
3569            "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 HTML//EN" => 1,            "-//SQ//DTD HTML 2.0 HOTMETAL + EXTENSIONS//",
3570            "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 TABLES//EN" => 1,            "-//SUN MICROSYSTEMS CORP.//DTD HOTJAVA HTML//",
3571            "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 HTML STRICT//EN" => 1,            "-//SUN MICROSYSTEMS CORP.//DTD HOTJAVA STRICT HTML//",
3572            "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 HTML//EN" => 1,            "-//W3C//DTD HTML 3 1995-03-24//",
3573            "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 TABLES//EN" => 1,            "-//W3C//DTD HTML 3.2 DRAFT//",
3574            "-//NETSCAPE COMM. CORP.//DTD HTML//EN" => 1,            "-//W3C//DTD HTML 3.2 FINAL//",
3575            "-//NETSCAPE COMM. CORP.//DTD STRICT HTML//EN" => 1,            "-//W3C//DTD HTML 3.2//",
3576            "-//O'REILLY AND ASSOCIATES//DTD HTML 2.0//EN" => 1,            "-//W3C//DTD HTML 3.2S DRAFT//",
3577            "-//O'REILLY AND ASSOCIATES//DTD HTML EXTENDED 1.0//EN" => 1,            "-//W3C//DTD HTML 4.0 FRAMESET//",
3578            "-//O'REILLY AND ASSOCIATES//DTD HTML EXTENDED RELAXED 1.0//EN" => 1,            "-//W3C//DTD HTML 4.0 TRANSITIONAL//",
3579            "-//SOFTQUAD SOFTWARE//DTD HOTMETAL PRO 6.0::19990601::EXTENSIONS TO HTML 4.0//EN" => 1,            "-//W3C//DTD HTML EXPERIMETNAL 19960712//",
3580            "-//SOFTQUAD//DTD HOTMETAL PRO 4.0::19971010::EXTENSIONS TO HTML 4.0//EN" => 1,            "-//W3C//DTD HTML EXPERIMENTAL 970421//",
3581            "-//SPYGLASS//DTD HTML 2.0 EXTENDED//EN" => 1,            "-//W3C//DTD W3 HTML//",
3582            "-//SQ//DTD HTML 2.0 HOTMETAL + EXTENSIONS//EN" => 1,            "-//W3O//DTD W3 HTML 3.0//",
3583            "-//SUN MICROSYSTEMS CORP.//DTD HOTJAVA HTML//EN" => 1,            "-//WEBTECHS//DTD MOZILLA HTML 2.0//",
3584            "-//SUN MICROSYSTEMS CORP.//DTD HOTJAVA STRICT HTML//EN" => 1,            "-//WEBTECHS//DTD MOZILLA HTML//",
3585            "-//W3C//DTD HTML 3 1995-03-24//EN" => 1,          ]; # $prefix
3586            "-//W3C//DTD HTML 3.2 DRAFT//EN" => 1,          my $match;
3587            "-//W3C//DTD HTML 3.2 FINAL//EN" => 1,          for (@$prefix) {
3588            "-//W3C//DTD HTML 3.2//EN" => 1,            if (substr ($prefix, 0, length $_) eq $_) {
3589            "-//W3C//DTD HTML 3.2S DRAFT//EN" => 1,              $match = 1;
3590            "-//W3C//DTD HTML 4.0 FRAMESET//EN" => 1,              last;
3591            "-//W3C//DTD HTML 4.0 TRANSITIONAL//EN" => 1,            }
3592            "-//W3C//DTD HTML EXPERIMETNAL 19960712//EN" => 1,          }
3593            "-//W3C//DTD HTML EXPERIMENTAL 970421//EN" => 1,          if ($match or
3594            "-//W3C//DTD W3 HTML//EN" => 1,              $pubid eq "-//W3O//DTD W3 HTML STRICT 3.0//EN//" or
3595            "-//W3O//DTD W3 HTML 3.0//EN" => 1,              $pubid eq "-/W3C/DTD HTML 4.0 TRANSITIONAL/EN" or
3596            "-//W3O//DTD W3 HTML 3.0//EN//" => 1,              $pubid eq "HTML") {
           "-//W3O//DTD W3 HTML STRICT 3.0//EN//" => 1,  
           "-//WEBTECHS//DTD MOZILLA HTML 2.0//EN" => 1,  
           "-//WEBTECHS//DTD MOZILLA HTML//EN" => 1,  
           "-/W3C/DTD HTML 4.0 TRANSITIONAL/EN" => 1,  
           "HTML" => 1,  
         }->{$pubid}) {  
3597            !!!cp ('t5');            !!!cp ('t5');
3598            $self->{document}->manakai_compat_mode ('quirks');            $self->{document}->manakai_compat_mode ('quirks');
3599          } elsif ($pubid eq "-//W3C//DTD HTML 4.01 FRAMESET//EN" or          } elsif ($pubid =~ m[^-//W3C//DTD HTML 4.01 FRAMESET//] or
3600                   $pubid eq "-//W3C//DTD HTML 4.01 TRANSITIONAL//EN") {                   $pubid =~ m[^-//W3C//DTD HTML 4.01 TRANSITIONAL//]) {
3601            if (defined $token->{system_identifier}) {            if (defined $token->{sysid}) {
3602              !!!cp ('t6');              !!!cp ('t6');
3603              $self->{document}->manakai_compat_mode ('quirks');              $self->{document}->manakai_compat_mode ('quirks');
3604            } else {            } else {
3605              !!!cp ('t7');              !!!cp ('t7');
3606              $self->{document}->manakai_compat_mode ('limited quirks');              $self->{document}->manakai_compat_mode ('limited quirks');
3607            }            }
3608          } elsif ($pubid eq "-//W3C//DTD XHTML 1.0 FRAMESET//EN" or          } elsif ($pubid =~ m[^-//W3C//DTD XHTML 1.0 FRAMESET//] or
3609                   $pubid eq "-//W3C//DTD XHTML 1.0 TRANSITIONAL//EN") {                   $pubid =~ m[^-//W3C//DTD XHTML 1.0 TRANSITIONAL//]) {
3610            !!!cp ('t8');            !!!cp ('t8');
3611            $self->{document}->manakai_compat_mode ('limited quirks');            $self->{document}->manakai_compat_mode ('limited quirks');
3612          } else {          } else {
# Line 3238  sub _tree_construction_initial ($) { Line 3615  sub _tree_construction_initial ($) {
3615        } else {        } else {
3616          !!!cp ('t10');          !!!cp ('t10');
3617        }        }
3618        if (defined $token->{system_identifier}) {        if (defined $token->{sysid}) {
3619          my $sysid = $token->{system_identifier};          my $sysid = $token->{sysid};
3620          $sysid =~ tr/A-Z/a-z/;          $sysid =~ tr/A-Z/a-z/;
3621          if ($sysid eq "http://www.ibm.com/data/dtd/v11/ibmxhtml1-transitional.dtd") {          if ($sysid eq "http://www.ibm.com/data/dtd/v11/ibmxhtml1-transitional.dtd") {
3622            ## TODO: Check the spec: PUBLIC "(limited quirks)" "(quirks)"            ## NOTE: Ensure that |PUBLIC "(limited quirks)" "(quirks)"| is
3623              ## marked as quirks.
3624            $self->{document}->manakai_compat_mode ('quirks');            $self->{document}->manakai_compat_mode ('quirks');
3625            !!!cp ('t11');            !!!cp ('t11');
3626          } else {          } else {
# Line 3268  sub _tree_construction_initial ($) { Line 3646  sub _tree_construction_initial ($) {
3646        !!!ack-later;        !!!ack-later;
3647        return;        return;
3648      } elsif ($token->{type} == CHARACTER_TOKEN) {      } elsif ($token->{type} == CHARACTER_TOKEN) {
3649        if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D        if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
3650          ## Ignore the token          ## Ignore the token
3651    
3652          unless (length $token->{data}) {          unless (length $token->{data}) {
# Line 3325  sub _tree_construction_root_element ($) Line 3703  sub _tree_construction_root_element ($)
3703          !!!next-token;          !!!next-token;
3704          redo B;          redo B;
3705        } elsif ($token->{type} == CHARACTER_TOKEN) {        } elsif ($token->{type} == CHARACTER_TOKEN) {
3706          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D          if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
3707            ## Ignore the token.            ## Ignore the token.
3708    
3709            unless (length $token->{data}) {            unless (length $token->{data}) {
# Line 3428  sub _reset_insertion_mode ($) { Line 3806  sub _reset_insertion_mode ($) {
3806          ## NOTE: Strictly spaking, the line below only applies to MathML and          ## NOTE: Strictly spaking, the line below only applies to MathML and
3807          ## SVG elements.  Currently the HTML syntax supports only MathML and          ## SVG elements.  Currently the HTML syntax supports only MathML and
3808          ## SVG elements as foreigners.          ## SVG elements as foreigners.
3809          $new_mode = $self->{insertion_mode} | IN_FOREIGN_CONTENT_IM;          $new_mode = IN_BODY_IM | IN_FOREIGN_CONTENT_IM;
         ## ISSUE: What is set as the secondary insertion mode?  
3810        } elsif ($node->[1] & TABLE_CELL_EL) {        } elsif ($node->[1] & TABLE_CELL_EL) {
3811          if ($last) {          if ($last) {
3812            !!!cp ('t28.2');            !!!cp ('t28.2');
# Line 3630  sub _tree_construction_main ($) { Line 4007  sub _tree_construction_main ($) {
4007        ## NOTE: An end-of-file token.        ## NOTE: An end-of-file token.
4008        if ($content_model_flag == CDATA_CONTENT_MODEL) {        if ($content_model_flag == CDATA_CONTENT_MODEL) {
4009          !!!cp ('t43');          !!!cp ('t43');
4010          !!!parse-error (type => 'in CDATA:#'.$token->{type}, token => $token);          !!!parse-error (type => 'in CDATA:#eof', token => $token);
4011        } elsif ($content_model_flag == RCDATA_CONTENT_MODEL) {        } elsif ($content_model_flag == RCDATA_CONTENT_MODEL) {
4012          !!!cp ('t44');          !!!cp ('t44');
4013          !!!parse-error (type => 'in RCDATA:#'.$token->{type}, token => $token);          !!!parse-error (type => 'in RCDATA:#eof', token => $token);
4014        } else {        } else {
4015          die "$0: $content_model_flag in parse_rcdata";          die "$0: $content_model_flag in parse_rcdata";
4016        }        }
# Line 3670  sub _tree_construction_main ($) { Line 4047  sub _tree_construction_main ($) {
4047        ## Ignore the token        ## Ignore the token
4048      } else {      } else {
4049        !!!cp ('t48');        !!!cp ('t48');
4050        !!!parse-error (type => 'in CDATA:#'.$token->{type}, token => $token);        !!!parse-error (type => 'in CDATA:#eof', token => $token);
4051        ## ISSUE: And ignore?        ## ISSUE: And ignore?
4052        ## TODO: mark as "already executed"        ## TODO: mark as "already executed"
4053      }      }
# Line 3721  sub _tree_construction_main ($) { Line 4098  sub _tree_construction_main ($) {
4098        } # AFE        } # AFE
4099        unless (defined $formatting_element) {        unless (defined $formatting_element) {
4100          !!!cp ('t53');          !!!cp ('t53');
4101          !!!parse-error (type => 'unmatched end tag:'.$tag_name, token => $end_tag_token);          !!!parse-error (type => 'unmatched end tag', text => $tag_name, token => $end_tag_token);
4102          ## Ignore the token          ## Ignore the token
4103          !!!next-token;          !!!next-token;
4104          return;          return;
# Line 3738  sub _tree_construction_main ($) { Line 4115  sub _tree_construction_main ($) {
4115              last INSCOPE;              last INSCOPE;
4116            } else { # in open elements but not in scope            } else { # in open elements but not in scope
4117              !!!cp ('t55');              !!!cp ('t55');
4118              !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name},              !!!parse-error (type => 'unmatched end tag',
4119                                text => $token->{tag_name},
4120                              token => $end_tag_token);                              token => $end_tag_token);
4121              ## Ignore the token              ## Ignore the token
4122              !!!next-token;              !!!next-token;
# Line 3751  sub _tree_construction_main ($) { Line 4129  sub _tree_construction_main ($) {
4129        } # INSCOPE        } # INSCOPE
4130        unless (defined $formatting_element_i_in_open) {        unless (defined $formatting_element_i_in_open) {
4131          !!!cp ('t57');          !!!cp ('t57');
4132          !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name},          !!!parse-error (type => 'unmatched end tag',
4133                            text => $token->{tag_name},
4134                          token => $end_tag_token);                          token => $end_tag_token);
4135          pop @$active_formatting_elements; # $formatting_element          pop @$active_formatting_elements; # $formatting_element
4136          !!!next-token; ## TODO: ok?          !!!next-token; ## TODO: ok?
# Line 3760  sub _tree_construction_main ($) { Line 4139  sub _tree_construction_main ($) {
4139        if (not $self->{open_elements}->[-1]->[0] eq $formatting_element->[0]) {        if (not $self->{open_elements}->[-1]->[0] eq $formatting_element->[0]) {
4140          !!!cp ('t58');          !!!cp ('t58');
4141          !!!parse-error (type => 'not closed',          !!!parse-error (type => 'not closed',
4142                          value => $self->{open_elements}->[-1]->[0]                          text => $self->{open_elements}->[-1]->[0]
4143                              ->manakai_local_name,                              ->manakai_local_name,
4144                          token => $end_tag_token);                          token => $end_tag_token);
4145        }        }
# Line 3969  sub _tree_construction_main ($) { Line 4348  sub _tree_construction_main ($) {
4348    B: while (1) {    B: while (1) {
4349      if ($token->{type} == DOCTYPE_TOKEN) {      if ($token->{type} == DOCTYPE_TOKEN) {
4350        !!!cp ('t73');        !!!cp ('t73');
4351        !!!parse-error (type => 'DOCTYPE in the middle', token => $token);        !!!parse-error (type => 'in html:#DOCTYPE', token => $token);
4352        ## Ignore the token        ## Ignore the token
4353        ## Stay in the phase        ## Stay in the phase
4354        !!!next-token;        !!!next-token;
# Line 3978  sub _tree_construction_main ($) { Line 4357  sub _tree_construction_main ($) {
4357               $token->{tag_name} eq 'html') {               $token->{tag_name} eq 'html') {
4358        if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {        if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
4359          !!!cp ('t79');          !!!cp ('t79');
4360          !!!parse-error (type => 'after html:html', token => $token);          !!!parse-error (type => 'after html', text => 'html', token => $token);
4361          $self->{insertion_mode} = AFTER_BODY_IM;          $self->{insertion_mode} = AFTER_BODY_IM;
4362        } elsif ($self->{insertion_mode} == AFTER_HTML_FRAMESET_IM) {        } elsif ($self->{insertion_mode} == AFTER_HTML_FRAMESET_IM) {
4363          !!!cp ('t80');          !!!cp ('t80');
4364          !!!parse-error (type => 'after html:html', token => $token);          !!!parse-error (type => 'after html', text => 'html', token => $token);
4365          $self->{insertion_mode} = AFTER_FRAMESET_IM;          $self->{insertion_mode} = AFTER_FRAMESET_IM;
4366        } else {        } else {
4367          !!!cp ('t81');          !!!cp ('t81');
# Line 4033  sub _tree_construction_main ($) { Line 4412  sub _tree_construction_main ($) {
4412            #            #
4413          } elsif ({          } elsif ({
4414                    b => 1, big => 1, blockquote => 1, body => 1, br => 1,                    b => 1, big => 1, blockquote => 1, body => 1, br => 1,
4415                    center => 1, code => 1, dd => 1, div => 1, dl => 1, em => 1,                    center => 1, code => 1, dd => 1, div => 1, dl => 1, dt => 1,
4416                    embed => 1, font => 1, h1 => 1, h2 => 1, h3 => 1, ## No h4!                    em => 1, embed => 1, font => 1, h1 => 1, h2 => 1, h3 => 1,
4417                    h5 => 1, h6 => 1, head => 1, hr => 1, i => 1, img => 1,                    h4 => 1, h5 => 1, h6 => 1, head => 1, hr => 1, i => 1,
4418                    li => 1, menu => 1, meta => 1, nobr => 1, p => 1, pre => 1,                    img => 1, li => 1, listing => 1, menu => 1, meta => 1,
4419                    ruby => 1, s => 1, small => 1, span => 1, strong => 1,                    nobr => 1, ol => 1, p => 1, pre => 1, ruby => 1, s => 1,
4420                    sub => 1, sup => 1, table => 1, tt => 1, u => 1, ul => 1,                    small => 1, span => 1, strong => 1, strike => 1, sub => 1,
4421                    var => 1,                    sup => 1, table => 1, tt => 1, u => 1, ul => 1, var => 1,
4422                   }->{$token->{tag_name}}) {                   }->{$token->{tag_name}}) {
4423            !!!cp ('t87.2');            !!!cp ('t87.2');
4424            !!!parse-error (type => 'not closed',            !!!parse-error (type => 'not closed',
4425                            value => $self->{open_elements}->[-1]->[0]                            text => $self->{open_elements}->[-1]->[0]
4426                                ->manakai_local_name,                                ->manakai_local_name,
4427                            token => $token);                            token => $token);
4428    
# Line 4119  sub _tree_construction_main ($) { Line 4498  sub _tree_construction_main ($) {
4498          !!!cp ('t87.5');          !!!cp ('t87.5');
4499          #          #
4500        } elsif ($token->{type} == END_OF_FILE_TOKEN) {        } elsif ($token->{type} == END_OF_FILE_TOKEN) {
         ## NOTE: "using the rules for secondary insertion mode" then "continue"  
4501          !!!cp ('t87.6');          !!!cp ('t87.6');
4502          #          !!!parse-error (type => 'not closed',
4503          ## TODO: ...                          text => $self->{open_elements}->[-1]->[0]
4504                                ->manakai_local_name,
4505                            token => $token);
4506    
4507            pop @{$self->{open_elements}}
4508                while $self->{open_elements}->[-1]->[1] & FOREIGN_EL;
4509    
4510            $self->{insertion_mode} &= ~ IN_FOREIGN_CONTENT_IM;
4511            ## Reprocess.
4512            next B;
4513        } else {        } else {
4514          die "$0: $token->{type}: Unknown token type";                  die "$0: $token->{type}: Unknown token type";        
4515        }        }
# Line 4130  sub _tree_construction_main ($) { Line 4517  sub _tree_construction_main ($) {
4517    
4518      if ($self->{insertion_mode} & HEAD_IMS) {      if ($self->{insertion_mode} & HEAD_IMS) {
4519        if ($token->{type} == CHARACTER_TOKEN) {        if ($token->{type} == CHARACTER_TOKEN) {
4520          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {          if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
4521            unless ($self->{insertion_mode} == BEFORE_HEAD_IM) {            unless ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4522              !!!cp ('t88.2');              !!!cp ('t88.2');
4523              $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);              $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
4524                #
4525            } else {            } else {
4526              !!!cp ('t88.1');              !!!cp ('t88.1');
4527              ## Ignore the token.              ## Ignore the token.
4528              !!!next-token;              #
             next B;  
4529            }            }
4530            unless (length $token->{data}) {            unless (length $token->{data}) {
4531              !!!cp ('t88');              !!!cp ('t88');
4532              !!!next-token;              !!!next-token;
4533              next B;              next B;
4534            }            }
4535    ## TODO: set $token->{column} appropriately
4536          }          }
4537    
4538          if ($self->{insertion_mode} == BEFORE_HEAD_IM) {          if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
# Line 4163  sub _tree_construction_main ($) { Line 4551  sub _tree_construction_main ($) {
4551            !!!cp ('t90');            !!!cp ('t90');
4552            ## As if </noscript>            ## As if </noscript>
4553            pop @{$self->{open_elements}};            pop @{$self->{open_elements}};
4554            !!!parse-error (type => 'in noscript:#character', token => $token);            !!!parse-error (type => 'in noscript:#text', token => $token);
4555                        
4556            ## Reprocess in the "in head" insertion mode...            ## Reprocess in the "in head" insertion mode...
4557            ## As if </head>            ## As if </head>
# Line 4200  sub _tree_construction_main ($) { Line 4588  sub _tree_construction_main ($) {
4588              next B;              next B;
4589            } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {            } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4590              !!!cp ('t93.2');              !!!cp ('t93.2');
4591              !!!parse-error (type => 'after head:head', token => $token); ## TODO: error type              !!!parse-error (type => 'after head', text => 'head',
4592                                token => $token);
4593              ## Ignore the token              ## Ignore the token
4594              !!!nack ('t93.3');              !!!nack ('t93.3');
4595              !!!next-token;              !!!next-token;
4596              next B;              next B;
4597            } else {            } else {
4598              !!!cp ('t95');              !!!cp ('t95');
4599              !!!parse-error (type => 'in head:head', token => $token); # or in head noscript              !!!parse-error (type => 'in head:head',
4600                                token => $token); # or in head noscript
4601              ## Ignore the token              ## Ignore the token
4602              !!!nack ('t95.1');              !!!nack ('t95.1');
4603              !!!next-token;              !!!next-token;
# Line 4232  sub _tree_construction_main ($) { Line 4622  sub _tree_construction_main ($) {
4622                  !!!cp ('t98');                  !!!cp ('t98');
4623                  ## As if </noscript>                  ## As if </noscript>
4624                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
4625                  !!!parse-error (type => 'in noscript:base', token => $token);                  !!!parse-error (type => 'in noscript', text => 'base',
4626                                    token => $token);
4627                                
4628                  $self->{insertion_mode} = IN_HEAD_IM;                  $self->{insertion_mode} = IN_HEAD_IM;
4629                  ## Reprocess in the "in head" insertion mode...                  ## Reprocess in the "in head" insertion mode...
# Line 4243  sub _tree_construction_main ($) { Line 4634  sub _tree_construction_main ($) {
4634                ## NOTE: There is a "as if in head" code clone.                ## NOTE: There is a "as if in head" code clone.
4635                if ($self->{insertion_mode} == AFTER_HEAD_IM) {                if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4636                  !!!cp ('t100');                  !!!cp ('t100');
4637                  !!!parse-error (type => 'after head:'.$token->{tag_name}, token => $token);                  !!!parse-error (type => 'after head',
4638                                    text => $token->{tag_name}, token => $token);
4639                  push @{$self->{open_elements}},                  push @{$self->{open_elements}},
4640                      [$self->{head_element}, $el_category->{head}];                      [$self->{head_element}, $el_category->{head}];
4641                } else {                } else {
# Line 4260  sub _tree_construction_main ($) { Line 4652  sub _tree_construction_main ($) {
4652                ## NOTE: There is a "as if in head" code clone.                ## NOTE: There is a "as if in head" code clone.
4653                if ($self->{insertion_mode} == AFTER_HEAD_IM) {                if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4654                  !!!cp ('t102');                  !!!cp ('t102');
4655                  !!!parse-error (type => 'after head:'.$token->{tag_name}, token => $token);                  !!!parse-error (type => 'after head',
4656                                    text => $token->{tag_name}, token => $token);
4657                  push @{$self->{open_elements}},                  push @{$self->{open_elements}},
4658                      [$self->{head_element}, $el_category->{head}];                      [$self->{head_element}, $el_category->{head}];
4659                } else {                } else {
# Line 4277  sub _tree_construction_main ($) { Line 4670  sub _tree_construction_main ($) {
4670                ## NOTE: There is a "as if in head" code clone.                ## NOTE: There is a "as if in head" code clone.
4671                if ($self->{insertion_mode} == AFTER_HEAD_IM) {                if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4672                  !!!cp ('t104');                  !!!cp ('t104');
4673                  !!!parse-error (type => 'after head:'.$token->{tag_name}, token => $token);                  !!!parse-error (type => 'after head',
4674                                    text => $token->{tag_name}, token => $token);
4675                  push @{$self->{open_elements}},                  push @{$self->{open_elements}},
4676                      [$self->{head_element}, $el_category->{head}];                      [$self->{head_element}, $el_category->{head}];
4677                } else {                } else {
# Line 4301  sub _tree_construction_main ($) { Line 4695  sub _tree_construction_main ($) {
4695                                                 ->{has_reference});                                                 ->{has_reference});
4696                  } elsif ($token->{attributes}->{content}) {                  } elsif ($token->{attributes}->{content}) {
4697                    if ($token->{attributes}->{content}->{value}                    if ($token->{attributes}->{content}->{value}
4698                        =~ /\A[^;]*;[\x09-\x0D\x20]*[Cc][Hh][Aa][Rr][Ss][Ee][Tt]                        =~ /[Cc][Hh][Aa][Rr][Ss][Ee][Tt]
4699                            [\x09-\x0D\x20]*=                            [\x09\x0A\x0C\x0D\x20]*=
4700                            [\x09-\x0D\x20]*(?>"([^"]*)"|'([^']*)'|                            [\x09\x0A\x0C\x0D\x20]*(?>"([^"]*)"|'([^']*)'|
4701                            ([^"'\x09-\x0D\x20][^\x09-\x0D\x20]*))/x) {                            ([^"'\x09\x0A\x0C\x0D\x20]
4702                               [^\x09\x0A\x0C\x0D\x20\x3B]*))/x) {
4703                      !!!cp ('t107');                      !!!cp ('t107');
4704                      ## NOTE: Whether the encoding is supported or not is handled                      ## NOTE: Whether the encoding is supported or not is handled
4705                      ## in the {change_encoding} callback.                      ## in the {change_encoding} callback.
# Line 4346  sub _tree_construction_main ($) { Line 4741  sub _tree_construction_main ($) {
4741                  !!!cp ('t111');                  !!!cp ('t111');
4742                  ## As if </noscript>                  ## As if </noscript>
4743                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
4744                  !!!parse-error (type => 'in noscript:title', token => $token);                  !!!parse-error (type => 'in noscript', text => 'title',
4745                                    token => $token);
4746                                
4747                  $self->{insertion_mode} = IN_HEAD_IM;                  $self->{insertion_mode} = IN_HEAD_IM;
4748                  ## Reprocess in the "in head" insertion mode...                  ## Reprocess in the "in head" insertion mode...
4749                } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {                } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4750                  !!!cp ('t112');                  !!!cp ('t112');
4751                  !!!parse-error (type => 'after head:'.$token->{tag_name}, token => $token);                  !!!parse-error (type => 'after head',
4752                                    text => $token->{tag_name}, token => $token);
4753                  push @{$self->{open_elements}},                  push @{$self->{open_elements}},
4754                      [$self->{head_element}, $el_category->{head}];                      [$self->{head_element}, $el_category->{head}];
4755                } else {                } else {
# Line 4366  sub _tree_construction_main ($) { Line 4763  sub _tree_construction_main ($) {
4763                pop @{$self->{open_elements}} # <head>                pop @{$self->{open_elements}} # <head>
4764                    if $self->{insertion_mode} == AFTER_HEAD_IM;                    if $self->{insertion_mode} == AFTER_HEAD_IM;
4765                next B;                next B;
4766              } elsif ($token->{tag_name} eq 'style') {              } elsif ($token->{tag_name} eq 'style' or
4767                         $token->{tag_name} eq 'noframes') {
4768                ## NOTE: Or (scripting is enabled and tag_name eq 'noscript' and                ## NOTE: Or (scripting is enabled and tag_name eq 'noscript' and
4769                ## insertion mode IN_HEAD_IM)                ## insertion mode IN_HEAD_IM)
4770                ## NOTE: There is a "as if in head" code clone.                ## NOTE: There is a "as if in head" code clone.
4771                if ($self->{insertion_mode} == AFTER_HEAD_IM) {                if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4772                  !!!cp ('t114');                  !!!cp ('t114');
4773                  !!!parse-error (type => 'after head:'.$token->{tag_name}, token => $token);                  !!!parse-error (type => 'after head',
4774                                    text => $token->{tag_name}, token => $token);
4775                  push @{$self->{open_elements}},                  push @{$self->{open_elements}},
4776                      [$self->{head_element}, $el_category->{head}];                      [$self->{head_element}, $el_category->{head}];
4777                } else {                } else {
# Line 4393  sub _tree_construction_main ($) { Line 4792  sub _tree_construction_main ($) {
4792                  next B;                  next B;
4793                } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {                } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4794                  !!!cp ('t117');                  !!!cp ('t117');
4795                  !!!parse-error (type => 'in noscript:noscript', token => $token);                  !!!parse-error (type => 'in noscript', text => 'noscript',
4796                                    token => $token);
4797                  ## Ignore the token                  ## Ignore the token
4798                  !!!nack ('t117.1');                  !!!nack ('t117.1');
4799                  !!!next-token;                  !!!next-token;
# Line 4407  sub _tree_construction_main ($) { Line 4807  sub _tree_construction_main ($) {
4807                  !!!cp ('t119');                  !!!cp ('t119');
4808                  ## As if </noscript>                  ## As if </noscript>
4809                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
4810                  !!!parse-error (type => 'in noscript:script', token => $token);                  !!!parse-error (type => 'in noscript', text => 'script',
4811                                    token => $token);
4812                                
4813                  $self->{insertion_mode} = IN_HEAD_IM;                  $self->{insertion_mode} = IN_HEAD_IM;
4814                  ## Reprocess in the "in head" insertion mode...                  ## Reprocess in the "in head" insertion mode...
4815                } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {                } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4816                  !!!cp ('t120');                  !!!cp ('t120');
4817                  !!!parse-error (type => 'after head:'.$token->{tag_name}, token => $token);                  !!!parse-error (type => 'after head',
4818                                    text => $token->{tag_name}, token => $token);
4819                  push @{$self->{open_elements}},                  push @{$self->{open_elements}},
4820                      [$self->{head_element}, $el_category->{head}];                      [$self->{head_element}, $el_category->{head}];
4821                } else {                } else {
# Line 4431  sub _tree_construction_main ($) { Line 4833  sub _tree_construction_main ($) {
4833                  !!!cp ('t122');                  !!!cp ('t122');
4834                  ## As if </noscript>                  ## As if </noscript>
4835                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
4836                  !!!parse-error (type => 'in noscript:'.$token->{tag_name}, token => $token);                  !!!parse-error (type => 'in noscript',
4837                                    text => $token->{tag_name}, token => $token);
4838                                    
4839                  ## Reprocess in the "in head" insertion mode...                  ## Reprocess in the "in head" insertion mode...
4840                  ## As if </head>                  ## As if </head>
# Line 4470  sub _tree_construction_main ($) { Line 4873  sub _tree_construction_main ($) {
4873                !!!cp ('t129');                !!!cp ('t129');
4874                ## As if </noscript>                ## As if </noscript>
4875                pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
4876                !!!parse-error (type => 'in noscript:/'.$token->{tag_name}, token => $token);                !!!parse-error (type => 'in noscript:/',
4877                                  text => $token->{tag_name}, token => $token);
4878                                
4879                ## Reprocess in the "in head" insertion mode...                ## Reprocess in the "in head" insertion mode...
4880                ## As if </head>                ## As if </head>
# Line 4513  sub _tree_construction_main ($) { Line 4917  sub _tree_construction_main ($) {
4917                  !!!cp ('t133');                  !!!cp ('t133');
4918                  ## As if </noscript>                  ## As if </noscript>
4919                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
4920                  !!!parse-error (type => 'in noscript:/head', token => $token);                  !!!parse-error (type => 'in noscript:/',
4921                                    text => 'head', token => $token);
4922                                    
4923                  ## Reprocess in the "in head" insertion mode...                  ## Reprocess in the "in head" insertion mode...
4924                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
# Line 4528  sub _tree_construction_main ($) { Line 4933  sub _tree_construction_main ($) {
4933                  next B;                  next B;
4934                } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {                } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4935                  !!!cp ('t134.1');                  !!!cp ('t134.1');
4936                  !!!parse-error (type => 'unmatched end tag:head', token => $token);                  !!!parse-error (type => 'unmatched end tag', text => 'head',
4937                                    token => $token);
4938                  ## Ignore the token                  ## Ignore the token
4939                  !!!next-token;                  !!!next-token;
4940                  next B;                  next B;
# Line 4545  sub _tree_construction_main ($) { Line 4951  sub _tree_construction_main ($) {
4951                } elsif ($self->{insertion_mode} == BEFORE_HEAD_IM or                } elsif ($self->{insertion_mode} == BEFORE_HEAD_IM or
4952                         $self->{insertion_mode} == AFTER_HEAD_IM) {                         $self->{insertion_mode} == AFTER_HEAD_IM) {
4953                  !!!cp ('t137');                  !!!cp ('t137');
4954                  !!!parse-error (type => 'unmatched end tag:noscript', token => $token);                  !!!parse-error (type => 'unmatched end tag',
4955                                    text => 'noscript', token => $token);
4956                  ## Ignore the token ## ISSUE: An issue in the spec.                  ## Ignore the token ## ISSUE: An issue in the spec.
4957                  !!!next-token;                  !!!next-token;
4958                  next B;                  next B;
# Line 4560  sub _tree_construction_main ($) { Line 4967  sub _tree_construction_main ($) {
4967                    $self->{insertion_mode} == IN_HEAD_IM or                    $self->{insertion_mode} == IN_HEAD_IM or
4968                    $self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {                    $self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4969                  !!!cp ('t140');                  !!!cp ('t140');
4970                  !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);                  !!!parse-error (type => 'unmatched end tag',
4971                                    text => $token->{tag_name}, token => $token);
4972                  ## Ignore the token                  ## Ignore the token
4973                  !!!next-token;                  !!!next-token;
4974                  next B;                  next B;
4975                } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {                } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4976                  !!!cp ('t140.1');                  !!!cp ('t140.1');
4977                  !!!parse-error (type => 'unmatched end tag:' . $token->{tag_name}, token => $token);                  !!!parse-error (type => 'unmatched end tag',
4978                                    text => $token->{tag_name}, token => $token);
4979                  ## Ignore the token                  ## Ignore the token
4980                  !!!next-token;                  !!!next-token;
4981                  next B;                  next B;
# Line 4575  sub _tree_construction_main ($) { Line 4984  sub _tree_construction_main ($) {
4984                }                }
4985              } elsif ($token->{tag_name} eq 'p') {              } elsif ($token->{tag_name} eq 'p') {
4986                !!!cp ('t142');                !!!cp ('t142');
4987                !!!parse-error (type => 'unmatched end tag:p', token => $token);                !!!parse-error (type => 'unmatched end tag',
4988                                  text => $token->{tag_name}, token => $token);
4989                ## Ignore the token                ## Ignore the token
4990                !!!next-token;                !!!next-token;
4991                next B;                next B;
# Line 4598  sub _tree_construction_main ($) { Line 5008  sub _tree_construction_main ($) {
5008                } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {                } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
5009                  !!!cp ('t143.3');                  !!!cp ('t143.3');
5010                  ## ISSUE: Two parse errors for <head><noscript></br>                  ## ISSUE: Two parse errors for <head><noscript></br>
5011                  !!!parse-error (type => 'unmatched end tag:br', token => $token);                  !!!parse-error (type => 'unmatched end tag',
5012                                    text => 'br', token => $token);
5013                  ## As if </noscript>                  ## As if </noscript>
5014                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
5015                  $self->{insertion_mode} = IN_HEAD_IM;                  $self->{insertion_mode} = IN_HEAD_IM;
# Line 4617  sub _tree_construction_main ($) { Line 5028  sub _tree_construction_main ($) {
5028                }                }
5029    
5030                ## ISSUE: does not agree with IE7 - it doesn't ignore </br>.                ## ISSUE: does not agree with IE7 - it doesn't ignore </br>.
5031                !!!parse-error (type => 'unmatched end tag:br', token => $token);                !!!parse-error (type => 'unmatched end tag',
5032                                  text => 'br', token => $token);
5033                ## Ignore the token                ## Ignore the token
5034                !!!next-token;                !!!next-token;
5035                next B;                next B;
5036              } else {              } else {
5037                !!!cp ('t145');                !!!cp ('t145');
5038                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);                !!!parse-error (type => 'unmatched end tag',
5039                                  text => $token->{tag_name}, token => $token);
5040                ## Ignore the token                ## Ignore the token
5041                !!!next-token;                !!!next-token;
5042                next B;                next B;
# Line 4633  sub _tree_construction_main ($) { Line 5046  sub _tree_construction_main ($) {
5046                !!!cp ('t146');                !!!cp ('t146');
5047                ## As if </noscript>                ## As if </noscript>
5048                pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
5049                !!!parse-error (type => 'in noscript:/'.$token->{tag_name}, token => $token);                !!!parse-error (type => 'in noscript:/',
5050                                  text => $token->{tag_name}, token => $token);
5051                                
5052                ## Reprocess in the "in head" insertion mode...                ## Reprocess in the "in head" insertion mode...
5053                ## As if </head>                ## As if </head>
# Line 4649  sub _tree_construction_main ($) { Line 5063  sub _tree_construction_main ($) {
5063              } elsif ($self->{insertion_mode} == BEFORE_HEAD_IM) {              } elsif ($self->{insertion_mode} == BEFORE_HEAD_IM) {
5064  ## ISSUE: This case cannot be reached?  ## ISSUE: This case cannot be reached?
5065                !!!cp ('t148');                !!!cp ('t148');
5066                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);                !!!parse-error (type => 'unmatched end tag',
5067                                  text => $token->{tag_name}, token => $token);
5068                ## Ignore the token ## ISSUE: An issue in the spec.                ## Ignore the token ## ISSUE: An issue in the spec.
5069                !!!next-token;                !!!next-token;
5070                next B;                next B;
# Line 4760  sub _tree_construction_main ($) { Line 5175  sub _tree_construction_main ($) {
5175    
5176                  !!!cp ('t153');                  !!!cp ('t153');
5177                  !!!parse-error (type => 'start tag not allowed',                  !!!parse-error (type => 'start tag not allowed',
5178                      value => $token->{tag_name}, token => $token);                      text => $token->{tag_name}, token => $token);
5179                  ## Ignore the token                  ## Ignore the token
5180                  !!!nack ('t153.1');                  !!!nack ('t153.1');
5181                  !!!next-token;                  !!!next-token;
5182                  next B;                  next B;
5183                } elsif ($self->{insertion_mode} == IN_CAPTION_IM) {                } elsif ($self->{insertion_mode} == IN_CAPTION_IM) {
5184                  !!!parse-error (type => 'not closed:caption', token => $token);                  !!!parse-error (type => 'not closed', text => 'caption',
5185                                    token => $token);
5186                                    
5187                  ## NOTE: As if </caption>.                  ## NOTE: As if </caption>.
5188                  ## have a table element in table scope                  ## have a table element in table scope
# Line 4786  sub _tree_construction_main ($) { Line 5202  sub _tree_construction_main ($) {
5202    
5203                    !!!cp ('t157');                    !!!cp ('t157');
5204                    !!!parse-error (type => 'start tag not allowed',                    !!!parse-error (type => 'start tag not allowed',
5205                                    value => $token->{tag_name}, token => $token);                                    text => $token->{tag_name}, token => $token);
5206                    ## Ignore the token                    ## Ignore the token
5207                    !!!nack ('t157.1');                    !!!nack ('t157.1');
5208                    !!!next-token;                    !!!next-token;
# Line 4803  sub _tree_construction_main ($) { Line 5219  sub _tree_construction_main ($) {
5219                  unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {                  unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {
5220                    !!!cp ('t159');                    !!!cp ('t159');
5221                    !!!parse-error (type => 'not closed',                    !!!parse-error (type => 'not closed',
5222                                    value => $self->{open_elements}->[-1]->[0]                                    text => $self->{open_elements}->[-1]->[0]
5223                                        ->manakai_local_name,                                        ->manakai_local_name,
5224                                    token => $token);                                    token => $token);
5225                  } else {                  } else {
# Line 4845  sub _tree_construction_main ($) { Line 5261  sub _tree_construction_main ($) {
5261                  } # INSCOPE                  } # INSCOPE
5262                    unless (defined $i) {                    unless (defined $i) {
5263                      !!!cp ('t165');                      !!!cp ('t165');
5264                      !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);                      !!!parse-error (type => 'unmatched end tag',
5265                                        text => $token->{tag_name},
5266                                        token => $token);
5267                      ## Ignore the token                      ## Ignore the token
5268                      !!!next-token;                      !!!next-token;
5269                      next B;                      next B;
# Line 4862  sub _tree_construction_main ($) { Line 5280  sub _tree_construction_main ($) {
5280                          ne $token->{tag_name}) {                          ne $token->{tag_name}) {
5281                    !!!cp ('t167');                    !!!cp ('t167');
5282                    !!!parse-error (type => 'not closed',                    !!!parse-error (type => 'not closed',
5283                                    value => $self->{open_elements}->[-1]->[0]                                    text => $self->{open_elements}->[-1]->[0]
5284                                        ->manakai_local_name,                                        ->manakai_local_name,
5285                                    token => $token);                                    token => $token);
5286                  } else {                  } else {
# Line 4879  sub _tree_construction_main ($) { Line 5297  sub _tree_construction_main ($) {
5297                  next B;                  next B;
5298                } elsif ($self->{insertion_mode} == IN_CAPTION_IM) {                } elsif ($self->{insertion_mode} == IN_CAPTION_IM) {
5299                  !!!cp ('t169');                  !!!cp ('t169');
5300                  !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);                  !!!parse-error (type => 'unmatched end tag',
5301                                    text => $token->{tag_name}, token => $token);
5302                  ## Ignore the token                  ## Ignore the token
5303                  !!!next-token;                  !!!next-token;
5304                  next B;                  next B;
# Line 4906  sub _tree_construction_main ($) { Line 5325  sub _tree_construction_main ($) {
5325    
5326                    !!!cp ('t173');                    !!!cp ('t173');
5327                    !!!parse-error (type => 'unmatched end tag',                    !!!parse-error (type => 'unmatched end tag',
5328                                    value => $token->{tag_name}, token => $token);                                    text => $token->{tag_name}, token => $token);
5329                    ## Ignore the token                    ## Ignore the token
5330                    !!!next-token;                    !!!next-token;
5331                    next B;                    next B;
# Line 4922  sub _tree_construction_main ($) { Line 5341  sub _tree_construction_main ($) {
5341                  unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {                  unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {
5342                    !!!cp ('t175');                    !!!cp ('t175');
5343                    !!!parse-error (type => 'not closed',                    !!!parse-error (type => 'not closed',
5344                                    value => $self->{open_elements}->[-1]->[0]                                    text => $self->{open_elements}->[-1]->[0]
5345                                        ->manakai_local_name,                                        ->manakai_local_name,
5346                                    token => $token);                                    token => $token);
5347                  } else {                  } else {
# Line 4939  sub _tree_construction_main ($) { Line 5358  sub _tree_construction_main ($) {
5358                  next B;                  next B;
5359                } elsif ($self->{insertion_mode} == IN_CELL_IM) {                } elsif ($self->{insertion_mode} == IN_CELL_IM) {
5360                  !!!cp ('t177');                  !!!cp ('t177');
5361                  !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);                  !!!parse-error (type => 'unmatched end tag',
5362                                    text => $token->{tag_name}, token => $token);
5363                  ## Ignore the token                  ## Ignore the token
5364                  !!!next-token;                  !!!next-token;
5365                  next B;                  next B;
# Line 4982  sub _tree_construction_main ($) { Line 5402  sub _tree_construction_main ($) {
5402    
5403                  !!!cp ('t182');                  !!!cp ('t182');
5404                  !!!parse-error (type => 'unmatched end tag',                  !!!parse-error (type => 'unmatched end tag',
5405                      value => $token->{tag_name}, token => $token);                      text => $token->{tag_name}, token => $token);
5406                  ## Ignore the token                  ## Ignore the token
5407                  !!!next-token;                  !!!next-token;
5408                  next B;                  next B;
5409                } # INSCOPE                } # INSCOPE
5410              } elsif ($token->{tag_name} eq 'table' and              } elsif ($token->{tag_name} eq 'table' and
5411                       $self->{insertion_mode} == IN_CAPTION_IM) {                       $self->{insertion_mode} == IN_CAPTION_IM) {
5412                !!!parse-error (type => 'not closed:caption', token => $token);                !!!parse-error (type => 'not closed', text => 'caption',
5413                                  token => $token);
5414    
5415                ## As if </caption>                ## As if </caption>
5416                ## have a table element in table scope                ## have a table element in table scope
# Line 5007  sub _tree_construction_main ($) { Line 5428  sub _tree_construction_main ($) {
5428                } # INSCOPE                } # INSCOPE
5429                unless (defined $i) {                unless (defined $i) {
5430                  !!!cp ('t186');                  !!!cp ('t186');
5431                  !!!parse-error (type => 'unmatched end tag:caption', token => $token);                  !!!parse-error (type => 'unmatched end tag',
5432                                    text => 'caption', token => $token);
5433                  ## Ignore the token                  ## Ignore the token
5434                  !!!next-token;                  !!!next-token;
5435                  next B;                  next B;
# Line 5022  sub _tree_construction_main ($) { Line 5444  sub _tree_construction_main ($) {
5444                unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {                unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {
5445                  !!!cp ('t188');                  !!!cp ('t188');
5446                  !!!parse-error (type => 'not closed',                  !!!parse-error (type => 'not closed',
5447                                  value => $self->{open_elements}->[-1]->[0]                                  text => $self->{open_elements}->[-1]->[0]
5448                                      ->manakai_local_name,                                      ->manakai_local_name,
5449                                  token => $token);                                  token => $token);
5450                } else {                } else {
# Line 5042  sub _tree_construction_main ($) { Line 5464  sub _tree_construction_main ($) {
5464                       }->{$token->{tag_name}}) {                       }->{$token->{tag_name}}) {
5465                if ($self->{insertion_mode} & BODY_TABLE_IMS) {                if ($self->{insertion_mode} & BODY_TABLE_IMS) {
5466                  !!!cp ('t190');                  !!!cp ('t190');
5467                  !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);                  !!!parse-error (type => 'unmatched end tag',
5468                                    text => $token->{tag_name}, token => $token);
5469                  ## Ignore the token                  ## Ignore the token
5470                  !!!next-token;                  !!!next-token;
5471                  next B;                  next B;
# Line 5056  sub _tree_construction_main ($) { Line 5479  sub _tree_construction_main ($) {
5479                       }->{$token->{tag_name}} and                       }->{$token->{tag_name}} and
5480                       $self->{insertion_mode} == IN_CAPTION_IM) {                       $self->{insertion_mode} == IN_CAPTION_IM) {
5481                !!!cp ('t192');                !!!cp ('t192');
5482                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);                !!!parse-error (type => 'unmatched end tag',
5483                                  text => $token->{tag_name}, token => $token);
5484                ## Ignore the token                ## Ignore the token
5485                !!!next-token;                !!!next-token;
5486                next B;                next B;
# Line 5084  sub _tree_construction_main ($) { Line 5508  sub _tree_construction_main ($) {
5508      } elsif ($self->{insertion_mode} & TABLE_IMS) {      } elsif ($self->{insertion_mode} & TABLE_IMS) {
5509        if ($token->{type} == CHARACTER_TOKEN) {        if ($token->{type} == CHARACTER_TOKEN) {
5510          if (not $open_tables->[-1]->[1] and # tainted          if (not $open_tables->[-1]->[1] and # tainted
5511              $token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {              $token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
5512            $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);            $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
5513                                
5514            unless (length $token->{data}) {            unless (length $token->{data}) {
# Line 5096  sub _tree_construction_main ($) { Line 5520  sub _tree_construction_main ($) {
5520            }            }
5521          }          }
5522    
5523              !!!parse-error (type => 'in table:#character', token => $token);          !!!parse-error (type => 'in table:#text', token => $token);
5524    
5525              ## As if in body, but insert into foster parent element              ## As if in body, but insert into foster parent element
5526              ## ISSUE: Spec says that "whenever a node would be inserted              ## ISSUE: Spec says that "whenever a node would be inserted
# Line 5147  sub _tree_construction_main ($) { Line 5571  sub _tree_construction_main ($) {
5571          !!!next-token;          !!!next-token;
5572          next B;          next B;
5573        } elsif ($token->{type} == START_TAG_TOKEN) {        } elsif ($token->{type} == START_TAG_TOKEN) {
5574              if ({          if ({
5575                   tr => ($self->{insertion_mode} != IN_ROW_IM),               tr => ($self->{insertion_mode} != IN_ROW_IM),
5576                   th => 1, td => 1,               th => 1, td => 1,
5577                  }->{$token->{tag_name}}) {              }->{$token->{tag_name}}) {
5578                if ($self->{insertion_mode} == IN_TABLE_IM) {            if ($self->{insertion_mode} == IN_TABLE_IM) {
5579                  ## Clear back to table context              ## Clear back to table context
5580                  while (not ($self->{open_elements}->[-1]->[1]              while (not ($self->{open_elements}->[-1]->[1]
5581                                  & TABLE_SCOPING_EL)) {                              & TABLE_SCOPING_EL)) {
5582                    !!!cp ('t201');                !!!cp ('t201');
5583                    pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
5584                  }              }
5585                                
5586                  !!!insert-element ('tbody',, $token);              !!!insert-element ('tbody',, $token);
5587                  $self->{insertion_mode} = IN_TABLE_BODY_IM;              $self->{insertion_mode} = IN_TABLE_BODY_IM;
5588                  ## reprocess in the "in table body" insertion mode...              ## reprocess in the "in table body" insertion mode...
5589                }            }
5590              
5591                if ($self->{insertion_mode} == IN_TABLE_BODY_IM) {            if ($self->{insertion_mode} == IN_TABLE_BODY_IM) {
5592                  unless ($token->{tag_name} eq 'tr') {              unless ($token->{tag_name} eq 'tr') {
5593                    !!!cp ('t202');                !!!cp ('t202');
5594                    !!!parse-error (type => 'missing start tag:tr', token => $token);                !!!parse-error (type => 'missing start tag:tr', token => $token);
5595                  }              }
5596                                    
5597                  ## Clear back to table body context              ## Clear back to table body context
5598                  while (not ($self->{open_elements}->[-1]->[1]              while (not ($self->{open_elements}->[-1]->[1]
5599                                  & TABLE_ROWS_SCOPING_EL)) {                              & TABLE_ROWS_SCOPING_EL)) {
5600                    !!!cp ('t203');                !!!cp ('t203');
5601                    ## ISSUE: Can this case be reached?                ## ISSUE: Can this case be reached?
5602                    pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
5603                  }              }
5604                                    
5605                  $self->{insertion_mode} = IN_ROW_IM;                  $self->{insertion_mode} = IN_ROW_IM;
5606                  if ($token->{tag_name} eq 'tr') {                  if ($token->{tag_name} eq 'tr') {
# Line 5232  sub _tree_construction_main ($) { Line 5656  sub _tree_construction_main ($) {
5656                  unless (defined $i) {                  unless (defined $i) {
5657                    !!!cp ('t210');                    !!!cp ('t210');
5658  ## TODO: This type is wrong.  ## TODO: This type is wrong.
5659                    !!!parse-error (type => 'unmacthed end tag:'.$token->{tag_name}, token => $token);                    !!!parse-error (type => 'unmacthed end tag',
5660                                      text => $token->{tag_name}, token => $token);
5661                    ## Ignore the token                    ## Ignore the token
5662                    !!!nack ('t210.1');                    !!!nack ('t210.1');
5663                    !!!next-token;                    !!!next-token;
# Line 5276  sub _tree_construction_main ($) { Line 5701  sub _tree_construction_main ($) {
5701                  } # INSCOPE                  } # INSCOPE
5702                  unless (defined $i) {                  unless (defined $i) {
5703                    !!!cp ('t216');                    !!!cp ('t216');
5704  ## TODO: This erorr type ios wrong.  ## TODO: This erorr type is wrong.
5705                    !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);                    !!!parse-error (type => 'unmatched end tag',
5706                                      text => $token->{tag_name}, token => $token);
5707                    ## Ignore the token                    ## Ignore the token
5708                    !!!nack ('t216.1');                    !!!nack ('t216.1');
5709                    !!!next-token;                    !!!next-token;
# Line 5352  sub _tree_construction_main ($) { Line 5778  sub _tree_construction_main ($) {
5778                }                }
5779              } elsif ($token->{tag_name} eq 'table') {              } elsif ($token->{tag_name} eq 'table') {
5780                !!!parse-error (type => 'not closed',                !!!parse-error (type => 'not closed',
5781                                value => $self->{open_elements}->[-1]->[0]                                text => $self->{open_elements}->[-1]->[0]
5782                                    ->manakai_local_name,                                    ->manakai_local_name,
5783                                token => $token);                                token => $token);
5784    
# Line 5373  sub _tree_construction_main ($) { Line 5799  sub _tree_construction_main ($) {
5799                unless (defined $i) {                unless (defined $i) {
5800                  !!!cp ('t223');                  !!!cp ('t223');
5801  ## TODO: The following is wrong, maybe.  ## TODO: The following is wrong, maybe.
5802                  !!!parse-error (type => 'unmatched end tag:table', token => $token);                  !!!parse-error (type => 'unmatched end tag', text => 'table',
5803                                    token => $token);
5804                  ## Ignore tokens </table><table>                  ## Ignore tokens </table><table>
5805                  !!!nack ('t223.1');                  !!!nack ('t223.1');
5806                  !!!next-token;                  !!!next-token;
5807                  next B;                  next B;
5808                }                }
5809                                
5810  ## TODO: Followings are removed from the latest spec.  ## TODO: Followings are removed from the latest spec.
5811                ## generate implied end tags                ## generate implied end tags
5812                while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {                while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
5813                  !!!cp ('t224');                  !!!cp ('t224');
# Line 5391  sub _tree_construction_main ($) { Line 5818  sub _tree_construction_main ($) {
5818                  !!!cp ('t225');                  !!!cp ('t225');
5819                  ## NOTE: |<table><tr><table>|                  ## NOTE: |<table><tr><table>|
5820                  !!!parse-error (type => 'not closed',                  !!!parse-error (type => 'not closed',
5821                                  value => $self->{open_elements}->[-1]->[0]                                  text => $self->{open_elements}->[-1]->[0]
5822                                      ->manakai_local_name,                                      ->manakai_local_name,
5823                                  token => $token);                                  token => $token);
5824                } else {                } else {
# Line 5432  sub _tree_construction_main ($) { Line 5859  sub _tree_construction_main ($) {
5859                my $type = lc $token->{attributes}->{type}->{value};                my $type = lc $token->{attributes}->{type}->{value};
5860                if ($type eq 'hidden') {                if ($type eq 'hidden') {
5861                  !!!cp ('t227.3');                  !!!cp ('t227.3');
5862                  !!!parse-error (type => 'in table:'.$token->{tag_name}, token => $token);                  !!!parse-error (type => 'in table',
5863                                    text => $token->{tag_name}, token => $token);
5864    
5865                  !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);                  !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5866    
# Line 5460  sub _tree_construction_main ($) { Line 5888  sub _tree_construction_main ($) {
5888            #            #
5889          }          }
5890    
5891          !!!parse-error (type => 'in table:'.$token->{tag_name}, token => $token);          !!!parse-error (type => 'in table', text => $token->{tag_name},
5892                            token => $token);
5893    
5894          $insert = $insert_to_foster;          $insert = $insert_to_foster;
5895          #          #
# Line 5482  sub _tree_construction_main ($) { Line 5911  sub _tree_construction_main ($) {
5911                } # INSCOPE                } # INSCOPE
5912                unless (defined $i) {                unless (defined $i) {
5913                  !!!cp ('t230');                  !!!cp ('t230');
5914                  !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);                  !!!parse-error (type => 'unmatched end tag',
5915                                    text => $token->{tag_name}, token => $token);
5916                  ## Ignore the token                  ## Ignore the token
5917                  !!!nack ('t230.1');                  !!!nack ('t230.1');
5918                  !!!next-token;                  !!!next-token;
# Line 5523  sub _tree_construction_main ($) { Line 5953  sub _tree_construction_main ($) {
5953                  unless (defined $i) {                  unless (defined $i) {
5954                    !!!cp ('t235');                    !!!cp ('t235');
5955  ## TODO: The following is wrong.  ## TODO: The following is wrong.
5956                    !!!parse-error (type => 'unmatched end tag:'.$token->{type}, token => $token);                    !!!parse-error (type => 'unmatched end tag',
5957                                      text => $token->{type}, token => $token);
5958                    ## Ignore the token                    ## Ignore the token
5959                    !!!nack ('t236.1');                    !!!nack ('t236.1');
5960                    !!!next-token;                    !!!next-token;
# Line 5559  sub _tree_construction_main ($) { Line 5990  sub _tree_construction_main ($) {
5990                  } # INSCOPE                  } # INSCOPE
5991                  unless (defined $i) {                  unless (defined $i) {
5992                    !!!cp ('t239');                    !!!cp ('t239');
5993                    !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);                    !!!parse-error (type => 'unmatched end tag',
5994                                      text => $token->{tag_name}, token => $token);
5995                    ## Ignore the token                    ## Ignore the token
5996                    !!!nack ('t239.1');                    !!!nack ('t239.1');
5997                    !!!next-token;                    !!!next-token;
# Line 5605  sub _tree_construction_main ($) { Line 6037  sub _tree_construction_main ($) {
6037                } # INSCOPE                } # INSCOPE
6038                unless (defined $i) {                unless (defined $i) {
6039                  !!!cp ('t243');                  !!!cp ('t243');
6040                  !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);                  !!!parse-error (type => 'unmatched end tag',
6041                                    text => $token->{tag_name}, token => $token);
6042                  ## Ignore the token                  ## Ignore the token
6043                  !!!nack ('t243.1');                  !!!nack ('t243.1');
6044                  !!!next-token;                  !!!next-token;
# Line 5639  sub _tree_construction_main ($) { Line 6072  sub _tree_construction_main ($) {
6072                  } # INSCOPE                  } # INSCOPE
6073                    unless (defined $i) {                    unless (defined $i) {
6074                      !!!cp ('t249');                      !!!cp ('t249');
6075                      !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);                      !!!parse-error (type => 'unmatched end tag',
6076                                        text => $token->{tag_name}, token => $token);
6077                      ## Ignore the token                      ## Ignore the token
6078                      !!!nack ('t249.1');                      !!!nack ('t249.1');
6079                      !!!next-token;                      !!!next-token;
# Line 5662  sub _tree_construction_main ($) { Line 6096  sub _tree_construction_main ($) {
6096                  } # INSCOPE                  } # INSCOPE
6097                    unless (defined $i) {                    unless (defined $i) {
6098                      !!!cp ('t252');                      !!!cp ('t252');
6099                      !!!parse-error (type => 'unmatched end tag:tr', token => $token);                      !!!parse-error (type => 'unmatched end tag',
6100                                        text => 'tr', token => $token);
6101                      ## Ignore the token                      ## Ignore the token
6102                      !!!nack ('t252.1');                      !!!nack ('t252.1');
6103                      !!!next-token;                      !!!next-token;
# Line 5697  sub _tree_construction_main ($) { Line 6132  sub _tree_construction_main ($) {
6132                } # INSCOPE                } # INSCOPE
6133                unless (defined $i) {                unless (defined $i) {
6134                  !!!cp ('t256');                  !!!cp ('t256');
6135                  !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);                  !!!parse-error (type => 'unmatched end tag',
6136                                    text => $token->{tag_name}, token => $token);
6137                  ## Ignore the token                  ## Ignore the token
6138                  !!!nack ('t256.1');                  !!!nack ('t256.1');
6139                  !!!next-token;                  !!!next-token;
# Line 5724  sub _tree_construction_main ($) { Line 6160  sub _tree_construction_main ($) {
6160                        tbody => 1, tfoot => 1, thead => 1, # $self->{insertion_mode} == IN_TABLE_IM                        tbody => 1, tfoot => 1, thead => 1, # $self->{insertion_mode} == IN_TABLE_IM
6161                       }->{$token->{tag_name}}) {                       }->{$token->{tag_name}}) {
6162            !!!cp ('t258');            !!!cp ('t258');
6163            !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);            !!!parse-error (type => 'unmatched end tag',
6164                              text => $token->{tag_name}, token => $token);
6165            ## Ignore the token            ## Ignore the token
6166            !!!nack ('t258.1');            !!!nack ('t258.1');
6167             !!!next-token;             !!!next-token;
6168            next B;            next B;
6169          } else {          } else {
6170            !!!cp ('t259');            !!!cp ('t259');
6171            !!!parse-error (type => 'in table:/'.$token->{tag_name}, token => $token);            !!!parse-error (type => 'in table:/',
6172                              text => $token->{tag_name}, token => $token);
6173    
6174            $insert = $insert_to_foster;            $insert = $insert_to_foster;
6175            #            #
# Line 5754  sub _tree_construction_main ($) { Line 6192  sub _tree_construction_main ($) {
6192        }        }
6193      } elsif ($self->{insertion_mode} == IN_COLUMN_GROUP_IM) {      } elsif ($self->{insertion_mode} == IN_COLUMN_GROUP_IM) {
6194            if ($token->{type} == CHARACTER_TOKEN) {            if ($token->{type} == CHARACTER_TOKEN) {
6195              if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {              if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
6196                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
6197                unless (length $token->{data}) {                unless (length $token->{data}) {
6198                  !!!cp ('t260');                  !!!cp ('t260');
# Line 5781  sub _tree_construction_main ($) { Line 6219  sub _tree_construction_main ($) {
6219              if ($token->{tag_name} eq 'colgroup') {              if ($token->{tag_name} eq 'colgroup') {
6220                if ($self->{open_elements}->[-1]->[1] & HTML_EL) {                if ($self->{open_elements}->[-1]->[1] & HTML_EL) {
6221                  !!!cp ('t264');                  !!!cp ('t264');
6222                  !!!parse-error (type => 'unmatched end tag:colgroup', token => $token);                  !!!parse-error (type => 'unmatched end tag',
6223                                    text => 'colgroup', token => $token);
6224                  ## Ignore the token                  ## Ignore the token
6225                  !!!next-token;                  !!!next-token;
6226                  next B;                  next B;
# Line 5794  sub _tree_construction_main ($) { Line 6233  sub _tree_construction_main ($) {
6233                }                }
6234              } elsif ($token->{tag_name} eq 'col') {              } elsif ($token->{tag_name} eq 'col') {
6235                !!!cp ('t266');                !!!cp ('t266');
6236                !!!parse-error (type => 'unmatched end tag:col', token => $token);                !!!parse-error (type => 'unmatched end tag',
6237                                  text => 'col', token => $token);
6238                ## Ignore the token                ## Ignore the token
6239                !!!next-token;                !!!next-token;
6240                next B;                next B;
# Line 5824  sub _tree_construction_main ($) { Line 6264  sub _tree_construction_main ($) {
6264            if ($self->{open_elements}->[-1]->[1] & HTML_EL) {            if ($self->{open_elements}->[-1]->[1] & HTML_EL) {
6265              !!!cp ('t269');              !!!cp ('t269');
6266  ## TODO: Wrong error type?  ## TODO: Wrong error type?
6267              !!!parse-error (type => 'unmatched end tag:colgroup', token => $token);              !!!parse-error (type => 'unmatched end tag',
6268                                text => 'colgroup', token => $token);
6269              ## Ignore the token              ## Ignore the token
6270              !!!nack ('t269.1');              !!!nack ('t269.1');
6271              !!!next-token;              !!!next-token;
# Line 5878  sub _tree_construction_main ($) { Line 6319  sub _tree_construction_main ($) {
6319            !!!nack ('t277.1');            !!!nack ('t277.1');
6320            !!!next-token;            !!!next-token;
6321            next B;            next B;
6322          } elsif ($token->{tag_name} eq 'select' or          } elsif ({
6323                   $token->{tag_name} eq 'input' or                     select => 1, input => 1, textarea => 1,
6324                     }->{$token->{tag_name}} or
6325                   ($self->{insertion_mode} == IN_SELECT_IN_TABLE_IM and                   ($self->{insertion_mode} == IN_SELECT_IN_TABLE_IM and
6326                    {                    {
6327                     caption => 1, table => 1,                     caption => 1, table => 1,
# Line 5887  sub _tree_construction_main ($) { Line 6329  sub _tree_construction_main ($) {
6329                     tr => 1, td => 1, th => 1,                     tr => 1, td => 1, th => 1,
6330                    }->{$token->{tag_name}})) {                    }->{$token->{tag_name}})) {
6331            ## TODO: The type below is not good - <select> is replaced by </select>            ## TODO: The type below is not good - <select> is replaced by </select>
6332            !!!parse-error (type => 'not closed:select', token => $token);            !!!parse-error (type => 'not closed', text => 'select',
6333                              token => $token);
6334            ## NOTE: As if the token were </select> (<select> case) or            ## NOTE: As if the token were </select> (<select> case) or
6335            ## as if there were </select> (otherwise).            ## as if there were </select> (otherwise).
6336            ## have an element in table scope            ## have an element in table scope
# Line 5905  sub _tree_construction_main ($) { Line 6348  sub _tree_construction_main ($) {
6348            } # INSCOPE            } # INSCOPE
6349            unless (defined $i) {            unless (defined $i) {
6350              !!!cp ('t280');              !!!cp ('t280');
6351              !!!parse-error (type => 'unmatched end tag:select', token => $token);              !!!parse-error (type => 'unmatched end tag',
6352                                text => 'select', token => $token);
6353              ## Ignore the token              ## Ignore the token
6354              !!!nack ('t280.1');              !!!nack ('t280.1');
6355              !!!next-token;              !!!next-token;
# Line 5929  sub _tree_construction_main ($) { Line 6373  sub _tree_construction_main ($) {
6373            }            }
6374          } else {          } else {
6375            !!!cp ('t282');            !!!cp ('t282');
6376            !!!parse-error (type => 'in select:'.$token->{tag_name}, token => $token);            !!!parse-error (type => 'in select',
6377                              text => $token->{tag_name}, token => $token);
6378            ## Ignore the token            ## Ignore the token
6379            !!!nack ('t282.1');            !!!nack ('t282.1');
6380            !!!next-token;            !!!next-token;
# Line 5947  sub _tree_construction_main ($) { Line 6392  sub _tree_construction_main ($) {
6392              pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
6393            } else {            } else {
6394              !!!cp ('t285');              !!!cp ('t285');
6395              !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);              !!!parse-error (type => 'unmatched end tag',
6396                                text => $token->{tag_name}, token => $token);
6397              ## Ignore the token              ## Ignore the token
6398            }            }
6399            !!!nack ('t285.1');            !!!nack ('t285.1');
# Line 5959  sub _tree_construction_main ($) { Line 6405  sub _tree_construction_main ($) {
6405              pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
6406            } else {            } else {
6407              !!!cp ('t287');              !!!cp ('t287');
6408              !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);              !!!parse-error (type => 'unmatched end tag',
6409                                text => $token->{tag_name}, token => $token);
6410              ## Ignore the token              ## Ignore the token
6411            }            }
6412            !!!nack ('t287.1');            !!!nack ('t287.1');
# Line 5981  sub _tree_construction_main ($) { Line 6428  sub _tree_construction_main ($) {
6428            } # INSCOPE            } # INSCOPE
6429            unless (defined $i) {            unless (defined $i) {
6430              !!!cp ('t290');              !!!cp ('t290');
6431              !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);              !!!parse-error (type => 'unmatched end tag',
6432                                text => $token->{tag_name}, token => $token);
6433              ## Ignore the token              ## Ignore the token
6434              !!!nack ('t290.1');              !!!nack ('t290.1');
6435              !!!next-token;              !!!next-token;
# Line 6002  sub _tree_construction_main ($) { Line 6450  sub _tree_construction_main ($) {
6450                    tfoot => 1, thead => 1, tr => 1, td => 1, th => 1,                    tfoot => 1, thead => 1, tr => 1, td => 1, th => 1,
6451                   }->{$token->{tag_name}}) {                   }->{$token->{tag_name}}) {
6452  ## TODO: The following is wrong?  ## TODO: The following is wrong?
6453            !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);            !!!parse-error (type => 'unmatched end tag',
6454                              text => $token->{tag_name}, token => $token);
6455                                
6456            ## have an element in table scope            ## have an element in table scope
6457            my $i;            my $i;
# Line 6043  sub _tree_construction_main ($) { Line 6492  sub _tree_construction_main ($) {
6492            unless (defined $i) {            unless (defined $i) {
6493              !!!cp ('t297');              !!!cp ('t297');
6494  ## TODO: The following error type is correct?  ## TODO: The following error type is correct?
6495              !!!parse-error (type => 'unmatched end tag:select', token => $token);              !!!parse-error (type => 'unmatched end tag',
6496                                text => 'select', token => $token);
6497              ## Ignore the </select> token              ## Ignore the </select> token
6498              !!!nack ('t297.1');              !!!nack ('t297.1');
6499              !!!next-token; ## TODO: ok?              !!!next-token; ## TODO: ok?
# Line 6060  sub _tree_construction_main ($) { Line 6510  sub _tree_construction_main ($) {
6510            next B;            next B;
6511          } else {          } else {
6512            !!!cp ('t299');            !!!cp ('t299');
6513            !!!parse-error (type => 'in select:/'.$token->{tag_name}, token => $token);            !!!parse-error (type => 'in select:/',
6514                              text => $token->{tag_name}, token => $token);
6515            ## Ignore the token            ## Ignore the token
6516            !!!nack ('t299.3');            !!!nack ('t299.3');
6517            !!!next-token;            !!!next-token;
# Line 6082  sub _tree_construction_main ($) { Line 6533  sub _tree_construction_main ($) {
6533        }        }
6534      } elsif ($self->{insertion_mode} & BODY_AFTER_IMS) {      } elsif ($self->{insertion_mode} & BODY_AFTER_IMS) {
6535        if ($token->{type} == CHARACTER_TOKEN) {        if ($token->{type} == CHARACTER_TOKEN) {
6536          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {          if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
6537            my $data = $1;            my $data = $1;
6538            ## As if in body            ## As if in body
6539            $reconstruct_active_formatting_elements->($insert_to_current);            $reconstruct_active_formatting_elements->($insert_to_current);
# Line 6098  sub _tree_construction_main ($) { Line 6549  sub _tree_construction_main ($) {
6549                    
6550          if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {          if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
6551            !!!cp ('t301');            !!!cp ('t301');
6552            !!!parse-error (type => 'after html:#character', token => $token);            !!!parse-error (type => 'after html:#text', token => $token);
6553              #
           ## Reprocess in the "after body" insertion mode.  
6554          } else {          } else {
6555            !!!cp ('t302');            !!!cp ('t302');
6556              ## "after body" insertion mode
6557              !!!parse-error (type => 'after body:#text', token => $token);
6558              #
6559          }          }
           
         ## "after body" insertion mode  
         !!!parse-error (type => 'after body:#character', token => $token);  
6560    
6561          $self->{insertion_mode} = IN_BODY_IM;          $self->{insertion_mode} = IN_BODY_IM;
6562          ## reprocess          ## reprocess
# Line 6114  sub _tree_construction_main ($) { Line 6564  sub _tree_construction_main ($) {
6564        } elsif ($token->{type} == START_TAG_TOKEN) {        } elsif ($token->{type} == START_TAG_TOKEN) {
6565          if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {          if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
6566            !!!cp ('t303');            !!!cp ('t303');
6567            !!!parse-error (type => 'after html:'.$token->{tag_name}, token => $token);            !!!parse-error (type => 'after html',
6568                                        text => $token->{tag_name}, token => $token);
6569            ## Reprocess in the "after body" insertion mode.            #
6570          } else {          } else {
6571            !!!cp ('t304');            !!!cp ('t304');
6572              ## "after body" insertion mode
6573              !!!parse-error (type => 'after body',
6574                              text => $token->{tag_name}, token => $token);
6575              #
6576          }          }
6577    
         ## "after body" insertion mode  
         !!!parse-error (type => 'after body:'.$token->{tag_name}, token => $token);  
   
6578          $self->{insertion_mode} = IN_BODY_IM;          $self->{insertion_mode} = IN_BODY_IM;
6579          !!!ack-later;          !!!ack-later;
6580          ## reprocess          ## reprocess
# Line 6131  sub _tree_construction_main ($) { Line 6582  sub _tree_construction_main ($) {
6582        } elsif ($token->{type} == END_TAG_TOKEN) {        } elsif ($token->{type} == END_TAG_TOKEN) {
6583          if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {          if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
6584            !!!cp ('t305');            !!!cp ('t305');
6585            !!!parse-error (type => 'after html:/'.$token->{tag_name}, token => $token);            !!!parse-error (type => 'after html:/',
6586                              text => $token->{tag_name}, token => $token);
6587                        
6588            $self->{insertion_mode} = AFTER_BODY_IM;            $self->{insertion_mode} = IN_BODY_IM;
6589            ## Reprocess in the "after body" insertion mode.            ## Reprocess.
6590              next B;
6591          } else {          } else {
6592            !!!cp ('t306');            !!!cp ('t306');
6593          }          }
# Line 6143  sub _tree_construction_main ($) { Line 6596  sub _tree_construction_main ($) {
6596          if ($token->{tag_name} eq 'html') {          if ($token->{tag_name} eq 'html') {
6597            if (defined $self->{inner_html_node}) {            if (defined $self->{inner_html_node}) {
6598              !!!cp ('t307');              !!!cp ('t307');
6599              !!!parse-error (type => 'unmatched end tag:html', token => $token);              !!!parse-error (type => 'unmatched end tag',
6600                                text => 'html', token => $token);
6601              ## Ignore the token              ## Ignore the token
6602              !!!next-token;              !!!next-token;
6603              next B;              next B;
# Line 6155  sub _tree_construction_main ($) { Line 6609  sub _tree_construction_main ($) {
6609            }            }
6610          } else {          } else {
6611            !!!cp ('t309');            !!!cp ('t309');
6612            !!!parse-error (type => 'after body:/'.$token->{tag_name}, token => $token);            !!!parse-error (type => 'after body:/',
6613                              text => $token->{tag_name}, token => $token);
6614    
6615            $self->{insertion_mode} = IN_BODY_IM;            $self->{insertion_mode} = IN_BODY_IM;
6616            ## reprocess            ## reprocess
# Line 6170  sub _tree_construction_main ($) { Line 6625  sub _tree_construction_main ($) {
6625        }        }
6626      } elsif ($self->{insertion_mode} & FRAME_IMS) {      } elsif ($self->{insertion_mode} & FRAME_IMS) {
6627        if ($token->{type} == CHARACTER_TOKEN) {        if ($token->{type} == CHARACTER_TOKEN) {
6628          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {          if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
6629            $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);            $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
6630                        
6631            unless (length $token->{data}) {            unless (length $token->{data}) {
# Line 6180  sub _tree_construction_main ($) { Line 6635  sub _tree_construction_main ($) {
6635            }            }
6636          }          }
6637                    
6638          if ($token->{data} =~ s/^[^\x09\x0A\x0B\x0C\x20]+//) {          if ($token->{data} =~ s/^[^\x09\x0A\x0C\x20]+//) {
6639            if ($self->{insertion_mode} == IN_FRAMESET_IM) {            if ($self->{insertion_mode} == IN_FRAMESET_IM) {
6640              !!!cp ('t311');              !!!cp ('t311');
6641              !!!parse-error (type => 'in frameset:#character', token => $token);              !!!parse-error (type => 'in frameset:#text', token => $token);
6642            } elsif ($self->{insertion_mode} == AFTER_FRAMESET_IM) {            } elsif ($self->{insertion_mode} == AFTER_FRAMESET_IM) {
6643              !!!cp ('t312');              !!!cp ('t312');
6644              !!!parse-error (type => 'after frameset:#character', token => $token);              !!!parse-error (type => 'after frameset:#text', token => $token);
6645            } else { # "after html frameset"            } else { # "after after frameset"
6646              !!!cp ('t313');              !!!cp ('t313');
6647              !!!parse-error (type => 'after html:#character', token => $token);              !!!parse-error (type => 'after html:#text', token => $token);
   
             $self->{insertion_mode} = AFTER_FRAMESET_IM;  
             ## Reprocess in the "after frameset" insertion mode.  
             !!!parse-error (type => 'after frameset:#character', token => $token);  
6648            }            }
6649                        
6650            ## Ignore the token.            ## Ignore the token.
# Line 6209  sub _tree_construction_main ($) { Line 6660  sub _tree_construction_main ($) {
6660                    
6661          die qq[$0: Character "$token->{data}"];          die qq[$0: Character "$token->{data}"];
6662        } elsif ($token->{type} == START_TAG_TOKEN) {        } elsif ($token->{type} == START_TAG_TOKEN) {
         if ($self->{insertion_mode} == AFTER_HTML_FRAMESET_IM) {  
           !!!cp ('t316');  
           !!!parse-error (type => 'after html:'.$token->{tag_name}, token => $token);  
   
           $self->{insertion_mode} = AFTER_FRAMESET_IM;  
           ## Process in the "after frameset" insertion mode.  
         } else {  
           !!!cp ('t317');  
         }  
   
6663          if ($token->{tag_name} eq 'frameset' and          if ($token->{tag_name} eq 'frameset' and
6664              $self->{insertion_mode} == IN_FRAMESET_IM) {              $self->{insertion_mode} == IN_FRAMESET_IM) {
6665            !!!cp ('t318');            !!!cp ('t318');
# Line 6236  sub _tree_construction_main ($) { Line 6677  sub _tree_construction_main ($) {
6677            next B;            next B;
6678          } elsif ($token->{tag_name} eq 'noframes') {          } elsif ($token->{tag_name} eq 'noframes') {
6679            !!!cp ('t320');            !!!cp ('t320');
6680            ## NOTE: As if in body.            ## NOTE: As if in head.
6681            $parse_rcdata->(CDATA_CONTENT_MODEL);            $parse_rcdata->(CDATA_CONTENT_MODEL);
6682            next B;            next B;
6683    
6684              ## NOTE: |<!DOCTYPE HTML><frameset></frameset></html><noframes></noframes>|
6685              ## has no parse error.
6686          } else {          } else {
6687            if ($self->{insertion_mode} == IN_FRAMESET_IM) {            if ($self->{insertion_mode} == IN_FRAMESET_IM) {
6688              !!!cp ('t321');              !!!cp ('t321');
6689              !!!parse-error (type => 'in frameset:'.$token->{tag_name}, token => $token);              !!!parse-error (type => 'in frameset',
6690            } else {                              text => $token->{tag_name}, token => $token);
6691              } elsif ($self->{insertion_mode} == AFTER_FRAMESET_IM) {
6692              !!!cp ('t322');              !!!cp ('t322');
6693              !!!parse-error (type => 'after frameset:'.$token->{tag_name}, token => $token);              !!!parse-error (type => 'after frameset',
6694                                text => $token->{tag_name}, token => $token);
6695              } else { # "after after frameset"
6696                !!!cp ('t322.2');
6697                !!!parse-error (type => 'after after frameset',
6698                                text => $token->{tag_name}, token => $token);
6699            }            }
6700            ## Ignore the token            ## Ignore the token
6701            !!!nack ('t322.1');            !!!nack ('t322.1');
# Line 6253  sub _tree_construction_main ($) { Line 6703  sub _tree_construction_main ($) {
6703            next B;            next B;
6704          }          }
6705        } elsif ($token->{type} == END_TAG_TOKEN) {        } elsif ($token->{type} == END_TAG_TOKEN) {
         if ($self->{insertion_mode} == AFTER_HTML_FRAMESET_IM) {  
           !!!cp ('t323');  
           !!!parse-error (type => 'after html:/'.$token->{tag_name}, token => $token);  
   
           $self->{insertion_mode} = AFTER_FRAMESET_IM;  
           ## Process in the "after frameset" insertion mode.  
         } else {  
           !!!cp ('t324');  
         }  
   
6706          if ($token->{tag_name} eq 'frameset' and          if ($token->{tag_name} eq 'frameset' and
6707              $self->{insertion_mode} == IN_FRAMESET_IM) {              $self->{insertion_mode} == IN_FRAMESET_IM) {
6708            if ($self->{open_elements}->[-1]->[1] & HTML_EL and            if ($self->{open_elements}->[-1]->[1] & HTML_EL and
6709                @{$self->{open_elements}} == 1) {                @{$self->{open_elements}} == 1) {
6710              !!!cp ('t325');              !!!cp ('t325');
6711              !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);              !!!parse-error (type => 'unmatched end tag',
6712                                text => $token->{tag_name}, token => $token);
6713              ## Ignore the token              ## Ignore the token
6714              !!!next-token;              !!!next-token;
6715            } else {            } else {
# Line 6294  sub _tree_construction_main ($) { Line 6735  sub _tree_construction_main ($) {
6735          } else {          } else {
6736            if ($self->{insertion_mode} == IN_FRAMESET_IM) {            if ($self->{insertion_mode} == IN_FRAMESET_IM) {
6737              !!!cp ('t330');              !!!cp ('t330');
6738              !!!parse-error (type => 'in frameset:/'.$token->{tag_name}, token => $token);              !!!parse-error (type => 'in frameset:/',
6739            } else {                              text => $token->{tag_name}, token => $token);
6740              } elsif ($self->{insertion_mode} == AFTER_FRAMESET_IM) {
6741                !!!cp ('t330.1');
6742                !!!parse-error (type => 'after frameset:/',
6743                                text => $token->{tag_name}, token => $token);
6744              } else { # "after after html"
6745              !!!cp ('t331');              !!!cp ('t331');
6746              !!!parse-error (type => 'after frameset:/'.$token->{tag_name}, token => $token);              !!!parse-error (type => 'after after frameset:/',
6747                                text => $token->{tag_name}, token => $token);
6748            }            }
6749            ## Ignore the token            ## Ignore the token
6750            !!!next-token;            !!!next-token;
# Line 6364  sub _tree_construction_main ($) { Line 6811  sub _tree_construction_main ($) {
6811                                           ->{has_reference});                                           ->{has_reference});
6812            } elsif ($token->{attributes}->{content}) {            } elsif ($token->{attributes}->{content}) {
6813              if ($token->{attributes}->{content}->{value}              if ($token->{attributes}->{content}->{value}
6814                  =~ /\A[^;]*;[\x09-\x0D\x20]*[Cc][Hh][Aa][Rr][Ss][Ee][Tt]                  =~ /[Cc][Hh][Aa][Rr][Ss][Ee][Tt]
6815                      [\x09-\x0D\x20]*=                      [\x09\x0A\x0C\x0D\x20]*=
6816                      [\x09-\x0D\x20]*(?>"([^"]*)"|'([^']*)'|                      [\x09\x0A\x0C\x0D\x20]*(?>"([^"]*)"|'([^']*)'|
6817                      ([^"'\x09-\x0D\x20][^\x09-\x0D\x20]*))/x) {                      ([^"'\x09\x0A\x0C\x0D\x20][^\x09\x0A\x0C\x0D\x20\x3B]*))
6818                       /x) {
6819                !!!cp ('t336');                !!!cp ('t336');
6820                ## NOTE: Whether the encoding is supported or not is handled                ## NOTE: Whether the encoding is supported or not is handled
6821                ## in the {change_encoding} callback.                ## in the {change_encoding} callback.
# Line 6405  sub _tree_construction_main ($) { Line 6853  sub _tree_construction_main ($) {
6853          $parse_rcdata->(RCDATA_CONTENT_MODEL);          $parse_rcdata->(RCDATA_CONTENT_MODEL);
6854          next B;          next B;
6855        } elsif ($token->{tag_name} eq 'body') {        } elsif ($token->{tag_name} eq 'body') {
6856          !!!parse-error (type => 'in body:body', token => $token);          !!!parse-error (type => 'in body', text => 'body', token => $token);
6857                                
6858          if (@{$self->{open_elements}} == 1 or          if (@{$self->{open_elements}} == 1 or
6859              not ($self->{open_elements}->[1]->[1] & BODY_EL)) {              not ($self->{open_elements}->[1]->[1] & BODY_EL)) {
# Line 6525  sub _tree_construction_main ($) { Line 6973  sub _tree_construction_main ($) {
6973              if ($i != -1) {              if ($i != -1) {
6974                !!!cp ('t355');                !!!cp ('t355');
6975                !!!parse-error (type => 'not closed',                !!!parse-error (type => 'not closed',
6976                                value => $self->{open_elements}->[-1]->[0]                                text => $self->{open_elements}->[-1]->[0]
6977                                    ->manakai_local_name,                                    ->manakai_local_name,
6978                                token => $token);                                token => $token);
6979              } else {              } else {
# Line 6679  sub _tree_construction_main ($) { Line 7127  sub _tree_construction_main ($) {
7127                  xmp => 1,                  xmp => 1,
7128                  iframe => 1,                  iframe => 1,
7129                  noembed => 1,                  noembed => 1,
7130                  noframes => 1,                  noframes => 1, ## NOTE: This is an "as if in head" code clone.
7131                  noscript => 0, ## TODO: 1 if scripting is enabled                  noscript => 0, ## TODO: 1 if scripting is enabled
7132                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
7133          if ($token->{tag_name} eq 'xmp') {          if ($token->{tag_name} eq 'xmp') {
# Line 6701  sub _tree_construction_main ($) { Line 7149  sub _tree_construction_main ($) {
7149            !!!next-token;            !!!next-token;
7150            next B;            next B;
7151          } else {          } else {
7152              !!!ack ('t391.1');
7153    
7154            my $at = $token->{attributes};            my $at = $token->{attributes};
7155            my $form_attrs;            my $form_attrs;
7156            $form_attrs->{action} = $at->{action} if $at->{action};            $form_attrs->{action} = $at->{action} if $at->{action};
# Line 6744  sub _tree_construction_main ($) { Line 7194  sub _tree_construction_main ($) {
7194                           line => $token->{line}, column => $token->{column}},                           line => $token->{line}, column => $token->{column}},
7195                          {type => END_TAG_TOKEN, tag_name => 'form',                          {type => END_TAG_TOKEN, tag_name => 'form',
7196                           line => $token->{line}, column => $token->{column}};                           line => $token->{line}, column => $token->{column}};
           !!!nack ('t391.1'); ## NOTE: Not acknowledged.  
7197            !!!back-token (@tokens);            !!!back-token (@tokens);
7198            !!!next-token;            !!!next-token;
7199            next B;            next B;
# Line 6792  sub _tree_construction_main ($) { Line 7241  sub _tree_construction_main ($) {
7241            ## Ignore the token            ## Ignore the token
7242          } else {          } else {
7243            !!!cp ('t398');            !!!cp ('t398');
7244            !!!parse-error (type => 'in RCDATA:#'.$token->{type}, token => $token);            !!!parse-error (type => 'in RCDATA:#eof', token => $token);
7245          }          }
7246          !!!next-token;          !!!next-token;
7247          next B;          next B;
7248          } elsif ($token->{tag_name} eq 'rt' or
7249                   $token->{tag_name} eq 'rp') {
7250            ## has a |ruby| element in scope
7251            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7252              my $node = $self->{open_elements}->[$_];
7253              if ($node->[1] & RUBY_EL) {
7254                !!!cp ('t398.1');
7255                ## generate implied end tags
7256                while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
7257                  !!!cp ('t398.2');
7258                  pop @{$self->{open_elements}};
7259                }
7260                unless ($self->{open_elements}->[-1]->[1] & RUBY_EL) {
7261                  !!!cp ('t398.3');
7262                  !!!parse-error (type => 'not closed',
7263                                  text => $self->{open_elements}->[-1]->[0]
7264                                      ->manakai_local_name,
7265                                  token => $token);
7266                  pop @{$self->{open_elements}}
7267                      while not $self->{open_elements}->[-1]->[1] & RUBY_EL;
7268                }
7269                last INSCOPE;
7270              } elsif ($node->[1] & SCOPING_EL) {
7271                !!!cp ('t398.4');
7272                last INSCOPE;
7273              }
7274            } # INSCOPE
7275    
7276            !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7277    
7278            !!!nack ('t398.5');
7279            !!!next-token;
7280            redo B;
7281        } elsif ($token->{tag_name} eq 'math' or        } elsif ($token->{tag_name} eq 'math' or
7282                 $token->{tag_name} eq 'svg') {                 $token->{tag_name} eq 'svg') {
7283          $reconstruct_active_formatting_elements->($insert_to_current);          $reconstruct_active_formatting_elements->($insert_to_current);
7284    
7285            ## "Adjust MathML attributes" ('math' only) - done in insert-element-f
7286    
7287          ## "adjust SVG attributes" ('svg' only) - done in insert-element-f          ## "adjust SVG attributes" ('svg' only) - done in insert-element-f
7288    
7289          ## "adjust foreign attributes" - done in insert-element-f          ## "adjust foreign attributes" - done in insert-element-f
# Line 6826  sub _tree_construction_main ($) { Line 7310  sub _tree_construction_main ($) {
7310                  thead => 1, tr => 1,                  thead => 1, tr => 1,
7311                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
7312          !!!cp ('t401');          !!!cp ('t401');
7313          !!!parse-error (type => 'in body:'.$token->{tag_name}, token => $token);          !!!parse-error (type => 'in body',
7314                            text => $token->{tag_name}, token => $token);
7315          ## Ignore the token          ## Ignore the token
7316          !!!nack ('t401.1'); ## NOTE: |<col/>| or |<frame/>| here is an error.          !!!nack ('t401.1'); ## NOTE: |<col/>| or |<frame/>| here is an error.
7317          !!!next-token;          !!!next-token;
# Line 6911  sub _tree_construction_main ($) { Line 7396  sub _tree_construction_main ($) {
7396            }            }
7397    
7398            !!!parse-error (type => 'start tag not allowed',            !!!parse-error (type => 'start tag not allowed',
7399                            value => $token->{tag_name}, token => $token);                            text => $token->{tag_name}, token => $token);
7400            ## NOTE: Ignore the token.            ## NOTE: Ignore the token.
7401            !!!next-token;            !!!next-token;
7402            next B;            next B;
# Line 6921  sub _tree_construction_main ($) { Line 7406  sub _tree_construction_main ($) {
7406            unless ($_->[1] & ALL_END_TAG_OPTIONAL_EL) {            unless ($_->[1] & ALL_END_TAG_OPTIONAL_EL) {
7407              !!!cp ('t403');              !!!cp ('t403');
7408              !!!parse-error (type => 'not closed',              !!!parse-error (type => 'not closed',
7409                              value => $_->[0]->manakai_local_name,                              text => $_->[0]->manakai_local_name,
7410                              token => $token);                              token => $token);
7411              last;              last;
7412            } else {            } else {
# Line 6941  sub _tree_construction_main ($) { Line 7426  sub _tree_construction_main ($) {
7426            unless ($self->{open_elements}->[-1]->[1] & BODY_EL) {            unless ($self->{open_elements}->[-1]->[1] & BODY_EL) {
7427              !!!cp ('t406');              !!!cp ('t406');
7428              !!!parse-error (type => 'not closed',              !!!parse-error (type => 'not closed',
7429                              value => $self->{open_elements}->[1]->[0]                              text => $self->{open_elements}->[1]->[0]
7430                                  ->manakai_local_name,                                  ->manakai_local_name,
7431                              token => $token);                              token => $token);
7432            } else {            } else {
# Line 6952  sub _tree_construction_main ($) { Line 7437  sub _tree_construction_main ($) {
7437            next B;            next B;
7438          } else {          } else {
7439            !!!cp ('t408');            !!!cp ('t408');
7440            !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);            !!!parse-error (type => 'unmatched end tag',
7441                              text => $token->{tag_name}, token => $token);
7442            ## Ignore the token            ## Ignore the token
7443            !!!next-token;            !!!next-token;
7444            next B;            next B;
# Line 6980  sub _tree_construction_main ($) { Line 7466  sub _tree_construction_main ($) {
7466    
7467          unless (defined $i) { # has an element in scope          unless (defined $i) { # has an element in scope
7468            !!!cp ('t413');            !!!cp ('t413');
7469            !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);            !!!parse-error (type => 'unmatched end tag',
7470                              text => $token->{tag_name}, token => $token);
7471              ## NOTE: Ignore the token.
7472          } else {          } else {
7473            ## Step 1. generate implied end tags            ## Step 1. generate implied end tags
7474            while ({            while ({
7475                      ## END_TAG_OPTIONAL_EL
7476                    dd => ($token->{tag_name} ne 'dd'),                    dd => ($token->{tag_name} ne 'dd'),
7477                    dt => ($token->{tag_name} ne 'dt'),                    dt => ($token->{tag_name} ne 'dt'),
7478                    li => ($token->{tag_name} ne 'li'),                    li => ($token->{tag_name} ne 'li'),
7479                    p => 1,                    p => 1,
7480                      rt => 1,
7481                      rp => 1,
7482                   }->{$self->{open_elements}->[-1]->[0]->manakai_local_name}) {                   }->{$self->{open_elements}->[-1]->[0]->manakai_local_name}) {
7483              !!!cp ('t409');              !!!cp ('t409');
7484              pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
# Line 6998  sub _tree_construction_main ($) { Line 7489  sub _tree_construction_main ($) {
7489                    ne $token->{tag_name}) {                    ne $token->{tag_name}) {
7490              !!!cp ('t412');              !!!cp ('t412');
7491              !!!parse-error (type => 'not closed',              !!!parse-error (type => 'not closed',
7492                              value => $self->{open_elements}->[-1]->[0]                              text => $self->{open_elements}->[-1]->[0]
7493                                  ->manakai_local_name,                                  ->manakai_local_name,
7494                              token => $token);                              token => $token);
7495            } else {            } else {
# Line 7035  sub _tree_construction_main ($) { Line 7526  sub _tree_construction_main ($) {
7526    
7527          unless (defined $i) { # has an element in scope          unless (defined $i) { # has an element in scope
7528            !!!cp ('t421');            !!!cp ('t421');
7529            !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);            !!!parse-error (type => 'unmatched end tag',
7530                              text => $token->{tag_name}, token => $token);
7531              ## NOTE: Ignore the token.
7532          } else {          } else {
7533            ## Step 1. generate implied end tags            ## Step 1. generate implied end tags
7534            while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {            while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
# Line 7048  sub _tree_construction_main ($) { Line 7541  sub _tree_construction_main ($) {
7541                    ne $token->{tag_name}) {                    ne $token->{tag_name}) {
7542              !!!cp ('t417.1');              !!!cp ('t417.1');
7543              !!!parse-error (type => 'not closed',              !!!parse-error (type => 'not closed',
7544                              value => $self->{open_elements}->[-1]->[0]                              text => $self->{open_elements}->[-1]->[0]
7545                                  ->manakai_local_name,                                  ->manakai_local_name,
7546                              token => $token);                              token => $token);
7547            } else {            } else {
# Line 7080  sub _tree_construction_main ($) { Line 7573  sub _tree_construction_main ($) {
7573    
7574          unless (defined $i) { # has an element in scope          unless (defined $i) { # has an element in scope
7575            !!!cp ('t425.1');            !!!cp ('t425.1');
7576            !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);            !!!parse-error (type => 'unmatched end tag',
7577                              text => $token->{tag_name}, token => $token);
7578              ## NOTE: Ignore the token.
7579          } else {          } else {
7580            ## Step 1. generate implied end tags            ## Step 1. generate implied end tags
7581            while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {            while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
# Line 7092  sub _tree_construction_main ($) { Line 7587  sub _tree_construction_main ($) {
7587            if ($self->{open_elements}->[-1]->[0]->manakai_local_name            if ($self->{open_elements}->[-1]->[0]->manakai_local_name
7588                    ne $token->{tag_name}) {                    ne $token->{tag_name}) {
7589              !!!cp ('t425');              !!!cp ('t425');
7590              !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);              !!!parse-error (type => 'unmatched end tag',
7591                                text => $token->{tag_name}, token => $token);
7592            } else {            } else {
7593              !!!cp ('t426');              !!!cp ('t426');
7594            }            }
# Line 7123  sub _tree_construction_main ($) { Line 7619  sub _tree_construction_main ($) {
7619                    ne $token->{tag_name}) {                    ne $token->{tag_name}) {
7620              !!!cp ('t412.1');              !!!cp ('t412.1');
7621              !!!parse-error (type => 'not closed',              !!!parse-error (type => 'not closed',
7622                              value => $self->{open_elements}->[-1]->[0]                              text => $self->{open_elements}->[-1]->[0]
7623                                  ->manakai_local_name,                                  ->manakai_local_name,
7624                              token => $token);                              token => $token);
7625            } else {            } else {
# Line 7133  sub _tree_construction_main ($) { Line 7629  sub _tree_construction_main ($) {
7629            splice @{$self->{open_elements}}, $i;            splice @{$self->{open_elements}}, $i;
7630          } else {          } else {
7631            !!!cp ('t413.1');            !!!cp ('t413.1');
7632            !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);            !!!parse-error (type => 'unmatched end tag',
7633                              text => $token->{tag_name}, token => $token);
7634    
7635            !!!cp ('t415.1');            !!!cp ('t415.1');
7636            ## As if <p>, then reprocess the current token            ## As if <p>, then reprocess the current token
# Line 7156  sub _tree_construction_main ($) { Line 7653  sub _tree_construction_main ($) {
7653          next B;          next B;
7654        } elsif ($token->{tag_name} eq 'br') {        } elsif ($token->{tag_name} eq 'br') {
7655          !!!cp ('t428');          !!!cp ('t428');
7656          !!!parse-error (type => 'unmatched end tag:br', token => $token);          !!!parse-error (type => 'unmatched end tag',
7657                            text => 'br', token => $token);
7658    
7659          ## As if <br>          ## As if <br>
7660          $reconstruct_active_formatting_elements->($insert_to_current);          $reconstruct_active_formatting_elements->($insert_to_current);
# Line 7181  sub _tree_construction_main ($) { Line 7679  sub _tree_construction_main ($) {
7679                  noscript => 0, ## TODO: if scripting is enabled                  noscript => 0, ## TODO: if scripting is enabled
7680                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
7681          !!!cp ('t429');          !!!cp ('t429');
7682          !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);          !!!parse-error (type => 'unmatched end tag',
7683                            text => $token->{tag_name}, token => $token);
7684          ## Ignore the token          ## Ignore the token
7685          !!!next-token;          !!!next-token;
7686          next B;          next B;
# Line 7200  sub _tree_construction_main ($) { Line 7699  sub _tree_construction_main ($) {
7699              ## generate implied end tags              ## generate implied end tags
7700              while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {              while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
7701                !!!cp ('t430');                !!!cp ('t430');
7702                ## ISSUE: Can this case be reached?                ## NOTE: |<ruby><rt></ruby>|.
7703                  ## ISSUE: <ruby><rt></rt> will also take this code path,
7704                  ## which seems wrong.
7705                pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
7706                  $node_i++;
7707              }              }
7708                    
7709              ## Step 2              ## Step 2
# Line 7210  sub _tree_construction_main ($) { Line 7712  sub _tree_construction_main ($) {
7712                !!!cp ('t431');                !!!cp ('t431');
7713                ## NOTE: <x><y></x>                ## NOTE: <x><y></x>
7714                !!!parse-error (type => 'not closed',                !!!parse-error (type => 'not closed',
7715                                value => $self->{open_elements}->[-1]->[0]                                text => $self->{open_elements}->[-1]->[0]
7716                                    ->manakai_local_name,                                    ->manakai_local_name,
7717                                token => $token);                                token => $token);
7718              } else {              } else {
# Line 7218  sub _tree_construction_main ($) { Line 7720  sub _tree_construction_main ($) {
7720              }              }
7721                            
7722              ## Step 3              ## Step 3
7723              splice @{$self->{open_elements}}, $node_i;              splice @{$self->{open_elements}}, $node_i if $node_i < 0;
7724    
7725              !!!next-token;              !!!next-token;
7726              last S2;              last S2;
# Line 7229  sub _tree_construction_main ($) { Line 7731  sub _tree_construction_main ($) {
7731                  ($node->[1] & SPECIAL_EL or                  ($node->[1] & SPECIAL_EL or
7732                   $node->[1] & SCOPING_EL)) {                   $node->[1] & SCOPING_EL)) {
7733                !!!cp ('t433');                !!!cp ('t433');
7734                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);                !!!parse-error (type => 'unmatched end tag',
7735                                  text => $token->{tag_name}, token => $token);
7736                ## Ignore the token                ## Ignore the token
7737                !!!next-token;                !!!next-token;
7738                last S2;                last S2;
# Line 7275  sub _tree_construction_main ($) { Line 7778  sub _tree_construction_main ($) {
7778    ## TODO: script stuffs    ## TODO: script stuffs
7779  } # _tree_construct_main  } # _tree_construct_main
7780    
7781  sub set_inner_html ($$$) {  sub set_inner_html ($$$$;$) {
7782    my $class = shift;    my $class = shift;
7783    my $node = shift;    my $node = shift;
7784    my $s = \$_[0];    #my $s = \$_[0];
7785    my $onerror = $_[1];    my $onerror = $_[1];
7786      my $get_wrapper = $_[2] || sub ($) { return $_[0] };
7787    
7788    ## ISSUE: Should {confident} be true?    ## ISSUE: Should {confident} be true?
7789    
# Line 7298  sub set_inner_html ($$$) { Line 7802  sub set_inner_html ($$$) {
7802      }      }
7803    
7804      ## Step 3, 4, 5 # MUST      ## Step 3, 4, 5 # MUST
7805      $class->parse_string ($$s => $node, $onerror);      $class->parse_char_string ($_[0] => $node, $onerror, $get_wrapper);
7806    } elsif ($nt == 1) {    } elsif ($nt == 1) {
7807      ## TODO: If non-html element      ## TODO: If non-html element
7808    
7809      ## NOTE: Most of this code is copied from |parse_string|      ## NOTE: Most of this code is copied from |parse_string|
7810    
7811    ## TODO: Support for $get_wrapper
7812    
7813      ## Step 1 # MUST      ## Step 1 # MUST
7814      my $this_doc = $node->owner_document;      my $this_doc = $node->owner_document;
7815      my $doc = $this_doc->implementation->create_document;      my $doc = $this_doc->implementation->create_document;
# Line 7315  sub set_inner_html ($$$) { Line 7821  sub set_inner_html ($$$) {
7821      my $i = 0;      my $i = 0;
7822      $p->{line_prev} = $p->{line} = 1;      $p->{line_prev} = $p->{line} = 1;
7823      $p->{column_prev} = $p->{column} = 0;      $p->{column_prev} = $p->{column} = 0;
7824      $p->{set_next_char} = sub {      require Whatpm::Charset::DecodeHandle;
7825        my $input = Whatpm::Charset::DecodeHandle::CharString->new (\($_[0]));
7826        $input = $get_wrapper->($input);
7827        $p->{set_nc} = sub {
7828        my $self = shift;        my $self = shift;
7829    
7830        pop @{$self->{prev_char}};        my $char = '';
7831        unshift @{$self->{prev_char}}, $self->{next_char};        if (defined $self->{next_nc}) {
7832            $char = $self->{next_nc};
7833        $self->{next_char} = -1 and return if $i >= length $$s;          delete $self->{next_nc};
7834        $self->{next_char} = ord substr $$s, $i++, 1;          $self->{nc} = ord $char;
7835          } else {
7836            $self->{char_buffer} = '';
7837            $self->{char_buffer_pos} = 0;
7838            
7839            my $count = $input->manakai_read_until
7840                ($self->{char_buffer}, qr/[^\x00\x0A\x0D]/,
7841                 $self->{char_buffer_pos});
7842            if ($count) {
7843              $self->{line_prev} = $self->{line};
7844              $self->{column_prev} = $self->{column};
7845              $self->{column}++;
7846              $self->{nc}
7847                  = ord substr ($self->{char_buffer},
7848                                $self->{char_buffer_pos}++, 1);
7849              return;
7850            }
7851            
7852            if ($input->read ($char, 1)) {
7853              $self->{nc} = ord $char;
7854            } else {
7855              $self->{nc} = -1;
7856              return;
7857            }
7858          }
7859    
7860        ($p->{line_prev}, $p->{column_prev}) = ($p->{line}, $p->{column});        ($p->{line_prev}, $p->{column_prev}) = ($p->{line}, $p->{column});
7861        $p->{column}++;        $p->{column}++;
7862    
7863        if ($self->{next_char} == 0x000A) { # LF        if ($self->{nc} == 0x000A) { # LF
7864          $p->{line}++;          $p->{line}++;
7865          $p->{column} = 0;          $p->{column} = 0;
7866          !!!cp ('i1');          !!!cp ('i1');
7867        } elsif ($self->{next_char} == 0x000D) { # CR        } elsif ($self->{nc} == 0x000D) { # CR
7868          $i++ if substr ($$s, $i, 1) eq "\x0A";  ## TODO: support for abort/streaming
7869          $self->{next_char} = 0x000A; # LF # MUST          my $next = '';
7870            if ($input->read ($next, 1) and $next ne "\x0A") {
7871              $self->{next_nc} = $next;
7872            }
7873            $self->{nc} = 0x000A; # LF # MUST
7874          $p->{line}++;          $p->{line}++;
7875          $p->{column} = 0;          $p->{column} = 0;
7876          !!!cp ('i2');          !!!cp ('i2');
7877        } elsif ($self->{next_char} > 0x10FFFF) {        } elsif ($self->{nc} == 0x0000) { # NULL
         $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST  
         !!!cp ('i3');  
       } elsif ($self->{next_char} == 0x0000) { # NULL  
7878          !!!cp ('i4');          !!!cp ('i4');
7879          !!!parse-error (type => 'NULL');          !!!parse-error (type => 'NULL');
7880          $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST          $self->{nc} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
       } elsif ($self->{next_char} <= 0x0008 or  
                (0x000E <= $self->{next_char} and  
                 $self->{next_char} <= 0x001F) or  
                (0x007F <= $self->{next_char} and  
                 $self->{next_char} <= 0x009F) or  
                (0xD800 <= $self->{next_char} and  
                 $self->{next_char} <= 0xDFFF) or  
                (0xFDD0 <= $self->{next_char} and  
                 $self->{next_char} <= 0xFDDF) or  
                {  
                 0xFFFE => 1, 0xFFFF => 1, 0x1FFFE => 1, 0x1FFFF => 1,  
                 0x2FFFE => 1, 0x2FFFF => 1, 0x3FFFE => 1, 0x3FFFF => 1,  
                 0x4FFFE => 1, 0x4FFFF => 1, 0x5FFFE => 1, 0x5FFFF => 1,  
                 0x6FFFE => 1, 0x6FFFF => 1, 0x7FFFE => 1, 0x7FFFF => 1,  
                 0x8FFFE => 1, 0x8FFFF => 1, 0x9FFFE => 1, 0x9FFFF => 1,  
                 0xAFFFE => 1, 0xAFFFF => 1, 0xBFFFE => 1, 0xBFFFF => 1,  
                 0xCFFFE => 1, 0xCFFFF => 1, 0xDFFFE => 1, 0xDFFFF => 1,  
                 0xEFFFE => 1, 0xEFFFF => 1, 0xFFFFE => 1, 0xFFFFF => 1,  
                 0x10FFFE => 1, 0x10FFFF => 1,  
                }->{$self->{next_char}}) {  
         !!!cp ('i4.1');  
         !!!parse-error (type => 'control char', level => $self->{must_level});  
 ## TODO: error type documentation  
7881        }        }
7882      };      };
7883      $p->{prev_char} = [-1, -1, -1];  
7884      $p->{next_char} = -1;      $p->{read_until} = sub {
7885              #my ($scalar, $specials_range, $offset) = @_;
7886          return 0 if defined $p->{next_nc};
7887    
7888          my $pattern = qr/[^$_[1]\x00\x0A\x0D]/;
7889          my $offset = $_[2] || 0;
7890          
7891          if ($p->{char_buffer_pos} < length $p->{char_buffer}) {
7892            pos ($p->{char_buffer}) = $p->{char_buffer_pos};
7893            if ($p->{char_buffer} =~ /\G(?>$pattern)+/) {
7894              substr ($_[0], $offset)
7895                  = substr ($p->{char_buffer}, $-[0], $+[0] - $-[0]);
7896              my $count = $+[0] - $-[0];
7897              if ($count) {
7898                $p->{column} += $count;
7899                $p->{char_buffer_pos} += $count;
7900                $p->{line_prev} = $p->{line};
7901                $p->{column_prev} = $p->{column} - 1;
7902                $p->{nc} = -1;
7903              }
7904              return $count;
7905            } else {
7906              return 0;
7907            }
7908          } else {
7909            my $count = $input->manakai_read_until ($_[0], $pattern, $_[2]);
7910            if ($count) {
7911              $p->{column} += $count;
7912              $p->{column_prev} += $count;
7913              $p->{nc} = -1;
7914            }
7915            return $count;
7916          }
7917        }; # $p->{read_until}
7918    
7919      my $ponerror = $onerror || sub {      my $ponerror = $onerror || sub {
7920        my (%opt) = @_;        my (%opt) = @_;
7921        my $line = $opt{line};        my $line = $opt{line};
# Line 7386  sub set_inner_html ($$$) { Line 7930  sub set_inner_html ($$$) {
7930        $ponerror->(line => $p->{line}, column => $p->{column}, @_);        $ponerror->(line => $p->{line}, column => $p->{column}, @_);
7931      };      };
7932            
7933        my $char_onerror = sub {
7934          my (undef, $type, %opt) = @_;
7935          $ponerror->(layer => 'encode',
7936                      line => $p->{line}, column => $p->{column} + 1,
7937                      %opt, type => $type);
7938        }; # $char_onerror
7939        $input->onerror ($char_onerror);
7940    
7941      $p->_initialize_tokenizer;      $p->_initialize_tokenizer;
7942      $p->_initialize_tree_constructor;      $p->_initialize_tree_constructor;
7943    

Legend:
Removed from v.1.141  
changed lines
  Added in v.1.192

[email protected]
ViewVC Help
Powered by ViewVC 1.1.24