/[suikacvs]/markup/html/whatpm/Whatpm/HTML.pm.src
Suika

Diff of /markup/html/whatpm/Whatpm/HTML.pm.src

Parent Directory Parent Directory | Revision Log Revision Log | View Patch Patch

revision 1.141 by wakaba, Sat May 24 10:18:26 2008 UTC revision 1.205 by wakaba, Mon Oct 13 06:18:31 2008 UTC
# Line 3  use strict; Line 3  use strict;
3  our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r};  our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r};
4  use Error qw(:try);  use Error qw(:try);
5    
6    ## NOTE: This module don't check all HTML5 parse errors; character
7    ## encoding related parse errors are expected to be handled by relevant
8    ## modules.
9    ## Parse errors for control characters that are not allowed in HTML5
10    ## documents, for surrogate code points, and for noncharacter code
11    ## points, as well as U+FFFD substitions for characters whose code points
12    ## is higher than U+10FFFF may be detected by combining the parser with
13    ## the checker implemented by Whatpm::Charset::UnicodeChecker (for its
14    ## usage example, see |t/HTML-tree.t| in the Whatpm package or the
15    ## WebHACC::Language::HTML module in the WebHACC package).
16    
17  ## ISSUE:  ## ISSUE:
18  ## var doc = implementation.createDocument (null, null, null);  ## var doc = implementation.createDocument (null, null, null);
19  ## doc.write ('');  ## doc.write ('');
20  ## alert (doc.compatMode);  ## alert (doc.compatMode);
21    
 ## TODO: 1252 parse error (revision 1264)  
 ## TODO: 8859-11 = 874 (revision 1271)  
   
22  require IO::Handle;  require IO::Handle;
23    
24  my $HTML_NS = q<http://www.w3.org/1999/xhtml>;  my $HTML_NS = q<http://www.w3.org/1999/xhtml>;
# Line 48  sub MISC_SPECIAL_EL () { 0b1000000000000 Line 56  sub MISC_SPECIAL_EL () { 0b1000000000000
56  sub FOREIGN_EL () { 0b10000000000000000000000000 }  sub FOREIGN_EL () { 0b10000000000000000000000000 }
57  sub FOREIGN_FLOW_CONTENT_EL () { 0b100000000000000000000000000 }  sub FOREIGN_FLOW_CONTENT_EL () { 0b100000000000000000000000000 }
58  sub MML_AXML_EL () { 0b1000000000000000000000000000 }  sub MML_AXML_EL () { 0b1000000000000000000000000000 }
59    sub RUBY_EL () { 0b10000000000000000000000000000 }
60    sub RUBY_COMPONENT_EL () { 0b100000000000000000000000000000 }
61    
62  sub TABLE_ROWS_EL () {  sub TABLE_ROWS_EL () {
63    TABLE_EL |    TABLE_EL |
# Line 55  sub TABLE_ROWS_EL () { Line 65  sub TABLE_ROWS_EL () {
65    TABLE_ROW_GROUP_EL    TABLE_ROW_GROUP_EL
66  }  }
67    
68    ## NOTE: Used in "generate implied end tags" algorithm.
69    ## NOTE: There is a code where a modified version of
70    ## END_TAG_OPTIONAL_EL is used in "generate implied end tags"
71    ## implementation (search for the algorithm name).
72  sub END_TAG_OPTIONAL_EL () {  sub END_TAG_OPTIONAL_EL () {
73    DD_EL |    DD_EL |
74    DT_EL |    DT_EL |
75    LI_EL |    LI_EL |
76    P_EL    OPTION_EL |
77      OPTGROUP_EL |
78      P_EL |
79      RUBY_COMPONENT_EL
80  }  }
81    
82    ## NOTE: Used in </body> and EOF algorithms.
83  sub ALL_END_TAG_OPTIONAL_EL () {  sub ALL_END_TAG_OPTIONAL_EL () {
84    END_TAG_OPTIONAL_EL |    DD_EL |
85      DT_EL |
86      LI_EL |
87      P_EL |
88    
89      ## ISSUE: option, optgroup, rt, rp?
90    
91    BODY_EL |    BODY_EL |
92    HTML_EL |    HTML_EL |
93    TABLE_CELL_EL |    TABLE_CELL_EL |
# Line 99  sub SPECIAL_EL () { Line 123  sub SPECIAL_EL () {
123    ADDRESS_EL |    ADDRESS_EL |
124    BODY_EL |    BODY_EL |
125    DIV_EL |    DIV_EL |
126    END_TAG_OPTIONAL_EL |  
127      DD_EL |
128      DT_EL |
129      LI_EL |
130      P_EL |
131    
132    FORM_EL |    FORM_EL |
133    FRAMESET_EL |    FRAMESET_EL |
134    HEADING_EL |    HEADING_EL |
   OPTION_EL |  
   OPTGROUP_EL |  
135    SELECT_EL |    SELECT_EL |
136    TABLE_ROW_EL |    TABLE_ROW_EL |
137    TABLE_ROW_GROUP_EL |    TABLE_ROW_GROUP_EL |
# Line 116  my $el_category = { Line 143  my $el_category = {
143    address => ADDRESS_EL,    address => ADDRESS_EL,
144    applet => MISC_SCOPING_EL,    applet => MISC_SCOPING_EL,
145    area => MISC_SPECIAL_EL,    area => MISC_SPECIAL_EL,
146      article => MISC_SPECIAL_EL,
147      aside => MISC_SPECIAL_EL,
148    b => FORMATTING_EL,    b => FORMATTING_EL,
149    base => MISC_SPECIAL_EL,    base => MISC_SPECIAL_EL,
150    basefont => MISC_SPECIAL_EL,    basefont => MISC_SPECIAL_EL,
# Line 129  my $el_category = { Line 158  my $el_category = {
158    center => MISC_SPECIAL_EL,    center => MISC_SPECIAL_EL,
159    col => MISC_SPECIAL_EL,    col => MISC_SPECIAL_EL,
160    colgroup => MISC_SPECIAL_EL,    colgroup => MISC_SPECIAL_EL,
161      command => MISC_SPECIAL_EL,
162      datagrid => MISC_SPECIAL_EL,
163    dd => DD_EL,    dd => DD_EL,
164      details => MISC_SPECIAL_EL,
165      dialog => MISC_SPECIAL_EL,
166    dir => MISC_SPECIAL_EL,    dir => MISC_SPECIAL_EL,
167    div => DIV_EL,    div => DIV_EL,
168    dl => MISC_SPECIAL_EL,    dl => MISC_SPECIAL_EL,
169    dt => DT_EL,    dt => DT_EL,
170    em => FORMATTING_EL,    em => FORMATTING_EL,
171    embed => MISC_SPECIAL_EL,    embed => MISC_SPECIAL_EL,
172      eventsource => MISC_SPECIAL_EL,
173    fieldset => MISC_SPECIAL_EL,    fieldset => MISC_SPECIAL_EL,
174      figure => MISC_SPECIAL_EL,
175    font => FORMATTING_EL,    font => FORMATTING_EL,
176      footer => MISC_SPECIAL_EL,
177    form => FORM_EL,    form => FORM_EL,
178    frame => MISC_SPECIAL_EL,    frame => MISC_SPECIAL_EL,
179    frameset => FRAMESET_EL,    frameset => FRAMESET_EL,
# Line 148  my $el_category = { Line 184  my $el_category = {
184    h5 => HEADING_EL,    h5 => HEADING_EL,
185    h6 => HEADING_EL,    h6 => HEADING_EL,
186    head => MISC_SPECIAL_EL,    head => MISC_SPECIAL_EL,
187      header => MISC_SPECIAL_EL,
188    hr => MISC_SPECIAL_EL,    hr => MISC_SPECIAL_EL,
189    html => HTML_EL,    html => HTML_EL,
190    i => FORMATTING_EL,    i => FORMATTING_EL,
191    iframe => MISC_SPECIAL_EL,    iframe => MISC_SPECIAL_EL,
192    img => MISC_SPECIAL_EL,    img => MISC_SPECIAL_EL,
193      #image => MISC_SPECIAL_EL, ## NOTE: Commented out in the spec.
194    input => MISC_SPECIAL_EL,    input => MISC_SPECIAL_EL,
195    isindex => MISC_SPECIAL_EL,    isindex => MISC_SPECIAL_EL,
196    li => LI_EL,    li => LI_EL,
# Line 161  my $el_category = { Line 199  my $el_category = {
199    marquee => MISC_SCOPING_EL,    marquee => MISC_SCOPING_EL,
200    menu => MISC_SPECIAL_EL,    menu => MISC_SPECIAL_EL,
201    meta => MISC_SPECIAL_EL,    meta => MISC_SPECIAL_EL,
202      nav => MISC_SPECIAL_EL,
203    nobr => NOBR_EL | FORMATTING_EL,    nobr => NOBR_EL | FORMATTING_EL,
204    noembed => MISC_SPECIAL_EL,    noembed => MISC_SPECIAL_EL,
205    noframes => MISC_SPECIAL_EL,    noframes => MISC_SPECIAL_EL,
# Line 173  my $el_category = { Line 212  my $el_category = {
212    param => MISC_SPECIAL_EL,    param => MISC_SPECIAL_EL,
213    plaintext => MISC_SPECIAL_EL,    plaintext => MISC_SPECIAL_EL,
214    pre => MISC_SPECIAL_EL,    pre => MISC_SPECIAL_EL,
215      rp => RUBY_COMPONENT_EL,
216      rt => RUBY_COMPONENT_EL,
217      ruby => RUBY_EL,
218    s => FORMATTING_EL,    s => FORMATTING_EL,
219    script => MISC_SPECIAL_EL,    script => MISC_SPECIAL_EL,
220    select => SELECT_EL,    select => SELECT_EL,
221      section => MISC_SPECIAL_EL,
222    small => FORMATTING_EL,    small => FORMATTING_EL,
223    spacer => MISC_SPECIAL_EL,    spacer => MISC_SPECIAL_EL,
224    strike => FORMATTING_EL,    strike => FORMATTING_EL,
# Line 206  my $el_category_f = { Line 249  my $el_category_f = {
249      mtext => FOREIGN_FLOW_CONTENT_EL,      mtext => FOREIGN_FLOW_CONTENT_EL,
250    },    },
251    $SVG_NS => {    $SVG_NS => {
252      foreignObject => FOREIGN_FLOW_CONTENT_EL,      foreignObject => FOREIGN_FLOW_CONTENT_EL | MISC_SCOPING_EL,
253      desc => FOREIGN_FLOW_CONTENT_EL,      desc => FOREIGN_FLOW_CONTENT_EL,
254      title => FOREIGN_FLOW_CONTENT_EL,      title => FOREIGN_FLOW_CONTENT_EL,
255    },    },
# Line 214  my $el_category_f = { Line 257  my $el_category_f = {
257  };  };
258    
259  my $svg_attr_name = {  my $svg_attr_name = {
260      attributename => 'attributeName',
261    attributetype => 'attributeType',    attributetype => 'attributeType',
262    basefrequency => 'baseFrequency',    basefrequency => 'baseFrequency',
263    baseprofile => 'baseProfile',    baseprofile => 'baseProfile',
# Line 224  my $svg_attr_name = { Line 268  my $svg_attr_name = {
268    diffuseconstant => 'diffuseConstant',    diffuseconstant => 'diffuseConstant',
269    edgemode => 'edgeMode',    edgemode => 'edgeMode',
270    externalresourcesrequired => 'externalResourcesRequired',    externalresourcesrequired => 'externalResourcesRequired',
   fecolormatrix => 'feColorMatrix',  
   fecomposite => 'feComposite',  
   fegaussianblur => 'feGaussianBlur',  
   femorphology => 'feMorphology',  
   fetile => 'feTile',  
271    filterres => 'filterRes',    filterres => 'filterRes',
272    filterunits => 'filterUnits',    filterunits => 'filterUnits',
273    glyphref => 'glyphRef',    glyphref => 'glyphRef',
# Line 262  my $svg_attr_name = { Line 301  my $svg_attr_name = {
301    repeatcount => 'repeatCount',    repeatcount => 'repeatCount',
302    repeatdur => 'repeatDur',    repeatdur => 'repeatDur',
303    requiredextensions => 'requiredExtensions',    requiredextensions => 'requiredExtensions',
304      requiredfeatures => 'requiredFeatures',
305    specularconstant => 'specularConstant',    specularconstant => 'specularConstant',
306    specularexponent => 'specularExponent',    specularexponent => 'specularExponent',
307    spreadmethod => 'spreadMethod',    spreadmethod => 'spreadMethod',
# Line 298  my $foreign_attr_xname = { Line 338  my $foreign_attr_xname = {
338    
339  ## ISSUE: xmlns:xlink="non-xlink-ns" is not an error.  ## ISSUE: xmlns:xlink="non-xlink-ns" is not an error.
340    
341  my $c1_entity_char = {  my $charref_map = {
342      0x0D => 0x000A,
343    0x80 => 0x20AC,    0x80 => 0x20AC,
344    0x81 => 0xFFFD,    0x81 => 0xFFFD,
345    0x82 => 0x201A,    0x82 => 0x201A,
# Line 331  my $c1_entity_char = { Line 372  my $c1_entity_char = {
372    0x9D => 0xFFFD,    0x9D => 0xFFFD,
373    0x9E => 0x017E,    0x9E => 0x017E,
374    0x9F => 0x0178,    0x9F => 0x0178,
375  }; # $c1_entity_char  }; # $charref_map
376    $charref_map->{$_} = 0xFFFD
377        for 0x0000..0x0008, 0x000B, 0x000E..0x001F, 0x007F,
378            0xD800..0xDFFF, 0xFDD0..0xFDDF, ## ISSUE: 0xFDEF
379            0xFFFE, 0xFFFF, 0x1FFFE, 0x1FFFF, 0x2FFFE, 0x2FFFF, 0x3FFFE, 0x3FFFF,
380            0x4FFFE, 0x4FFFF, 0x5FFFE, 0x5FFFF, 0x6FFFE, 0x6FFFF, 0x7FFFE,
381            0x7FFFF, 0x8FFFE, 0x8FFFF, 0x9FFFE, 0x9FFFF, 0xAFFFE, 0xAFFFF,
382            0xBFFFE, 0xBFFFF, 0xCFFFE, 0xCFFFF, 0xDFFFE, 0xDFFFF, 0xEFFFE,
383            0xEFFFF, 0xFFFFE, 0xFFFFF, 0x10FFFE, 0x10FFFF;
384    
385    ## TODO: Invoke the reset algorithm when a resettable element is
386    ## created (cf. HTML5 revision 2259).
387    
388  sub parse_byte_string ($$$$;$) {  sub parse_byte_string ($$$$;$) {
389    my $self = shift;    my $self = shift;
# Line 340  sub parse_byte_string ($$$$;$) { Line 392  sub parse_byte_string ($$$$;$) {
392    return $self->parse_byte_stream ($charset_name, $input, @_[1..$#_]);    return $self->parse_byte_stream ($charset_name, $input, @_[1..$#_]);
393  } # parse_byte_string  } # parse_byte_string
394    
395  sub parse_byte_stream ($$$$;$) {  sub parse_byte_stream ($$$$;$$) {
396      # my ($self, $charset_name, $byte_stream, $doc, $onerror, $get_wrapper) = @_;
397    my $self = ref $_[0] ? shift : shift->new;    my $self = ref $_[0] ? shift : shift->new;
398    my $charset_name = shift;    my $charset_name = shift;
399    my $byte_stream = $_[0];    my $byte_stream = $_[0];
# Line 351  sub parse_byte_stream ($$$$;$) { Line 404  sub parse_byte_stream ($$$$;$) {
404    };    };
405    $self->{parse_error} = $onerror; # updated later by parse_char_string    $self->{parse_error} = $onerror; # updated later by parse_char_string
406    
407      my $get_wrapper = $_[3] || sub ($) {
408        return $_[0]; # $_[0] = byte stream handle, returned = arg to char handle
409      };
410    
411    ## HTML5 encoding sniffing algorithm    ## HTML5 encoding sniffing algorithm
412    require Message::Charset::Info;    require Message::Charset::Info;
413    my $charset;    my $charset;
# Line 358  sub parse_byte_stream ($$$$;$) { Line 415  sub parse_byte_stream ($$$$;$) {
415    my ($char_stream, $e_status);    my ($char_stream, $e_status);
416    
417    SNIFFING: {    SNIFFING: {
418        ## NOTE: By setting |allow_fallback| option true when the
419        ## |get_decode_handle| method is invoked, we ignore what the HTML5
420        ## spec requires, i.e. unsupported encoding should be ignored.
421          ## TODO: We should not do this unless the parser is invoked
422          ## in the conformance checking mode, in which this behavior
423          ## would be useful.
424    
425      ## Step 1      ## Step 1
426      if (defined $charset_name) {      if (defined $charset_name) {
427        $charset = Message::Charset::Info->get_by_iana_name ($charset_name);        $charset = Message::Charset::Info->get_by_html_name ($charset_name);
428              ## TODO: Is this ok?  Transfer protocol's parameter should be
429              ## interpreted in its semantics?
430    
       ## ISSUE: Unsupported encoding is not ignored according to the spec.  
431        ($char_stream, $e_status) = $charset->get_decode_handle        ($char_stream, $e_status) = $charset->get_decode_handle
432            ($byte_stream, allow_error_reporting => 1,            ($byte_stream, allow_error_reporting => 1,
433             allow_fallback => 1);             allow_fallback => 1);
# Line 371  sub parse_byte_stream ($$$$;$) { Line 435  sub parse_byte_stream ($$$$;$) {
435          $self->{confident} = 1;          $self->{confident} = 1;
436          last SNIFFING;          last SNIFFING;
437        } else {        } else {
438          ## TODO: unsupported error          !!!parse-error (type => 'charset:not supported',
439                            layer => 'encode',
440                            line => 1, column => 1,
441                            value => $charset_name,
442                            level => $self->{level}->{uncertain});
443        }        }
444      }      }
445    
# Line 385  sub parse_byte_stream ($$$$;$) { Line 453  sub parse_byte_stream ($$$$;$) {
453    
454      ## Step 3      ## Step 3
455      if ($byte_buffer =~ /^\xFE\xFF/) {      if ($byte_buffer =~ /^\xFE\xFF/) {
456        $charset = Message::Charset::Info->get_by_iana_name ('utf-16be');        $charset = Message::Charset::Info->get_by_html_name ('utf-16be');
457        ($char_stream, $e_status) = $charset->get_decode_handle        ($char_stream, $e_status) = $charset->get_decode_handle
458            ($byte_stream, allow_error_reporting => 1,            ($byte_stream, allow_error_reporting => 1,
459             allow_fallback => 1, byte_buffer => \$byte_buffer);             allow_fallback => 1, byte_buffer => \$byte_buffer);
460        $self->{confident} = 1;        $self->{confident} = 1;
461        last SNIFFING;        last SNIFFING;
462      } elsif ($byte_buffer =~ /^\xFF\xFE/) {      } elsif ($byte_buffer =~ /^\xFF\xFE/) {
463        $charset = Message::Charset::Info->get_by_iana_name ('utf-16le');        $charset = Message::Charset::Info->get_by_html_name ('utf-16le');
464        ($char_stream, $e_status) = $charset->get_decode_handle        ($char_stream, $e_status) = $charset->get_decode_handle
465            ($byte_stream, allow_error_reporting => 1,            ($byte_stream, allow_error_reporting => 1,
466             allow_fallback => 1, byte_buffer => \$byte_buffer);             allow_fallback => 1, byte_buffer => \$byte_buffer);
467        $self->{confident} = 1;        $self->{confident} = 1;
468        last SNIFFING;        last SNIFFING;
469      } elsif ($byte_buffer =~ /^\xEF\xBB\xBF/) {      } elsif ($byte_buffer =~ /^\xEF\xBB\xBF/) {
470        $charset = Message::Charset::Info->get_by_iana_name ('utf-8');        $charset = Message::Charset::Info->get_by_html_name ('utf-8');
471        ($char_stream, $e_status) = $charset->get_decode_handle        ($char_stream, $e_status) = $charset->get_decode_handle
472            ($byte_stream, allow_error_reporting => 1,            ($byte_stream, allow_error_reporting => 1,
473             allow_fallback => 1, byte_buffer => \$byte_buffer);             allow_fallback => 1, byte_buffer => \$byte_buffer);
# Line 418  sub parse_byte_stream ($$$$;$) { Line 486  sub parse_byte_stream ($$$$;$) {
486      $charset_name = Whatpm::Charset::UniversalCharDet->detect_byte_string      $charset_name = Whatpm::Charset::UniversalCharDet->detect_byte_string
487          ($byte_buffer);          ($byte_buffer);
488      if (defined $charset_name) {      if (defined $charset_name) {
489        $charset = Message::Charset::Info->get_by_iana_name ($charset_name);        $charset = Message::Charset::Info->get_by_html_name ($charset_name);
490    
       ## ISSUE: Unsupported encoding is not ignored according to the spec.  
491        require Whatpm::Charset::DecodeHandle;        require Whatpm::Charset::DecodeHandle;
492        $buffer = Whatpm::Charset::DecodeHandle::ByteBuffer->new        $buffer = Whatpm::Charset::DecodeHandle::ByteBuffer->new
493            ($byte_stream);            ($byte_stream);
# Line 429  sub parse_byte_stream ($$$$;$) { Line 496  sub parse_byte_stream ($$$$;$) {
496             allow_fallback => 1, byte_buffer => \$byte_buffer);             allow_fallback => 1, byte_buffer => \$byte_buffer);
497        if ($char_stream) {        if ($char_stream) {
498          $buffer->{buffer} = $byte_buffer;          $buffer->{buffer} = $byte_buffer;
499          !!!parse-error (type => 'sniffing:chardet', ## TODO: type name          !!!parse-error (type => 'sniffing:chardet',
500                          value => $charset_name,                          text => $charset_name,
501                          level => $self->{info_level},                          level => $self->{level}->{info},
502                            layer => 'encode',
503                          line => 1, column => 1);                          line => 1, column => 1);
504          $self->{confident} = 0;          $self->{confident} = 0;
505          last SNIFFING;          last SNIFFING;
# Line 440  sub parse_byte_stream ($$$$;$) { Line 508  sub parse_byte_stream ($$$$;$) {
508    
509      ## Step 7: default      ## Step 7: default
510      ## TODO: Make this configurable.      ## TODO: Make this configurable.
511      $charset = Message::Charset::Info->get_by_iana_name ('windows-1252');      $charset = Message::Charset::Info->get_by_html_name ('windows-1252');
512          ## NOTE: We choose |windows-1252| here, since |utf-8| should be          ## NOTE: We choose |windows-1252| here, since |utf-8| should be
513          ## detectable in the step 6.          ## detectable in the step 6.
514      require Whatpm::Charset::DecodeHandle;      require Whatpm::Charset::DecodeHandle;
# Line 452  sub parse_byte_stream ($$$$;$) { Line 520  sub parse_byte_stream ($$$$;$) {
520                                         allow_fallback => 1,                                         allow_fallback => 1,
521                                         byte_buffer => \$byte_buffer);                                         byte_buffer => \$byte_buffer);
522      $buffer->{buffer} = $byte_buffer;      $buffer->{buffer} = $byte_buffer;
523      !!!parse-error (type => 'sniffing:default', ## TODO: type name      !!!parse-error (type => 'sniffing:default',
524                      value => 'windows-1252',                      text => 'windows-1252',
525                      level => $self->{info_level},                      level => $self->{level}->{info},
526                      line => 1, column => 1);                      line => 1, column => 1,
527                        layer => 'encode');
528      $self->{confident} = 0;      $self->{confident} = 0;
529    } # SNIFFING    } # SNIFFING
530    
   $self->{input_encoding} = $charset->get_iana_name;  
531    if ($e_status & Message::Charset::Info::FALLBACK_ENCODING_IMPL ()) {    if ($e_status & Message::Charset::Info::FALLBACK_ENCODING_IMPL ()) {
532      !!!parse-error (type => 'chardecode:fallback', ## TODO: type name      $self->{input_encoding} = $charset->get_iana_name; ## TODO: Should we set actual charset decoder's encoding name?
533                      value => $self->{input_encoding},      !!!parse-error (type => 'chardecode:fallback',
534                      level => $self->{unsupported_level},                      #text => $self->{input_encoding},
535                      line => 1, column => 1);                      level => $self->{level}->{uncertain},
536                        line => 1, column => 1,
537                        layer => 'encode');
538    } elsif (not ($e_status &    } elsif (not ($e_status &
539                  Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL())) {                  Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL ())) {
540      !!!parse-error (type => 'chardecode:no error', ## TODO: type name      $self->{input_encoding} = $charset->get_iana_name;
541                      value => $self->{input_encoding},      !!!parse-error (type => 'chardecode:no error',
542                      level => $self->{unsupported_level},                      text => $self->{input_encoding},
543                      line => 1, column => 1);                      level => $self->{level}->{uncertain},
544                        line => 1, column => 1,
545                        layer => 'encode');
546      } else {
547        $self->{input_encoding} = $charset->get_iana_name;
548    }    }
549    
550    $self->{change_encoding} = sub {    $self->{change_encoding} = sub {
# Line 478  sub parse_byte_stream ($$$$;$) { Line 552  sub parse_byte_stream ($$$$;$) {
552      $charset_name = shift;      $charset_name = shift;
553      my $token = shift;      my $token = shift;
554    
555      $charset = Message::Charset::Info->get_by_iana_name ($charset_name);      $charset = Message::Charset::Info->get_by_html_name ($charset_name);
556      ($char_stream, $e_status) = $charset->get_decode_handle      ($char_stream, $e_status) = $charset->get_decode_handle
557          ($byte_stream, allow_error_reporting => 1, allow_fallback => 1,          ($byte_stream, allow_error_reporting => 1, allow_fallback => 1,
558           byte_buffer => \ $buffer->{buffer});           byte_buffer => \ $buffer->{buffer});
# Line 487  sub parse_byte_stream ($$$$;$) { Line 561  sub parse_byte_stream ($$$$;$) {
561        ## "Change the encoding" algorithm:        ## "Change the encoding" algorithm:
562    
563        ## Step 1            ## Step 1    
564        if ($charset->{iana_names}->{'utf-16'}) { ## ISSUE: UTF-16BE -> UTF-8? UTF-16LE -> UTF-8?        if ($charset->{category} &
565          $charset = Message::Charset::Info->get_by_iana_name ('utf-8');            Message::Charset::Info::CHARSET_CATEGORY_UTF16 ()) {
566            $charset = Message::Charset::Info->get_by_html_name ('utf-8');
567          ($char_stream, $e_status) = $charset->get_decode_handle          ($char_stream, $e_status) = $charset->get_decode_handle
568              ($byte_stream,              ($byte_stream,
569               byte_buffer => \ $buffer->{buffer});               byte_buffer => \ $buffer->{buffer});
# Line 498  sub parse_byte_stream ($$$$;$) { Line 573  sub parse_byte_stream ($$$$;$) {
573        ## Step 2        ## Step 2
574        if (defined $self->{input_encoding} and        if (defined $self->{input_encoding} and
575            $self->{input_encoding} eq $charset_name) {            $self->{input_encoding} eq $charset_name) {
576          !!!parse-error (type => 'charset label:matching', ## TODO: type          !!!parse-error (type => 'charset label:matching',
577                          value => $charset_name,                          text => $charset_name,
578                          level => $self->{info_level});                          level => $self->{level}->{info});
579          $self->{confident} = 1;          $self->{confident} = 1;
580          return;          return;
581        }        }
582    
583        !!!parse-error (type => 'charset label detected:'.$self->{input_encoding}.        !!!parse-error (type => 'charset label detected',
584            ':'.$charset_name, level => 'w', token => $token);                        text => $self->{input_encoding},
585                          value => $charset_name,
586                          level => $self->{level}->{warn},
587                          token => $token);
588                
589        ## Step 3        ## Step 3
590        # if (can) {        # if (can) {
# Line 522  sub parse_byte_stream ($$$$;$) { Line 600  sub parse_byte_stream ($$$$;$) {
600    
601    my $char_onerror = sub {    my $char_onerror = sub {
602      my (undef, $type, %opt) = @_;      my (undef, $type, %opt) = @_;
603      !!!parse-error (%opt, type => $type,      !!!parse-error (layer => 'encode',
604                      line => $self->{line}, column => $self->{column} + 1);                      line => $self->{line}, column => $self->{column} + 1,
605                        %opt, type => $type);
606      if ($opt{octets}) {      if ($opt{octets}) {
607        ${$opt{octets}} = "\x{FFFD}"; # relacement character        ${$opt{octets}} = "\x{FFFD}"; # relacement character
608      }      }
609    };    };
   $char_stream->onerror ($char_onerror);  
610    
611    my @args = @_; shift @args; # $s    my $wrapped_char_stream = $get_wrapper->($char_stream);
612      $wrapped_char_stream->onerror ($char_onerror);
613    
614      my @args = ($_[1], $_[2]); # $doc, $onerror - $get_wrapper = undef;
615    my $return;    my $return;
616    try {    try {
617      $return = $self->parse_char_stream ($char_stream, @args);        $return = $self->parse_char_stream ($wrapped_char_stream, @args);  
618    } catch Whatpm::HTML::RestartParser with {    } catch Whatpm::HTML::RestartParser with {
619      ## NOTE: Invoked after {change_encoding}.      ## NOTE: Invoked after {change_encoding}.
620    
     $self->{input_encoding} = $charset->get_iana_name;  
621      if ($e_status & Message::Charset::Info::FALLBACK_ENCODING_IMPL ()) {      if ($e_status & Message::Charset::Info::FALLBACK_ENCODING_IMPL ()) {
622        !!!parse-error (type => 'chardecode:fallback', ## TODO: type name        $self->{input_encoding} = $charset->get_iana_name; ## TODO: Should we set actual charset decoder's encoding name?
623                        value => $self->{input_encoding},        !!!parse-error (type => 'chardecode:fallback',
624                        level => $self->{unsupported_level},                        level => $self->{level}->{uncertain},
625                        line => 1, column => 1);                        #text => $self->{input_encoding},
626                          line => 1, column => 1,
627                          layer => 'encode');
628      } elsif (not ($e_status &      } elsif (not ($e_status &
629                    Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL())) {                    Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL ())) {
630        !!!parse-error (type => 'chardecode:no error', ## TODO: type name        $self->{input_encoding} = $charset->get_iana_name;
631                        value => $self->{input_encoding},        !!!parse-error (type => 'chardecode:no error',
632                        level => $self->{unsupported_level},                        text => $self->{input_encoding},
633                        line => 1, column => 1);                        level => $self->{level}->{uncertain},
634                          line => 1, column => 1,
635                          layer => 'encode');
636        } else {
637          $self->{input_encoding} = $charset->get_iana_name;
638      }      }
639      $self->{confident} = 1;      $self->{confident} = 1;
640      $char_stream->onerror ($char_onerror);  
641      $return = $self->parse_char_stream ($char_stream, @args);      $wrapped_char_stream = $get_wrapper->($char_stream);
642        $wrapped_char_stream->onerror ($char_onerror);
643    
644        $return = $self->parse_char_stream ($wrapped_char_stream, @args);
645    };    };
646    return $return;    return $return;
647  } # parse_byte_stream  } # parse_byte_stream
# Line 566  sub parse_byte_stream ($$$$;$) { Line 655  sub parse_byte_stream ($$$$;$) {
655  ## such as |parse_byte_string| in this module, must ensure that it does  ## such as |parse_byte_string| in this module, must ensure that it does
656  ## strip the BOM and never strip any ZWNBSP.  ## strip the BOM and never strip any ZWNBSP.
657    
658  sub parse_char_string ($$$;$) {  sub parse_char_string ($$$;$$) {
659      #my ($self, $s, $doc, $onerror, $get_wrapper) = @_;
660    my $self = shift;    my $self = shift;
   require utf8;  
661    my $s = ref $_[0] ? $_[0] : \($_[0]);    my $s = ref $_[0] ? $_[0] : \($_[0]);
662    open my $input, '<' . (utf8::is_utf8 ($$s) ? ':utf8' : ''), $s;    require Whatpm::Charset::DecodeHandle;
663      my $input = Whatpm::Charset::DecodeHandle::CharString->new ($s);
664    return $self->parse_char_stream ($input, @_[1..$#_]);    return $self->parse_char_stream ($input, @_[1..$#_]);
665  } # parse_char_string  } # parse_char_string
666  *parse_string = \&parse_char_string;  *parse_string = \&parse_char_string; ## NOTE: Alias for backward compatibility.
667    
668  sub parse_char_stream ($$$;$) {  sub parse_char_stream ($$$;$$) {
669    my $self = ref $_[0] ? shift : shift->new;    my $self = ref $_[0] ? shift : shift->new;
670    my $input = $_[0];    my $input = $_[0];
671    $self->{document} = $_[1];    $self->{document} = $_[1];
# Line 586  sub parse_char_stream ($$$;$) { Line 676  sub parse_char_stream ($$$;$) {
676    $self->{confident} = 1 unless exists $self->{confident};    $self->{confident} = 1 unless exists $self->{confident};
677    $self->{document}->input_encoding ($self->{input_encoding})    $self->{document}->input_encoding ($self->{input_encoding})
678        if defined $self->{input_encoding};        if defined $self->{input_encoding};
679    ## TODO: |{input_encoding}| is needless?
680    
   my $i = 0;  
681    $self->{line_prev} = $self->{line} = 1;    $self->{line_prev} = $self->{line} = 1;
682    $self->{column_prev} = $self->{column} = 0;    $self->{column_prev} = -1;
683    $self->{set_next_char} = sub {    $self->{column} = 0;
684      $self->{set_nc} = sub {
685      my $self = shift;      my $self = shift;
686    
687      pop @{$self->{prev_char}};      my $char = '';
688      unshift @{$self->{prev_char}}, $self->{next_char};      if (defined $self->{next_nc}) {
689          $char = $self->{next_nc};
690      my $char;        delete $self->{next_nc};
691      if (defined $self->{next_next_char}) {        $self->{nc} = ord $char;
       $char = $self->{next_next_char};  
       delete $self->{next_next_char};  
692      } else {      } else {
693        $char = $input->getc;        $self->{char_buffer} = '';
694          $self->{char_buffer_pos} = 0;
695    
696          my $count = $input->manakai_read_until
697             ($self->{char_buffer}, qr/[^\x00\x0A\x0D]/, $self->{char_buffer_pos});
698          if ($count) {
699            $self->{line_prev} = $self->{line};
700            $self->{column_prev} = $self->{column};
701            $self->{column}++;
702            $self->{nc}
703                = ord substr ($self->{char_buffer}, $self->{char_buffer_pos}++, 1);
704            return;
705          }
706    
707          if ($input->read ($char, 1)) {
708            $self->{nc} = ord $char;
709          } else {
710            $self->{nc} = -1;
711            return;
712          }
713      }      }
     $self->{next_char} = -1 and return unless defined $char;  
     $self->{next_char} = ord $char;  
714    
715      ($self->{line_prev}, $self->{column_prev})      ($self->{line_prev}, $self->{column_prev})
716          = ($self->{line}, $self->{column});          = ($self->{line}, $self->{column});
717      $self->{column}++;      $self->{column}++;
718            
719      if ($self->{next_char} == 0x000A) { # LF      if ($self->{nc} == 0x000A) { # LF
720        !!!cp ('j1');        !!!cp ('j1');
721        $self->{line}++;        $self->{line}++;
722        $self->{column} = 0;        $self->{column} = 0;
723      } elsif ($self->{next_char} == 0x000D) { # CR      } elsif ($self->{nc} == 0x000D) { # CR
724        !!!cp ('j2');        !!!cp ('j2');
725        my $next = $input->getc;  ## TODO: support for abort/streaming
726        if (defined $next and $next ne "\x0A") {        my $next = '';
727          $self->{next_next_char} = $next;        if ($input->read ($next, 1) and $next ne "\x0A") {
728            $self->{next_nc} = $next;
729        }        }
730        $self->{next_char} = 0x000A; # LF # MUST        $self->{nc} = 0x000A; # LF # MUST
731        $self->{line}++;        $self->{line}++;
732        $self->{column} = 0;        $self->{column} = 0;
733      } elsif ($self->{next_char} > 0x10FFFF) {      } elsif ($self->{nc} == 0x0000) { # NULL
       !!!cp ('j3');  
       $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST  
     } elsif ($self->{next_char} == 0x0000) { # NULL  
734        !!!cp ('j4');        !!!cp ('j4');
735        !!!parse-error (type => 'NULL');        !!!parse-error (type => 'NULL');
736        $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST        $self->{nc} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
     } elsif ($self->{next_char} <= 0x0008 or  
              (0x000E <= $self->{next_char} and $self->{next_char} <= 0x001F) or  
              (0x007F <= $self->{next_char} and $self->{next_char} <= 0x009F) or  
              (0xD800 <= $self->{next_char} and $self->{next_char} <= 0xDFFF) or  
              (0xFDD0 <= $self->{next_char} and $self->{next_char} <= 0xFDDF) or  
              {  
               0xFFFE => 1, 0xFFFF => 1, 0x1FFFE => 1, 0x1FFFF => 1,  
               0x2FFFE => 1, 0x2FFFF => 1, 0x3FFFE => 1, 0x3FFFF => 1,  
               0x4FFFE => 1, 0x4FFFF => 1, 0x5FFFE => 1, 0x5FFFF => 1,  
               0x6FFFE => 1, 0x6FFFF => 1, 0x7FFFE => 1, 0x7FFFF => 1,  
               0x8FFFE => 1, 0x8FFFF => 1, 0x9FFFE => 1, 0x9FFFF => 1,  
               0xAFFFE => 1, 0xAFFFF => 1, 0xBFFFE => 1, 0xBFFFF => 1,  
               0xCFFFE => 1, 0xCFFFF => 1, 0xDFFFE => 1, 0xDFFFF => 1,  
               0xEFFFE => 1, 0xEFFFF => 1, 0xFFFFE => 1, 0xFFFFF => 1,  
               0x10FFFE => 1, 0x10FFFF => 1,  
              }->{$self->{next_char}}) {  
       !!!cp ('j5');  
       !!!parse-error (type => 'control char', level => $self->{must_level});  
 ## TODO: error type documentation  
737      }      }
738    };    };
739    $self->{prev_char} = [-1, -1, -1];  
740    $self->{next_char} = -1;    $self->{read_until} = sub {
741        #my ($scalar, $specials_range, $offset) = @_;
742        return 0 if defined $self->{next_nc};
743    
744        my $pattern = qr/[^$_[1]\x00\x0A\x0D]/;
745        my $offset = $_[2] || 0;
746    
747        if ($self->{char_buffer_pos} < length $self->{char_buffer}) {
748          pos ($self->{char_buffer}) = $self->{char_buffer_pos};
749          if ($self->{char_buffer} =~ /\G(?>$pattern)+/) {
750            substr ($_[0], $offset)
751                = substr ($self->{char_buffer}, $-[0], $+[0] - $-[0]);
752            my $count = $+[0] - $-[0];
753            if ($count) {
754              $self->{column} += $count;
755              $self->{char_buffer_pos} += $count;
756              $self->{line_prev} = $self->{line};
757              $self->{column_prev} = $self->{column} - 1;
758              $self->{nc} = -1;
759            }
760            return $count;
761          } else {
762            return 0;
763          }
764        } else {
765          my $count = $input->manakai_read_until ($_[0], $pattern, $_[2]);
766          if ($count) {
767            $self->{column} += $count;
768            $self->{line_prev} = $self->{line};
769            $self->{column_prev} = $self->{column} - 1;
770            $self->{nc} = -1;
771          }
772          return $count;
773        }
774      }; # $self->{read_until}
775    
776    my $onerror = $_[2] || sub {    my $onerror = $_[2] || sub {
777      my (%opt) = @_;      my (%opt) = @_;
# Line 664  sub parse_char_stream ($$$;$) { Line 783  sub parse_char_stream ($$$;$) {
783      $onerror->(line => $self->{line}, column => $self->{column}, @_);      $onerror->(line => $self->{line}, column => $self->{column}, @_);
784    };    };
785    
786      my $char_onerror = sub {
787        my (undef, $type, %opt) = @_;
788        !!!parse-error (layer => 'encode',
789                        line => $self->{line}, column => $self->{column} + 1,
790                        %opt, type => $type);
791      }; # $char_onerror
792    
793      if ($_[3]) {
794        $input = $_[3]->($input);
795        $input->onerror ($char_onerror);
796      } else {
797        $input->onerror ($char_onerror) unless defined $input->onerror;
798      }
799    
800    $self->_initialize_tokenizer;    $self->_initialize_tokenizer;
801    $self->_initialize_tree_constructor;    $self->_initialize_tree_constructor;
802    $self->_construct_tree;    $self->_construct_tree;
# Line 677  sub parse_char_stream ($$$;$) { Line 810  sub parse_char_stream ($$$;$) {
810  sub new ($) {  sub new ($) {
811    my $class = shift;    my $class = shift;
812    my $self = bless {    my $self = bless {
813      must_level => 'm',      level => {must => 'm',
814      should_level => 's',                should => 's',
815      good_level => 'w',                warn => 'w',
816      warn_level => 'w',                info => 'i',
817      info_level => 'i',                uncertain => 'u'},
     unsupported_level => 'u',  
818    }, $class;    }, $class;
819    $self->{set_next_char} = sub {    $self->{set_nc} = sub {
820      $self->{next_char} = -1;      $self->{nc} = -1;
821    };    };
822    $self->{parse_error} = sub {    $self->{parse_error} = sub {
823      #      #
# Line 712  sub RCDATA_CONTENT_MODEL () { CM_ENTITY Line 844  sub RCDATA_CONTENT_MODEL () { CM_ENTITY
844  sub PCDATA_CONTENT_MODEL () { CM_ENTITY | CM_FULL_MARKUP }  sub PCDATA_CONTENT_MODEL () { CM_ENTITY | CM_FULL_MARKUP }
845    
846  sub DATA_STATE () { 0 }  sub DATA_STATE () { 0 }
847  sub ENTITY_DATA_STATE () { 1 }  #sub ENTITY_DATA_STATE () { 1 }
848  sub TAG_OPEN_STATE () { 2 }  sub TAG_OPEN_STATE () { 2 }
849  sub CLOSE_TAG_OPEN_STATE () { 3 }  sub CLOSE_TAG_OPEN_STATE () { 3 }
850  sub TAG_NAME_STATE () { 4 }  sub TAG_NAME_STATE () { 4 }
# Line 723  sub BEFORE_ATTRIBUTE_VALUE_STATE () { 8 Line 855  sub BEFORE_ATTRIBUTE_VALUE_STATE () { 8
855  sub ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE () { 9 }  sub ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE () { 9 }
856  sub ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE () { 10 }  sub ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE () { 10 }
857  sub ATTRIBUTE_VALUE_UNQUOTED_STATE () { 11 }  sub ATTRIBUTE_VALUE_UNQUOTED_STATE () { 11 }
858  sub ENTITY_IN_ATTRIBUTE_VALUE_STATE () { 12 }  #sub ENTITY_IN_ATTRIBUTE_VALUE_STATE () { 12 }
859  sub MARKUP_DECLARATION_OPEN_STATE () { 13 }  sub MARKUP_DECLARATION_OPEN_STATE () { 13 }
860  sub COMMENT_START_STATE () { 14 }  sub COMMENT_START_STATE () { 14 }
861  sub COMMENT_START_DASH_STATE () { 15 }  sub COMMENT_START_DASH_STATE () { 15 }
# Line 746  sub AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STAT Line 878  sub AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STAT
878  sub BOGUS_DOCTYPE_STATE () { 32 }  sub BOGUS_DOCTYPE_STATE () { 32 }
879  sub AFTER_ATTRIBUTE_VALUE_QUOTED_STATE () { 33 }  sub AFTER_ATTRIBUTE_VALUE_QUOTED_STATE () { 33 }
880  sub SELF_CLOSING_START_TAG_STATE () { 34 }  sub SELF_CLOSING_START_TAG_STATE () { 34 }
881  sub CDATA_BLOCK_STATE () { 35 }  sub CDATA_SECTION_STATE () { 35 }
882    sub MD_HYPHEN_STATE () { 36 } # "markup declaration open state" in the spec
883    sub MD_DOCTYPE_STATE () { 37 } # "markup declaration open state" in the spec
884    sub MD_CDATA_STATE () { 38 } # "markup declaration open state" in the spec
885    sub CDATA_RCDATA_CLOSE_TAG_STATE () { 39 } # "close tag open state" in the spec
886    sub CDATA_SECTION_MSE1_STATE () { 40 } # "CDATA section state" in the spec
887    sub CDATA_SECTION_MSE2_STATE () { 41 } # "CDATA section state" in the spec
888    sub PUBLIC_STATE () { 42 } # "after DOCTYPE name state" in the spec
889    sub SYSTEM_STATE () { 43 } # "after DOCTYPE name state" in the spec
890    ## NOTE: "Entity data state", "entity in attribute value state", and
891    ## "consume a character reference" algorithm are jointly implemented
892    ## using the following six states:
893    sub ENTITY_STATE () { 44 }
894    sub ENTITY_HASH_STATE () { 45 }
895    sub NCR_NUM_STATE () { 46 }
896    sub HEXREF_X_STATE () { 47 }
897    sub HEXREF_HEX_STATE () { 48 }
898    sub ENTITY_NAME_STATE () { 49 }
899    sub PCDATA_STATE () { 50 } # "data state" in the spec
900    
901  sub DOCTYPE_TOKEN () { 1 }  sub DOCTYPE_TOKEN () { 1 }
902  sub COMMENT_TOKEN () { 2 }  sub COMMENT_TOKEN () { 2 }
# Line 768  sub IN_FOREIGN_CONTENT_IM () { 0b1000000 Line 918  sub IN_FOREIGN_CONTENT_IM () { 0b1000000
918      ## NOTE: "in foreign content" insertion mode is special; it is combined      ## NOTE: "in foreign content" insertion mode is special; it is combined
919      ## with the secondary insertion mode.  In this parser, they are stored      ## with the secondary insertion mode.  In this parser, they are stored
920      ## together in the bit-or'ed form.      ## together in the bit-or'ed form.
921    sub IN_CDATA_RCDATA_IM () { 0b1000000000000 }
922        ## NOTE: "in CDATA/RCDATA" insertion mode is also special; it is
923        ## combined with the original insertion mode.  In thie parser,
924        ## they are stored together in the bit-or'ed form.
925    
926  ## NOTE: "initial" and "before html" insertion modes have no constants.  ## NOTE: "initial" and "before html" insertion modes have no constants.
927    
# Line 799  sub IN_COLUMN_GROUP_IM () { 0b10 } Line 953  sub IN_COLUMN_GROUP_IM () { 0b10 }
953  sub _initialize_tokenizer ($) {  sub _initialize_tokenizer ($) {
954    my $self = shift;    my $self = shift;
955    $self->{state} = DATA_STATE; # MUST    $self->{state} = DATA_STATE; # MUST
956      #$self->{s_kwd}; # state keyword - initialized when used
957      #$self->{entity__value}; # initialized when used
958      #$self->{entity__match}; # initialized when used
959    $self->{content_model} = PCDATA_CONTENT_MODEL; # be    $self->{content_model} = PCDATA_CONTENT_MODEL; # be
960    undef $self->{current_token}; # start tag, end tag, comment, or DOCTYPE    undef $self->{ct}; # current token
961    undef $self->{current_attribute};    undef $self->{ca}; # current attribute
962    undef $self->{last_emitted_start_tag_name};    undef $self->{last_stag_name}; # last emitted start tag name
963    undef $self->{last_attribute_value_state};    #$self->{prev_state}; # initialized when used
964    delete $self->{self_closing};    delete $self->{self_closing};
965    $self->{char} = [];    $self->{char_buffer} = '';
966    # $self->{next_char}    $self->{char_buffer_pos} = 0;
967      $self->{nc} = -1; # next input character
968      #$self->{next_nc}
969    !!!next-input-character;    !!!next-input-character;
970    $self->{token} = [];    $self->{token} = [];
971    # $self->{escape}    # $self->{escape}
# Line 817  sub _initialize_tokenizer ($) { Line 976  sub _initialize_tokenizer ($) {
976  ##       CHARACTER_TOKEN, or END_OF_FILE_TOKEN  ##       CHARACTER_TOKEN, or END_OF_FILE_TOKEN
977  ##   ->{name} (DOCTYPE_TOKEN)  ##   ->{name} (DOCTYPE_TOKEN)
978  ##   ->{tag_name} (START_TAG_TOKEN, END_TAG_TOKEN)  ##   ->{tag_name} (START_TAG_TOKEN, END_TAG_TOKEN)
979  ##   ->{public_identifier} (DOCTYPE_TOKEN)  ##   ->{pubid} (DOCTYPE_TOKEN)
980  ##   ->{system_identifier} (DOCTYPE_TOKEN)  ##   ->{sysid} (DOCTYPE_TOKEN)
981  ##   ->{quirks} == 1 or 0 (DOCTYPE_TOKEN): "force-quirks" flag  ##   ->{quirks} == 1 or 0 (DOCTYPE_TOKEN): "force-quirks" flag
982  ##   ->{attributes} isa HASH (START_TAG_TOKEN, END_TAG_TOKEN)  ##   ->{attributes} isa HASH (START_TAG_TOKEN, END_TAG_TOKEN)
983  ##        ->{name}  ##        ->{name}
# Line 829  sub _initialize_tokenizer ($) { Line 988  sub _initialize_tokenizer ($) {
988  ##     |->{self_closing}| is used to save the value of |$self->{self_closing}|  ##     |->{self_closing}| is used to save the value of |$self->{self_closing}|
989  ##     while the token is pushed back to the stack.  ##     while the token is pushed back to the stack.
990    
 ## ISSUE: "When a DOCTYPE token is created, its  
 ## <i>self-closing flag</i> must be unset (its other state is that it  
 ## be set), and its attributes list must be empty.": Wrong subject?  
   
991  ## Emitted token MUST immediately be handled by the tree construction state.  ## Emitted token MUST immediately be handled by the tree construction state.
992    
993  ## Before each step, UA MAY check to see if either one of the scripts in  ## Before each step, UA MAY check to see if either one of the scripts in
# Line 841  sub _initialize_tokenizer ($) { Line 996  sub _initialize_tokenizer ($) {
996  ## has completed loading.  If one has, then it MUST be executed  ## has completed loading.  If one has, then it MUST be executed
997  ## and removed from the list.  ## and removed from the list.
998    
999  ## NOTE: HTML5 "Writing HTML documents" section, applied to  ## TODO: Polytheistic slash SHOULD NOT be used. (Applied only to atheists.)
1000  ## documents and not to user agents and conformance checkers,  ## (This requirement was dropped from HTML5 spec, unfortunately.)
1001  ## contains some requirements that are not detected by the  
1002  ## parsing algorithm:  my $is_space = {
1003  ## - Some requirements on character encoding declarations. ## TODO    0x0009 => 1, # CHARACTER TABULATION (HT)
1004  ## - "Elements MUST NOT contain content that their content model disallows."    0x000A => 1, # LINE FEED (LF)
1005  ##   ... Some are parse error, some are not (will be reported by c.c.).    #0x000B => 0, # LINE TABULATION (VT)
1006  ## - Polytheistic slash SHOULD NOT be used. (Applied only to atheists.) ## TODO    0x000C => 1, # FORM FEED (FF)
1007  ## - Text (in elements, attributes, and comments) SHOULD NOT contain    #0x000D => 1, # CARRIAGE RETURN (CR)
1008  ##   control characters other than space characters. ## TODO: (what is control character? C0, C1 and DEL?  Unicode control character?)    0x0020 => 1, # SPACE (SP)
1009    };
 ## TODO: HTML5 poses authors two SHOULD-level requirements that cannot  
 ## be detected by the HTML5 parsing algorithm:  
 ## - Text,  
1010    
1011  sub _get_next_token ($) {  sub _get_next_token ($) {
1012    my $self = shift;    my $self = shift;
1013    
1014    if ($self->{self_closing}) {    if ($self->{self_closing}) {
1015      !!!parse-error (type => 'nestc', token => $self->{current_token});      !!!parse-error (type => 'nestc', token => $self->{ct});
1016      ## NOTE: The |self_closing| flag is only set by start tag token.      ## NOTE: The |self_closing| flag is only set by start tag token.
1017      ## In addition, when a start tag token is emitted, it is always set to      ## In addition, when a start tag token is emitted, it is always set to
1018      ## |current_token|.      ## |ct|.
1019      delete $self->{self_closing};      delete $self->{self_closing};
1020    }    }
1021    
# Line 873  sub _get_next_token ($) { Line 1025  sub _get_next_token ($) {
1025    }    }
1026    
1027    A: {    A: {
1028      if ($self->{state} == DATA_STATE) {      if ($self->{state} == PCDATA_STATE) {
1029        if ($self->{next_char} == 0x0026) { # &        ## NOTE: Same as |DATA_STATE|, but only for |PCDATA| content model.
1030    
1031          if ($self->{nc} == 0x0026) { # &
1032            !!!cp (0.1);
1033            ## NOTE: In the spec, the tokenizer is switched to the
1034            ## "entity data state".  In this implementation, the tokenizer
1035            ## is switched to the |ENTITY_STATE|, which is an implementation
1036            ## of the "consume a character reference" algorithm.
1037            $self->{entity_add} = -1;
1038            $self->{prev_state} = DATA_STATE;
1039            $self->{state} = ENTITY_STATE;
1040            !!!next-input-character;
1041            redo A;
1042          } elsif ($self->{nc} == 0x003C) { # <
1043            !!!cp (0.2);
1044            $self->{state} = TAG_OPEN_STATE;
1045            !!!next-input-character;
1046            redo A;
1047          } elsif ($self->{nc} == -1) {
1048            !!!cp (0.3);
1049            !!!emit ({type => END_OF_FILE_TOKEN,
1050                      line => $self->{line}, column => $self->{column}});
1051            last A; ## TODO: ok?
1052          } else {
1053            !!!cp (0.4);
1054            #
1055          }
1056    
1057          # Anything else
1058          my $token = {type => CHARACTER_TOKEN,
1059                       data => chr $self->{nc},
1060                       line => $self->{line}, column => $self->{column},
1061                      };
1062          $self->{read_until}->($token->{data}, q[<&], length $token->{data});
1063    
1064          ## Stay in the state.
1065          !!!next-input-character;
1066          !!!emit ($token);
1067          redo A;
1068        } elsif ($self->{state} == DATA_STATE) {
1069          $self->{s_kwd} = '' unless defined $self->{s_kwd};
1070          if ($self->{nc} == 0x0026) { # &
1071            $self->{s_kwd} = '';
1072          if ($self->{content_model} & CM_ENTITY and # PCDATA | RCDATA          if ($self->{content_model} & CM_ENTITY and # PCDATA | RCDATA
1073              not $self->{escape}) {              not $self->{escape}) {
1074            !!!cp (1);            !!!cp (1);
1075            $self->{state} = ENTITY_DATA_STATE;            ## NOTE: In the spec, the tokenizer is switched to the
1076              ## "entity data state".  In this implementation, the tokenizer
1077              ## is switched to the |ENTITY_STATE|, which is an implementation
1078              ## of the "consume a character reference" algorithm.
1079              $self->{entity_add} = -1;
1080              $self->{prev_state} = DATA_STATE;
1081              $self->{state} = ENTITY_STATE;
1082            !!!next-input-character;            !!!next-input-character;
1083            redo A;            redo A;
1084          } else {          } else {
1085            !!!cp (2);            !!!cp (2);
1086            #            #
1087          }          }
1088        } elsif ($self->{next_char} == 0x002D) { # -        } elsif ($self->{nc} == 0x002D) { # -
1089          if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA          if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
1090            unless ($self->{escape}) {            $self->{s_kwd} .= '-';
1091              if ($self->{prev_char}->[0] == 0x002D and # -            
1092                  $self->{prev_char}->[1] == 0x0021 and # !            if ($self->{s_kwd} eq '<!--') {
1093                  $self->{prev_char}->[2] == 0x003C) { # <              !!!cp (3);
1094                !!!cp (3);              $self->{escape} = 1; # unless $self->{escape};
1095                $self->{escape} = 1;              $self->{s_kwd} = '--';
1096              } else {              #
1097                !!!cp (4);            } elsif ($self->{s_kwd} eq '---') {
1098              }              !!!cp (4);
1099                $self->{s_kwd} = '--';
1100                #
1101            } else {            } else {
1102              !!!cp (5);              !!!cp (5);
1103                #
1104            }            }
1105          }          }
1106                    
1107          #          #
1108        } elsif ($self->{next_char} == 0x003C) { # <        } elsif ($self->{nc} == 0x0021) { # !
1109            if (length $self->{s_kwd}) {
1110              !!!cp (5.1);
1111              $self->{s_kwd} .= '!';
1112              #
1113            } else {
1114              !!!cp (5.2);
1115              #$self->{s_kwd} = '';
1116              #
1117            }
1118            #
1119          } elsif ($self->{nc} == 0x003C) { # <
1120          if ($self->{content_model} & CM_FULL_MARKUP or # PCDATA          if ($self->{content_model} & CM_FULL_MARKUP or # PCDATA
1121              (($self->{content_model} & CM_LIMITED_MARKUP) and # CDATA | RCDATA              (($self->{content_model} & CM_LIMITED_MARKUP) and # CDATA | RCDATA
1122               not $self->{escape})) {               not $self->{escape})) {
# Line 912  sub _get_next_token ($) { Line 1126  sub _get_next_token ($) {
1126            redo A;            redo A;
1127          } else {          } else {
1128            !!!cp (7);            !!!cp (7);
1129              $self->{s_kwd} = '';
1130            #            #
1131          }          }
1132        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1133          if ($self->{escape} and          if ($self->{escape} and
1134              ($self->{content_model} & CM_LIMITED_MARKUP)) { # RCDATA | CDATA              ($self->{content_model} & CM_LIMITED_MARKUP)) { # RCDATA | CDATA
1135            if ($self->{prev_char}->[0] == 0x002D and # -            if ($self->{s_kwd} eq '--') {
               $self->{prev_char}->[1] == 0x002D) { # -  
1136              !!!cp (8);              !!!cp (8);
1137              delete $self->{escape};              delete $self->{escape};
1138            } else {            } else {
# Line 928  sub _get_next_token ($) { Line 1142  sub _get_next_token ($) {
1142            !!!cp (10);            !!!cp (10);
1143          }          }
1144                    
1145            $self->{s_kwd} = '';
1146          #          #
1147        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1148          !!!cp (11);          !!!cp (11);
1149            $self->{s_kwd} = '';
1150          !!!emit ({type => END_OF_FILE_TOKEN,          !!!emit ({type => END_OF_FILE_TOKEN,
1151                    line => $self->{line}, column => $self->{column}});                    line => $self->{line}, column => $self->{column}});
1152          last A; ## TODO: ok?          last A; ## TODO: ok?
1153        } else {        } else {
1154          !!!cp (12);          !!!cp (12);
1155            $self->{s_kwd} = '';
1156            #
1157        }        }
1158    
1159        # Anything else        # Anything else
1160        my $token = {type => CHARACTER_TOKEN,        my $token = {type => CHARACTER_TOKEN,
1161                     data => chr $self->{next_char},                     data => chr $self->{nc},
1162                     line => $self->{line}, column => $self->{column},                     line => $self->{line}, column => $self->{column},
1163                    };                    };
1164        ## Stay in the data state        if ($self->{read_until}->($token->{data}, q[-!<>&],
1165        !!!next-input-character;                                  length $token->{data})) {
1166            $self->{s_kwd} = '';
1167        !!!emit ($token);        }
   
       redo A;  
     } elsif ($self->{state} == ENTITY_DATA_STATE) {  
       ## (cannot happen in CDATA state)  
   
       my ($l, $c) = ($self->{line_prev}, $self->{column_prev});  
         
       my $token = $self->_tokenize_attempt_to_consume_an_entity (0, -1);  
   
       $self->{state} = DATA_STATE;  
       # next-input-character is already done  
1168    
1169        unless (defined $token) {        ## Stay in the data state.
1170          if ($self->{content_model} == PCDATA_CONTENT_MODEL) {
1171          !!!cp (13);          !!!cp (13);
1172          !!!emit ({type => CHARACTER_TOKEN, data => '&',          $self->{state} = PCDATA_STATE;
                   line => $l, column => $c,  
                  });  
1173        } else {        } else {
1174          !!!cp (14);          !!!cp (14);
1175          !!!emit ($token);          ## Stay in the state.
1176        }        }
1177          !!!next-input-character;
1178          !!!emit ($token);
1179        redo A;        redo A;
1180      } elsif ($self->{state} == TAG_OPEN_STATE) {      } elsif ($self->{state} == TAG_OPEN_STATE) {
1181        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
1182          if ($self->{next_char} == 0x002F) { # /          if ($self->{nc} == 0x002F) { # /
1183            !!!cp (15);            !!!cp (15);
1184            !!!next-input-character;            !!!next-input-character;
1185            $self->{state} = CLOSE_TAG_OPEN_STATE;            $self->{state} = CLOSE_TAG_OPEN_STATE;
1186            redo A;            redo A;
1187            } elsif ($self->{nc} == 0x0021) { # !
1188              !!!cp (15.1);
1189              $self->{s_kwd} = '<' unless $self->{escape};
1190              #
1191          } else {          } else {
1192            !!!cp (16);            !!!cp (16);
1193            ## reconsume            #
           $self->{state} = DATA_STATE;  
   
           !!!emit ({type => CHARACTER_TOKEN, data => '<',  
                     line => $self->{line_prev},  
                     column => $self->{column_prev},  
                    });  
   
           redo A;  
1194          }          }
1195    
1196            ## reconsume
1197            $self->{state} = DATA_STATE;
1198            !!!emit ({type => CHARACTER_TOKEN, data => '<',
1199                      line => $self->{line_prev},
1200                      column => $self->{column_prev},
1201                     });
1202            redo A;
1203        } elsif ($self->{content_model} & CM_FULL_MARKUP) { # PCDATA        } elsif ($self->{content_model} & CM_FULL_MARKUP) { # PCDATA
1204          if ($self->{next_char} == 0x0021) { # !          if ($self->{nc} == 0x0021) { # !
1205            !!!cp (17);            !!!cp (17);
1206            $self->{state} = MARKUP_DECLARATION_OPEN_STATE;            $self->{state} = MARKUP_DECLARATION_OPEN_STATE;
1207            !!!next-input-character;            !!!next-input-character;
1208            redo A;            redo A;
1209          } elsif ($self->{next_char} == 0x002F) { # /          } elsif ($self->{nc} == 0x002F) { # /
1210            !!!cp (18);            !!!cp (18);
1211            $self->{state} = CLOSE_TAG_OPEN_STATE;            $self->{state} = CLOSE_TAG_OPEN_STATE;
1212            !!!next-input-character;            !!!next-input-character;
1213            redo A;            redo A;
1214          } elsif (0x0041 <= $self->{next_char} and          } elsif (0x0041 <= $self->{nc} and
1215                   $self->{next_char} <= 0x005A) { # A..Z                   $self->{nc} <= 0x005A) { # A..Z
1216            !!!cp (19);            !!!cp (19);
1217            $self->{current_token}            $self->{ct}
1218              = {type => START_TAG_TOKEN,              = {type => START_TAG_TOKEN,
1219                 tag_name => chr ($self->{next_char} + 0x0020),                 tag_name => chr ($self->{nc} + 0x0020),
1220                 line => $self->{line_prev},                 line => $self->{line_prev},
1221                 column => $self->{column_prev}};                 column => $self->{column_prev}};
1222            $self->{state} = TAG_NAME_STATE;            $self->{state} = TAG_NAME_STATE;
1223            !!!next-input-character;            !!!next-input-character;
1224            redo A;            redo A;
1225          } elsif (0x0061 <= $self->{next_char} and          } elsif (0x0061 <= $self->{nc} and
1226                   $self->{next_char} <= 0x007A) { # a..z                   $self->{nc} <= 0x007A) { # a..z
1227            !!!cp (20);            !!!cp (20);
1228            $self->{current_token} = {type => START_TAG_TOKEN,            $self->{ct} = {type => START_TAG_TOKEN,
1229                                      tag_name => chr ($self->{next_char}),                                      tag_name => chr ($self->{nc}),
1230                                      line => $self->{line_prev},                                      line => $self->{line_prev},
1231                                      column => $self->{column_prev}};                                      column => $self->{column_prev}};
1232            $self->{state} = TAG_NAME_STATE;            $self->{state} = TAG_NAME_STATE;
1233            !!!next-input-character;            !!!next-input-character;
1234            redo A;            redo A;
1235          } elsif ($self->{next_char} == 0x003E) { # >          } elsif ($self->{nc} == 0x003E) { # >
1236            !!!cp (21);            !!!cp (21);
1237            !!!parse-error (type => 'empty start tag',            !!!parse-error (type => 'empty start tag',
1238                            line => $self->{line_prev},                            line => $self->{line_prev},
# Line 1034  sub _get_next_token ($) { Line 1246  sub _get_next_token ($) {
1246                     });                     });
1247    
1248            redo A;            redo A;
1249          } elsif ($self->{next_char} == 0x003F) { # ?          } elsif ($self->{nc} == 0x003F) { # ?
1250            !!!cp (22);            !!!cp (22);
1251            !!!parse-error (type => 'pio',            !!!parse-error (type => 'pio',
1252                            line => $self->{line_prev},                            line => $self->{line_prev},
1253                            column => $self->{column_prev});                            column => $self->{column_prev});
1254            $self->{state} = BOGUS_COMMENT_STATE;            $self->{state} = BOGUS_COMMENT_STATE;
1255            $self->{current_token} = {type => COMMENT_TOKEN, data => '',            $self->{ct} = {type => COMMENT_TOKEN, data => '',
1256                                      line => $self->{line_prev},                                      line => $self->{line_prev},
1257                                      column => $self->{column_prev},                                      column => $self->{column_prev},
1258                                     };                                     };
1259            ## $self->{next_char} is intentionally left as is            ## $self->{nc} is intentionally left as is
1260            redo A;            redo A;
1261          } else {          } else {
1262            !!!cp (23);            !!!cp (23);
# Line 1065  sub _get_next_token ($) { Line 1277  sub _get_next_token ($) {
1277          die "$0: $self->{content_model} in tag open";          die "$0: $self->{content_model} in tag open";
1278        }        }
1279      } elsif ($self->{state} == CLOSE_TAG_OPEN_STATE) {      } elsif ($self->{state} == CLOSE_TAG_OPEN_STATE) {
1280          ## NOTE: The "close tag open state" in the spec is implemented as
1281          ## |CLOSE_TAG_OPEN_STATE| and |CDATA_RCDATA_CLOSE_TAG_STATE|.
1282    
1283        my ($l, $c) = ($self->{line_prev}, $self->{column_prev} - 1); # "<"of"</"        my ($l, $c) = ($self->{line_prev}, $self->{column_prev} - 1); # "<"of"</"
1284        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
1285          if (defined $self->{last_emitted_start_tag_name}) {          if (defined $self->{last_stag_name}) {
1286              $self->{state} = CDATA_RCDATA_CLOSE_TAG_STATE;
1287            ## NOTE: <http://krijnhoetmer.nl/irc-logs/whatwg/20070626#l-564>            $self->{s_kwd} = '';
1288            my @next_char;            ## Reconsume.
1289            TAGNAME: for (my $i = 0; $i < length $self->{last_emitted_start_tag_name}; $i++) {            redo A;
             push @next_char, $self->{next_char};  
             my $c = ord substr ($self->{last_emitted_start_tag_name}, $i, 1);  
             my $C = 0x0061 <= $c && $c <= 0x007A ? $c - 0x0020 : $c;  
             if ($self->{next_char} == $c or $self->{next_char} == $C) {  
               !!!cp (24);  
               !!!next-input-character;  
               next TAGNAME;  
             } else {  
               !!!cp (25);  
               $self->{next_char} = shift @next_char; # reconsume  
               !!!back-next-input-character (@next_char);  
               $self->{state} = DATA_STATE;  
   
               !!!emit ({type => CHARACTER_TOKEN, data => '</',  
                         line => $l, column => $c,  
                        });  
     
               redo A;  
             }  
           }  
           push @next_char, $self->{next_char};  
         
           unless ($self->{next_char} == 0x0009 or # HT  
                   $self->{next_char} == 0x000A or # LF  
                   $self->{next_char} == 0x000B or # VT  
                   $self->{next_char} == 0x000C or # FF  
                   $self->{next_char} == 0x0020 or # SP  
                   $self->{next_char} == 0x003E or # >  
                   $self->{next_char} == 0x002F or # /  
                   $self->{next_char} == -1) {  
             !!!cp (26);  
             $self->{next_char} = shift @next_char; # reconsume  
             !!!back-next-input-character (@next_char);  
             $self->{state} = DATA_STATE;  
             !!!emit ({type => CHARACTER_TOKEN, data => '</',  
                       line => $l, column => $c,  
                      });  
             redo A;  
           } else {  
             !!!cp (27);  
             $self->{next_char} = shift @next_char;  
             !!!back-next-input-character (@next_char);  
             # and consume...  
           }  
1290          } else {          } else {
1291            ## No start tag token has ever been emitted            ## No start tag token has ever been emitted
1292              ## NOTE: See <http://krijnhoetmer.nl/irc-logs/whatwg/20070626#l-564>.
1293            !!!cp (28);            !!!cp (28);
           # next-input-character is already done  
1294            $self->{state} = DATA_STATE;            $self->{state} = DATA_STATE;
1295              ## Reconsume.
1296            !!!emit ({type => CHARACTER_TOKEN, data => '</',            !!!emit ({type => CHARACTER_TOKEN, data => '</',
1297                      line => $l, column => $c,                      line => $l, column => $c,
1298                     });                     });
1299            redo A;            redo A;
1300          }          }
1301        }        }
1302          
1303        if (0x0041 <= $self->{next_char} and        if (0x0041 <= $self->{nc} and
1304            $self->{next_char} <= 0x005A) { # A..Z            $self->{nc} <= 0x005A) { # A..Z
1305          !!!cp (29);          !!!cp (29);
1306          $self->{current_token}          $self->{ct}
1307              = {type => END_TAG_TOKEN,              = {type => END_TAG_TOKEN,
1308                 tag_name => chr ($self->{next_char} + 0x0020),                 tag_name => chr ($self->{nc} + 0x0020),
1309                 line => $l, column => $c};                 line => $l, column => $c};
1310          $self->{state} = TAG_NAME_STATE;          $self->{state} = TAG_NAME_STATE;
1311          !!!next-input-character;          !!!next-input-character;
1312          redo A;          redo A;
1313        } elsif (0x0061 <= $self->{next_char} and        } elsif (0x0061 <= $self->{nc} and
1314                 $self->{next_char} <= 0x007A) { # a..z                 $self->{nc} <= 0x007A) { # a..z
1315          !!!cp (30);          !!!cp (30);
1316          $self->{current_token} = {type => END_TAG_TOKEN,          $self->{ct} = {type => END_TAG_TOKEN,
1317                                    tag_name => chr ($self->{next_char}),                                    tag_name => chr ($self->{nc}),
1318                                    line => $l, column => $c};                                    line => $l, column => $c};
1319          $self->{state} = TAG_NAME_STATE;          $self->{state} = TAG_NAME_STATE;
1320          !!!next-input-character;          !!!next-input-character;
1321          redo A;          redo A;
1322        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1323          !!!cp (31);          !!!cp (31);
1324          !!!parse-error (type => 'empty end tag',          !!!parse-error (type => 'empty end tag',
1325                          line => $self->{line_prev}, ## "<" in "</>"                          line => $self->{line_prev}, ## "<" in "</>"
# Line 1155  sub _get_next_token ($) { Line 1327  sub _get_next_token ($) {
1327          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1328          !!!next-input-character;          !!!next-input-character;
1329          redo A;          redo A;
1330        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1331          !!!cp (32);          !!!cp (32);
1332          !!!parse-error (type => 'bare etago');          !!!parse-error (type => 'bare etago');
1333          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
# Line 1170  sub _get_next_token ($) { Line 1342  sub _get_next_token ($) {
1342          !!!cp (33);          !!!cp (33);
1343          !!!parse-error (type => 'bogus end tag');          !!!parse-error (type => 'bogus end tag');
1344          $self->{state} = BOGUS_COMMENT_STATE;          $self->{state} = BOGUS_COMMENT_STATE;
1345          $self->{current_token} = {type => COMMENT_TOKEN, data => '',          $self->{ct} = {type => COMMENT_TOKEN, data => '',
1346                                    line => $self->{line_prev}, # "<" of "</"                                    line => $self->{line_prev}, # "<" of "</"
1347                                    column => $self->{column_prev} - 1,                                    column => $self->{column_prev} - 1,
1348                                   };                                   };
1349          ## $self->{next_char} is intentionally left as is          ## NOTE: $self->{nc} is intentionally left as is.
1350          redo A;          ## Although the "anything else" case of the spec not explicitly
1351            ## states that the next input character is to be reconsumed,
1352            ## it will be included to the |data| of the comment token
1353            ## generated from the bogus end tag, as defined in the
1354            ## "bogus comment state" entry.
1355            redo A;
1356          }
1357        } elsif ($self->{state} == CDATA_RCDATA_CLOSE_TAG_STATE) {
1358          my $ch = substr $self->{last_stag_name}, length $self->{s_kwd}, 1;
1359          if (length $ch) {
1360            my $CH = $ch;
1361            $ch =~ tr/a-z/A-Z/;
1362            my $nch = chr $self->{nc};
1363            if ($nch eq $ch or $nch eq $CH) {
1364              !!!cp (24);
1365              ## Stay in the state.
1366              $self->{s_kwd} .= $nch;
1367              !!!next-input-character;
1368              redo A;
1369            } else {
1370              !!!cp (25);
1371              $self->{state} = DATA_STATE;
1372              ## Reconsume.
1373              !!!emit ({type => CHARACTER_TOKEN,
1374                        data => '</' . $self->{s_kwd},
1375                        line => $self->{line_prev},
1376                        column => $self->{column_prev} - 1 - length $self->{s_kwd},
1377                       });
1378              redo A;
1379            }
1380          } else { # after "<{tag-name}"
1381            unless ($is_space->{$self->{nc}} or
1382                    {
1383                     0x003E => 1, # >
1384                     0x002F => 1, # /
1385                     -1 => 1, # EOF
1386                    }->{$self->{nc}}) {
1387              !!!cp (26);
1388              ## Reconsume.
1389              $self->{state} = DATA_STATE;
1390              !!!emit ({type => CHARACTER_TOKEN,
1391                        data => '</' . $self->{s_kwd},
1392                        line => $self->{line_prev},
1393                        column => $self->{column_prev} - 1 - length $self->{s_kwd},
1394                       });
1395              redo A;
1396            } else {
1397              !!!cp (27);
1398              $self->{ct}
1399                  = {type => END_TAG_TOKEN,
1400                     tag_name => $self->{last_stag_name},
1401                     line => $self->{line_prev},
1402                     column => $self->{column_prev} - 1 - length $self->{s_kwd}};
1403              $self->{state} = TAG_NAME_STATE;
1404              ## Reconsume.
1405              redo A;
1406            }
1407        }        }
1408      } elsif ($self->{state} == TAG_NAME_STATE) {      } elsif ($self->{state} == TAG_NAME_STATE) {
1409        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
1410          !!!cp (34);          !!!cp (34);
1411          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1412          !!!next-input-character;          !!!next-input-character;
1413          redo A;          redo A;
1414        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1415          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1416            !!!cp (35);            !!!cp (35);
1417            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1418          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1419            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1420            #if ($self->{current_token}->{attributes}) {            #if ($self->{ct}->{attributes}) {
1421            #  ## NOTE: This should never be reached.            #  ## NOTE: This should never be reached.
1422            #  !!! cp (36);            #  !!! cp (36);
1423            #  !!! parse-error (type => 'end tag attribute');            #  !!! parse-error (type => 'end tag attribute');
# Line 1201  sub _get_next_token ($) { Line 1425  sub _get_next_token ($) {
1425              !!!cp (37);              !!!cp (37);
1426            #}            #}
1427          } else {          } else {
1428            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1429          }          }
1430          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1431          !!!next-input-character;          !!!next-input-character;
1432    
1433          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1434    
1435          redo A;          redo A;
1436        } elsif (0x0041 <= $self->{next_char} and        } elsif (0x0041 <= $self->{nc} and
1437                 $self->{next_char} <= 0x005A) { # A..Z                 $self->{nc} <= 0x005A) { # A..Z
1438          !!!cp (38);          !!!cp (38);
1439          $self->{current_token}->{tag_name} .= chr ($self->{next_char} + 0x0020);          $self->{ct}->{tag_name} .= chr ($self->{nc} + 0x0020);
1440            # start tag or end tag            # start tag or end tag
1441          ## Stay in this state          ## Stay in this state
1442          !!!next-input-character;          !!!next-input-character;
1443          redo A;          redo A;
1444        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1445          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1446          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1447            !!!cp (39);            !!!cp (39);
1448            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1449          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1450            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1451            #if ($self->{current_token}->{attributes}) {            #if ($self->{ct}->{attributes}) {
1452            #  ## NOTE: This state should never be reached.            #  ## NOTE: This state should never be reached.
1453            #  !!! cp (40);            #  !!! cp (40);
1454            #  !!! parse-error (type => 'end tag attribute');            #  !!! parse-error (type => 'end tag attribute');
# Line 1232  sub _get_next_token ($) { Line 1456  sub _get_next_token ($) {
1456              !!!cp (41);              !!!cp (41);
1457            #}            #}
1458          } else {          } else {
1459            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1460          }          }
1461          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1462          # reconsume          # reconsume
1463    
1464          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1465    
1466          redo A;          redo A;
1467        } elsif ($self->{next_char} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
1468          !!!cp (42);          !!!cp (42);
1469          $self->{state} = SELF_CLOSING_START_TAG_STATE;          $self->{state} = SELF_CLOSING_START_TAG_STATE;
1470          !!!next-input-character;          !!!next-input-character;
1471          redo A;          redo A;
1472        } else {        } else {
1473          !!!cp (44);          !!!cp (44);
1474          $self->{current_token}->{tag_name} .= chr $self->{next_char};          $self->{ct}->{tag_name} .= chr $self->{nc};
1475            # start tag or end tag            # start tag or end tag
1476          ## Stay in the state          ## Stay in the state
1477          !!!next-input-character;          !!!next-input-character;
1478          redo A;          redo A;
1479        }        }
1480      } elsif ($self->{state} == BEFORE_ATTRIBUTE_NAME_STATE) {      } elsif ($self->{state} == BEFORE_ATTRIBUTE_NAME_STATE) {
1481        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
1482          !!!cp (45);          !!!cp (45);
1483          ## Stay in the state          ## Stay in the state
1484          !!!next-input-character;          !!!next-input-character;
1485          redo A;          redo A;
1486        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1487          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1488            !!!cp (46);            !!!cp (46);
1489            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1490          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1491            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1492            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1493              !!!cp (47);              !!!cp (47);
1494              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1495            } else {            } else {
1496              !!!cp (48);              !!!cp (48);
1497            }            }
1498          } else {          } else {
1499            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1500          }          }
1501          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1502          !!!next-input-character;          !!!next-input-character;
1503    
1504          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1505    
1506          redo A;          redo A;
1507        } elsif (0x0041 <= $self->{next_char} and        } elsif (0x0041 <= $self->{nc} and
1508                 $self->{next_char} <= 0x005A) { # A..Z                 $self->{nc} <= 0x005A) { # A..Z
1509          !!!cp (49);          !!!cp (49);
1510          $self->{current_attribute}          $self->{ca}
1511              = {name => chr ($self->{next_char} + 0x0020),              = {name => chr ($self->{nc} + 0x0020),
1512                 value => '',                 value => '',
1513                 line => $self->{line}, column => $self->{column}};                 line => $self->{line}, column => $self->{column}};
1514          $self->{state} = ATTRIBUTE_NAME_STATE;          $self->{state} = ATTRIBUTE_NAME_STATE;
1515          !!!next-input-character;          !!!next-input-character;
1516          redo A;          redo A;
1517        } elsif ($self->{next_char} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
1518          !!!cp (50);          !!!cp (50);
1519          $self->{state} = SELF_CLOSING_START_TAG_STATE;          $self->{state} = SELF_CLOSING_START_TAG_STATE;
1520          !!!next-input-character;          !!!next-input-character;
1521          redo A;          redo A;
1522        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1523          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1524          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1525            !!!cp (52);            !!!cp (52);
1526            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1527          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1528            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1529            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1530              !!!cp (53);              !!!cp (53);
1531              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1532            } else {            } else {
1533              !!!cp (54);              !!!cp (54);
1534            }            }
1535          } else {          } else {
1536            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1537          }          }
1538          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1539          # reconsume          # reconsume
1540    
1541          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1542    
1543          redo A;          redo A;
1544        } else {        } else {
# Line 1326  sub _get_next_token ($) { Line 1546  sub _get_next_token ($) {
1546               0x0022 => 1, # "               0x0022 => 1, # "
1547               0x0027 => 1, # '               0x0027 => 1, # '
1548               0x003D => 1, # =               0x003D => 1, # =
1549              }->{$self->{next_char}}) {              }->{$self->{nc}}) {
1550            !!!cp (55);            !!!cp (55);
1551            !!!parse-error (type => 'bad attribute name');            !!!parse-error (type => 'bad attribute name');
1552          } else {          } else {
1553            !!!cp (56);            !!!cp (56);
1554          }          }
1555          $self->{current_attribute}          $self->{ca}
1556              = {name => chr ($self->{next_char}),              = {name => chr ($self->{nc}),
1557                 value => '',                 value => '',
1558                 line => $self->{line}, column => $self->{column}};                 line => $self->{line}, column => $self->{column}};
1559          $self->{state} = ATTRIBUTE_NAME_STATE;          $self->{state} = ATTRIBUTE_NAME_STATE;
# Line 1342  sub _get_next_token ($) { Line 1562  sub _get_next_token ($) {
1562        }        }
1563      } elsif ($self->{state} == ATTRIBUTE_NAME_STATE) {      } elsif ($self->{state} == ATTRIBUTE_NAME_STATE) {
1564        my $before_leave = sub {        my $before_leave = sub {
1565          if (exists $self->{current_token}->{attributes} # start tag or end tag          if (exists $self->{ct}->{attributes} # start tag or end tag
1566              ->{$self->{current_attribute}->{name}}) { # MUST              ->{$self->{ca}->{name}}) { # MUST
1567            !!!cp (57);            !!!cp (57);
1568            !!!parse-error (type => 'duplicate attribute:'.$self->{current_attribute}->{name}, line => $self->{current_attribute}->{line}, column => $self->{current_attribute}->{column});            !!!parse-error (type => 'duplicate attribute', text => $self->{ca}->{name}, line => $self->{ca}->{line}, column => $self->{ca}->{column});
1569            ## Discard $self->{current_attribute} # MUST            ## Discard $self->{ca} # MUST
1570          } else {          } else {
1571            !!!cp (58);            !!!cp (58);
1572            $self->{current_token}->{attributes}->{$self->{current_attribute}->{name}}            $self->{ct}->{attributes}->{$self->{ca}->{name}}
1573              = $self->{current_attribute};              = $self->{ca};
1574          }          }
1575        }; # $before_leave        }; # $before_leave
1576    
1577        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
1578          !!!cp (59);          !!!cp (59);
1579          $before_leave->();          $before_leave->();
1580          $self->{state} = AFTER_ATTRIBUTE_NAME_STATE;          $self->{state} = AFTER_ATTRIBUTE_NAME_STATE;
1581          !!!next-input-character;          !!!next-input-character;
1582          redo A;          redo A;
1583        } elsif ($self->{next_char} == 0x003D) { # =        } elsif ($self->{nc} == 0x003D) { # =
1584          !!!cp (60);          !!!cp (60);
1585          $before_leave->();          $before_leave->();
1586          $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;          $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;
1587          !!!next-input-character;          !!!next-input-character;
1588          redo A;          redo A;
1589        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1590          $before_leave->();          $before_leave->();
1591          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1592            !!!cp (61);            !!!cp (61);
1593            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1594          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1595            !!!cp (62);            !!!cp (62);
1596            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1597            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1598              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1599            }            }
1600          } else {          } else {
1601            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1602          }          }
1603          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1604          !!!next-input-character;          !!!next-input-character;
1605    
1606          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1607    
1608          redo A;          redo A;
1609        } elsif (0x0041 <= $self->{next_char} and        } elsif (0x0041 <= $self->{nc} and
1610                 $self->{next_char} <= 0x005A) { # A..Z                 $self->{nc} <= 0x005A) { # A..Z
1611          !!!cp (63);          !!!cp (63);
1612          $self->{current_attribute}->{name} .= chr ($self->{next_char} + 0x0020);          $self->{ca}->{name} .= chr ($self->{nc} + 0x0020);
1613          ## Stay in the state          ## Stay in the state
1614          !!!next-input-character;          !!!next-input-character;
1615          redo A;          redo A;
1616        } elsif ($self->{next_char} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
1617          !!!cp (64);          !!!cp (64);
1618          $before_leave->();          $before_leave->();
1619          $self->{state} = SELF_CLOSING_START_TAG_STATE;          $self->{state} = SELF_CLOSING_START_TAG_STATE;
1620          !!!next-input-character;          !!!next-input-character;
1621          redo A;          redo A;
1622        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1623          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1624          $before_leave->();          $before_leave->();
1625          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1626            !!!cp (66);            !!!cp (66);
1627            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1628          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1629            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1630            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1631              !!!cp (67);              !!!cp (67);
1632              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1633            } else {            } else {
# Line 1419  sub _get_next_token ($) { Line 1635  sub _get_next_token ($) {
1635              !!!cp (68);              !!!cp (68);
1636            }            }
1637          } else {          } else {
1638            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1639          }          }
1640          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1641          # reconsume          # reconsume
1642    
1643          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1644    
1645          redo A;          redo A;
1646        } else {        } else {
1647          if ($self->{next_char} == 0x0022 or # "          if ($self->{nc} == 0x0022 or # "
1648              $self->{next_char} == 0x0027) { # '              $self->{nc} == 0x0027) { # '
1649            !!!cp (69);            !!!cp (69);
1650            !!!parse-error (type => 'bad attribute name');            !!!parse-error (type => 'bad attribute name');
1651          } else {          } else {
1652            !!!cp (70);            !!!cp (70);
1653          }          }
1654          $self->{current_attribute}->{name} .= chr ($self->{next_char});          $self->{ca}->{name} .= chr ($self->{nc});
1655          ## Stay in the state          ## Stay in the state
1656          !!!next-input-character;          !!!next-input-character;
1657          redo A;          redo A;
1658        }        }
1659      } elsif ($self->{state} == AFTER_ATTRIBUTE_NAME_STATE) {      } elsif ($self->{state} == AFTER_ATTRIBUTE_NAME_STATE) {
1660        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
1661          !!!cp (71);          !!!cp (71);
1662          ## Stay in the state          ## Stay in the state
1663          !!!next-input-character;          !!!next-input-character;
1664          redo A;          redo A;
1665        } elsif ($self->{next_char} == 0x003D) { # =        } elsif ($self->{nc} == 0x003D) { # =
1666          !!!cp (72);          !!!cp (72);
1667          $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;          $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;
1668          !!!next-input-character;          !!!next-input-character;
1669          redo A;          redo A;
1670        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1671          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1672            !!!cp (73);            !!!cp (73);
1673            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1674          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1675            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1676            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1677              !!!cp (74);              !!!cp (74);
1678              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1679            } else {            } else {
# Line 1469  sub _get_next_token ($) { Line 1681  sub _get_next_token ($) {
1681              !!!cp (75);              !!!cp (75);
1682            }            }
1683          } else {          } else {
1684            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1685          }          }
1686          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1687          !!!next-input-character;          !!!next-input-character;
1688    
1689          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1690    
1691          redo A;          redo A;
1692        } elsif (0x0041 <= $self->{next_char} and        } elsif (0x0041 <= $self->{nc} and
1693                 $self->{next_char} <= 0x005A) { # A..Z                 $self->{nc} <= 0x005A) { # A..Z
1694          !!!cp (76);          !!!cp (76);
1695          $self->{current_attribute}          $self->{ca}
1696              = {name => chr ($self->{next_char} + 0x0020),              = {name => chr ($self->{nc} + 0x0020),
1697                 value => '',                 value => '',
1698                 line => $self->{line}, column => $self->{column}};                 line => $self->{line}, column => $self->{column}};
1699          $self->{state} = ATTRIBUTE_NAME_STATE;          $self->{state} = ATTRIBUTE_NAME_STATE;
1700          !!!next-input-character;          !!!next-input-character;
1701          redo A;          redo A;
1702        } elsif ($self->{next_char} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
1703          !!!cp (77);          !!!cp (77);
1704          $self->{state} = SELF_CLOSING_START_TAG_STATE;          $self->{state} = SELF_CLOSING_START_TAG_STATE;
1705          !!!next-input-character;          !!!next-input-character;
1706          redo A;          redo A;
1707        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1708          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1709          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1710            !!!cp (79);            !!!cp (79);
1711            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1712          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1713            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1714            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1715              !!!cp (80);              !!!cp (80);
1716              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1717            } else {            } else {
# Line 1507  sub _get_next_token ($) { Line 1719  sub _get_next_token ($) {
1719              !!!cp (81);              !!!cp (81);
1720            }            }
1721          } else {          } else {
1722            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1723          }          }
1724          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1725          # reconsume          # reconsume
1726    
1727          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1728    
1729          redo A;          redo A;
1730        } else {        } else {
1731          !!!cp (82);          if ($self->{nc} == 0x0022 or # "
1732          $self->{current_attribute}              $self->{nc} == 0x0027) { # '
1733              = {name => chr ($self->{next_char}),            !!!cp (78);
1734              !!!parse-error (type => 'bad attribute name');
1735            } else {
1736              !!!cp (82);
1737            }
1738            $self->{ca}
1739                = {name => chr ($self->{nc}),
1740                 value => '',                 value => '',
1741                 line => $self->{line}, column => $self->{column}};                 line => $self->{line}, column => $self->{column}};
1742          $self->{state} = ATTRIBUTE_NAME_STATE;          $self->{state} = ATTRIBUTE_NAME_STATE;
# Line 1526  sub _get_next_token ($) { Line 1744  sub _get_next_token ($) {
1744          redo A;                  redo A;        
1745        }        }
1746      } elsif ($self->{state} == BEFORE_ATTRIBUTE_VALUE_STATE) {      } elsif ($self->{state} == BEFORE_ATTRIBUTE_VALUE_STATE) {
1747        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP        
1748          !!!cp (83);          !!!cp (83);
1749          ## Stay in the state          ## Stay in the state
1750          !!!next-input-character;          !!!next-input-character;
1751          redo A;          redo A;
1752        } elsif ($self->{next_char} == 0x0022) { # "        } elsif ($self->{nc} == 0x0022) { # "
1753          !!!cp (84);          !!!cp (84);
1754          $self->{state} = ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE;          $self->{state} = ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE;
1755          !!!next-input-character;          !!!next-input-character;
1756          redo A;          redo A;
1757        } elsif ($self->{next_char} == 0x0026) { # &        } elsif ($self->{nc} == 0x0026) { # &
1758          !!!cp (85);          !!!cp (85);
1759          $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;          $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;
1760          ## reconsume          ## reconsume
1761          redo A;          redo A;
1762        } elsif ($self->{next_char} == 0x0027) { # '        } elsif ($self->{nc} == 0x0027) { # '
1763          !!!cp (86);          !!!cp (86);
1764          $self->{state} = ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE;          $self->{state} = ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE;
1765          !!!next-input-character;          !!!next-input-character;
1766          redo A;          redo A;
1767        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1768          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          !!!parse-error (type => 'empty unquoted attribute value');
1769            if ($self->{ct}->{type} == START_TAG_TOKEN) {
1770            !!!cp (87);            !!!cp (87);
1771            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1772          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1773            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1774            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1775              !!!cp (88);              !!!cp (88);
1776              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1777            } else {            } else {
# Line 1564  sub _get_next_token ($) { Line 1779  sub _get_next_token ($) {
1779              !!!cp (89);              !!!cp (89);
1780            }            }
1781          } else {          } else {
1782            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1783          }          }
1784          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1785          !!!next-input-character;          !!!next-input-character;
1786    
1787          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1788    
1789          redo A;          redo A;
1790        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1791          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1792          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1793            !!!cp (90);            !!!cp (90);
1794            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1795          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1796            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1797            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1798              !!!cp (91);              !!!cp (91);
1799              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1800            } else {            } else {
# Line 1587  sub _get_next_token ($) { Line 1802  sub _get_next_token ($) {
1802              !!!cp (92);              !!!cp (92);
1803            }            }
1804          } else {          } else {
1805            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1806          }          }
1807          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1808          ## reconsume          ## reconsume
1809    
1810          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1811    
1812          redo A;          redo A;
1813        } else {        } else {
1814          if ($self->{next_char} == 0x003D) { # =          if ($self->{nc} == 0x003D) { # =
1815            !!!cp (93);            !!!cp (93);
1816            !!!parse-error (type => 'bad attribute value');            !!!parse-error (type => 'bad attribute value');
1817          } else {          } else {
1818            !!!cp (94);            !!!cp (94);
1819          }          }
1820          $self->{current_attribute}->{value} .= chr ($self->{next_char});          $self->{ca}->{value} .= chr ($self->{nc});
1821          $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;          $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;
1822          !!!next-input-character;          !!!next-input-character;
1823          redo A;          redo A;
1824        }        }
1825      } elsif ($self->{state} == ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE) {      } elsif ($self->{state} == ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE) {
1826        if ($self->{next_char} == 0x0022) { # "        if ($self->{nc} == 0x0022) { # "
1827          !!!cp (95);          !!!cp (95);
1828          $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;          $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;
1829          !!!next-input-character;          !!!next-input-character;
1830          redo A;          redo A;
1831        } elsif ($self->{next_char} == 0x0026) { # &        } elsif ($self->{nc} == 0x0026) { # &
1832          !!!cp (96);          !!!cp (96);
1833          $self->{last_attribute_value_state} = $self->{state};          ## NOTE: In the spec, the tokenizer is switched to the
1834          $self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE;          ## "entity in attribute value state".  In this implementation, the
1835            ## tokenizer is switched to the |ENTITY_STATE|, which is an
1836            ## implementation of the "consume a character reference" algorithm.
1837            $self->{prev_state} = $self->{state};
1838            $self->{entity_add} = 0x0022; # "
1839            $self->{state} = ENTITY_STATE;
1840          !!!next-input-character;          !!!next-input-character;
1841          redo A;          redo A;
1842        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1843          !!!parse-error (type => 'unclosed attribute value');          !!!parse-error (type => 'unclosed attribute value');
1844          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1845            !!!cp (97);            !!!cp (97);
1846            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1847          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1848            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1849            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1850              !!!cp (98);              !!!cp (98);
1851              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1852            } else {            } else {
# Line 1634  sub _get_next_token ($) { Line 1854  sub _get_next_token ($) {
1854              !!!cp (99);              !!!cp (99);
1855            }            }
1856          } else {          } else {
1857            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1858          }          }
1859          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1860          ## reconsume          ## reconsume
1861    
1862          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1863    
1864          redo A;          redo A;
1865        } else {        } else {
1866          !!!cp (100);          !!!cp (100);
1867          $self->{current_attribute}->{value} .= chr ($self->{next_char});          $self->{ca}->{value} .= chr ($self->{nc});
1868            $self->{read_until}->($self->{ca}->{value},
1869                                  q["&],
1870                                  length $self->{ca}->{value});
1871    
1872          ## Stay in the state          ## Stay in the state
1873          !!!next-input-character;          !!!next-input-character;
1874          redo A;          redo A;
1875        }        }
1876      } elsif ($self->{state} == ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE) {      } elsif ($self->{state} == ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE) {
1877        if ($self->{next_char} == 0x0027) { # '        if ($self->{nc} == 0x0027) { # '
1878          !!!cp (101);          !!!cp (101);
1879          $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;          $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;
1880          !!!next-input-character;          !!!next-input-character;
1881          redo A;          redo A;
1882        } elsif ($self->{next_char} == 0x0026) { # &        } elsif ($self->{nc} == 0x0026) { # &
1883          !!!cp (102);          !!!cp (102);
1884          $self->{last_attribute_value_state} = $self->{state};          ## NOTE: In the spec, the tokenizer is switched to the
1885          $self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE;          ## "entity in attribute value state".  In this implementation, the
1886            ## tokenizer is switched to the |ENTITY_STATE|, which is an
1887            ## implementation of the "consume a character reference" algorithm.
1888            $self->{entity_add} = 0x0027; # '
1889            $self->{prev_state} = $self->{state};
1890            $self->{state} = ENTITY_STATE;
1891          !!!next-input-character;          !!!next-input-character;
1892          redo A;          redo A;
1893        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1894          !!!parse-error (type => 'unclosed attribute value');          !!!parse-error (type => 'unclosed attribute value');
1895          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1896            !!!cp (103);            !!!cp (103);
1897            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1898          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1899            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1900            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1901              !!!cp (104);              !!!cp (104);
1902              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1903            } else {            } else {
# Line 1676  sub _get_next_token ($) { Line 1905  sub _get_next_token ($) {
1905              !!!cp (105);              !!!cp (105);
1906            }            }
1907          } else {          } else {
1908            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1909          }          }
1910          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1911          ## reconsume          ## reconsume
1912    
1913          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1914    
1915          redo A;          redo A;
1916        } else {        } else {
1917          !!!cp (106);          !!!cp (106);
1918          $self->{current_attribute}->{value} .= chr ($self->{next_char});          $self->{ca}->{value} .= chr ($self->{nc});
1919            $self->{read_until}->($self->{ca}->{value},
1920                                  q['&],
1921                                  length $self->{ca}->{value});
1922    
1923          ## Stay in the state          ## Stay in the state
1924          !!!next-input-character;          !!!next-input-character;
1925          redo A;          redo A;
1926        }        }
1927      } elsif ($self->{state} == ATTRIBUTE_VALUE_UNQUOTED_STATE) {      } elsif ($self->{state} == ATTRIBUTE_VALUE_UNQUOTED_STATE) {
1928        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # HT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
1929          !!!cp (107);          !!!cp (107);
1930          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1931          !!!next-input-character;          !!!next-input-character;
1932          redo A;          redo A;
1933        } elsif ($self->{next_char} == 0x0026) { # &        } elsif ($self->{nc} == 0x0026) { # &
1934          !!!cp (108);          !!!cp (108);
1935          $self->{last_attribute_value_state} = $self->{state};          ## NOTE: In the spec, the tokenizer is switched to the
1936          $self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE;          ## "entity in attribute value state".  In this implementation, the
1937            ## tokenizer is switched to the |ENTITY_STATE|, which is an
1938            ## implementation of the "consume a character reference" algorithm.
1939            $self->{entity_add} = -1;
1940            $self->{prev_state} = $self->{state};
1941            $self->{state} = ENTITY_STATE;
1942          !!!next-input-character;          !!!next-input-character;
1943          redo A;          redo A;
1944        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1945          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1946            !!!cp (109);            !!!cp (109);
1947            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1948          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1949            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1950            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1951              !!!cp (110);              !!!cp (110);
1952              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1953            } else {            } else {
# Line 1721  sub _get_next_token ($) { Line 1955  sub _get_next_token ($) {
1955              !!!cp (111);              !!!cp (111);
1956            }            }
1957          } else {          } else {
1958            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1959          }          }
1960          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1961          !!!next-input-character;          !!!next-input-character;
1962    
1963          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1964    
1965          redo A;          redo A;
1966        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1967          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1968          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1969            !!!cp (112);            !!!cp (112);
1970            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1971          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1972            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1973            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1974              !!!cp (113);              !!!cp (113);
1975              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1976            } else {            } else {
# Line 1744  sub _get_next_token ($) { Line 1978  sub _get_next_token ($) {
1978              !!!cp (114);              !!!cp (114);
1979            }            }
1980          } else {          } else {
1981            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1982          }          }
1983          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1984          ## reconsume          ## reconsume
1985    
1986          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1987    
1988          redo A;          redo A;
1989        } else {        } else {
# Line 1757  sub _get_next_token ($) { Line 1991  sub _get_next_token ($) {
1991               0x0022 => 1, # "               0x0022 => 1, # "
1992               0x0027 => 1, # '               0x0027 => 1, # '
1993               0x003D => 1, # =               0x003D => 1, # =
1994              }->{$self->{next_char}}) {              }->{$self->{nc}}) {
1995            !!!cp (115);            !!!cp (115);
1996            !!!parse-error (type => 'bad attribute value');            !!!parse-error (type => 'bad attribute value');
1997          } else {          } else {
1998            !!!cp (116);            !!!cp (116);
1999          }          }
2000          $self->{current_attribute}->{value} .= chr ($self->{next_char});          $self->{ca}->{value} .= chr ($self->{nc});
2001            $self->{read_until}->($self->{ca}->{value},
2002                                  q["'=& >],
2003                                  length $self->{ca}->{value});
2004    
2005          ## Stay in the state          ## Stay in the state
2006          !!!next-input-character;          !!!next-input-character;
2007          redo A;          redo A;
2008        }        }
     } elsif ($self->{state} == ENTITY_IN_ATTRIBUTE_VALUE_STATE) {  
       my $token = $self->_tokenize_attempt_to_consume_an_entity  
           (1,  
            $self->{last_attribute_value_state}  
              == ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE ? 0x0022 : # "  
            $self->{last_attribute_value_state}  
              == ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE ? 0x0027 : # '  
            -1);  
   
       unless (defined $token) {  
         !!!cp (117);  
         $self->{current_attribute}->{value} .= '&';  
       } else {  
         !!!cp (118);  
         $self->{current_attribute}->{value} .= $token->{data};  
         $self->{current_attribute}->{has_reference} = $token->{has_reference};  
         ## ISSUE: spec says "append the returned character token to the current attribute's value"  
       }  
   
       $self->{state} = $self->{last_attribute_value_state};  
       # next-input-character is already done  
       redo A;  
2009      } elsif ($self->{state} == AFTER_ATTRIBUTE_VALUE_QUOTED_STATE) {      } elsif ($self->{state} == AFTER_ATTRIBUTE_VALUE_QUOTED_STATE) {
2010        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
2011          !!!cp (118);          !!!cp (118);
2012          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
2013          !!!next-input-character;          !!!next-input-character;
2014          redo A;          redo A;
2015        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2016          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
2017            !!!cp (119);            !!!cp (119);
2018            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
2019          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
2020            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
2021            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
2022              !!!cp (120);              !!!cp (120);
2023              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
2024            } else {            } else {
# Line 1814  sub _get_next_token ($) { Line 2026  sub _get_next_token ($) {
2026              !!!cp (121);              !!!cp (121);
2027            }            }
2028          } else {          } else {
2029            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
2030          }          }
2031          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2032          !!!next-input-character;          !!!next-input-character;
2033    
2034          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
2035    
2036          redo A;          redo A;
2037        } elsif ($self->{next_char} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
2038          !!!cp (122);          !!!cp (122);
2039          $self->{state} = SELF_CLOSING_START_TAG_STATE;          $self->{state} = SELF_CLOSING_START_TAG_STATE;
2040          !!!next-input-character;          !!!next-input-character;
2041          redo A;          redo A;
2042        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2043          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
2044          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
2045            !!!cp (122.3);            !!!cp (122.3);
2046            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
2047          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
2048            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
2049              !!!cp (122.1);              !!!cp (122.1);
2050              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
2051            } else {            } else {
# Line 1841  sub _get_next_token ($) { Line 2053  sub _get_next_token ($) {
2053              !!!cp (122.2);              !!!cp (122.2);
2054            }            }
2055          } else {          } else {
2056            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
2057          }          }
2058          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2059          ## Reconsume.          ## Reconsume.
2060          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
2061          redo A;          redo A;
2062        } else {        } else {
2063          !!!cp ('124.1');          !!!cp ('124.1');
# Line 1855  sub _get_next_token ($) { Line 2067  sub _get_next_token ($) {
2067          redo A;          redo A;
2068        }        }
2069      } elsif ($self->{state} == SELF_CLOSING_START_TAG_STATE) {      } elsif ($self->{state} == SELF_CLOSING_START_TAG_STATE) {
2070        if ($self->{next_char} == 0x003E) { # >        if ($self->{nc} == 0x003E) { # >
2071          if ($self->{current_token}->{type} == END_TAG_TOKEN) {          if ($self->{ct}->{type} == END_TAG_TOKEN) {
2072            !!!cp ('124.2');            !!!cp ('124.2');
2073            !!!parse-error (type => 'nestc', token => $self->{current_token});            !!!parse-error (type => 'nestc', token => $self->{ct});
2074            ## TODO: Different type than slash in start tag            ## TODO: Different type than slash in start tag
2075            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
2076            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
2077              !!!cp ('124.4');              !!!cp ('124.4');
2078              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
2079            } else {            } else {
# Line 1876  sub _get_next_token ($) { Line 2088  sub _get_next_token ($) {
2088          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2089          !!!next-input-character;          !!!next-input-character;
2090    
2091          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
2092    
2093          redo A;          redo A;
2094        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2095          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
2096          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
2097            !!!cp (124.7);            !!!cp (124.7);
2098            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
2099          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
2100            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
2101              !!!cp (124.5);              !!!cp (124.5);
2102              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
2103            } else {            } else {
# Line 1893  sub _get_next_token ($) { Line 2105  sub _get_next_token ($) {
2105              !!!cp (124.6);              !!!cp (124.6);
2106            }            }
2107          } else {          } else {
2108            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
2109          }          }
2110          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2111          ## Reconsume.          ## Reconsume.
2112          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
2113          redo A;          redo A;
2114        } else {        } else {
2115          !!!cp ('124.4');          !!!cp ('124.4');
# Line 1909  sub _get_next_token ($) { Line 2121  sub _get_next_token ($) {
2121        }        }
2122      } elsif ($self->{state} == BOGUS_COMMENT_STATE) {      } elsif ($self->{state} == BOGUS_COMMENT_STATE) {
2123        ## (only happen if PCDATA state)        ## (only happen if PCDATA state)
         
       ## NOTE: Set by the previous state  
       #my $token = {type => COMMENT_TOKEN, data => ''};  
   
       BC: {  
         if ($self->{next_char} == 0x003E) { # >  
           !!!cp (124);  
           $self->{state} = DATA_STATE;  
           !!!next-input-character;  
2124    
2125            !!!emit ($self->{current_token}); # comment        ## NOTE: Unlike spec's "bogus comment state", this implementation
2126          ## consumes characters one-by-one basis.
2127            redo A;        
2128          } elsif ($self->{next_char} == -1) {        if ($self->{nc} == 0x003E) { # >
2129            !!!cp (125);          !!!cp (124);
2130            $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2131            ## reconsume          !!!next-input-character;
2132    
2133            !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2134            redo A;
2135          } elsif ($self->{nc} == -1) {
2136            !!!cp (125);
2137            $self->{state} = DATA_STATE;
2138            ## reconsume
2139    
2140            redo A;          !!!emit ($self->{ct}); # comment
2141          } else {          redo A;
2142            !!!cp (126);        } else {
2143            $self->{current_token}->{data} .= chr ($self->{next_char}); # comment          !!!cp (126);
2144            !!!next-input-character;          $self->{ct}->{data} .= chr ($self->{nc}); # comment
2145            redo BC;          $self->{read_until}->($self->{ct}->{data},
2146          }                                q[>],
2147        } # BC                                length $self->{ct}->{data});
2148    
2149        die "$0: _get_next_token: unexpected case [BC]";          ## Stay in the state.
2150            !!!next-input-character;
2151            redo A;
2152          }
2153      } elsif ($self->{state} == MARKUP_DECLARATION_OPEN_STATE) {      } elsif ($self->{state} == MARKUP_DECLARATION_OPEN_STATE) {
2154        ## (only happen if PCDATA state)        ## (only happen if PCDATA state)
   
       my ($l, $c) = ($self->{line_prev}, $self->{column_prev} - 1);  
   
       my @next_char;  
       push @next_char, $self->{next_char};  
2155                
2156        if ($self->{next_char} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2157            !!!cp (133);
2158            $self->{state} = MD_HYPHEN_STATE;
2159          !!!next-input-character;          !!!next-input-character;
2160          push @next_char, $self->{next_char};          redo A;
2161          if ($self->{next_char} == 0x002D) { # -        } elsif ($self->{nc} == 0x0044 or # D
2162            !!!cp (127);                 $self->{nc} == 0x0064) { # d
2163            $self->{current_token} = {type => COMMENT_TOKEN, data => '',          ## ASCII case-insensitive.
2164                                      line => $l, column => $c,          !!!cp (130);
2165                                     };          $self->{state} = MD_DOCTYPE_STATE;
2166            $self->{state} = COMMENT_START_STATE;          $self->{s_kwd} = chr $self->{nc};
           !!!next-input-character;  
           redo A;  
         } else {  
           !!!cp (128);  
         }  
       } elsif ($self->{next_char} == 0x0044 or # D  
                $self->{next_char} == 0x0064) { # d  
2167          !!!next-input-character;          !!!next-input-character;
2168          push @next_char, $self->{next_char};          redo A;
         if ($self->{next_char} == 0x004F or # O  
             $self->{next_char} == 0x006F) { # o  
           !!!next-input-character;  
           push @next_char, $self->{next_char};  
           if ($self->{next_char} == 0x0043 or # C  
               $self->{next_char} == 0x0063) { # c  
             !!!next-input-character;  
             push @next_char, $self->{next_char};  
             if ($self->{next_char} == 0x0054 or # T  
                 $self->{next_char} == 0x0074) { # t  
               !!!next-input-character;  
               push @next_char, $self->{next_char};  
               if ($self->{next_char} == 0x0059 or # Y  
                   $self->{next_char} == 0x0079) { # y  
                 !!!next-input-character;  
                 push @next_char, $self->{next_char};  
                 if ($self->{next_char} == 0x0050 or # P  
                     $self->{next_char} == 0x0070) { # p  
                   !!!next-input-character;  
                   push @next_char, $self->{next_char};  
                   if ($self->{next_char} == 0x0045 or # E  
                       $self->{next_char} == 0x0065) { # e  
                     !!!cp (129);  
                     ## TODO: What a stupid code this is!  
                     $self->{state} = DOCTYPE_STATE;  
                     $self->{current_token} = {type => DOCTYPE_TOKEN,  
                                               quirks => 1,  
                                               line => $l, column => $c,  
                                              };  
                     !!!next-input-character;  
                     redo A;  
                   } else {  
                     !!!cp (130);  
                   }  
                 } else {  
                   !!!cp (131);  
                 }  
               } else {  
                 !!!cp (132);  
               }  
             } else {  
               !!!cp (133);  
             }  
           } else {  
             !!!cp (134);  
           }  
         } else {  
           !!!cp (135);  
         }  
2169        } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM and        } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM and
2170                 $self->{open_elements}->[-1]->[1] & FOREIGN_EL and                 $self->{open_elements}->[-1]->[1] & FOREIGN_EL and
2171                 $self->{next_char} == 0x005B) { # [                 $self->{nc} == 0x005B) { # [
2172            !!!cp (135.4);                
2173            $self->{state} = MD_CDATA_STATE;
2174            $self->{s_kwd} = '[';
2175          !!!next-input-character;          !!!next-input-character;
2176          push @next_char, $self->{next_char};          redo A;
         if ($self->{next_char} == 0x0043) { # C  
           !!!next-input-character;  
           push @next_char, $self->{next_char};  
           if ($self->{next_char} == 0x0044) { # D  
             !!!next-input-character;  
             push @next_char, $self->{next_char};  
             if ($self->{next_char} == 0x0041) { # A  
               !!!next-input-character;  
               push @next_char, $self->{next_char};  
               if ($self->{next_char} == 0x0054) { # T  
                 !!!next-input-character;  
                 push @next_char, $self->{next_char};  
                 if ($self->{next_char} == 0x0041) { # A  
                   !!!next-input-character;  
                   push @next_char, $self->{next_char};  
                   if ($self->{next_char} == 0x005B) { # [  
                     !!!cp (135.1);  
                     $self->{state} = CDATA_BLOCK_STATE;  
                     !!!next-input-character;  
                     redo A;  
                   } else {  
                     !!!cp (135.2);  
                   }  
                 } else {  
                   !!!cp (135.3);  
                 }  
               } else {  
                 !!!cp (135.4);                  
               }  
             } else {  
               !!!cp (135.5);  
             }  
           } else {  
             !!!cp (135.6);  
           }  
         } else {  
           !!!cp (135.7);  
         }  
2177        } else {        } else {
2178          !!!cp (136);          !!!cp (136);
2179        }        }
2180    
2181        !!!parse-error (type => 'bogus comment');        !!!parse-error (type => 'bogus comment',
2182        $self->{next_char} = shift @next_char;                        line => $self->{line_prev},
2183        !!!back-next-input-character (@next_char);                        column => $self->{column_prev} - 1);
2184          ## Reconsume.
2185        $self->{state} = BOGUS_COMMENT_STATE;        $self->{state} = BOGUS_COMMENT_STATE;
2186        $self->{current_token} = {type => COMMENT_TOKEN, data => '',        $self->{ct} = {type => COMMENT_TOKEN, data => '',
2187                                  line => $l, column => $c,                                  line => $self->{line_prev},
2188                                    column => $self->{column_prev} - 1,
2189                                 };                                 };
2190        redo A;        redo A;
2191              } elsif ($self->{state} == MD_HYPHEN_STATE) {
2192        ## ISSUE: typos in spec: chacacters, is is a parse error        if ($self->{nc} == 0x002D) { # -
2193        ## ISSUE: spec is somewhat unclear on "is the first character that will be in the comment"; what is "that will be in the comment" is what the algorithm defines, isn't it?          !!!cp (127);
2194            $self->{ct} = {type => COMMENT_TOKEN, data => '',
2195                                      line => $self->{line_prev},
2196                                      column => $self->{column_prev} - 2,
2197                                     };
2198            $self->{state} = COMMENT_START_STATE;
2199            !!!next-input-character;
2200            redo A;
2201          } else {
2202            !!!cp (128);
2203            !!!parse-error (type => 'bogus comment',
2204                            line => $self->{line_prev},
2205                            column => $self->{column_prev} - 2);
2206            $self->{state} = BOGUS_COMMENT_STATE;
2207            ## Reconsume.
2208            $self->{ct} = {type => COMMENT_TOKEN,
2209                                      data => '-',
2210                                      line => $self->{line_prev},
2211                                      column => $self->{column_prev} - 2,
2212                                     };
2213            redo A;
2214          }
2215        } elsif ($self->{state} == MD_DOCTYPE_STATE) {
2216          ## ASCII case-insensitive.
2217          if ($self->{nc} == [
2218                undef,
2219                0x004F, # O
2220                0x0043, # C
2221                0x0054, # T
2222                0x0059, # Y
2223                0x0050, # P
2224              ]->[length $self->{s_kwd}] or
2225              $self->{nc} == [
2226                undef,
2227                0x006F, # o
2228                0x0063, # c
2229                0x0074, # t
2230                0x0079, # y
2231                0x0070, # p
2232              ]->[length $self->{s_kwd}]) {
2233            !!!cp (131);
2234            ## Stay in the state.
2235            $self->{s_kwd} .= chr $self->{nc};
2236            !!!next-input-character;
2237            redo A;
2238          } elsif ((length $self->{s_kwd}) == 6 and
2239                   ($self->{nc} == 0x0045 or # E
2240                    $self->{nc} == 0x0065)) { # e
2241            !!!cp (129);
2242            $self->{state} = DOCTYPE_STATE;
2243            $self->{ct} = {type => DOCTYPE_TOKEN,
2244                                      quirks => 1,
2245                                      line => $self->{line_prev},
2246                                      column => $self->{column_prev} - 7,
2247                                     };
2248            !!!next-input-character;
2249            redo A;
2250          } else {
2251            !!!cp (132);        
2252            !!!parse-error (type => 'bogus comment',
2253                            line => $self->{line_prev},
2254                            column => $self->{column_prev} - 1 - length $self->{s_kwd});
2255            $self->{state} = BOGUS_COMMENT_STATE;
2256            ## Reconsume.
2257            $self->{ct} = {type => COMMENT_TOKEN,
2258                                      data => $self->{s_kwd},
2259                                      line => $self->{line_prev},
2260                                      column => $self->{column_prev} - 1 - length $self->{s_kwd},
2261                                     };
2262            redo A;
2263          }
2264        } elsif ($self->{state} == MD_CDATA_STATE) {
2265          if ($self->{nc} == {
2266                '[' => 0x0043, # C
2267                '[C' => 0x0044, # D
2268                '[CD' => 0x0041, # A
2269                '[CDA' => 0x0054, # T
2270                '[CDAT' => 0x0041, # A
2271              }->{$self->{s_kwd}}) {
2272            !!!cp (135.1);
2273            ## Stay in the state.
2274            $self->{s_kwd} .= chr $self->{nc};
2275            !!!next-input-character;
2276            redo A;
2277          } elsif ($self->{s_kwd} eq '[CDATA' and
2278                   $self->{nc} == 0x005B) { # [
2279            !!!cp (135.2);
2280            $self->{ct} = {type => CHARACTER_TOKEN,
2281                                      data => '',
2282                                      line => $self->{line_prev},
2283                                      column => $self->{column_prev} - 7};
2284            $self->{state} = CDATA_SECTION_STATE;
2285            !!!next-input-character;
2286            redo A;
2287          } else {
2288            !!!cp (135.3);
2289            !!!parse-error (type => 'bogus comment',
2290                            line => $self->{line_prev},
2291                            column => $self->{column_prev} - 1 - length $self->{s_kwd});
2292            $self->{state} = BOGUS_COMMENT_STATE;
2293            ## Reconsume.
2294            $self->{ct} = {type => COMMENT_TOKEN,
2295                                      data => $self->{s_kwd},
2296                                      line => $self->{line_prev},
2297                                      column => $self->{column_prev} - 1 - length $self->{s_kwd},
2298                                     };
2299            redo A;
2300          }
2301      } elsif ($self->{state} == COMMENT_START_STATE) {      } elsif ($self->{state} == COMMENT_START_STATE) {
2302        if ($self->{next_char} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2303          !!!cp (137);          !!!cp (137);
2304          $self->{state} = COMMENT_START_DASH_STATE;          $self->{state} = COMMENT_START_DASH_STATE;
2305          !!!next-input-character;          !!!next-input-character;
2306          redo A;          redo A;
2307        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2308          !!!cp (138);          !!!cp (138);
2309          !!!parse-error (type => 'bogus comment');          !!!parse-error (type => 'bogus comment');
2310          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2311          !!!next-input-character;          !!!next-input-character;
2312    
2313          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2314    
2315          redo A;          redo A;
2316        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2317          !!!cp (139);          !!!cp (139);
2318          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2319          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2320          ## reconsume          ## reconsume
2321    
2322          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2323    
2324          redo A;          redo A;
2325        } else {        } else {
2326          !!!cp (140);          !!!cp (140);
2327          $self->{current_token}->{data} # comment          $self->{ct}->{data} # comment
2328              .= chr ($self->{next_char});              .= chr ($self->{nc});
2329          $self->{state} = COMMENT_STATE;          $self->{state} = COMMENT_STATE;
2330          !!!next-input-character;          !!!next-input-character;
2331          redo A;          redo A;
2332        }        }
2333      } elsif ($self->{state} == COMMENT_START_DASH_STATE) {      } elsif ($self->{state} == COMMENT_START_DASH_STATE) {
2334        if ($self->{next_char} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2335          !!!cp (141);          !!!cp (141);
2336          $self->{state} = COMMENT_END_STATE;          $self->{state} = COMMENT_END_STATE;
2337          !!!next-input-character;          !!!next-input-character;
2338          redo A;          redo A;
2339        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2340          !!!cp (142);          !!!cp (142);
2341          !!!parse-error (type => 'bogus comment');          !!!parse-error (type => 'bogus comment');
2342          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2343          !!!next-input-character;          !!!next-input-character;
2344    
2345          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2346    
2347          redo A;          redo A;
2348        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2349          !!!cp (143);          !!!cp (143);
2350          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2351          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2352          ## reconsume          ## reconsume
2353    
2354          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2355    
2356          redo A;          redo A;
2357        } else {        } else {
2358          !!!cp (144);          !!!cp (144);
2359          $self->{current_token}->{data} # comment          $self->{ct}->{data} # comment
2360              .= '-' . chr ($self->{next_char});              .= '-' . chr ($self->{nc});
2361          $self->{state} = COMMENT_STATE;          $self->{state} = COMMENT_STATE;
2362          !!!next-input-character;          !!!next-input-character;
2363          redo A;          redo A;
2364        }        }
2365      } elsif ($self->{state} == COMMENT_STATE) {      } elsif ($self->{state} == COMMENT_STATE) {
2366        if ($self->{next_char} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2367          !!!cp (145);          !!!cp (145);
2368          $self->{state} = COMMENT_END_DASH_STATE;          $self->{state} = COMMENT_END_DASH_STATE;
2369          !!!next-input-character;          !!!next-input-character;
2370          redo A;          redo A;
2371        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2372          !!!cp (146);          !!!cp (146);
2373          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2374          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2375          ## reconsume          ## reconsume
2376    
2377          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2378    
2379          redo A;          redo A;
2380        } else {        } else {
2381          !!!cp (147);          !!!cp (147);
2382          $self->{current_token}->{data} .= chr ($self->{next_char}); # comment          $self->{ct}->{data} .= chr ($self->{nc}); # comment
2383            $self->{read_until}->($self->{ct}->{data},
2384                                  q[-],
2385                                  length $self->{ct}->{data});
2386    
2387          ## Stay in the state          ## Stay in the state
2388          !!!next-input-character;          !!!next-input-character;
2389          redo A;          redo A;
2390        }        }
2391      } elsif ($self->{state} == COMMENT_END_DASH_STATE) {      } elsif ($self->{state} == COMMENT_END_DASH_STATE) {
2392        if ($self->{next_char} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2393          !!!cp (148);          !!!cp (148);
2394          $self->{state} = COMMENT_END_STATE;          $self->{state} = COMMENT_END_STATE;
2395          !!!next-input-character;          !!!next-input-character;
2396          redo A;          redo A;
2397        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2398          !!!cp (149);          !!!cp (149);
2399          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2400          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2401          ## reconsume          ## reconsume
2402    
2403          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2404    
2405          redo A;          redo A;
2406        } else {        } else {
2407          !!!cp (150);          !!!cp (150);
2408          $self->{current_token}->{data} .= '-' . chr ($self->{next_char}); # comment          $self->{ct}->{data} .= '-' . chr ($self->{nc}); # comment
2409          $self->{state} = COMMENT_STATE;          $self->{state} = COMMENT_STATE;
2410          !!!next-input-character;          !!!next-input-character;
2411          redo A;          redo A;
2412        }        }
2413      } elsif ($self->{state} == COMMENT_END_STATE) {      } elsif ($self->{state} == COMMENT_END_STATE) {
2414        if ($self->{next_char} == 0x003E) { # >        if ($self->{nc} == 0x003E) { # >
2415          !!!cp (151);          !!!cp (151);
2416          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2417          !!!next-input-character;          !!!next-input-character;
2418    
2419          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2420    
2421          redo A;          redo A;
2422        } elsif ($self->{next_char} == 0x002D) { # -        } elsif ($self->{nc} == 0x002D) { # -
2423          !!!cp (152);          !!!cp (152);
2424          !!!parse-error (type => 'dash in comment',          !!!parse-error (type => 'dash in comment',
2425                          line => $self->{line_prev},                          line => $self->{line_prev},
2426                          column => $self->{column_prev});                          column => $self->{column_prev});
2427          $self->{current_token}->{data} .= '-'; # comment          $self->{ct}->{data} .= '-'; # comment
2428          ## Stay in the state          ## Stay in the state
2429          !!!next-input-character;          !!!next-input-character;
2430          redo A;          redo A;
2431        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2432          !!!cp (153);          !!!cp (153);
2433          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2434          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2435          ## reconsume          ## reconsume
2436    
2437          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2438    
2439          redo A;          redo A;
2440        } else {        } else {
# Line 2212  sub _get_next_token ($) { Line 2442  sub _get_next_token ($) {
2442          !!!parse-error (type => 'dash in comment',          !!!parse-error (type => 'dash in comment',
2443                          line => $self->{line_prev},                          line => $self->{line_prev},
2444                          column => $self->{column_prev});                          column => $self->{column_prev});
2445          $self->{current_token}->{data} .= '--' . chr ($self->{next_char}); # comment          $self->{ct}->{data} .= '--' . chr ($self->{nc}); # comment
2446          $self->{state} = COMMENT_STATE;          $self->{state} = COMMENT_STATE;
2447          !!!next-input-character;          !!!next-input-character;
2448          redo A;          redo A;
2449        }        }
2450      } elsif ($self->{state} == DOCTYPE_STATE) {      } elsif ($self->{state} == DOCTYPE_STATE) {
2451        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
2452          !!!cp (155);          !!!cp (155);
2453          $self->{state} = BEFORE_DOCTYPE_NAME_STATE;          $self->{state} = BEFORE_DOCTYPE_NAME_STATE;
2454          !!!next-input-character;          !!!next-input-character;
# Line 2235  sub _get_next_token ($) { Line 2461  sub _get_next_token ($) {
2461          redo A;          redo A;
2462        }        }
2463      } elsif ($self->{state} == BEFORE_DOCTYPE_NAME_STATE) {      } elsif ($self->{state} == BEFORE_DOCTYPE_NAME_STATE) {
2464        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
2465          !!!cp (157);          !!!cp (157);
2466          ## Stay in the state          ## Stay in the state
2467          !!!next-input-character;          !!!next-input-character;
2468          redo A;          redo A;
2469        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2470          !!!cp (158);          !!!cp (158);
2471          !!!parse-error (type => 'no DOCTYPE name');          !!!parse-error (type => 'no DOCTYPE name');
2472          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2473          !!!next-input-character;          !!!next-input-character;
2474    
2475          !!!emit ($self->{current_token}); # DOCTYPE (quirks)          !!!emit ($self->{ct}); # DOCTYPE (quirks)
2476    
2477          redo A;          redo A;
2478        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2479          !!!cp (159);          !!!cp (159);
2480          !!!parse-error (type => 'no DOCTYPE name');          !!!parse-error (type => 'no DOCTYPE name');
2481          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2482          ## reconsume          ## reconsume
2483    
2484          !!!emit ($self->{current_token}); # DOCTYPE (quirks)          !!!emit ($self->{ct}); # DOCTYPE (quirks)
2485    
2486          redo A;          redo A;
2487        } else {        } else {
2488          !!!cp (160);          !!!cp (160);
2489          $self->{current_token}->{name} = chr $self->{next_char};          $self->{ct}->{name} = chr $self->{nc};
2490          delete $self->{current_token}->{quirks};          delete $self->{ct}->{quirks};
 ## ISSUE: "Set the token's name name to the" in the spec  
2491          $self->{state} = DOCTYPE_NAME_STATE;          $self->{state} = DOCTYPE_NAME_STATE;
2492          !!!next-input-character;          !!!next-input-character;
2493          redo A;          redo A;
2494        }        }
2495      } elsif ($self->{state} == DOCTYPE_NAME_STATE) {      } elsif ($self->{state} == DOCTYPE_NAME_STATE) {
2496  ## ISSUE: Redundant "First," in the spec.  ## ISSUE: Redundant "First," in the spec.
2497        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
2498          !!!cp (161);          !!!cp (161);
2499          $self->{state} = AFTER_DOCTYPE_NAME_STATE;          $self->{state} = AFTER_DOCTYPE_NAME_STATE;
2500          !!!next-input-character;          !!!next-input-character;
2501          redo A;          redo A;
2502        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2503          !!!cp (162);          !!!cp (162);
2504          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2505          !!!next-input-character;          !!!next-input-character;
2506    
2507          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2508    
2509          redo A;          redo A;
2510        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2511          !!!cp (163);          !!!cp (163);
2512          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2513          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2514          ## reconsume          ## reconsume
2515    
2516          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2517          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2518    
2519          redo A;          redo A;
2520        } else {        } else {
2521          !!!cp (164);          !!!cp (164);
2522          $self->{current_token}->{name}          $self->{ct}->{name}
2523            .= chr ($self->{next_char}); # DOCTYPE            .= chr ($self->{nc}); # DOCTYPE
2524          ## Stay in the state          ## Stay in the state
2525          !!!next-input-character;          !!!next-input-character;
2526          redo A;          redo A;
2527        }        }
2528      } elsif ($self->{state} == AFTER_DOCTYPE_NAME_STATE) {      } elsif ($self->{state} == AFTER_DOCTYPE_NAME_STATE) {
2529        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
2530          !!!cp (165);          !!!cp (165);
2531          ## Stay in the state          ## Stay in the state
2532          !!!next-input-character;          !!!next-input-character;
2533          redo A;          redo A;
2534        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2535          !!!cp (166);          !!!cp (166);
2536          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2537          !!!next-input-character;          !!!next-input-character;
2538    
2539          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2540    
2541          redo A;          redo A;
2542        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2543          !!!cp (167);          !!!cp (167);
2544          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2545          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2546          ## reconsume          ## reconsume
2547    
2548          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2549          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2550    
2551          redo A;          redo A;
2552        } elsif ($self->{next_char} == 0x0050 or # P        } elsif ($self->{nc} == 0x0050 or # P
2553                 $self->{next_char} == 0x0070) { # p                 $self->{nc} == 0x0070) { # p
2554            $self->{state} = PUBLIC_STATE;
2555            $self->{s_kwd} = chr $self->{nc};
2556          !!!next-input-character;          !!!next-input-character;
2557          if ($self->{next_char} == 0x0055 or # U          redo A;
2558              $self->{next_char} == 0x0075) { # u        } elsif ($self->{nc} == 0x0053 or # S
2559            !!!next-input-character;                 $self->{nc} == 0x0073) { # s
2560            if ($self->{next_char} == 0x0042 or # B          $self->{state} = SYSTEM_STATE;
2561                $self->{next_char} == 0x0062) { # b          $self->{s_kwd} = chr $self->{nc};
             !!!next-input-character;  
             if ($self->{next_char} == 0x004C or # L  
                 $self->{next_char} == 0x006C) { # l  
               !!!next-input-character;  
               if ($self->{next_char} == 0x0049 or # I  
                   $self->{next_char} == 0x0069) { # i  
                 !!!next-input-character;  
                 if ($self->{next_char} == 0x0043 or # C  
                     $self->{next_char} == 0x0063) { # c  
                   !!!cp (168);  
                   $self->{state} = BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE;  
                   !!!next-input-character;  
                   redo A;  
                 } else {  
                   !!!cp (169);  
                 }  
               } else {  
                 !!!cp (170);  
               }  
             } else {  
               !!!cp (171);  
             }  
           } else {  
             !!!cp (172);  
           }  
         } else {  
           !!!cp (173);  
         }  
   
         #  
       } elsif ($self->{next_char} == 0x0053 or # S  
                $self->{next_char} == 0x0073) { # s  
2562          !!!next-input-character;          !!!next-input-character;
2563          if ($self->{next_char} == 0x0059 or # Y          redo A;
             $self->{next_char} == 0x0079) { # y  
           !!!next-input-character;  
           if ($self->{next_char} == 0x0053 or # S  
               $self->{next_char} == 0x0073) { # s  
             !!!next-input-character;  
             if ($self->{next_char} == 0x0054 or # T  
                 $self->{next_char} == 0x0074) { # t  
               !!!next-input-character;  
               if ($self->{next_char} == 0x0045 or # E  
                   $self->{next_char} == 0x0065) { # e  
                 !!!next-input-character;  
                 if ($self->{next_char} == 0x004D or # M  
                     $self->{next_char} == 0x006D) { # m  
                   !!!cp (174);  
                   $self->{state} = BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE;  
                   !!!next-input-character;  
                   redo A;  
                 } else {  
                   !!!cp (175);  
                 }  
               } else {  
                 !!!cp (176);  
               }  
             } else {  
               !!!cp (177);  
             }  
           } else {  
             !!!cp (178);  
           }  
         } else {  
           !!!cp (179);  
         }  
   
         #  
2564        } else {        } else {
2565          !!!cp (180);          !!!cp (180);
2566            !!!parse-error (type => 'string after DOCTYPE name');
2567            $self->{ct}->{quirks} = 1;
2568    
2569            $self->{state} = BOGUS_DOCTYPE_STATE;
2570          !!!next-input-character;          !!!next-input-character;
2571          #          redo A;
2572        }        }
2573        } elsif ($self->{state} == PUBLIC_STATE) {
2574          ## ASCII case-insensitive
2575          if ($self->{nc} == [
2576                undef,
2577                0x0055, # U
2578                0x0042, # B
2579                0x004C, # L
2580                0x0049, # I
2581              ]->[length $self->{s_kwd}] or
2582              $self->{nc} == [
2583                undef,
2584                0x0075, # u
2585                0x0062, # b
2586                0x006C, # l
2587                0x0069, # i
2588              ]->[length $self->{s_kwd}]) {
2589            !!!cp (175);
2590            ## Stay in the state.
2591            $self->{s_kwd} .= chr $self->{nc};
2592            !!!next-input-character;
2593            redo A;
2594          } elsif ((length $self->{s_kwd}) == 5 and
2595                   ($self->{nc} == 0x0043 or # C
2596                    $self->{nc} == 0x0063)) { # c
2597            !!!cp (168);
2598            $self->{state} = BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2599            !!!next-input-character;
2600            redo A;
2601          } else {
2602            !!!cp (169);
2603            !!!parse-error (type => 'string after DOCTYPE name',
2604                            line => $self->{line_prev},
2605                            column => $self->{column_prev} + 1 - length $self->{s_kwd});
2606            $self->{ct}->{quirks} = 1;
2607    
2608        !!!parse-error (type => 'string after DOCTYPE name');          $self->{state} = BOGUS_DOCTYPE_STATE;
2609        $self->{current_token}->{quirks} = 1;          ## Reconsume.
2610            redo A;
2611          }
2612        } elsif ($self->{state} == SYSTEM_STATE) {
2613          ## ASCII case-insensitive
2614          if ($self->{nc} == [
2615                undef,
2616                0x0059, # Y
2617                0x0053, # S
2618                0x0054, # T
2619                0x0045, # E
2620              ]->[length $self->{s_kwd}] or
2621              $self->{nc} == [
2622                undef,
2623                0x0079, # y
2624                0x0073, # s
2625                0x0074, # t
2626                0x0065, # e
2627              ]->[length $self->{s_kwd}]) {
2628            !!!cp (170);
2629            ## Stay in the state.
2630            $self->{s_kwd} .= chr $self->{nc};
2631            !!!next-input-character;
2632            redo A;
2633          } elsif ((length $self->{s_kwd}) == 5 and
2634                   ($self->{nc} == 0x004D or # M
2635                    $self->{nc} == 0x006D)) { # m
2636            !!!cp (171);
2637            $self->{state} = BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2638            !!!next-input-character;
2639            redo A;
2640          } else {
2641            !!!cp (172);
2642            !!!parse-error (type => 'string after DOCTYPE name',
2643                            line => $self->{line_prev},
2644                            column => $self->{column_prev} + 1 - length $self->{s_kwd});
2645            $self->{ct}->{quirks} = 1;
2646    
2647        $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2648        # next-input-character is already done          ## Reconsume.
2649        redo A;          redo A;
2650          }
2651      } elsif ($self->{state} == BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {      } elsif ($self->{state} == BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {
2652        if ({        if ($is_space->{$self->{nc}}) {
             0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,  
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_char}}) {  
2653          !!!cp (181);          !!!cp (181);
2654          ## Stay in the state          ## Stay in the state
2655          !!!next-input-character;          !!!next-input-character;
2656          redo A;          redo A;
2657        } elsif ($self->{next_char} eq 0x0022) { # "        } elsif ($self->{nc} eq 0x0022) { # "
2658          !!!cp (182);          !!!cp (182);
2659          $self->{current_token}->{public_identifier} = ''; # DOCTYPE          $self->{ct}->{pubid} = ''; # DOCTYPE
2660          $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE;          $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE;
2661          !!!next-input-character;          !!!next-input-character;
2662          redo A;          redo A;
2663        } elsif ($self->{next_char} eq 0x0027) { # '        } elsif ($self->{nc} eq 0x0027) { # '
2664          !!!cp (183);          !!!cp (183);
2665          $self->{current_token}->{public_identifier} = ''; # DOCTYPE          $self->{ct}->{pubid} = ''; # DOCTYPE
2666          $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE;          $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE;
2667          !!!next-input-character;          !!!next-input-character;
2668          redo A;          redo A;
2669        } elsif ($self->{next_char} eq 0x003E) { # >        } elsif ($self->{nc} eq 0x003E) { # >
2670          !!!cp (184);          !!!cp (184);
2671          !!!parse-error (type => 'no PUBLIC literal');          !!!parse-error (type => 'no PUBLIC literal');
2672    
2673          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2674          !!!next-input-character;          !!!next-input-character;
2675    
2676          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2677          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2678    
2679          redo A;          redo A;
2680        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2681          !!!cp (185);          !!!cp (185);
2682          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2683    
2684          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2685          ## reconsume          ## reconsume
2686    
2687          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2688          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2689    
2690          redo A;          redo A;
2691        } else {        } else {
2692          !!!cp (186);          !!!cp (186);
2693          !!!parse-error (type => 'string after PUBLIC');          !!!parse-error (type => 'string after PUBLIC');
2694          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2695    
2696          $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2697          !!!next-input-character;          !!!next-input-character;
2698          redo A;          redo A;
2699        }        }
2700      } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE) {      } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE) {
2701        if ($self->{next_char} == 0x0022) { # "        if ($self->{nc} == 0x0022) { # "
2702          !!!cp (187);          !!!cp (187);
2703          $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;          $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2704          !!!next-input-character;          !!!next-input-character;
2705          redo A;          redo A;
2706        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2707          !!!cp (188);          !!!cp (188);
2708          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed PUBLIC literal');
2709    
2710          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2711          !!!next-input-character;          !!!next-input-character;
2712    
2713          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2714          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2715    
2716          redo A;          redo A;
2717        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2718          !!!cp (189);          !!!cp (189);
2719          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed PUBLIC literal');
2720    
2721          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2722          ## reconsume          ## reconsume
2723    
2724          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2725          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2726    
2727          redo A;          redo A;
2728        } else {        } else {
2729          !!!cp (190);          !!!cp (190);
2730          $self->{current_token}->{public_identifier} # DOCTYPE          $self->{ct}->{pubid} # DOCTYPE
2731              .= chr $self->{next_char};              .= chr $self->{nc};
2732            $self->{read_until}->($self->{ct}->{pubid}, q[">],
2733                                  length $self->{ct}->{pubid});
2734    
2735          ## Stay in the state          ## Stay in the state
2736          !!!next-input-character;          !!!next-input-character;
2737          redo A;          redo A;
2738        }        }
2739      } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE) {      } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE) {
2740        if ($self->{next_char} == 0x0027) { # '        if ($self->{nc} == 0x0027) { # '
2741          !!!cp (191);          !!!cp (191);
2742          $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;          $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2743          !!!next-input-character;          !!!next-input-character;
2744          redo A;          redo A;
2745        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2746          !!!cp (192);          !!!cp (192);
2747          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed PUBLIC literal');
2748    
2749          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2750          !!!next-input-character;          !!!next-input-character;
2751    
2752          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2753          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2754    
2755          redo A;          redo A;
2756        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2757          !!!cp (193);          !!!cp (193);
2758          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed PUBLIC literal');
2759    
2760          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2761          ## reconsume          ## reconsume
2762    
2763          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2764          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2765    
2766          redo A;          redo A;
2767        } else {        } else {
2768          !!!cp (194);          !!!cp (194);
2769          $self->{current_token}->{public_identifier} # DOCTYPE          $self->{ct}->{pubid} # DOCTYPE
2770              .= chr $self->{next_char};              .= chr $self->{nc};
2771            $self->{read_until}->($self->{ct}->{pubid}, q['>],
2772                                  length $self->{ct}->{pubid});
2773    
2774          ## Stay in the state          ## Stay in the state
2775          !!!next-input-character;          !!!next-input-character;
2776          redo A;          redo A;
2777        }        }
2778      } elsif ($self->{state} == AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {      } elsif ($self->{state} == AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {
2779        if ({        if ($is_space->{$self->{nc}}) {
             0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,  
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_char}}) {  
2780          !!!cp (195);          !!!cp (195);
2781          ## Stay in the state          ## Stay in the state
2782          !!!next-input-character;          !!!next-input-character;
2783          redo A;          redo A;
2784        } elsif ($self->{next_char} == 0x0022) { # "        } elsif ($self->{nc} == 0x0022) { # "
2785          !!!cp (196);          !!!cp (196);
2786          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          $self->{ct}->{sysid} = ''; # DOCTYPE
2787          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;
2788          !!!next-input-character;          !!!next-input-character;
2789          redo A;          redo A;
2790        } elsif ($self->{next_char} == 0x0027) { # '        } elsif ($self->{nc} == 0x0027) { # '
2791          !!!cp (197);          !!!cp (197);
2792          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          $self->{ct}->{sysid} = ''; # DOCTYPE
2793          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;
2794          !!!next-input-character;          !!!next-input-character;
2795          redo A;          redo A;
2796        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2797          !!!cp (198);          !!!cp (198);
2798          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2799          !!!next-input-character;          !!!next-input-character;
2800    
2801          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2802    
2803          redo A;          redo A;
2804        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2805          !!!cp (199);          !!!cp (199);
2806          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2807    
2808          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2809          ## reconsume          ## reconsume
2810    
2811          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2812          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2813    
2814          redo A;          redo A;
2815        } else {        } else {
2816          !!!cp (200);          !!!cp (200);
2817          !!!parse-error (type => 'string after PUBLIC literal');          !!!parse-error (type => 'string after PUBLIC literal');
2818          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2819    
2820          $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2821          !!!next-input-character;          !!!next-input-character;
2822          redo A;          redo A;
2823        }        }
2824      } elsif ($self->{state} == BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {      } elsif ($self->{state} == BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {
2825        if ({        if ($is_space->{$self->{nc}}) {
             0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,  
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_char}}) {  
2826          !!!cp (201);          !!!cp (201);
2827          ## Stay in the state          ## Stay in the state
2828          !!!next-input-character;          !!!next-input-character;
2829          redo A;          redo A;
2830        } elsif ($self->{next_char} == 0x0022) { # "        } elsif ($self->{nc} == 0x0022) { # "
2831          !!!cp (202);          !!!cp (202);
2832          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          $self->{ct}->{sysid} = ''; # DOCTYPE
2833          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;
2834          !!!next-input-character;          !!!next-input-character;
2835          redo A;          redo A;
2836        } elsif ($self->{next_char} == 0x0027) { # '        } elsif ($self->{nc} == 0x0027) { # '
2837          !!!cp (203);          !!!cp (203);
2838          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          $self->{ct}->{sysid} = ''; # DOCTYPE
2839          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;
2840          !!!next-input-character;          !!!next-input-character;
2841          redo A;          redo A;
2842        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2843          !!!cp (204);          !!!cp (204);
2844          !!!parse-error (type => 'no SYSTEM literal');          !!!parse-error (type => 'no SYSTEM literal');
2845          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2846          !!!next-input-character;          !!!next-input-character;
2847    
2848          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2849          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2850    
2851          redo A;          redo A;
2852        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2853          !!!cp (205);          !!!cp (205);
2854          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2855    
2856          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2857          ## reconsume          ## reconsume
2858    
2859          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2860          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2861    
2862          redo A;          redo A;
2863        } else {        } else {
2864          !!!cp (206);          !!!cp (206);
2865          !!!parse-error (type => 'string after SYSTEM');          !!!parse-error (type => 'string after SYSTEM');
2866          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2867    
2868          $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2869          !!!next-input-character;          !!!next-input-character;
2870          redo A;          redo A;
2871        }        }
2872      } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE) {      } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE) {
2873        if ($self->{next_char} == 0x0022) { # "        if ($self->{nc} == 0x0022) { # "
2874          !!!cp (207);          !!!cp (207);
2875          $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;          $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2876          !!!next-input-character;          !!!next-input-character;
2877          redo A;          redo A;
2878        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2879          !!!cp (208);          !!!cp (208);
2880          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed SYSTEM literal');
2881    
2882          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2883          !!!next-input-character;          !!!next-input-character;
2884    
2885          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2886          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2887    
2888          redo A;          redo A;
2889        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2890          !!!cp (209);          !!!cp (209);
2891          !!!parse-error (type => 'unclosed SYSTEM literal');          !!!parse-error (type => 'unclosed SYSTEM literal');
2892    
2893          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2894          ## reconsume          ## reconsume
2895    
2896          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2897          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2898    
2899          redo A;          redo A;
2900        } else {        } else {
2901          !!!cp (210);          !!!cp (210);
2902          $self->{current_token}->{system_identifier} # DOCTYPE          $self->{ct}->{sysid} # DOCTYPE
2903              .= chr $self->{next_char};              .= chr $self->{nc};
2904            $self->{read_until}->($self->{ct}->{sysid}, q[">],
2905                                  length $self->{ct}->{sysid});
2906    
2907          ## Stay in the state          ## Stay in the state
2908          !!!next-input-character;          !!!next-input-character;
2909          redo A;          redo A;
2910        }        }
2911      } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE) {      } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE) {
2912        if ($self->{next_char} == 0x0027) { # '        if ($self->{nc} == 0x0027) { # '
2913          !!!cp (211);          !!!cp (211);
2914          $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;          $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2915          !!!next-input-character;          !!!next-input-character;
2916          redo A;          redo A;
2917        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2918          !!!cp (212);          !!!cp (212);
2919          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed SYSTEM literal');
2920    
2921          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2922          !!!next-input-character;          !!!next-input-character;
2923    
2924          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2925          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2926    
2927          redo A;          redo A;
2928        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2929          !!!cp (213);          !!!cp (213);
2930          !!!parse-error (type => 'unclosed SYSTEM literal');          !!!parse-error (type => 'unclosed SYSTEM literal');
2931    
2932          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2933          ## reconsume          ## reconsume
2934    
2935          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2936          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2937    
2938          redo A;          redo A;
2939        } else {        } else {
2940          !!!cp (214);          !!!cp (214);
2941          $self->{current_token}->{system_identifier} # DOCTYPE          $self->{ct}->{sysid} # DOCTYPE
2942              .= chr $self->{next_char};              .= chr $self->{nc};
2943            $self->{read_until}->($self->{ct}->{sysid}, q['>],
2944                                  length $self->{ct}->{sysid});
2945    
2946          ## Stay in the state          ## Stay in the state
2947          !!!next-input-character;          !!!next-input-character;
2948          redo A;          redo A;
2949        }        }
2950      } elsif ($self->{state} == AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {      } elsif ($self->{state} == AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {
2951        if ({        if ($is_space->{$self->{nc}}) {
             0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,  
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_char}}) {  
2952          !!!cp (215);          !!!cp (215);
2953          ## Stay in the state          ## Stay in the state
2954          !!!next-input-character;          !!!next-input-character;
2955          redo A;          redo A;
2956        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2957          !!!cp (216);          !!!cp (216);
2958          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2959          !!!next-input-character;          !!!next-input-character;
2960    
2961          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2962    
2963          redo A;          redo A;
2964        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2965          !!!cp (217);          !!!cp (217);
2966          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
   
2967          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2968          ## reconsume          ## reconsume
2969    
2970          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2971          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2972    
2973          redo A;          redo A;
2974        } else {        } else {
2975          !!!cp (218);          !!!cp (218);
2976          !!!parse-error (type => 'string after SYSTEM literal');          !!!parse-error (type => 'string after SYSTEM literal');
2977          #$self->{current_token}->{quirks} = 1;          #$self->{ct}->{quirks} = 1;
2978    
2979          $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2980          !!!next-input-character;          !!!next-input-character;
2981          redo A;          redo A;
2982        }        }
2983      } elsif ($self->{state} == BOGUS_DOCTYPE_STATE) {      } elsif ($self->{state} == BOGUS_DOCTYPE_STATE) {
2984        if ($self->{next_char} == 0x003E) { # >        if ($self->{nc} == 0x003E) { # >
2985          !!!cp (219);          !!!cp (219);
2986          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2987          !!!next-input-character;          !!!next-input-character;
2988    
2989          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2990    
2991          redo A;          redo A;
2992        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2993          !!!cp (220);          !!!cp (220);
         !!!parse-error (type => 'unclosed DOCTYPE');  
2994          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2995          ## reconsume          ## reconsume
2996    
2997          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2998    
2999          redo A;          redo A;
3000        } else {        } else {
3001          !!!cp (221);          !!!cp (221);
3002            my $s = '';
3003            $self->{read_until}->($s, q[>], 0);
3004    
3005          ## Stay in the state          ## Stay in the state
3006          !!!next-input-character;          !!!next-input-character;
3007          redo A;          redo A;
3008        }        }
3009      } elsif ($self->{state} == CDATA_BLOCK_STATE) {      } elsif ($self->{state} == CDATA_SECTION_STATE) {
3010        my $s = '';        ## NOTE: "CDATA section state" in the state is jointly implemented
3011          ## by three states, |CDATA_SECTION_STATE|, |CDATA_SECTION_MSE1_STATE|,
3012          ## and |CDATA_SECTION_MSE2_STATE|.
3013                
3014        my ($l, $c) = ($self->{line}, $self->{column});        if ($self->{nc} == 0x005D) { # ]
3015            !!!cp (221.1);
3016        CS: while ($self->{next_char} != -1) {          $self->{state} = CDATA_SECTION_MSE1_STATE;
3017          if ($self->{next_char} == 0x005D) { # ]          !!!next-input-character;
3018            !!!next-input-character;          redo A;
3019            if ($self->{next_char} == 0x005D) { # ]        } elsif ($self->{nc} == -1) {
3020              !!!next-input-character;          $self->{state} = DATA_STATE;
             MDC: {  
               if ($self->{next_char} == 0x003E) { # >  
                 !!!cp (221.1);  
                 !!!next-input-character;  
                 last CS;  
               } elsif ($self->{next_char} == 0x005D) { # ]  
                 !!!cp (221.2);  
                 $s .= ']';  
                 !!!next-input-character;  
                 redo MDC;  
               } else {  
                 !!!cp (221.3);  
                 $s .= ']]';  
                 #  
               }  
             } # MDC  
           } else {  
             !!!cp (221.4);  
             $s .= ']';  
             #  
           }  
         } else {  
           !!!cp (221.5);  
           #  
         }  
         $s .= chr $self->{next_char};  
3021          !!!next-input-character;          !!!next-input-character;
3022        } # CS          if (length $self->{ct}->{data}) { # character
3023              !!!cp (221.2);
3024              !!!emit ($self->{ct}); # character
3025            } else {
3026              !!!cp (221.3);
3027              ## No token to emit. $self->{ct} is discarded.
3028            }        
3029            redo A;
3030          } else {
3031            !!!cp (221.4);
3032            $self->{ct}->{data} .= chr $self->{nc};
3033            $self->{read_until}->($self->{ct}->{data},
3034                                  q<]>,
3035                                  length $self->{ct}->{data});
3036    
3037        $self->{state} = DATA_STATE;          ## Stay in the state.
3038        ## next-input-character done or EOF, which is reconsumed.          !!!next-input-character;
3039            redo A;
3040          }
3041    
3042        if (length $s) {        ## ISSUE: "text tokens" in spec.
3043        } elsif ($self->{state} == CDATA_SECTION_MSE1_STATE) {
3044          if ($self->{nc} == 0x005D) { # ]
3045            !!!cp (221.5);
3046            $self->{state} = CDATA_SECTION_MSE2_STATE;
3047            !!!next-input-character;
3048            redo A;
3049          } else {
3050          !!!cp (221.6);          !!!cp (221.6);
3051          !!!emit ({type => CHARACTER_TOKEN, data => $s,          $self->{ct}->{data} .= ']';
3052                    line => $l, column => $c});          $self->{state} = CDATA_SECTION_STATE;
3053            ## Reconsume.
3054            redo A;
3055          }
3056        } elsif ($self->{state} == CDATA_SECTION_MSE2_STATE) {
3057          if ($self->{nc} == 0x003E) { # >
3058            $self->{state} = DATA_STATE;
3059            !!!next-input-character;
3060            if (length $self->{ct}->{data}) { # character
3061              !!!cp (221.7);
3062              !!!emit ($self->{ct}); # character
3063            } else {
3064              !!!cp (221.8);
3065              ## No token to emit. $self->{ct} is discarded.
3066            }
3067            redo A;
3068          } elsif ($self->{nc} == 0x005D) { # ]
3069            !!!cp (221.9); # character
3070            $self->{ct}->{data} .= ']'; ## Add first "]" of "]]]".
3071            ## Stay in the state.
3072            !!!next-input-character;
3073            redo A;
3074        } else {        } else {
3075          !!!cp (221.7);          !!!cp (221.11);
3076            $self->{ct}->{data} .= ']]'; # character
3077            $self->{state} = CDATA_SECTION_STATE;
3078            ## Reconsume.
3079            redo A;
3080          }
3081        } elsif ($self->{state} == ENTITY_STATE) {
3082          if ($is_space->{$self->{nc}} or
3083              {
3084                0x003C => 1, 0x0026 => 1, -1 => 1, # <, &
3085                $self->{entity_add} => 1,
3086              }->{$self->{nc}}) {
3087            !!!cp (1001);
3088            ## Don't consume
3089            ## No error
3090            ## Return nothing.
3091            #
3092          } elsif ($self->{nc} == 0x0023) { # #
3093            !!!cp (999);
3094            $self->{state} = ENTITY_HASH_STATE;
3095            $self->{s_kwd} = '#';
3096            !!!next-input-character;
3097            redo A;
3098          } elsif ((0x0041 <= $self->{nc} and
3099                    $self->{nc} <= 0x005A) or # A..Z
3100                   (0x0061 <= $self->{nc} and
3101                    $self->{nc} <= 0x007A)) { # a..z
3102            !!!cp (998);
3103            require Whatpm::_NamedEntityList;
3104            $self->{state} = ENTITY_NAME_STATE;
3105            $self->{s_kwd} = chr $self->{nc};
3106            $self->{entity__value} = $self->{s_kwd};
3107            $self->{entity__match} = 0;
3108            !!!next-input-character;
3109            redo A;
3110          } else {
3111            !!!cp (1027);
3112            !!!parse-error (type => 'bare ero');
3113            ## Return nothing.
3114            #
3115        }        }
3116    
3117        redo A;        ## NOTE: No character is consumed by the "consume a character
3118          ## reference" algorithm.  In other word, there is an "&" character
3119        ## ISSUE: "text tokens" in spec.        ## that does not introduce a character reference, which would be
3120        ## TODO: Streaming support        ## appended to the parent element or the attribute value in later
3121      } else {        ## process of the tokenizer.
3122        die "$0: $self->{state}: Unknown state";  
3123      }        if ($self->{prev_state} == DATA_STATE) {
3124    } # A            !!!cp (997);
3125            $self->{state} = $self->{prev_state};
3126    die "$0: _get_next_token: unexpected case";          ## Reconsume.
3127  } # _get_next_token          !!!emit ({type => CHARACTER_TOKEN, data => '&',
3128                      line => $self->{line_prev},
3129  sub _tokenize_attempt_to_consume_an_entity ($$$) {                    column => $self->{column_prev},
3130    my ($self, $in_attr, $additional) = @_;                   });
3131            redo A;
3132    my ($l, $c) = ($self->{line_prev}, $self->{column_prev});        } else {
3133            !!!cp (996);
3134            $self->{ca}->{value} .= '&';
3135            $self->{state} = $self->{prev_state};
3136            ## Reconsume.
3137            redo A;
3138          }
3139        } elsif ($self->{state} == ENTITY_HASH_STATE) {
3140          if ($self->{nc} == 0x0078 or # x
3141              $self->{nc} == 0x0058) { # X
3142            !!!cp (995);
3143            $self->{state} = HEXREF_X_STATE;
3144            $self->{s_kwd} .= chr $self->{nc};
3145            !!!next-input-character;
3146            redo A;
3147          } elsif (0x0030 <= $self->{nc} and
3148                   $self->{nc} <= 0x0039) { # 0..9
3149            !!!cp (994);
3150            $self->{state} = NCR_NUM_STATE;
3151            $self->{s_kwd} = $self->{nc} - 0x0030;
3152            !!!next-input-character;
3153            redo A;
3154          } else {
3155            !!!parse-error (type => 'bare nero',
3156                            line => $self->{line_prev},
3157                            column => $self->{column_prev} - 1);
3158    
3159    if ({          ## NOTE: According to the spec algorithm, nothing is returned,
3160         0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, # HT, LF, VT, FF,          ## and then "&#" is appended to the parent element or the attribute
3161         0x0020 => 1, 0x003C => 1, 0x0026 => 1, -1 => 1, # SP, <, & # 0x000D # CR          ## value in the later processing.
3162         $additional => 1,  
3163        }->{$self->{next_char}}) {          if ($self->{prev_state} == DATA_STATE) {
3164      !!!cp (1001);            !!!cp (1019);
3165      ## Don't consume            $self->{state} = $self->{prev_state};
3166      ## No error            ## Reconsume.
3167      return undef;            !!!emit ({type => CHARACTER_TOKEN,
3168    } elsif ($self->{next_char} == 0x0023) { # #                      data => '&#',
3169      !!!next-input-character;                      line => $self->{line_prev},
3170      if ($self->{next_char} == 0x0078 or # x                      column => $self->{column_prev} - 1,
3171          $self->{next_char} == 0x0058) { # X                     });
3172        my $code;            redo A;
       X: {  
         my $x_char = $self->{next_char};  
         !!!next-input-character;  
         if (0x0030 <= $self->{next_char} and  
             $self->{next_char} <= 0x0039) { # 0..9  
           !!!cp (1002);  
           $code ||= 0;  
           $code *= 0x10;  
           $code += $self->{next_char} - 0x0030;  
           redo X;  
         } elsif (0x0061 <= $self->{next_char} and  
                  $self->{next_char} <= 0x0066) { # a..f  
           !!!cp (1003);  
           $code ||= 0;  
           $code *= 0x10;  
           $code += $self->{next_char} - 0x0060 + 9;  
           redo X;  
         } elsif (0x0041 <= $self->{next_char} and  
                  $self->{next_char} <= 0x0046) { # A..F  
           !!!cp (1004);  
           $code ||= 0;  
           $code *= 0x10;  
           $code += $self->{next_char} - 0x0040 + 9;  
           redo X;  
         } elsif (not defined $code) { # no hexadecimal digit  
           !!!cp (1005);  
           !!!parse-error (type => 'bare hcro', line => $l, column => $c);  
           !!!back-next-input-character ($x_char, $self->{next_char});  
           $self->{next_char} = 0x0023; # #  
           return undef;  
         } elsif ($self->{next_char} == 0x003B) { # ;  
           !!!cp (1006);  
           !!!next-input-character;  
3173          } else {          } else {
3174            !!!cp (1007);            !!!cp (993);
3175            !!!parse-error (type => 'no refc', line => $l, column => $c);            $self->{ca}->{value} .= '&#';
3176              $self->{state} = $self->{prev_state};
3177              ## Reconsume.
3178              redo A;
3179          }          }
3180          }
3181          if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) {      } elsif ($self->{state} == NCR_NUM_STATE) {
3182            !!!cp (1008);        if (0x0030 <= $self->{nc} and
3183            !!!parse-error (type => (sprintf 'invalid character reference:U+%04X', $code), line => $l, column => $c);            $self->{nc} <= 0x0039) { # 0..9
           $code = 0xFFFD;  
         } elsif ($code > 0x10FFFF) {  
           !!!cp (1009);  
           !!!parse-error (type => (sprintf 'invalid character reference:U-%08X', $code), line => $l, column => $c);  
           $code = 0xFFFD;  
         } elsif ($code == 0x000D) {  
           !!!cp (1010);  
           !!!parse-error (type => 'CR character reference', line => $l, column => $c);  
           $code = 0x000A;  
         } elsif (0x80 <= $code and $code <= 0x9F) {  
           !!!cp (1011);  
           !!!parse-error (type => (sprintf 'C1 character reference:U+%04X', $code), line => $l, column => $c);  
           $code = $c1_entity_char->{$code};  
         }  
   
         return {type => CHARACTER_TOKEN, data => chr $code,  
                 has_reference => 1,  
                 line => $l, column => $c,  
                };  
       } # X  
     } elsif (0x0030 <= $self->{next_char} and  
              $self->{next_char} <= 0x0039) { # 0..9  
       my $code = $self->{next_char} - 0x0030;  
       !!!next-input-character;  
         
       while (0x0030 <= $self->{next_char} and  
                 $self->{next_char} <= 0x0039) { # 0..9  
3184          !!!cp (1012);          !!!cp (1012);
3185          $code *= 10;          $self->{s_kwd} *= 10;
3186          $code += $self->{next_char} - 0x0030;          $self->{s_kwd} += $self->{nc} - 0x0030;
3187                    
3188            ## Stay in the state.
3189          !!!next-input-character;          !!!next-input-character;
3190        }          redo A;
3191          } elsif ($self->{nc} == 0x003B) { # ;
       if ($self->{next_char} == 0x003B) { # ;  
3192          !!!cp (1013);          !!!cp (1013);
3193          !!!next-input-character;          !!!next-input-character;
3194            #
3195        } else {        } else {
3196          !!!cp (1014);          !!!cp (1014);
3197          !!!parse-error (type => 'no refc', line => $l, column => $c);          !!!parse-error (type => 'no refc');
3198            ## Reconsume.
3199            #
3200        }        }
3201    
3202        if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) {        my $code = $self->{s_kwd};
3203          my $l = $self->{line_prev};
3204          my $c = $self->{column_prev};
3205          if ($charref_map->{$code}) {
3206          !!!cp (1015);          !!!cp (1015);
3207          !!!parse-error (type => (sprintf 'invalid character reference:U+%04X', $code), line => $l, column => $c);          !!!parse-error (type => 'invalid character reference',
3208          $code = 0xFFFD;                          text => (sprintf 'U+%04X', $code),
3209                            line => $l, column => $c);
3210            $code = $charref_map->{$code};
3211        } elsif ($code > 0x10FFFF) {        } elsif ($code > 0x10FFFF) {
3212          !!!cp (1016);          !!!cp (1016);
3213          !!!parse-error (type => (sprintf 'invalid character reference:U-%08X', $code), line => $l, column => $c);          !!!parse-error (type => 'invalid character reference',
3214                            text => (sprintf 'U-%08X', $code),
3215                            line => $l, column => $c);
3216          $code = 0xFFFD;          $code = 0xFFFD;
       } elsif ($code == 0x000D) {  
         !!!cp (1017);  
         !!!parse-error (type => 'CR character reference', line => $l, column => $c);  
         $code = 0x000A;  
       } elsif (0x80 <= $code and $code <= 0x9F) {  
         !!!cp (1018);  
         !!!parse-error (type => (sprintf 'C1 character reference:U+%04X', $code), line => $l, column => $c);  
         $code = $c1_entity_char->{$code};  
3217        }        }
3218          
3219        return {type => CHARACTER_TOKEN, data => chr $code, has_reference => 1,        if ($self->{prev_state} == DATA_STATE) {
3220                line => $l, column => $c,          !!!cp (992);
3221               };          $self->{state} = $self->{prev_state};
3222      } else {          ## Reconsume.
3223        !!!cp (1019);          !!!emit ({type => CHARACTER_TOKEN, data => chr $code,
3224        !!!parse-error (type => 'bare nero', line => $l, column => $c);                    line => $l, column => $c,
3225        !!!back-next-input-character ($self->{next_char});                   });
3226        $self->{next_char} = 0x0023; # #          redo A;
3227        return undef;        } else {
3228      }          !!!cp (991);
3229    } elsif ((0x0041 <= $self->{next_char} and          $self->{ca}->{value} .= chr $code;
3230              $self->{next_char} <= 0x005A) or          $self->{ca}->{has_reference} = 1;
3231             (0x0061 <= $self->{next_char} and          $self->{state} = $self->{prev_state};
3232              $self->{next_char} <= 0x007A)) {          ## Reconsume.
3233      my $entity_name = chr $self->{next_char};          redo A;
3234      !!!next-input-character;        }
3235        } elsif ($self->{state} == HEXREF_X_STATE) {
3236      my $value = $entity_name;        if ((0x0030 <= $self->{nc} and $self->{nc} <= 0x0039) or
3237      my $match = 0;            (0x0041 <= $self->{nc} and $self->{nc} <= 0x0046) or
3238      require Whatpm::_NamedEntityList;            (0x0061 <= $self->{nc} and $self->{nc} <= 0x0066)) {
3239      our $EntityChar;          # 0..9, A..F, a..f
3240            !!!cp (990);
3241      while (length $entity_name < 30 and          $self->{state} = HEXREF_HEX_STATE;
3242             ## NOTE: Some number greater than the maximum length of entity name          $self->{s_kwd} = 0;
3243             ((0x0041 <= $self->{next_char} and # a          ## Reconsume.
3244               $self->{next_char} <= 0x005A) or # x          redo A;
3245              (0x0061 <= $self->{next_char} and # a        } else {
3246               $self->{next_char} <= 0x007A) or # z          !!!parse-error (type => 'bare hcro',
3247              (0x0030 <= $self->{next_char} and # 0                          line => $self->{line_prev},
3248               $self->{next_char} <= 0x0039) or # 9                          column => $self->{column_prev} - 2);
3249              $self->{next_char} == 0x003B)) { # ;  
3250        $entity_name .= chr $self->{next_char};          ## NOTE: According to the spec algorithm, nothing is returned,
3251        if (defined $EntityChar->{$entity_name}) {          ## and then "&#" followed by "X" or "x" is appended to the parent
3252          if ($self->{next_char} == 0x003B) { # ;          ## element or the attribute value in the later processing.
3253            !!!cp (1020);  
3254            $value = $EntityChar->{$entity_name};          if ($self->{prev_state} == DATA_STATE) {
3255            $match = 1;            !!!cp (1005);
3256            !!!next-input-character;            $self->{state} = $self->{prev_state};
3257            last;            ## Reconsume.
3258              !!!emit ({type => CHARACTER_TOKEN,
3259                        data => '&' . $self->{s_kwd},
3260                        line => $self->{line_prev},
3261                        column => $self->{column_prev} - length $self->{s_kwd},
3262                       });
3263              redo A;
3264          } else {          } else {
3265            !!!cp (1021);            !!!cp (989);
3266            $value = $EntityChar->{$entity_name};            $self->{ca}->{value} .= '&' . $self->{s_kwd};
3267            $match = -1;            $self->{state} = $self->{prev_state};
3268              ## Reconsume.
3269              redo A;
3270            }
3271          }
3272        } elsif ($self->{state} == HEXREF_HEX_STATE) {
3273          if (0x0030 <= $self->{nc} and $self->{nc} <= 0x0039) {
3274            # 0..9
3275            !!!cp (1002);
3276            $self->{s_kwd} *= 0x10;
3277            $self->{s_kwd} += $self->{nc} - 0x0030;
3278            ## Stay in the state.
3279            !!!next-input-character;
3280            redo A;
3281          } elsif (0x0061 <= $self->{nc} and
3282                   $self->{nc} <= 0x0066) { # a..f
3283            !!!cp (1003);
3284            $self->{s_kwd} *= 0x10;
3285            $self->{s_kwd} += $self->{nc} - 0x0060 + 9;
3286            ## Stay in the state.
3287            !!!next-input-character;
3288            redo A;
3289          } elsif (0x0041 <= $self->{nc} and
3290                   $self->{nc} <= 0x0046) { # A..F
3291            !!!cp (1004);
3292            $self->{s_kwd} *= 0x10;
3293            $self->{s_kwd} += $self->{nc} - 0x0040 + 9;
3294            ## Stay in the state.
3295            !!!next-input-character;
3296            redo A;
3297          } elsif ($self->{nc} == 0x003B) { # ;
3298            !!!cp (1006);
3299            !!!next-input-character;
3300            #
3301          } else {
3302            !!!cp (1007);
3303            !!!parse-error (type => 'no refc',
3304                            line => $self->{line},
3305                            column => $self->{column});
3306            ## Reconsume.
3307            #
3308          }
3309    
3310          my $code = $self->{s_kwd};
3311          my $l = $self->{line_prev};
3312          my $c = $self->{column_prev};
3313          if ($charref_map->{$code}) {
3314            !!!cp (1008);
3315            !!!parse-error (type => 'invalid character reference',
3316                            text => (sprintf 'U+%04X', $code),
3317                            line => $l, column => $c);
3318            $code = $charref_map->{$code};
3319          } elsif ($code > 0x10FFFF) {
3320            !!!cp (1009);
3321            !!!parse-error (type => 'invalid character reference',
3322                            text => (sprintf 'U-%08X', $code),
3323                            line => $l, column => $c);
3324            $code = 0xFFFD;
3325          }
3326    
3327          if ($self->{prev_state} == DATA_STATE) {
3328            !!!cp (988);
3329            $self->{state} = $self->{prev_state};
3330            ## Reconsume.
3331            !!!emit ({type => CHARACTER_TOKEN, data => chr $code,
3332                      line => $l, column => $c,
3333                     });
3334            redo A;
3335          } else {
3336            !!!cp (987);
3337            $self->{ca}->{value} .= chr $code;
3338            $self->{ca}->{has_reference} = 1;
3339            $self->{state} = $self->{prev_state};
3340            ## Reconsume.
3341            redo A;
3342          }
3343        } elsif ($self->{state} == ENTITY_NAME_STATE) {
3344          if (length $self->{s_kwd} < 30 and
3345              ## NOTE: Some number greater than the maximum length of entity name
3346              ((0x0041 <= $self->{nc} and # a
3347                $self->{nc} <= 0x005A) or # x
3348               (0x0061 <= $self->{nc} and # a
3349                $self->{nc} <= 0x007A) or # z
3350               (0x0030 <= $self->{nc} and # 0
3351                $self->{nc} <= 0x0039) or # 9
3352               $self->{nc} == 0x003B)) { # ;
3353            our $EntityChar;
3354            $self->{s_kwd} .= chr $self->{nc};
3355            if (defined $EntityChar->{$self->{s_kwd}}) {
3356              if ($self->{nc} == 0x003B) { # ;
3357                !!!cp (1020);
3358                $self->{entity__value} = $EntityChar->{$self->{s_kwd}};
3359                $self->{entity__match} = 1;
3360                !!!next-input-character;
3361                #
3362              } else {
3363                !!!cp (1021);
3364                $self->{entity__value} = $EntityChar->{$self->{s_kwd}};
3365                $self->{entity__match} = -1;
3366                ## Stay in the state.
3367                !!!next-input-character;
3368                redo A;
3369              }
3370            } else {
3371              !!!cp (1022);
3372              $self->{entity__value} .= chr $self->{nc};
3373              $self->{entity__match} *= 2;
3374              ## Stay in the state.
3375            !!!next-input-character;            !!!next-input-character;
3376              redo A;
3377            }
3378          }
3379    
3380          my $data;
3381          my $has_ref;
3382          if ($self->{entity__match} > 0) {
3383            !!!cp (1023);
3384            $data = $self->{entity__value};
3385            $has_ref = 1;
3386            #
3387          } elsif ($self->{entity__match} < 0) {
3388            !!!parse-error (type => 'no refc');
3389            if ($self->{prev_state} != DATA_STATE and # in attribute
3390                $self->{entity__match} < -1) {
3391              !!!cp (1024);
3392              $data = '&' . $self->{s_kwd};
3393              #
3394            } else {
3395              !!!cp (1025);
3396              $data = $self->{entity__value};
3397              $has_ref = 1;
3398              #
3399          }          }
3400        } else {        } else {
3401          !!!cp (1022);          !!!cp (1026);
3402          $value .= chr $self->{next_char};          !!!parse-error (type => 'bare ero',
3403          $match *= 2;                          line => $self->{line_prev},
3404          !!!next-input-character;                          column => $self->{column_prev} - length $self->{s_kwd});
3405            $data = '&' . $self->{s_kwd};
3406            #
3407        }        }
3408      }    
3409              ## NOTE: In these cases, when a character reference is found,
3410      if ($match > 0) {        ## it is consumed and a character token is returned, or, otherwise,
3411        !!!cp (1023);        ## nothing is consumed and returned, according to the spec algorithm.
3412        return {type => CHARACTER_TOKEN, data => $value, has_reference => 1,        ## In this implementation, anything that has been examined by the
3413                line => $l, column => $c,        ## tokenizer is appended to the parent element or the attribute value
3414               };        ## as string, either literal string when no character reference or
3415      } elsif ($match < 0) {        ## entity-replaced string otherwise, in this stage, since any characters
3416        !!!parse-error (type => 'no refc', line => $l, column => $c);        ## that would not be consumed are appended in the data state or in an
3417        if ($in_attr and $match < -1) {        ## appropriate attribute value state anyway.
3418          !!!cp (1024);  
3419          return {type => CHARACTER_TOKEN, data => '&'.$entity_name,        if ($self->{prev_state} == DATA_STATE) {
3420                  line => $l, column => $c,          !!!cp (986);
3421                 };          $self->{state} = $self->{prev_state};
3422        } else {          ## Reconsume.
3423          !!!cp (1025);          !!!emit ({type => CHARACTER_TOKEN,
3424          return {type => CHARACTER_TOKEN, data => $value, has_reference => 1,                    data => $data,
3425                  line => $l, column => $c,                    line => $self->{line_prev},
3426                 };                    column => $self->{column_prev} + 1 - length $self->{s_kwd},
3427                     });
3428            redo A;
3429          } else {
3430            !!!cp (985);
3431            $self->{ca}->{value} .= $data;
3432            $self->{ca}->{has_reference} = 1 if $has_ref;
3433            $self->{state} = $self->{prev_state};
3434            ## Reconsume.
3435            redo A;
3436        }        }
3437      } else {      } else {
3438        !!!cp (1026);        die "$0: $self->{state}: Unknown state";
       !!!parse-error (type => 'bare ero', line => $l, column => $c);  
       ## NOTE: "No characters are consumed" in the spec.  
       return {type => CHARACTER_TOKEN, data => '&'.$value,  
               line => $l, column => $c,  
              };  
3439      }      }
3440    } else {    } # A  
3441      !!!cp (1027);  
3442      ## no characters are consumed    die "$0: _get_next_token: unexpected case";
3443      !!!parse-error (type => 'bare ero', line => $l, column => $c);  } # _get_next_token
     return undef;  
   }  
 } # _tokenize_attempt_to_consume_an_entity  
3444    
3445  sub _initialize_tree_constructor ($) {  sub _initialize_tree_constructor ($) {
3446    my $self = shift;    my $self = shift;
# Line 3057  sub _initialize_tree_constructor ($) { Line 3449  sub _initialize_tree_constructor ($) {
3449    ## TODO: Turn mutation events off # MUST    ## TODO: Turn mutation events off # MUST
3450    ## TODO: Turn loose Document option (manakai extension) on    ## TODO: Turn loose Document option (manakai extension) on
3451    $self->{document}->manakai_is_html (1); # MUST    $self->{document}->manakai_is_html (1); # MUST
3452      $self->{document}->set_user_data (manakai_source_line => 1);
3453      $self->{document}->set_user_data (manakai_source_column => 1);
3454  } # _initialize_tree_constructor  } # _initialize_tree_constructor
3455    
3456  sub _terminate_tree_constructor ($) {  sub _terminate_tree_constructor ($) {
# Line 3076  sub _construct_tree ($) { Line 3470  sub _construct_tree ($) {
3470    ## When an interactive UA render the $self->{document} available    ## When an interactive UA render the $self->{document} available
3471    ## to the user, or when it begin accepting user input, are    ## to the user, or when it begin accepting user input, are
3472    ## not defined.    ## not defined.
   
   ## Append a character: collect it and all subsequent consecutive  
   ## characters and insert one Text node whose data is concatenation  
   ## of all those characters. # MUST  
3473        
3474    !!!next-token;    !!!next-token;
3475    
3476    undef $self->{form_element};    undef $self->{form_element};
3477    undef $self->{head_element};    undef $self->{head_element};
3478      undef $self->{head_element_inserted};
3479    $self->{open_elements} = [];    $self->{open_elements} = [];
3480    undef $self->{inner_html_node};    undef $self->{inner_html_node};
3481    
# Line 3111  sub _tree_construction_initial ($) { Line 3502  sub _tree_construction_initial ($) {
3502        ## language.        ## language.
3503        my $doctype_name = $token->{name};        my $doctype_name = $token->{name};
3504        $doctype_name = '' unless defined $doctype_name;        $doctype_name = '' unless defined $doctype_name;
3505        $doctype_name =~ tr/a-z/A-Z/;        $doctype_name =~ tr/a-z/A-Z/; # ASCII case-insensitive
3506        if (not defined $token->{name} or # <!DOCTYPE>        if (not defined $token->{name} or # <!DOCTYPE>
3507            defined $token->{public_identifier} or            defined $token->{sysid}) {
           defined $token->{system_identifier}) {  
3508          !!!cp ('t1');          !!!cp ('t1');
3509          !!!parse-error (type => 'not HTML5', token => $token);          !!!parse-error (type => 'not HTML5', token => $token);
3510        } elsif ($doctype_name ne 'HTML') {        } elsif ($doctype_name ne 'HTML') {
3511          !!!cp ('t2');          !!!cp ('t2');
         ## ISSUE: ASCII case-insensitive? (in fact it does not matter)  
3512          !!!parse-error (type => 'not HTML5', token => $token);          !!!parse-error (type => 'not HTML5', token => $token);
3513          } elsif (defined $token->{pubid}) {
3514            if ($token->{pubid} eq 'XSLT-compat') {
3515              !!!cp ('t1.2');
3516              !!!parse-error (type => 'XSLT-compat', token => $token,
3517                              level => $self->{level}->{should});
3518            } else {
3519              !!!parse-error (type => 'not HTML5', token => $token);
3520            }
3521        } else {        } else {
3522          !!!cp ('t3');          !!!cp ('t3');
3523            #
3524        }        }
3525                
3526        my $doctype = $self->{document}->create_document_type_definition        my $doctype = $self->{document}->create_document_type_definition
3527          ($token->{name}); ## ISSUE: If name is missing (e.g. <!DOCTYPE>)?          ($token->{name}); ## ISSUE: If name is missing (e.g. <!DOCTYPE>)?
3528        ## NOTE: Default value for both |public_id| and |system_id| attributes        ## NOTE: Default value for both |public_id| and |system_id| attributes
3529        ## are empty strings, so that we don't set any value in missing cases.        ## are empty strings, so that we don't set any value in missing cases.
3530        $doctype->public_id ($token->{public_identifier})        $doctype->public_id ($token->{pubid}) if defined $token->{pubid};
3531            if defined $token->{public_identifier};        $doctype->system_id ($token->{sysid}) if defined $token->{sysid};
       $doctype->system_id ($token->{system_identifier})  
           if defined $token->{system_identifier};  
3532        ## NOTE: Other DocumentType attributes are null or empty lists.        ## NOTE: Other DocumentType attributes are null or empty lists.
3533        ## ISSUE: internalSubset = null??        ## ISSUE: internalSubset = null??
3534        $self->{document}->append_child ($doctype);        $self->{document}->append_child ($doctype);
# Line 3140  sub _tree_construction_initial ($) { Line 3536  sub _tree_construction_initial ($) {
3536        if ($token->{quirks} or $doctype_name ne 'HTML') {        if ($token->{quirks} or $doctype_name ne 'HTML') {
3537          !!!cp ('t4');          !!!cp ('t4');
3538          $self->{document}->manakai_compat_mode ('quirks');          $self->{document}->manakai_compat_mode ('quirks');
3539        } elsif (defined $token->{public_identifier}) {        } elsif (defined $token->{pubid}) {
3540          my $pubid = $token->{public_identifier};          my $pubid = $token->{pubid};
3541          $pubid =~ tr/a-z/A-z/;          $pubid =~ tr/a-z/A-z/;
3542          if ({          my $prefix = [
3543            "+//SILMARIL//DTD HTML PRO V0R11 19970101//EN" => 1,            "+//SILMARIL//DTD HTML PRO V0R11 19970101//",
3544            "-//ADVASOFT LTD//DTD HTML 3.0 ASWEDIT + EXTENSIONS//EN" => 1,            "-//ADVASOFT LTD//DTD HTML 3.0 ASWEDIT + EXTENSIONS//",
3545            "-//AS//DTD HTML 3.0 ASWEDIT + EXTENSIONS//EN" => 1,            "-//AS//DTD HTML 3.0 ASWEDIT + EXTENSIONS//",
3546            "-//IETF//DTD HTML 2.0 LEVEL 1//EN" => 1,            "-//IETF//DTD HTML 2.0 LEVEL 1//",
3547            "-//IETF//DTD HTML 2.0 LEVEL 2//EN" => 1,            "-//IETF//DTD HTML 2.0 LEVEL 2//",
3548            "-//IETF//DTD HTML 2.0 STRICT LEVEL 1//EN" => 1,            "-//IETF//DTD HTML 2.0 STRICT LEVEL 1//",
3549            "-//IETF//DTD HTML 2.0 STRICT LEVEL 2//EN" => 1,            "-//IETF//DTD HTML 2.0 STRICT LEVEL 2//",
3550            "-//IETF//DTD HTML 2.0 STRICT//EN" => 1,            "-//IETF//DTD HTML 2.0 STRICT//",
3551            "-//IETF//DTD HTML 2.0//EN" => 1,            "-//IETF//DTD HTML 2.0//",
3552            "-//IETF//DTD HTML 2.1E//EN" => 1,            "-//IETF//DTD HTML 2.1E//",
3553            "-//IETF//DTD HTML 3.0//EN" => 1,            "-//IETF//DTD HTML 3.0//",
3554            "-//IETF//DTD HTML 3.0//EN//" => 1,            "-//IETF//DTD HTML 3.2 FINAL//",
3555            "-//IETF//DTD HTML 3.2 FINAL//EN" => 1,            "-//IETF//DTD HTML 3.2//",
3556            "-//IETF//DTD HTML 3.2//EN" => 1,            "-//IETF//DTD HTML 3//",
3557            "-//IETF//DTD HTML 3//EN" => 1,            "-//IETF//DTD HTML LEVEL 0//",
3558            "-//IETF//DTD HTML LEVEL 0//EN" => 1,            "-//IETF//DTD HTML LEVEL 1//",
3559            "-//IETF//DTD HTML LEVEL 0//EN//2.0" => 1,            "-//IETF//DTD HTML LEVEL 2//",
3560            "-//IETF//DTD HTML LEVEL 1//EN" => 1,            "-//IETF//DTD HTML LEVEL 3//",
3561            "-//IETF//DTD HTML LEVEL 1//EN//2.0" => 1,            "-//IETF//DTD HTML STRICT LEVEL 0//",
3562            "-//IETF//DTD HTML LEVEL 2//EN" => 1,            "-//IETF//DTD HTML STRICT LEVEL 1//",
3563            "-//IETF//DTD HTML LEVEL 2//EN//2.0" => 1,            "-//IETF//DTD HTML STRICT LEVEL 2//",
3564            "-//IETF//DTD HTML LEVEL 3//EN" => 1,            "-//IETF//DTD HTML STRICT LEVEL 3//",
3565            "-//IETF//DTD HTML LEVEL 3//EN//3.0" => 1,            "-//IETF//DTD HTML STRICT//",
3566            "-//IETF//DTD HTML STRICT LEVEL 0//EN" => 1,            "-//IETF//DTD HTML//",
3567            "-//IETF//DTD HTML STRICT LEVEL 0//EN//2.0" => 1,            "-//METRIUS//DTD METRIUS PRESENTATIONAL//",
3568            "-//IETF//DTD HTML STRICT LEVEL 1//EN" => 1,            "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 HTML STRICT//",
3569            "-//IETF//DTD HTML STRICT LEVEL 1//EN//2.0" => 1,            "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 HTML//",
3570            "-//IETF//DTD HTML STRICT LEVEL 2//EN" => 1,            "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 TABLES//",
3571            "-//IETF//DTD HTML STRICT LEVEL 2//EN//2.0" => 1,            "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 HTML STRICT//",
3572            "-//IETF//DTD HTML STRICT LEVEL 3//EN" => 1,            "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 HTML//",
3573            "-//IETF//DTD HTML STRICT LEVEL 3//EN//3.0" => 1,            "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 TABLES//",
3574            "-//IETF//DTD HTML STRICT//EN" => 1,            "-//NETSCAPE COMM. CORP.//DTD HTML//",
3575            "-//IETF//DTD HTML STRICT//EN//2.0" => 1,            "-//NETSCAPE COMM. CORP.//DTD STRICT HTML//",
3576            "-//IETF//DTD HTML STRICT//EN//3.0" => 1,            "-//O'REILLY AND ASSOCIATES//DTD HTML 2.0//",
3577            "-//IETF//DTD HTML//EN" => 1,            "-//O'REILLY AND ASSOCIATES//DTD HTML EXTENDED 1.0//",
3578            "-//IETF//DTD HTML//EN//2.0" => 1,            "-//O'REILLY AND ASSOCIATES//DTD HTML EXTENDED RELAXED 1.0//",
3579            "-//IETF//DTD HTML//EN//3.0" => 1,            "-//SOFTQUAD SOFTWARE//DTD HOTMETAL PRO 6.0::19990601::EXTENSIONS TO HTML 4.0//",
3580            "-//METRIUS//DTD METRIUS PRESENTATIONAL//EN" => 1,            "-//SOFTQUAD//DTD HOTMETAL PRO 4.0::19971010::EXTENSIONS TO HTML 4.0//",
3581            "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 HTML STRICT//EN" => 1,            "-//SPYGLASS//DTD HTML 2.0 EXTENDED//",
3582            "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 HTML//EN" => 1,            "-//SQ//DTD HTML 2.0 HOTMETAL + EXTENSIONS//",
3583            "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 TABLES//EN" => 1,            "-//SUN MICROSYSTEMS CORP.//DTD HOTJAVA HTML//",
3584            "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 HTML STRICT//EN" => 1,            "-//SUN MICROSYSTEMS CORP.//DTD HOTJAVA STRICT HTML//",
3585            "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 HTML//EN" => 1,            "-//W3C//DTD HTML 3 1995-03-24//",
3586            "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 TABLES//EN" => 1,            "-//W3C//DTD HTML 3.2 DRAFT//",
3587            "-//NETSCAPE COMM. CORP.//DTD HTML//EN" => 1,            "-//W3C//DTD HTML 3.2 FINAL//",
3588            "-//NETSCAPE COMM. CORP.//DTD STRICT HTML//EN" => 1,            "-//W3C//DTD HTML 3.2//",
3589            "-//O'REILLY AND ASSOCIATES//DTD HTML 2.0//EN" => 1,            "-//W3C//DTD HTML 3.2S DRAFT//",
3590            "-//O'REILLY AND ASSOCIATES//DTD HTML EXTENDED 1.0//EN" => 1,            "-//W3C//DTD HTML 4.0 FRAMESET//",
3591            "-//O'REILLY AND ASSOCIATES//DTD HTML EXTENDED RELAXED 1.0//EN" => 1,            "-//W3C//DTD HTML 4.0 TRANSITIONAL//",
3592            "-//SOFTQUAD SOFTWARE//DTD HOTMETAL PRO 6.0::19990601::EXTENSIONS TO HTML 4.0//EN" => 1,            "-//W3C//DTD HTML EXPERIMETNAL 19960712//",
3593            "-//SOFTQUAD//DTD HOTMETAL PRO 4.0::19971010::EXTENSIONS TO HTML 4.0//EN" => 1,            "-//W3C//DTD HTML EXPERIMENTAL 970421//",
3594            "-//SPYGLASS//DTD HTML 2.0 EXTENDED//EN" => 1,            "-//W3C//DTD W3 HTML//",
3595            "-//SQ//DTD HTML 2.0 HOTMETAL + EXTENSIONS//EN" => 1,            "-//W3O//DTD W3 HTML 3.0//",
3596            "-//SUN MICROSYSTEMS CORP.//DTD HOTJAVA HTML//EN" => 1,            "-//WEBTECHS//DTD MOZILLA HTML 2.0//",
3597            "-//SUN MICROSYSTEMS CORP.//DTD HOTJAVA STRICT HTML//EN" => 1,            "-//WEBTECHS//DTD MOZILLA HTML//",
3598            "-//W3C//DTD HTML 3 1995-03-24//EN" => 1,          ]; # $prefix
3599            "-//W3C//DTD HTML 3.2 DRAFT//EN" => 1,          my $match;
3600            "-//W3C//DTD HTML 3.2 FINAL//EN" => 1,          for (@$prefix) {
3601            "-//W3C//DTD HTML 3.2//EN" => 1,            if (substr ($prefix, 0, length $_) eq $_) {
3602            "-//W3C//DTD HTML 3.2S DRAFT//EN" => 1,              $match = 1;
3603            "-//W3C//DTD HTML 4.0 FRAMESET//EN" => 1,              last;
3604            "-//W3C//DTD HTML 4.0 TRANSITIONAL//EN" => 1,            }
3605            "-//W3C//DTD HTML EXPERIMETNAL 19960712//EN" => 1,          }
3606            "-//W3C//DTD HTML EXPERIMENTAL 970421//EN" => 1,          if ($match or
3607            "-//W3C//DTD W3 HTML//EN" => 1,              $pubid eq "-//W3O//DTD W3 HTML STRICT 3.0//EN//" or
3608            "-//W3O//DTD W3 HTML 3.0//EN" => 1,              $pubid eq "-/W3C/DTD HTML 4.0 TRANSITIONAL/EN" or
3609            "-//W3O//DTD W3 HTML 3.0//EN//" => 1,              $pubid eq "HTML") {
           "-//W3O//DTD W3 HTML STRICT 3.0//EN//" => 1,  
           "-//WEBTECHS//DTD MOZILLA HTML 2.0//EN" => 1,  
           "-//WEBTECHS//DTD MOZILLA HTML//EN" => 1,  
           "-/W3C/DTD HTML 4.0 TRANSITIONAL/EN" => 1,  
           "HTML" => 1,  
         }->{$pubid}) {  
3610            !!!cp ('t5');            !!!cp ('t5');
3611            $self->{document}->manakai_compat_mode ('quirks');            $self->{document}->manakai_compat_mode ('quirks');
3612          } elsif ($pubid eq "-//W3C//DTD HTML 4.01 FRAMESET//EN" or          } elsif ($pubid =~ m[^-//W3C//DTD HTML 4.01 FRAMESET//] or
3613                   $pubid eq "-//W3C//DTD HTML 4.01 TRANSITIONAL//EN") {                   $pubid =~ m[^-//W3C//DTD HTML 4.01 TRANSITIONAL//]) {
3614            if (defined $token->{system_identifier}) {            if (defined $token->{sysid}) {
3615              !!!cp ('t6');              !!!cp ('t6');
3616              $self->{document}->manakai_compat_mode ('quirks');              $self->{document}->manakai_compat_mode ('quirks');
3617            } else {            } else {
3618              !!!cp ('t7');              !!!cp ('t7');
3619              $self->{document}->manakai_compat_mode ('limited quirks');              $self->{document}->manakai_compat_mode ('limited quirks');
3620            }            }
3621          } elsif ($pubid eq "-//W3C//DTD XHTML 1.0 FRAMESET//EN" or          } elsif ($pubid =~ m[^-//W3C//DTD XHTML 1.0 FRAMESET//] or
3622                   $pubid eq "-//W3C//DTD XHTML 1.0 TRANSITIONAL//EN") {                   $pubid =~ m[^-//W3C//DTD XHTML 1.0 TRANSITIONAL//]) {
3623            !!!cp ('t8');            !!!cp ('t8');
3624            $self->{document}->manakai_compat_mode ('limited quirks');            $self->{document}->manakai_compat_mode ('limited quirks');
3625          } else {          } else {
# Line 3238  sub _tree_construction_initial ($) { Line 3628  sub _tree_construction_initial ($) {
3628        } else {        } else {
3629          !!!cp ('t10');          !!!cp ('t10');
3630        }        }
3631        if (defined $token->{system_identifier}) {        if (defined $token->{sysid}) {
3632          my $sysid = $token->{system_identifier};          my $sysid = $token->{sysid};
3633          $sysid =~ tr/A-Z/a-z/;          $sysid =~ tr/A-Z/a-z/;
3634          if ($sysid eq "http://www.ibm.com/data/dtd/v11/ibmxhtml1-transitional.dtd") {          if ($sysid eq "http://www.ibm.com/data/dtd/v11/ibmxhtml1-transitional.dtd") {
3635            ## TODO: Check the spec: PUBLIC "(limited quirks)" "(quirks)"            ## NOTE: Ensure that |PUBLIC "(limited quirks)" "(quirks)"| is
3636              ## marked as quirks.
3637            $self->{document}->manakai_compat_mode ('quirks');            $self->{document}->manakai_compat_mode ('quirks');
3638            !!!cp ('t11');            !!!cp ('t11');
3639          } else {          } else {
# Line 3268  sub _tree_construction_initial ($) { Line 3659  sub _tree_construction_initial ($) {
3659        !!!ack-later;        !!!ack-later;
3660        return;        return;
3661      } elsif ($token->{type} == CHARACTER_TOKEN) {      } elsif ($token->{type} == CHARACTER_TOKEN) {
3662        if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D        if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
3663          ## Ignore the token          ## Ignore the token
3664    
3665          unless (length $token->{data}) {          unless (length $token->{data}) {
# Line 3325  sub _tree_construction_root_element ($) Line 3716  sub _tree_construction_root_element ($)
3716          !!!next-token;          !!!next-token;
3717          redo B;          redo B;
3718        } elsif ($token->{type} == CHARACTER_TOKEN) {        } elsif ($token->{type} == CHARACTER_TOKEN) {
3719          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D          if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
3720            ## Ignore the token.            ## Ignore the token.
3721    
3722            unless (length $token->{data}) {            unless (length $token->{data}) {
# Line 3392  sub _tree_construction_root_element ($) Line 3783  sub _tree_construction_root_element ($)
3783      ## NOTE: Reprocess the token.      ## NOTE: Reprocess the token.
3784      !!!ack-later;      !!!ack-later;
3785      return; ## Go to the "before head" insertion mode.      return; ## Go to the "before head" insertion mode.
   
     ## ISSUE: There is an issue in the spec  
3786    } # B    } # B
3787    
3788    die "$0: _tree_construction_root_element: This should never be reached";    die "$0: _tree_construction_root_element: This should never be reached";
# Line 3428  sub _reset_insertion_mode ($) { Line 3817  sub _reset_insertion_mode ($) {
3817          ## NOTE: Strictly spaking, the line below only applies to MathML and          ## NOTE: Strictly spaking, the line below only applies to MathML and
3818          ## SVG elements.  Currently the HTML syntax supports only MathML and          ## SVG elements.  Currently the HTML syntax supports only MathML and
3819          ## SVG elements as foreigners.          ## SVG elements as foreigners.
3820          $new_mode = $self->{insertion_mode} | IN_FOREIGN_CONTENT_IM;          $new_mode = IN_BODY_IM | IN_FOREIGN_CONTENT_IM;
         ## ISSUE: What is set as the secondary insertion mode?  
3821        } elsif ($node->[1] & TABLE_CELL_EL) {        } elsif ($node->[1] & TABLE_CELL_EL) {
3822          if ($last) {          if ($last) {
3823            !!!cp ('t28.2');            !!!cp ('t28.2');
# Line 3591  sub _tree_construction_main ($) { Line 3979  sub _tree_construction_main ($) {
3979    
3980      ## Step 1      ## Step 1
3981      my $start_tag_name = $token->{tag_name};      my $start_tag_name = $token->{tag_name};
3982      my $el;      !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
     !!!create-element ($el, $HTML_NS, $start_tag_name, $token->{attributes}, $token);  
3983    
3984      ## Step 2      ## Step 2
     $insert->($el);  
   
     ## Step 3  
3985      $self->{content_model} = $content_model_flag; # CDATA or RCDATA      $self->{content_model} = $content_model_flag; # CDATA or RCDATA
3986      delete $self->{escape}; # MUST      delete $self->{escape}; # MUST
3987    
3988      ## Step 4      ## Step 3, 4
3989      my $text = '';      $self->{insertion_mode} |= IN_CDATA_RCDATA_IM;
     !!!nack ('t40.1');  
     !!!next-token;  
     while ($token->{type} == CHARACTER_TOKEN) { # or until stop tokenizing  
       !!!cp ('t40');  
       $text .= $token->{data};  
       !!!next-token;  
     }  
   
     ## Step 5  
     if (length $text) {  
       !!!cp ('t41');  
       my $text = $self->{document}->create_text_node ($text);  
       $el->append_child ($text);  
     }  
   
     ## Step 6  
     $self->{content_model} = PCDATA_CONTENT_MODEL;  
3990    
3991      ## Step 7      !!!nack ('t40.1');
     if ($token->{type} == END_TAG_TOKEN and  
         $token->{tag_name} eq $start_tag_name) {  
       !!!cp ('t42');  
       ## Ignore the token  
     } else {  
       ## NOTE: An end-of-file token.  
       if ($content_model_flag == CDATA_CONTENT_MODEL) {  
         !!!cp ('t43');  
         !!!parse-error (type => 'in CDATA:#'.$token->{type}, token => $token);  
       } elsif ($content_model_flag == RCDATA_CONTENT_MODEL) {  
         !!!cp ('t44');  
         !!!parse-error (type => 'in RCDATA:#'.$token->{type}, token => $token);  
       } else {  
         die "$0: $content_model_flag in parse_rcdata";  
       }  
     }  
3992      !!!next-token;      !!!next-token;
3993    }; # $parse_rcdata    }; # $parse_rcdata
3994    
3995    my $script_start_tag = sub () {    my $script_start_tag = sub () {
3996        ## Step 1
3997      my $script_el;      my $script_el;
3998      !!!create-element ($script_el, $HTML_NS, 'script', $token->{attributes}, $token);      !!!create-element ($script_el, $HTML_NS, 'script', $token->{attributes}, $token);
3999    
4000        ## Step 2
4001      ## TODO: mark as "parser-inserted"      ## TODO: mark as "parser-inserted"
4002    
4003        ## Step 3
4004        ## TODO: Mark as "already executed", if ...
4005    
4006        ## Step 4
4007        $insert->($script_el);
4008    
4009        ## ISSUE: $script_el is not put into the stack
4010        push @{$self->{open_elements}}, [$script_el, $el_category->{script}];
4011    
4012        ## Step 5
4013      $self->{content_model} = CDATA_CONTENT_MODEL;      $self->{content_model} = CDATA_CONTENT_MODEL;
4014      delete $self->{escape}; # MUST      delete $self->{escape}; # MUST
       
     my $text = '';  
     !!!nack ('t45.1');  
     !!!next-token;  
     while ($token->{type} == CHARACTER_TOKEN) {  
       !!!cp ('t45');  
       $text .= $token->{data};  
       !!!next-token;  
     } # stop if non-character token or tokenizer stops tokenising  
     if (length $text) {  
       !!!cp ('t46');  
       $script_el->manakai_append_text ($text);  
     }  
                 
     $self->{content_model} = PCDATA_CONTENT_MODEL;  
4015    
4016      if ($token->{type} == END_TAG_TOKEN and      ## Step 6-7
4017          $token->{tag_name} eq 'script') {      $self->{insertion_mode} |= IN_CDATA_RCDATA_IM;
       !!!cp ('t47');  
       ## Ignore the token  
     } else {  
       !!!cp ('t48');  
       !!!parse-error (type => 'in CDATA:#'.$token->{type}, token => $token);  
       ## ISSUE: And ignore?  
       ## TODO: mark as "already executed"  
     }  
       
     if (defined $self->{inner_html_node}) {  
       !!!cp ('t49');  
       ## TODO: mark as "already executed"  
     } else {  
       !!!cp ('t50');  
       ## TODO: $old_insertion_point = current insertion point  
       ## TODO: insertion point = just before the next input character  
4018    
4019        $insert->($script_el);      !!!nack ('t40.2');
         
       ## TODO: insertion point = $old_insertion_point (might be "undefined")  
         
       ## TODO: if there is a script that will execute as soon as the parser resume, then...  
     }  
       
4020      !!!next-token;      !!!next-token;
4021    }; # $script_start_tag    }; # $script_start_tag
4022    
4023    ## NOTE: $open_tables->[-1]->[0] is the "current table" element node.    ## NOTE: $open_tables->[-1]->[0] is the "current table" element node.
4024    ## NOTE: $open_tables->[-1]->[1] is the "tainted" flag.    ## NOTE: $open_tables->[-1]->[1] is the "tainted" flag.
4025      ## NOTE: $open_tables->[-1]->[2] is set false when non-Text node inserted.
4026    my $open_tables = [[$self->{open_elements}->[0]->[0]]];    my $open_tables = [[$self->{open_elements}->[0]->[0]]];
4027    
4028    my $formatting_end_tag = sub {    my $formatting_end_tag = sub {
# Line 3721  sub _tree_construction_main ($) { Line 4049  sub _tree_construction_main ($) {
4049        } # AFE        } # AFE
4050        unless (defined $formatting_element) {        unless (defined $formatting_element) {
4051          !!!cp ('t53');          !!!cp ('t53');
4052          !!!parse-error (type => 'unmatched end tag:'.$tag_name, token => $end_tag_token);          !!!parse-error (type => 'unmatched end tag', text => $tag_name, token => $end_tag_token);
4053          ## Ignore the token          ## Ignore the token
4054          !!!next-token;          !!!next-token;
4055          return;          return;
# Line 3738  sub _tree_construction_main ($) { Line 4066  sub _tree_construction_main ($) {
4066              last INSCOPE;              last INSCOPE;
4067            } else { # in open elements but not in scope            } else { # in open elements but not in scope
4068              !!!cp ('t55');              !!!cp ('t55');
4069              !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name},              !!!parse-error (type => 'unmatched end tag',
4070                                text => $token->{tag_name},
4071                              token => $end_tag_token);                              token => $end_tag_token);
4072              ## Ignore the token              ## Ignore the token
4073              !!!next-token;              !!!next-token;
# Line 3751  sub _tree_construction_main ($) { Line 4080  sub _tree_construction_main ($) {
4080        } # INSCOPE        } # INSCOPE
4081        unless (defined $formatting_element_i_in_open) {        unless (defined $formatting_element_i_in_open) {
4082          !!!cp ('t57');          !!!cp ('t57');
4083          !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name},          !!!parse-error (type => 'unmatched end tag',
4084                            text => $token->{tag_name},
4085                          token => $end_tag_token);                          token => $end_tag_token);
4086          pop @$active_formatting_elements; # $formatting_element          pop @$active_formatting_elements; # $formatting_element
4087          !!!next-token; ## TODO: ok?          !!!next-token; ## TODO: ok?
# Line 3760  sub _tree_construction_main ($) { Line 4090  sub _tree_construction_main ($) {
4090        if (not $self->{open_elements}->[-1]->[0] eq $formatting_element->[0]) {        if (not $self->{open_elements}->[-1]->[0] eq $formatting_element->[0]) {
4091          !!!cp ('t58');          !!!cp ('t58');
4092          !!!parse-error (type => 'not closed',          !!!parse-error (type => 'not closed',
4093                          value => $self->{open_elements}->[-1]->[0]                          text => $self->{open_elements}->[-1]->[0]
4094                              ->manakai_local_name,                              ->manakai_local_name,
4095                          token => $end_tag_token);                          token => $end_tag_token);
4096        }        }
# Line 3777  sub _tree_construction_main ($) { Line 4107  sub _tree_construction_main ($) {
4107            !!!cp ('t59');            !!!cp ('t59');
4108            $furthest_block = $node;            $furthest_block = $node;
4109            $furthest_block_i_in_open = $_;            $furthest_block_i_in_open = $_;
4110              ## NOTE: The topmost (eldest) node.
4111          } elsif ($node->[0] eq $formatting_element->[0]) {          } elsif ($node->[0] eq $formatting_element->[0]) {
4112            !!!cp ('t60');            !!!cp ('t60');
4113            last OE;            last OE;
# Line 3923  sub _tree_construction_main ($) { Line 4254  sub _tree_construction_main ($) {
4254            $i = $_;            $i = $_;
4255          }          }
4256        } # OE        } # OE
4257        splice @{$self->{open_elements}}, $i + 1, 1, $clone;        splice @{$self->{open_elements}}, $i + 1, 0, $clone;
4258                
4259        ## Step 14        ## Step 14
4260        redo FET;        redo FET;
# Line 3966  sub _tree_construction_main ($) { Line 4297  sub _tree_construction_main ($) {
4297      }      }
4298    }; # $insert_to_foster    }; # $insert_to_foster
4299    
4300      ## NOTE: Insert a character (MUST): When a character is inserted, if
4301      ## the last node that was inserted by the parser is a Text node and
4302      ## the character has to be inserted after that node, then the
4303      ## character is appended to the Text node.  However, if any other
4304      ## node is inserted by the parser, then a new Text node is created
4305      ## and the character is appended as that Text node.  If I'm not
4306      ## wrong, for a parser with scripting disabled, there are only two
4307      ## cases where this occurs.  One is the case where an element node
4308      ## is inserted to the |head| element.  This is covered by using the
4309      ## |$self->{head_element_inserted}| flag.  Another is the case where
4310      ## an element or comment is inserted into the |table| subtree while
4311      ## foster parenting happens.  This is covered by using the [2] flag
4312      ## of the |$open_tables| structure.  All other cases are handled
4313      ## simply by calling |manakai_append_text| method.
4314    
4315      ## TODO: |<body><script>document.write("a<br>");
4316      ## document.body.removeChild (document.body.lastChild);
4317      ## document.write ("b")</script>|
4318    
4319    B: while (1) {    B: while (1) {
4320      if ($token->{type} == DOCTYPE_TOKEN) {      if ($token->{type} == DOCTYPE_TOKEN) {
4321        !!!cp ('t73');        !!!cp ('t73');
4322        !!!parse-error (type => 'DOCTYPE in the middle', token => $token);        !!!parse-error (type => 'in html:#DOCTYPE', token => $token);
4323        ## Ignore the token        ## Ignore the token
4324        ## Stay in the phase        ## Stay in the phase
4325        !!!next-token;        !!!next-token;
# Line 3978  sub _tree_construction_main ($) { Line 4328  sub _tree_construction_main ($) {
4328               $token->{tag_name} eq 'html') {               $token->{tag_name} eq 'html') {
4329        if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {        if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
4330          !!!cp ('t79');          !!!cp ('t79');
4331          !!!parse-error (type => 'after html:html', token => $token);          !!!parse-error (type => 'after html', text => 'html', token => $token);
4332          $self->{insertion_mode} = AFTER_BODY_IM;          $self->{insertion_mode} = AFTER_BODY_IM;
4333        } elsif ($self->{insertion_mode} == AFTER_HTML_FRAMESET_IM) {        } elsif ($self->{insertion_mode} == AFTER_HTML_FRAMESET_IM) {
4334          !!!cp ('t80');          !!!cp ('t80');
4335          !!!parse-error (type => 'after html:html', token => $token);          !!!parse-error (type => 'after html', text => 'html', token => $token);
4336          $self->{insertion_mode} = AFTER_FRAMESET_IM;          $self->{insertion_mode} = AFTER_FRAMESET_IM;
4337        } else {        } else {
4338          !!!cp ('t81');          !!!cp ('t81');
# Line 4013  sub _tree_construction_main ($) { Line 4363  sub _tree_construction_main ($) {
4363        } else {        } else {
4364          !!!cp ('t87');          !!!cp ('t87');
4365          $self->{open_elements}->[-1]->[0]->append_child ($comment);          $self->{open_elements}->[-1]->[0]->append_child ($comment);
4366            $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
4367        }        }
4368        !!!next-token;        !!!next-token;
4369        next B;        next B;
4370        } elsif ($self->{insertion_mode} & IN_CDATA_RCDATA_IM) {
4371          if ($token->{type} == CHARACTER_TOKEN) {
4372            $token->{data} =~ s/^\x0A// if $self->{ignore_newline};
4373            delete $self->{ignore_newline};
4374    
4375            if (length $token->{data}) {
4376              !!!cp ('t43');
4377              $self->{open_elements}->[-1]->[0]->manakai_append_text
4378                  ($token->{data});
4379            } else {
4380              !!!cp ('t43.1');
4381            }
4382            !!!next-token;
4383            next B;
4384          } elsif ($token->{type} == END_TAG_TOKEN) {
4385            delete $self->{ignore_newline};
4386    
4387            if ($token->{tag_name} eq 'script') {
4388              !!!cp ('t50');
4389              
4390              ## Para 1-2
4391              my $script = pop @{$self->{open_elements}};
4392              
4393              ## Para 3
4394              $self->{insertion_mode} &= ~ IN_CDATA_RCDATA_IM;
4395    
4396              ## Para 4
4397              ## TODO: $old_insertion_point = $current_insertion_point;
4398              ## TODO: $current_insertion_point = just before $self->{nc};
4399    
4400              ## Para 5
4401              ## TODO: Run the $script->[0].
4402    
4403              ## Para 6
4404              ## TODO: $current_insertion_point = $old_insertion_point;
4405    
4406              ## Para 7
4407              ## TODO: if ($pending_external_script) {
4408                ## TODO: ...
4409              ## TODO: }
4410    
4411              !!!next-token;
4412              next B;
4413            } else {
4414              !!!cp ('t42');
4415    
4416              pop @{$self->{open_elements}};
4417    
4418              $self->{insertion_mode} &= ~ IN_CDATA_RCDATA_IM;
4419              !!!next-token;
4420              next B;
4421            }
4422          } elsif ($token->{type} == END_OF_FILE_TOKEN) {
4423            delete $self->{ignore_newline};
4424    
4425            !!!cp ('t44');
4426            !!!parse-error (type => 'not closed',
4427                            text => $self->{open_elements}->[-1]->[0]
4428                                ->manakai_local_name,
4429                            token => $token);
4430    
4431            #if ($self->{open_elements}->[-1]->[1] & SCRIPT_EL) {
4432            #  ## TODO: Mark as "already executed"
4433            #}
4434    
4435            pop @{$self->{open_elements}};
4436    
4437            $self->{insertion_mode} &= ~ IN_CDATA_RCDATA_IM;
4438            ## Reprocess.
4439            next B;
4440          } else {
4441            die "$0: $token->{type}: In CDATA/RCDATA: Unknown token type";        
4442          }
4443      } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM) {      } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM) {
4444        if ($token->{type} == CHARACTER_TOKEN) {        if ($token->{type} == CHARACTER_TOKEN) {
4445          !!!cp ('t87.1');          !!!cp ('t87.1');
# Line 4033  sub _tree_construction_main ($) { Line 4457  sub _tree_construction_main ($) {
4457            #            #
4458          } elsif ({          } elsif ({
4459                    b => 1, big => 1, blockquote => 1, body => 1, br => 1,                    b => 1, big => 1, blockquote => 1, body => 1, br => 1,
4460                    center => 1, code => 1, dd => 1, div => 1, dl => 1, em => 1,                    center => 1, code => 1, dd => 1, div => 1, dl => 1, dt => 1,
4461                    embed => 1, font => 1, h1 => 1, h2 => 1, h3 => 1, ## No h4!                    em => 1, embed => 1, font => 1, h1 => 1, h2 => 1, h3 => 1,
4462                    h5 => 1, h6 => 1, head => 1, hr => 1, i => 1, img => 1,                    h4 => 1, h5 => 1, h6 => 1, head => 1, hr => 1, i => 1,
4463                    li => 1, menu => 1, meta => 1, nobr => 1, p => 1, pre => 1,                    img => 1, li => 1, listing => 1, menu => 1, meta => 1,
4464                    ruby => 1, s => 1, small => 1, span => 1, strong => 1,                    nobr => 1, ol => 1, p => 1, pre => 1, ruby => 1, s => 1,
4465                    sub => 1, sup => 1, table => 1, tt => 1, u => 1, ul => 1,                    small => 1, span => 1, strong => 1, strike => 1, sub => 1,
4466                    var => 1,                    sup => 1, table => 1, tt => 1, u => 1, ul => 1, var => 1,
4467                   }->{$token->{tag_name}}) {                   }->{$token->{tag_name}}) {
4468            !!!cp ('t87.2');            !!!cp ('t87.2');
4469            !!!parse-error (type => 'not closed',            !!!parse-error (type => 'not closed',
4470                            value => $self->{open_elements}->[-1]->[0]                            text => $self->{open_elements}->[-1]->[0]
4471                                ->manakai_local_name,                                ->manakai_local_name,
4472                            token => $token);                            token => $token);
4473    
# Line 4119  sub _tree_construction_main ($) { Line 4543  sub _tree_construction_main ($) {
4543          !!!cp ('t87.5');          !!!cp ('t87.5');
4544          #          #
4545        } elsif ($token->{type} == END_OF_FILE_TOKEN) {        } elsif ($token->{type} == END_OF_FILE_TOKEN) {
         ## NOTE: "using the rules for secondary insertion mode" then "continue"  
4546          !!!cp ('t87.6');          !!!cp ('t87.6');
4547          #          !!!parse-error (type => 'not closed',
4548          ## TODO: ...                          text => $self->{open_elements}->[-1]->[0]
4549                                ->manakai_local_name,
4550                            token => $token);
4551    
4552            pop @{$self->{open_elements}}
4553                while $self->{open_elements}->[-1]->[1] & FOREIGN_EL;
4554    
4555            ## NOTE: |<span><svg>| ... two parse errors, |<svg>| ... a parse error.
4556    
4557            $self->{insertion_mode} &= ~ IN_FOREIGN_CONTENT_IM;
4558            ## Reprocess.
4559            next B;
4560        } else {        } else {
4561          die "$0: $token->{type}: Unknown token type";                  die "$0: $token->{type}: Unknown token type";        
4562        }        }
# Line 4130  sub _tree_construction_main ($) { Line 4564  sub _tree_construction_main ($) {
4564    
4565      if ($self->{insertion_mode} & HEAD_IMS) {      if ($self->{insertion_mode} & HEAD_IMS) {
4566        if ($token->{type} == CHARACTER_TOKEN) {        if ($token->{type} == CHARACTER_TOKEN) {
4567          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {          if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
4568            unless ($self->{insertion_mode} == BEFORE_HEAD_IM) {            unless ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4569              !!!cp ('t88.2');              if ($self->{head_element_inserted}) {
4570              $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);                !!!cp ('t88.3');
4571                  $self->{open_elements}->[-1]->[0]->append_child
4572                    ($self->{document}->create_text_node ($1));
4573                  delete $self->{head_element_inserted};
4574                  ## NOTE: |</head> <link> |
4575                  #
4576                } else {
4577                  !!!cp ('t88.2');
4578                  $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
4579                  ## NOTE: |</head> &#x20;|
4580                  #
4581                }
4582            } else {            } else {
4583              !!!cp ('t88.1');              !!!cp ('t88.1');
4584              ## Ignore the token.              ## Ignore the token.
4585              !!!next-token;              #
             next B;  
4586            }            }
4587            unless (length $token->{data}) {            unless (length $token->{data}) {
4588              !!!cp ('t88');              !!!cp ('t88');
4589              !!!next-token;              !!!next-token;
4590              next B;              next B;
4591            }            }
4592    ## TODO: set $token->{column} appropriately
4593          }          }
4594    
4595          if ($self->{insertion_mode} == BEFORE_HEAD_IM) {          if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
# Line 4163  sub _tree_construction_main ($) { Line 4608  sub _tree_construction_main ($) {
4608            !!!cp ('t90');            !!!cp ('t90');
4609            ## As if </noscript>            ## As if </noscript>
4610            pop @{$self->{open_elements}};            pop @{$self->{open_elements}};
4611            !!!parse-error (type => 'in noscript:#character', token => $token);            !!!parse-error (type => 'in noscript:#text', token => $token);
4612                        
4613            ## Reprocess in the "in head" insertion mode...            ## Reprocess in the "in head" insertion mode...
4614            ## As if </head>            ## As if </head>
# Line 4200  sub _tree_construction_main ($) { Line 4645  sub _tree_construction_main ($) {
4645              next B;              next B;
4646            } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {            } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4647              !!!cp ('t93.2');              !!!cp ('t93.2');
4648              !!!parse-error (type => 'after head:head', token => $token); ## TODO: error type              !!!parse-error (type => 'after head', text => 'head',
4649                                token => $token);
4650              ## Ignore the token              ## Ignore the token
4651              !!!nack ('t93.3');              !!!nack ('t93.3');
4652              !!!next-token;              !!!next-token;
4653              next B;              next B;
4654            } else {            } else {
4655              !!!cp ('t95');              !!!cp ('t95');
4656              !!!parse-error (type => 'in head:head', token => $token); # or in head noscript              !!!parse-error (type => 'in head:head',
4657                                token => $token); # or in head noscript
4658              ## Ignore the token              ## Ignore the token
4659              !!!nack ('t95.1');              !!!nack ('t95.1');
4660              !!!next-token;              !!!next-token;
# Line 4227  sub _tree_construction_main ($) { Line 4674  sub _tree_construction_main ($) {
4674            !!!cp ('t97');            !!!cp ('t97');
4675          }          }
4676    
4677              if ($token->{tag_name} eq 'base') {          if ($token->{tag_name} eq 'base') {
4678                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {            if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4679                  !!!cp ('t98');              !!!cp ('t98');
4680                  ## As if </noscript>              ## As if </noscript>
4681                  pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
4682                  !!!parse-error (type => 'in noscript:base', token => $token);              !!!parse-error (type => 'in noscript', text => 'base',
4683                                              token => $token);
4684                  $self->{insertion_mode} = IN_HEAD_IM;            
4685                  ## Reprocess in the "in head" insertion mode...              $self->{insertion_mode} = IN_HEAD_IM;
4686                } else {              ## Reprocess in the "in head" insertion mode...
4687                  !!!cp ('t99');            } else {
4688                }              !!!cp ('t99');
4689              }
4690    
4691                ## NOTE: There is a "as if in head" code clone.            ## NOTE: There is a "as if in head" code clone.
4692                if ($self->{insertion_mode} == AFTER_HEAD_IM) {            if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4693                  !!!cp ('t100');              !!!cp ('t100');
4694                  !!!parse-error (type => 'after head:'.$token->{tag_name}, token => $token);              !!!parse-error (type => 'after head',
4695                  push @{$self->{open_elements}},                              text => $token->{tag_name}, token => $token);
4696                      [$self->{head_element}, $el_category->{head}];              push @{$self->{open_elements}},
4697                } else {                  [$self->{head_element}, $el_category->{head}];
4698                  !!!cp ('t101');              $self->{head_element_inserted} = 1;
4699                }            } else {
4700                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);              !!!cp ('t101');
4701                pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.            }
4702                pop @{$self->{open_elements}} # <head>            !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4703                    if $self->{insertion_mode} == AFTER_HEAD_IM;            pop @{$self->{open_elements}};
4704                !!!nack ('t101.1');            pop @{$self->{open_elements}} # <head>
4705                !!!next-token;                if $self->{insertion_mode} == AFTER_HEAD_IM;
4706                next B;            !!!nack ('t101.1');
4707              } elsif ($token->{tag_name} eq 'link') {            !!!next-token;
4708                ## NOTE: There is a "as if in head" code clone.            next B;
4709                if ($self->{insertion_mode} == AFTER_HEAD_IM) {          } elsif ($token->{tag_name} eq 'link') {
4710                  !!!cp ('t102');            ## NOTE: There is a "as if in head" code clone.
4711                  !!!parse-error (type => 'after head:'.$token->{tag_name}, token => $token);            if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4712                  push @{$self->{open_elements}},              !!!cp ('t102');
4713                      [$self->{head_element}, $el_category->{head}];              !!!parse-error (type => 'after head',
4714                } else {                              text => $token->{tag_name}, token => $token);
4715                  !!!cp ('t103');              push @{$self->{open_elements}},
4716                }                  [$self->{head_element}, $el_category->{head}];
4717                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);              $self->{head_element_inserted} = 1;
4718                pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.            } else {
4719                pop @{$self->{open_elements}} # <head>              !!!cp ('t103');
4720                    if $self->{insertion_mode} == AFTER_HEAD_IM;            }
4721                !!!ack ('t103.1');            !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4722                !!!next-token;            pop @{$self->{open_elements}};
4723                next B;            pop @{$self->{open_elements}} # <head>
4724              } elsif ($token->{tag_name} eq 'meta') {                if $self->{insertion_mode} == AFTER_HEAD_IM;
4725                ## NOTE: There is a "as if in head" code clone.            !!!ack ('t103.1');
4726                if ($self->{insertion_mode} == AFTER_HEAD_IM) {            !!!next-token;
4727                  !!!cp ('t104');            next B;
4728                  !!!parse-error (type => 'after head:'.$token->{tag_name}, token => $token);          } elsif ($token->{tag_name} eq 'command' or
4729                  push @{$self->{open_elements}},                   $token->{tag_name} eq 'eventsource') {
4730                      [$self->{head_element}, $el_category->{head}];            if ($self->{insertion_mode} == IN_HEAD_IM) {
4731                } else {              ## NOTE: If the insertion mode at the time of the emission
4732                  !!!cp ('t105');              ## of the token was "before head", $self->{insertion_mode}
4733                }              ## is already changed to |IN_HEAD_IM|.
4734                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);  
4735                my $meta_el = pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.              ## NOTE: There is a "as if in head" code clone.
4736                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4737                pop @{$self->{open_elements}};
4738                pop @{$self->{open_elements}} # <head>
4739                    if $self->{insertion_mode} == AFTER_HEAD_IM;
4740                !!!ack ('t103.2');
4741                !!!next-token;
4742                next B;
4743              } else {
4744                ## NOTE: "in head noscript" or "after head" insertion mode
4745                ## - in these cases, these tags are treated as same as
4746                ## normal in-body tags.
4747                !!!cp ('t103.3');
4748                #
4749              }
4750            } elsif ($token->{tag_name} eq 'meta') {
4751              ## NOTE: There is a "as if in head" code clone.
4752              if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4753                !!!cp ('t104');
4754                !!!parse-error (type => 'after head',
4755                                text => $token->{tag_name}, token => $token);
4756                push @{$self->{open_elements}},
4757                    [$self->{head_element}, $el_category->{head}];
4758                $self->{head_element_inserted} = 1;
4759              } else {
4760                !!!cp ('t105');
4761              }
4762              !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4763              my $meta_el = pop @{$self->{open_elements}};
4764    
4765                unless ($self->{confident}) {                unless ($self->{confident}) {
4766                  if ($token->{attributes}->{charset}) {                  if ($token->{attributes}->{charset}) {
# Line 4301  sub _tree_construction_main ($) { Line 4777  sub _tree_construction_main ($) {
4777                                                 ->{has_reference});                                                 ->{has_reference});
4778                  } elsif ($token->{attributes}->{content}) {                  } elsif ($token->{attributes}->{content}) {
4779                    if ($token->{attributes}->{content}->{value}                    if ($token->{attributes}->{content}->{value}
4780                        =~ /\A[^;]*;[\x09-\x0D\x20]*[Cc][Hh][Aa][Rr][Ss][Ee][Tt]                        =~ /[Cc][Hh][Aa][Rr][Ss][Ee][Tt]
4781                            [\x09-\x0D\x20]*=                            [\x09\x0A\x0C\x0D\x20]*=
4782                            [\x09-\x0D\x20]*(?>"([^"]*)"|'([^']*)'|                            [\x09\x0A\x0C\x0D\x20]*(?>"([^"]*)"|'([^']*)'|
4783                            ([^"'\x09-\x0D\x20][^\x09-\x0D\x20]*))/x) {                            ([^"'\x09\x0A\x0C\x0D\x20]
4784                               [^\x09\x0A\x0C\x0D\x20\x3B]*))/x) {
4785                      !!!cp ('t107');                      !!!cp ('t107');
4786                      ## NOTE: Whether the encoding is supported or not is handled                      ## NOTE: Whether the encoding is supported or not is handled
4787                      ## in the {change_encoding} callback.                      ## in the {change_encoding} callback.
# Line 4341  sub _tree_construction_main ($) { Line 4818  sub _tree_construction_main ($) {
4818                !!!ack ('t110.1');                !!!ack ('t110.1');
4819                !!!next-token;                !!!next-token;
4820                next B;                next B;
4821              } elsif ($token->{tag_name} eq 'title') {          } elsif ($token->{tag_name} eq 'title') {
4822                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {            if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4823                  !!!cp ('t111');              !!!cp ('t111');
4824                  ## As if </noscript>              ## As if </noscript>
4825                  pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
4826                  !!!parse-error (type => 'in noscript:title', token => $token);              !!!parse-error (type => 'in noscript', text => 'title',
4827                                              token => $token);
4828                  $self->{insertion_mode} = IN_HEAD_IM;            
4829                  ## Reprocess in the "in head" insertion mode...              $self->{insertion_mode} = IN_HEAD_IM;
4830                } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {              ## Reprocess in the "in head" insertion mode...
4831                  !!!cp ('t112');            } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4832                  !!!parse-error (type => 'after head:'.$token->{tag_name}, token => $token);              !!!cp ('t112');
4833                  push @{$self->{open_elements}},              !!!parse-error (type => 'after head',
4834                      [$self->{head_element}, $el_category->{head}];                              text => $token->{tag_name}, token => $token);
4835                } else {              push @{$self->{open_elements}},
4836                  !!!cp ('t113');                  [$self->{head_element}, $el_category->{head}];
4837                }              $self->{head_element_inserted} = 1;
4838              } else {
4839                !!!cp ('t113');
4840              }
4841    
4842                ## NOTE: There is a "as if in head" code clone.            ## NOTE: There is a "as if in head" code clone.
4843                my $parent = defined $self->{head_element} ? $self->{head_element}            $parse_rcdata->(RCDATA_CONTENT_MODEL);
4844                    : $self->{open_elements}->[-1]->[0];            ## ISSUE: A spec bug [Bug 6038]
4845                $parse_rcdata->(RCDATA_CONTENT_MODEL);            splice @{$self->{open_elements}}, -2, 1, () # <head>
4846                pop @{$self->{open_elements}} # <head>                if ($self->{insertion_mode} & AFTER_HEAD_IM) == AFTER_HEAD_IM;
4847                    if $self->{insertion_mode} == AFTER_HEAD_IM;            next B;
4848                next B;          } elsif ($token->{tag_name} eq 'style' or
4849              } elsif ($token->{tag_name} eq 'style') {                   $token->{tag_name} eq 'noframes') {
4850                ## NOTE: Or (scripting is enabled and tag_name eq 'noscript' and            ## NOTE: Or (scripting is enabled and tag_name eq 'noscript' and
4851                ## insertion mode IN_HEAD_IM)            ## insertion mode IN_HEAD_IM)
4852                ## NOTE: There is a "as if in head" code clone.            ## NOTE: There is a "as if in head" code clone.
4853                if ($self->{insertion_mode} == AFTER_HEAD_IM) {            if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4854                  !!!cp ('t114');              !!!cp ('t114');
4855                  !!!parse-error (type => 'after head:'.$token->{tag_name}, token => $token);              !!!parse-error (type => 'after head',
4856                  push @{$self->{open_elements}},                              text => $token->{tag_name}, token => $token);
4857                      [$self->{head_element}, $el_category->{head}];              push @{$self->{open_elements}},
4858                } else {                  [$self->{head_element}, $el_category->{head}];
4859                  !!!cp ('t115');              $self->{head_element_inserted} = 1;
4860                }            } else {
4861                $parse_rcdata->(CDATA_CONTENT_MODEL);              !!!cp ('t115');
4862                pop @{$self->{open_elements}} # <head>            }
4863                    if $self->{insertion_mode} == AFTER_HEAD_IM;            $parse_rcdata->(CDATA_CONTENT_MODEL);
4864                next B;            ## ISSUE: A spec bug [Bug 6038]
4865              } elsif ($token->{tag_name} eq 'noscript') {            splice @{$self->{open_elements}}, -2, 1, () # <head>
4866                  if ($self->{insertion_mode} & AFTER_HEAD_IM) == AFTER_HEAD_IM;
4867              next B;
4868            } elsif ($token->{tag_name} eq 'noscript') {
4869                if ($self->{insertion_mode} == IN_HEAD_IM) {                if ($self->{insertion_mode} == IN_HEAD_IM) {
4870                  !!!cp ('t116');                  !!!cp ('t116');
4871                  ## NOTE: and scripting is disalbed                  ## NOTE: and scripting is disalbed
# Line 4393  sub _tree_construction_main ($) { Line 4876  sub _tree_construction_main ($) {
4876                  next B;                  next B;
4877                } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {                } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4878                  !!!cp ('t117');                  !!!cp ('t117');
4879                  !!!parse-error (type => 'in noscript:noscript', token => $token);                  !!!parse-error (type => 'in noscript', text => 'noscript',
4880                                    token => $token);
4881                  ## Ignore the token                  ## Ignore the token
4882                  !!!nack ('t117.1');                  !!!nack ('t117.1');
4883                  !!!next-token;                  !!!next-token;
# Line 4402  sub _tree_construction_main ($) { Line 4886  sub _tree_construction_main ($) {
4886                  !!!cp ('t118');                  !!!cp ('t118');
4887                  #                  #
4888                }                }
4889              } elsif ($token->{tag_name} eq 'script') {          } elsif ($token->{tag_name} eq 'script') {
4890                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {            if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4891                  !!!cp ('t119');              !!!cp ('t119');
4892                  ## As if </noscript>              ## As if </noscript>
4893                  pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
4894                  !!!parse-error (type => 'in noscript:script', token => $token);              !!!parse-error (type => 'in noscript', text => 'script',
4895                                              token => $token);
4896                  $self->{insertion_mode} = IN_HEAD_IM;            
4897                  ## Reprocess in the "in head" insertion mode...              $self->{insertion_mode} = IN_HEAD_IM;
4898                } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {              ## Reprocess in the "in head" insertion mode...
4899                  !!!cp ('t120');            } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4900                  !!!parse-error (type => 'after head:'.$token->{tag_name}, token => $token);              !!!cp ('t120');
4901                  push @{$self->{open_elements}},              !!!parse-error (type => 'after head',
4902                      [$self->{head_element}, $el_category->{head}];                              text => $token->{tag_name}, token => $token);
4903                } else {              push @{$self->{open_elements}},
4904                  !!!cp ('t121');                  [$self->{head_element}, $el_category->{head}];
4905                }              $self->{head_element_inserted} = 1;
4906              } else {
4907                !!!cp ('t121');
4908              }
4909    
4910                ## NOTE: There is a "as if in head" code clone.            ## NOTE: There is a "as if in head" code clone.
4911                $script_start_tag->();            $script_start_tag->();
4912                pop @{$self->{open_elements}} # <head>            ## ISSUE: A spec bug  [Bug 6038]
4913                    if $self->{insertion_mode} == AFTER_HEAD_IM;            splice @{$self->{open_elements}}, -2, 1 # <head>
4914                next B;                if ($self->{insertion_mode} & AFTER_HEAD_IM) == AFTER_HEAD_IM;
4915              } elsif ($token->{tag_name} eq 'body' or            next B;
4916                       $token->{tag_name} eq 'frameset') {          } elsif ($token->{tag_name} eq 'body' or
4917                     $token->{tag_name} eq 'frameset') {
4918                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4919                  !!!cp ('t122');                  !!!cp ('t122');
4920                  ## As if </noscript>                  ## As if </noscript>
4921                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
4922                  !!!parse-error (type => 'in noscript:'.$token->{tag_name}, token => $token);                  !!!parse-error (type => 'in noscript',
4923                                    text => $token->{tag_name}, token => $token);
4924                                    
4925                  ## Reprocess in the "in head" insertion mode...                  ## Reprocess in the "in head" insertion mode...
4926                  ## As if </head>                  ## As if </head>
# Line 4470  sub _tree_construction_main ($) { Line 4959  sub _tree_construction_main ($) {
4959                !!!cp ('t129');                !!!cp ('t129');
4960                ## As if </noscript>                ## As if </noscript>
4961                pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
4962                !!!parse-error (type => 'in noscript:/'.$token->{tag_name}, token => $token);                !!!parse-error (type => 'in noscript:/',
4963                                  text => $token->{tag_name}, token => $token);
4964                                
4965                ## Reprocess in the "in head" insertion mode...                ## Reprocess in the "in head" insertion mode...
4966                ## As if </head>                ## As if </head>
# Line 4513  sub _tree_construction_main ($) { Line 5003  sub _tree_construction_main ($) {
5003                  !!!cp ('t133');                  !!!cp ('t133');
5004                  ## As if </noscript>                  ## As if </noscript>
5005                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
5006                  !!!parse-error (type => 'in noscript:/head', token => $token);                  !!!parse-error (type => 'in noscript:/',
5007                                    text => 'head', token => $token);
5008                                    
5009                  ## Reprocess in the "in head" insertion mode...                  ## Reprocess in the "in head" insertion mode...
5010                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
# Line 4528  sub _tree_construction_main ($) { Line 5019  sub _tree_construction_main ($) {
5019                  next B;                  next B;
5020                } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {                } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
5021                  !!!cp ('t134.1');                  !!!cp ('t134.1');
5022                  !!!parse-error (type => 'unmatched end tag:head', token => $token);                  !!!parse-error (type => 'unmatched end tag', text => 'head',
5023                                    token => $token);
5024                  ## Ignore the token                  ## Ignore the token
5025                  !!!next-token;                  !!!next-token;
5026                  next B;                  next B;
# Line 4545  sub _tree_construction_main ($) { Line 5037  sub _tree_construction_main ($) {
5037                } elsif ($self->{insertion_mode} == BEFORE_HEAD_IM or                } elsif ($self->{insertion_mode} == BEFORE_HEAD_IM or
5038                         $self->{insertion_mode} == AFTER_HEAD_IM) {                         $self->{insertion_mode} == AFTER_HEAD_IM) {
5039                  !!!cp ('t137');                  !!!cp ('t137');
5040                  !!!parse-error (type => 'unmatched end tag:noscript', token => $token);                  !!!parse-error (type => 'unmatched end tag',
5041                                    text => 'noscript', token => $token);
5042                  ## Ignore the token ## ISSUE: An issue in the spec.                  ## Ignore the token ## ISSUE: An issue in the spec.
5043                  !!!next-token;                  !!!next-token;
5044                  next B;                  next B;
# Line 4556  sub _tree_construction_main ($) { Line 5049  sub _tree_construction_main ($) {
5049              } elsif ({              } elsif ({
5050                        body => 1, html => 1,                        body => 1, html => 1,
5051                       }->{$token->{tag_name}}) {                       }->{$token->{tag_name}}) {
5052                if ($self->{insertion_mode} == BEFORE_HEAD_IM or                ## TODO: This branch is entirely redundant.
5053                  if ($self->{insertion_mode} == BEFORE_HEAD_IM or
5054                    $self->{insertion_mode} == IN_HEAD_IM or                    $self->{insertion_mode} == IN_HEAD_IM or
5055                    $self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {                    $self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
5056                  !!!cp ('t140');                  !!!cp ('t140');
5057                  !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);                  !!!parse-error (type => 'unmatched end tag',
5058                                    text => $token->{tag_name}, token => $token);
5059                  ## Ignore the token                  ## Ignore the token
5060                  !!!next-token;                  !!!next-token;
5061                  next B;                  next B;
5062                } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {                } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
5063                  !!!cp ('t140.1');                  !!!cp ('t140.1');
5064                  !!!parse-error (type => 'unmatched end tag:' . $token->{tag_name}, token => $token);                  !!!parse-error (type => 'unmatched end tag',
5065                                    text => $token->{tag_name}, token => $token);
5066                  ## Ignore the token                  ## Ignore the token
5067                  !!!next-token;                  !!!next-token;
5068                  next B;                  next B;
# Line 4575  sub _tree_construction_main ($) { Line 5071  sub _tree_construction_main ($) {
5071                }                }
5072              } elsif ($token->{tag_name} eq 'p') {              } elsif ($token->{tag_name} eq 'p') {
5073                !!!cp ('t142');                !!!cp ('t142');
5074                !!!parse-error (type => 'unmatched end tag:p', token => $token);                !!!parse-error (type => 'unmatched end tag',
5075                                  text => $token->{tag_name}, token => $token);
5076                ## Ignore the token                ## Ignore the token
5077                !!!next-token;                !!!next-token;
5078                next B;                next B;
# Line 4598  sub _tree_construction_main ($) { Line 5095  sub _tree_construction_main ($) {
5095                } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {                } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
5096                  !!!cp ('t143.3');                  !!!cp ('t143.3');
5097                  ## ISSUE: Two parse errors for <head><noscript></br>                  ## ISSUE: Two parse errors for <head><noscript></br>
5098                  !!!parse-error (type => 'unmatched end tag:br', token => $token);                  !!!parse-error (type => 'unmatched end tag',
5099                                    text => 'br', token => $token);
5100                  ## As if </noscript>                  ## As if </noscript>
5101                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
5102                  $self->{insertion_mode} = IN_HEAD_IM;                  $self->{insertion_mode} = IN_HEAD_IM;
# Line 4617  sub _tree_construction_main ($) { Line 5115  sub _tree_construction_main ($) {
5115                }                }
5116    
5117                ## ISSUE: does not agree with IE7 - it doesn't ignore </br>.                ## ISSUE: does not agree with IE7 - it doesn't ignore </br>.
5118                !!!parse-error (type => 'unmatched end tag:br', token => $token);                !!!parse-error (type => 'unmatched end tag',
5119                                  text => 'br', token => $token);
5120                ## Ignore the token                ## Ignore the token
5121                !!!next-token;                !!!next-token;
5122                next B;                next B;
5123              } else {              } else {
5124                !!!cp ('t145');                !!!cp ('t145');
5125                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);                !!!parse-error (type => 'unmatched end tag',
5126                                  text => $token->{tag_name}, token => $token);
5127                ## Ignore the token                ## Ignore the token
5128                !!!next-token;                !!!next-token;
5129                next B;                next B;
# Line 4633  sub _tree_construction_main ($) { Line 5133  sub _tree_construction_main ($) {
5133                !!!cp ('t146');                !!!cp ('t146');
5134                ## As if </noscript>                ## As if </noscript>
5135                pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
5136                !!!parse-error (type => 'in noscript:/'.$token->{tag_name}, token => $token);                !!!parse-error (type => 'in noscript:/',
5137                                  text => $token->{tag_name}, token => $token);
5138                                
5139                ## Reprocess in the "in head" insertion mode...                ## Reprocess in the "in head" insertion mode...
5140                ## As if </head>                ## As if </head>
# Line 4649  sub _tree_construction_main ($) { Line 5150  sub _tree_construction_main ($) {
5150              } elsif ($self->{insertion_mode} == BEFORE_HEAD_IM) {              } elsif ($self->{insertion_mode} == BEFORE_HEAD_IM) {
5151  ## ISSUE: This case cannot be reached?  ## ISSUE: This case cannot be reached?
5152                !!!cp ('t148');                !!!cp ('t148');
5153                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);                !!!parse-error (type => 'unmatched end tag',
5154                                  text => $token->{tag_name}, token => $token);
5155                ## Ignore the token ## ISSUE: An issue in the spec.                ## Ignore the token ## ISSUE: An issue in the spec.
5156                !!!next-token;                !!!next-token;
5157                next B;                next B;
# Line 4720  sub _tree_construction_main ($) { Line 5222  sub _tree_construction_main ($) {
5222        } else {        } else {
5223          die "$0: $token->{type}: Unknown token type";          die "$0: $token->{type}: Unknown token type";
5224        }        }
   
           ## ISSUE: An issue in the spec.  
5225      } elsif ($self->{insertion_mode} & BODY_IMS) {      } elsif ($self->{insertion_mode} & BODY_IMS) {
5226            if ($token->{type} == CHARACTER_TOKEN) {            if ($token->{type} == CHARACTER_TOKEN) {
5227              !!!cp ('t150');              !!!cp ('t150');
# Line 4760  sub _tree_construction_main ($) { Line 5260  sub _tree_construction_main ($) {
5260    
5261                  !!!cp ('t153');                  !!!cp ('t153');
5262                  !!!parse-error (type => 'start tag not allowed',                  !!!parse-error (type => 'start tag not allowed',
5263                      value => $token->{tag_name}, token => $token);                      text => $token->{tag_name}, token => $token);
5264                  ## Ignore the token                  ## Ignore the token
5265                  !!!nack ('t153.1');                  !!!nack ('t153.1');
5266                  !!!next-token;                  !!!next-token;
5267                  next B;                  next B;
5268                } elsif ($self->{insertion_mode} == IN_CAPTION_IM) {                } elsif ($self->{insertion_mode} == IN_CAPTION_IM) {
5269                  !!!parse-error (type => 'not closed:caption', token => $token);                  !!!parse-error (type => 'not closed', text => 'caption',
5270                                    token => $token);
5271                                    
5272                  ## NOTE: As if </caption>.                  ## NOTE: As if </caption>.
5273                  ## have a table element in table scope                  ## have a table element in table scope
# Line 4786  sub _tree_construction_main ($) { Line 5287  sub _tree_construction_main ($) {
5287    
5288                    !!!cp ('t157');                    !!!cp ('t157');
5289                    !!!parse-error (type => 'start tag not allowed',                    !!!parse-error (type => 'start tag not allowed',
5290                                    value => $token->{tag_name}, token => $token);                                    text => $token->{tag_name}, token => $token);
5291                    ## Ignore the token                    ## Ignore the token
5292                    !!!nack ('t157.1');                    !!!nack ('t157.1');
5293                    !!!next-token;                    !!!next-token;
# Line 4803  sub _tree_construction_main ($) { Line 5304  sub _tree_construction_main ($) {
5304                  unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {                  unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {
5305                    !!!cp ('t159');                    !!!cp ('t159');
5306                    !!!parse-error (type => 'not closed',                    !!!parse-error (type => 'not closed',
5307                                    value => $self->{open_elements}->[-1]->[0]                                    text => $self->{open_elements}->[-1]->[0]
5308                                        ->manakai_local_name,                                        ->manakai_local_name,
5309                                    token => $token);                                    token => $token);
5310                  } else {                  } else {
# Line 4845  sub _tree_construction_main ($) { Line 5346  sub _tree_construction_main ($) {
5346                  } # INSCOPE                  } # INSCOPE
5347                    unless (defined $i) {                    unless (defined $i) {
5348                      !!!cp ('t165');                      !!!cp ('t165');
5349                      !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);                      !!!parse-error (type => 'unmatched end tag',
5350                                        text => $token->{tag_name},
5351                                        token => $token);
5352                      ## Ignore the token                      ## Ignore the token
5353                      !!!next-token;                      !!!next-token;
5354                      next B;                      next B;
# Line 4862  sub _tree_construction_main ($) { Line 5365  sub _tree_construction_main ($) {
5365                          ne $token->{tag_name}) {                          ne $token->{tag_name}) {
5366                    !!!cp ('t167');                    !!!cp ('t167');
5367                    !!!parse-error (type => 'not closed',                    !!!parse-error (type => 'not closed',
5368                                    value => $self->{open_elements}->[-1]->[0]                                    text => $self->{open_elements}->[-1]->[0]
5369                                        ->manakai_local_name,                                        ->manakai_local_name,
5370                                    token => $token);                                    token => $token);
5371                  } else {                  } else {
# Line 4879  sub _tree_construction_main ($) { Line 5382  sub _tree_construction_main ($) {
5382                  next B;                  next B;
5383                } elsif ($self->{insertion_mode} == IN_CAPTION_IM) {                } elsif ($self->{insertion_mode} == IN_CAPTION_IM) {
5384                  !!!cp ('t169');                  !!!cp ('t169');
5385                  !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);                  !!!parse-error (type => 'unmatched end tag',
5386                                    text => $token->{tag_name}, token => $token);
5387                  ## Ignore the token                  ## Ignore the token
5388                  !!!next-token;                  !!!next-token;
5389                  next B;                  next B;
# Line 4906  sub _tree_construction_main ($) { Line 5410  sub _tree_construction_main ($) {
5410    
5411                    !!!cp ('t173');                    !!!cp ('t173');
5412                    !!!parse-error (type => 'unmatched end tag',                    !!!parse-error (type => 'unmatched end tag',
5413                                    value => $token->{tag_name}, token => $token);                                    text => $token->{tag_name}, token => $token);
5414                    ## Ignore the token                    ## Ignore the token
5415                    !!!next-token;                    !!!next-token;
5416                    next B;                    next B;
# Line 4922  sub _tree_construction_main ($) { Line 5426  sub _tree_construction_main ($) {
5426                  unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {                  unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {
5427                    !!!cp ('t175');                    !!!cp ('t175');
5428                    !!!parse-error (type => 'not closed',                    !!!parse-error (type => 'not closed',
5429                                    value => $self->{open_elements}->[-1]->[0]                                    text => $self->{open_elements}->[-1]->[0]
5430                                        ->manakai_local_name,                                        ->manakai_local_name,
5431                                    token => $token);                                    token => $token);
5432                  } else {                  } else {
# Line 4939  sub _tree_construction_main ($) { Line 5443  sub _tree_construction_main ($) {
5443                  next B;                  next B;
5444                } elsif ($self->{insertion_mode} == IN_CELL_IM) {                } elsif ($self->{insertion_mode} == IN_CELL_IM) {
5445                  !!!cp ('t177');                  !!!cp ('t177');
5446                  !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);                  !!!parse-error (type => 'unmatched end tag',
5447                                    text => $token->{tag_name}, token => $token);
5448                  ## Ignore the token                  ## Ignore the token
5449                  !!!next-token;                  !!!next-token;
5450                  next B;                  next B;
# Line 4982  sub _tree_construction_main ($) { Line 5487  sub _tree_construction_main ($) {
5487    
5488                  !!!cp ('t182');                  !!!cp ('t182');
5489                  !!!parse-error (type => 'unmatched end tag',                  !!!parse-error (type => 'unmatched end tag',
5490                      value => $token->{tag_name}, token => $token);                      text => $token->{tag_name}, token => $token);
5491                  ## Ignore the token                  ## Ignore the token
5492                  !!!next-token;                  !!!next-token;
5493                  next B;                  next B;
5494                } # INSCOPE                } # INSCOPE
5495              } elsif ($token->{tag_name} eq 'table' and              } elsif ($token->{tag_name} eq 'table' and
5496                       $self->{insertion_mode} == IN_CAPTION_IM) {                       $self->{insertion_mode} == IN_CAPTION_IM) {
5497                !!!parse-error (type => 'not closed:caption', token => $token);                !!!parse-error (type => 'not closed', text => 'caption',
5498                                  token => $token);
5499    
5500                ## As if </caption>                ## As if </caption>
5501                ## have a table element in table scope                ## have a table element in table scope
# Line 5007  sub _tree_construction_main ($) { Line 5513  sub _tree_construction_main ($) {
5513                } # INSCOPE                } # INSCOPE
5514                unless (defined $i) {                unless (defined $i) {
5515                  !!!cp ('t186');                  !!!cp ('t186');
5516                  !!!parse-error (type => 'unmatched end tag:caption', token => $token);                  !!!parse-error (type => 'unmatched end tag',
5517                                    text => 'caption', token => $token);
5518                  ## Ignore the token                  ## Ignore the token
5519                  !!!next-token;                  !!!next-token;
5520                  next B;                  next B;
# Line 5022  sub _tree_construction_main ($) { Line 5529  sub _tree_construction_main ($) {
5529                unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {                unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {
5530                  !!!cp ('t188');                  !!!cp ('t188');
5531                  !!!parse-error (type => 'not closed',                  !!!parse-error (type => 'not closed',
5532                                  value => $self->{open_elements}->[-1]->[0]                                  text => $self->{open_elements}->[-1]->[0]
5533                                      ->manakai_local_name,                                      ->manakai_local_name,
5534                                  token => $token);                                  token => $token);
5535                } else {                } else {
# Line 5042  sub _tree_construction_main ($) { Line 5549  sub _tree_construction_main ($) {
5549                       }->{$token->{tag_name}}) {                       }->{$token->{tag_name}}) {
5550                if ($self->{insertion_mode} & BODY_TABLE_IMS) {                if ($self->{insertion_mode} & BODY_TABLE_IMS) {
5551                  !!!cp ('t190');                  !!!cp ('t190');
5552                  !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);                  !!!parse-error (type => 'unmatched end tag',
5553                                    text => $token->{tag_name}, token => $token);
5554                  ## Ignore the token                  ## Ignore the token
5555                  !!!next-token;                  !!!next-token;
5556                  next B;                  next B;
# Line 5056  sub _tree_construction_main ($) { Line 5564  sub _tree_construction_main ($) {
5564                       }->{$token->{tag_name}} and                       }->{$token->{tag_name}} and
5565                       $self->{insertion_mode} == IN_CAPTION_IM) {                       $self->{insertion_mode} == IN_CAPTION_IM) {
5566                !!!cp ('t192');                !!!cp ('t192');
5567                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);                !!!parse-error (type => 'unmatched end tag',
5568                                  text => $token->{tag_name}, token => $token);
5569                ## Ignore the token                ## Ignore the token
5570                !!!next-token;                !!!next-token;
5571                next B;                next B;
# Line 5084  sub _tree_construction_main ($) { Line 5593  sub _tree_construction_main ($) {
5593      } elsif ($self->{insertion_mode} & TABLE_IMS) {      } elsif ($self->{insertion_mode} & TABLE_IMS) {
5594        if ($token->{type} == CHARACTER_TOKEN) {        if ($token->{type} == CHARACTER_TOKEN) {
5595          if (not $open_tables->[-1]->[1] and # tainted          if (not $open_tables->[-1]->[1] and # tainted
5596              $token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {              $token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
5597            $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);            $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
5598                                
5599            unless (length $token->{data}) {            unless (length $token->{data}) {
# Line 5096  sub _tree_construction_main ($) { Line 5605  sub _tree_construction_main ($) {
5605            }            }
5606          }          }
5607    
5608              !!!parse-error (type => 'in table:#character', token => $token);          !!!parse-error (type => 'in table:#text', token => $token);
5609    
5610              ## As if in body, but insert into foster parent element          ## NOTE: As if in body, but insert into the foster parent element.
5611              ## ISSUE: Spec says that "whenever a node would be inserted          $reconstruct_active_formatting_elements->($insert_to_foster);
             ## into the current node" while characters might not be  
             ## result in a new Text node.  
             $reconstruct_active_formatting_elements->($insert_to_foster);  
5612                            
5613              if ($self->{open_elements}->[-1]->[1] & TABLE_ROWS_EL) {          if ($self->{open_elements}->[-1]->[1] & TABLE_ROWS_EL) {
5614                # MUST            # MUST
5615                my $foster_parent_element;            my $foster_parent_element;
5616                my $next_sibling;            my $next_sibling;
5617                my $prev_sibling;            my $prev_sibling;
5618                OE: for (reverse 0..$#{$self->{open_elements}}) {            OE: for (reverse 0..$#{$self->{open_elements}}) {
5619                  if ($self->{open_elements}->[$_]->[1] & TABLE_EL) {              if ($self->{open_elements}->[$_]->[1] & TABLE_EL) {
5620                    my $parent = $self->{open_elements}->[$_]->[0]->parent_node;                my $parent = $self->{open_elements}->[$_]->[0]->parent_node;
5621                    if (defined $parent and $parent->node_type == 1) {                if (defined $parent and $parent->node_type == 1) {
5622                      !!!cp ('t196');                  $foster_parent_element = $parent;
5623                      $foster_parent_element = $parent;                  !!!cp ('t196');
5624                      $next_sibling = $self->{open_elements}->[$_]->[0];                  $next_sibling = $self->{open_elements}->[$_]->[0];
5625                      $prev_sibling = $next_sibling->previous_sibling;                  $prev_sibling = $next_sibling->previous_sibling;
5626                    } else {                  #
                     !!!cp ('t197');  
                     $foster_parent_element = $self->{open_elements}->[$_ - 1]->[0];  
                     $prev_sibling = $foster_parent_element->last_child;  
                   }  
                   last OE;  
                 }  
               } # OE  
               $foster_parent_element = $self->{open_elements}->[0]->[0] and  
               $prev_sibling = $foster_parent_element->last_child  
                 unless defined $foster_parent_element;  
               if (defined $prev_sibling and  
                   $prev_sibling->node_type == 3) {  
                 !!!cp ('t198');  
                 $prev_sibling->manakai_append_text ($token->{data});  
5627                } else {                } else {
5628                  !!!cp ('t199');                  !!!cp ('t197');
5629                  $foster_parent_element->insert_before                  $foster_parent_element = $self->{open_elements}->[$_ - 1]->[0];
5630                    ($self->{document}->create_text_node ($token->{data}),                  $prev_sibling = $foster_parent_element->last_child;
5631                     $next_sibling);                  #
5632                }                }
5633                  last OE;
5634                }
5635              } # OE
5636              $foster_parent_element = $self->{open_elements}->[0]->[0] and
5637              $prev_sibling = $foster_parent_element->last_child
5638                  unless defined $foster_parent_element;
5639              undef $prev_sibling unless $open_tables->[-1]->[2]; # ~node inserted
5640              if (defined $prev_sibling and
5641                  $prev_sibling->node_type == 3) {
5642                !!!cp ('t198');
5643                $prev_sibling->manakai_append_text ($token->{data});
5644              } else {
5645                !!!cp ('t199');
5646                $foster_parent_element->insert_before
5647                    ($self->{document}->create_text_node ($token->{data}),
5648                     $next_sibling);
5649              }
5650            $open_tables->[-1]->[1] = 1; # tainted            $open_tables->[-1]->[1] = 1; # tainted
5651              $open_tables->[-1]->[2] = 1; # ~node inserted
5652          } else {          } else {
5653              ## NOTE: Fragment case or in a foster parent'ed element
5654              ## (e.g. |<table><span>a|).  In fragment case, whether the
5655              ## character is appended to existing node or a new node is
5656              ## created is irrelevant, since the foster parent'ed nodes
5657              ## are discarded and fragment parsing does not invoke any
5658              ## script.
5659            !!!cp ('t200');            !!!cp ('t200');
5660            $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});            $self->{open_elements}->[-1]->[0]->manakai_append_text
5661                  ($token->{data});
5662          }          }
5663                            
5664          !!!next-token;          !!!next-token;
5665          next B;          next B;
5666        } elsif ($token->{type} == START_TAG_TOKEN) {        } elsif ($token->{type} == START_TAG_TOKEN) {
5667              if ({          if ({
5668                   tr => ($self->{insertion_mode} != IN_ROW_IM),               tr => ($self->{insertion_mode} != IN_ROW_IM),
5669                   th => 1, td => 1,               th => 1, td => 1,
5670                  }->{$token->{tag_name}}) {              }->{$token->{tag_name}}) {
5671                if ($self->{insertion_mode} == IN_TABLE_IM) {            if ($self->{insertion_mode} == IN_TABLE_IM) {
5672                  ## Clear back to table context              ## Clear back to table context
5673                  while (not ($self->{open_elements}->[-1]->[1]              while (not ($self->{open_elements}->[-1]->[1]
5674                                  & TABLE_SCOPING_EL)) {                              & TABLE_SCOPING_EL)) {
5675                    !!!cp ('t201');                !!!cp ('t201');
5676                    pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
5677                  }              }
5678                                
5679                  !!!insert-element ('tbody',, $token);              !!!insert-element ('tbody',, $token);
5680                  $self->{insertion_mode} = IN_TABLE_BODY_IM;              $self->{insertion_mode} = IN_TABLE_BODY_IM;
5681                  ## reprocess in the "in table body" insertion mode...              ## reprocess in the "in table body" insertion mode...
5682                }            }
5683              
5684                if ($self->{insertion_mode} == IN_TABLE_BODY_IM) {            if ($self->{insertion_mode} == IN_TABLE_BODY_IM) {
5685                  unless ($token->{tag_name} eq 'tr') {              unless ($token->{tag_name} eq 'tr') {
5686                    !!!cp ('t202');                !!!cp ('t202');
5687                    !!!parse-error (type => 'missing start tag:tr', token => $token);                !!!parse-error (type => 'missing start tag:tr', token => $token);
5688                  }              }
5689                                    
5690                  ## Clear back to table body context              ## Clear back to table body context
5691                  while (not ($self->{open_elements}->[-1]->[1]              while (not ($self->{open_elements}->[-1]->[1]
5692                                  & TABLE_ROWS_SCOPING_EL)) {                              & TABLE_ROWS_SCOPING_EL)) {
5693                    !!!cp ('t203');                !!!cp ('t203');
5694                    ## ISSUE: Can this case be reached?                ## ISSUE: Can this case be reached?
5695                    pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
5696                  }              }
5697                                    
5698                  $self->{insertion_mode} = IN_ROW_IM;              $self->{insertion_mode} = IN_ROW_IM;
5699                  if ($token->{tag_name} eq 'tr') {              if ($token->{tag_name} eq 'tr') {
5700                    !!!cp ('t204');                !!!cp ('t204');
5701                    !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5702                    !!!nack ('t204');                $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5703                    !!!next-token;                !!!nack ('t204');
5704                    next B;                !!!next-token;
5705                  } else {                next B;
5706                    !!!cp ('t205');              } else {
5707                    !!!insert-element ('tr',, $token);                !!!cp ('t205');
5708                    ## reprocess in the "in row" insertion mode                !!!insert-element ('tr',, $token);
5709                  }                ## reprocess in the "in row" insertion mode
5710                } else {              }
5711                  !!!cp ('t206');            } else {
5712                }              !!!cp ('t206');
5713              }
5714    
5715                ## Clear back to table row context                ## Clear back to table row context
5716                while (not ($self->{open_elements}->[-1]->[1]                while (not ($self->{open_elements}->[-1]->[1]
# Line 5201  sub _tree_construction_main ($) { Line 5719  sub _tree_construction_main ($) {
5719                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
5720                }                }
5721                                
5722                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);            !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5723                $self->{insertion_mode} = IN_CELL_IM;            $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5724              $self->{insertion_mode} = IN_CELL_IM;
5725    
5726                push @$active_formatting_elements, ['#marker', ''];            push @$active_formatting_elements, ['#marker', ''];
5727                                
5728                !!!nack ('t207.1');            !!!nack ('t207.1');
5729              !!!next-token;
5730              next B;
5731            } elsif ({
5732                      caption => 1, col => 1, colgroup => 1,
5733                      tbody => 1, tfoot => 1, thead => 1,
5734                      tr => 1, # $self->{insertion_mode} == IN_ROW_IM
5735                     }->{$token->{tag_name}}) {
5736              if ($self->{insertion_mode} == IN_ROW_IM) {
5737                ## As if </tr>
5738                ## have an element in table scope
5739                my $i;
5740                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5741                  my $node = $self->{open_elements}->[$_];
5742                  if ($node->[1] & TABLE_ROW_EL) {
5743                    !!!cp ('t208');
5744                    $i = $_;
5745                    last INSCOPE;
5746                  } elsif ($node->[1] & TABLE_SCOPING_EL) {
5747                    !!!cp ('t209');
5748                    last INSCOPE;
5749                  }
5750                } # INSCOPE
5751                unless (defined $i) {
5752                  !!!cp ('t210');
5753                  ## TODO: This type is wrong.
5754                  !!!parse-error (type => 'unmacthed end tag',
5755                                  text => $token->{tag_name}, token => $token);
5756                  ## Ignore the token
5757                  !!!nack ('t210.1');
5758                !!!next-token;                !!!next-token;
5759                next B;                next B;
5760              } elsif ({              }
                       caption => 1, col => 1, colgroup => 1,  
                       tbody => 1, tfoot => 1, thead => 1,  
                       tr => 1, # $self->{insertion_mode} == IN_ROW_IM  
                      }->{$token->{tag_name}}) {  
               if ($self->{insertion_mode} == IN_ROW_IM) {  
                 ## As if </tr>  
                 ## have an element in table scope  
                 my $i;  
                 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
                   my $node = $self->{open_elements}->[$_];  
                   if ($node->[1] & TABLE_ROW_EL) {  
                     !!!cp ('t208');  
                     $i = $_;  
                     last INSCOPE;  
                   } elsif ($node->[1] & TABLE_SCOPING_EL) {  
                     !!!cp ('t209');  
                     last INSCOPE;  
                   }  
                 } # INSCOPE  
                 unless (defined $i) {  
                   !!!cp ('t210');  
 ## TODO: This type is wrong.  
                   !!!parse-error (type => 'unmacthed end tag:'.$token->{tag_name}, token => $token);  
                   ## Ignore the token  
                   !!!nack ('t210.1');  
                   !!!next-token;  
                   next B;  
                 }  
5761                                    
5762                  ## Clear back to table row context                  ## Clear back to table row context
5763                  while (not ($self->{open_elements}->[-1]->[1]                  while (not ($self->{open_elements}->[-1]->[1]
# Line 5276  sub _tree_construction_main ($) { Line 5796  sub _tree_construction_main ($) {
5796                  } # INSCOPE                  } # INSCOPE
5797                  unless (defined $i) {                  unless (defined $i) {
5798                    !!!cp ('t216');                    !!!cp ('t216');
5799  ## TODO: This erorr type ios wrong.  ## TODO: This erorr type is wrong.
5800                    !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);                    !!!parse-error (type => 'unmatched end tag',
5801                                      text => $token->{tag_name}, token => $token);
5802                    ## Ignore the token                    ## Ignore the token
5803                    !!!nack ('t216.1');                    !!!nack ('t216.1');
5804                    !!!next-token;                    !!!next-token;
# Line 5306  sub _tree_construction_main ($) { Line 5827  sub _tree_construction_main ($) {
5827                  !!!cp ('t218');                  !!!cp ('t218');
5828                }                }
5829    
5830                if ($token->{tag_name} eq 'col') {            if ($token->{tag_name} eq 'col') {
5831                  ## Clear back to table context              ## Clear back to table context
5832                  while (not ($self->{open_elements}->[-1]->[1]              while (not ($self->{open_elements}->[-1]->[1]
5833                                  & TABLE_SCOPING_EL)) {                              & TABLE_SCOPING_EL)) {
5834                    !!!cp ('t219');                !!!cp ('t219');
5835                    ## ISSUE: Can this state be reached?                ## ISSUE: Can this state be reached?
5836                    pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
5837                  }              }
5838                                
5839                  !!!insert-element ('colgroup',, $token);              !!!insert-element ('colgroup',, $token);
5840                  $self->{insertion_mode} = IN_COLUMN_GROUP_IM;              $self->{insertion_mode} = IN_COLUMN_GROUP_IM;
5841                  ## reprocess              ## reprocess
5842                  !!!ack-later;              $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5843                  next B;              !!!ack-later;
5844                } elsif ({              next B;
5845                          caption => 1,            } elsif ({
5846                          colgroup => 1,                      caption => 1,
5847                          tbody => 1, tfoot => 1, thead => 1,                      colgroup => 1,
5848                         }->{$token->{tag_name}}) {                      tbody => 1, tfoot => 1, thead => 1,
5849                  ## Clear back to table context                     }->{$token->{tag_name}}) {
5850                ## Clear back to table context
5851                  while (not ($self->{open_elements}->[-1]->[1]                  while (not ($self->{open_elements}->[-1]->[1]
5852                                  & TABLE_SCOPING_EL)) {                                  & TABLE_SCOPING_EL)) {
5853                    !!!cp ('t220');                    !!!cp ('t220');
# Line 5333  sub _tree_construction_main ($) { Line 5855  sub _tree_construction_main ($) {
5855                    pop @{$self->{open_elements}};                    pop @{$self->{open_elements}};
5856                  }                  }
5857                                    
5858                  push @$active_formatting_elements, ['#marker', '']              push @$active_formatting_elements, ['#marker', '']
5859                      if $token->{tag_name} eq 'caption';                  if $token->{tag_name} eq 'caption';
5860                                    
5861                  !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);              !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5862                  $self->{insertion_mode} = {              $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5863                                             caption => IN_CAPTION_IM,              $self->{insertion_mode} = {
5864                                             colgroup => IN_COLUMN_GROUP_IM,                                         caption => IN_CAPTION_IM,
5865                                             tbody => IN_TABLE_BODY_IM,                                         colgroup => IN_COLUMN_GROUP_IM,
5866                                             tfoot => IN_TABLE_BODY_IM,                                         tbody => IN_TABLE_BODY_IM,
5867                                             thead => IN_TABLE_BODY_IM,                                         tfoot => IN_TABLE_BODY_IM,
5868                                            }->{$token->{tag_name}};                                         thead => IN_TABLE_BODY_IM,
5869                  !!!next-token;                                        }->{$token->{tag_name}};
5870                  !!!nack ('t220.1');              !!!next-token;
5871                  next B;              !!!nack ('t220.1');
5872                } else {              next B;
5873                  die "$0: in table: <>: $token->{tag_name}";            } else {
5874                }              die "$0: in table: <>: $token->{tag_name}";
5875              }
5876              } elsif ($token->{tag_name} eq 'table') {              } elsif ($token->{tag_name} eq 'table') {
5877                !!!parse-error (type => 'not closed',                !!!parse-error (type => 'not closed',
5878                                value => $self->{open_elements}->[-1]->[0]                                text => $self->{open_elements}->[-1]->[0]
5879                                    ->manakai_local_name,                                    ->manakai_local_name,
5880                                token => $token);                                token => $token);
5881    
# Line 5373  sub _tree_construction_main ($) { Line 5896  sub _tree_construction_main ($) {
5896                unless (defined $i) {                unless (defined $i) {
5897                  !!!cp ('t223');                  !!!cp ('t223');
5898  ## TODO: The following is wrong, maybe.  ## TODO: The following is wrong, maybe.
5899                  !!!parse-error (type => 'unmatched end tag:table', token => $token);                  !!!parse-error (type => 'unmatched end tag', text => 'table',
5900                                    token => $token);
5901                  ## Ignore tokens </table><table>                  ## Ignore tokens </table><table>
5902                  !!!nack ('t223.1');                  !!!nack ('t223.1');
5903                  !!!next-token;                  !!!next-token;
5904                  next B;                  next B;
5905                }                }
5906                                
5907  ## TODO: Followings are removed from the latest spec.  ## TODO: Followings are removed from the latest spec.
5908                ## generate implied end tags                ## generate implied end tags
5909                while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {                while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
5910                  !!!cp ('t224');                  !!!cp ('t224');
# Line 5391  sub _tree_construction_main ($) { Line 5915  sub _tree_construction_main ($) {
5915                  !!!cp ('t225');                  !!!cp ('t225');
5916                  ## NOTE: |<table><tr><table>|                  ## NOTE: |<table><tr><table>|
5917                  !!!parse-error (type => 'not closed',                  !!!parse-error (type => 'not closed',
5918                                  value => $self->{open_elements}->[-1]->[0]                                  text => $self->{open_elements}->[-1]->[0]
5919                                      ->manakai_local_name,                                      ->manakai_local_name,
5920                                  token => $token);                                  token => $token);
5921                } else {                } else {
# Line 5411  sub _tree_construction_main ($) { Line 5935  sub _tree_construction_main ($) {
5935              !!!cp ('t227.8');              !!!cp ('t227.8');
5936              ## NOTE: This is a "as if in head" code clone.              ## NOTE: This is a "as if in head" code clone.
5937              $parse_rcdata->(CDATA_CONTENT_MODEL);              $parse_rcdata->(CDATA_CONTENT_MODEL);
5938                $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5939              next B;              next B;
5940            } else {            } else {
5941              !!!cp ('t227.7');              !!!cp ('t227.7');
# Line 5421  sub _tree_construction_main ($) { Line 5946  sub _tree_construction_main ($) {
5946              !!!cp ('t227.6');              !!!cp ('t227.6');
5947              ## NOTE: This is a "as if in head" code clone.              ## NOTE: This is a "as if in head" code clone.
5948              $script_start_tag->();              $script_start_tag->();
5949                $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5950              next B;              next B;
5951            } else {            } else {
5952              !!!cp ('t227.5');              !!!cp ('t227.5');
# Line 5432  sub _tree_construction_main ($) { Line 5958  sub _tree_construction_main ($) {
5958                my $type = lc $token->{attributes}->{type}->{value};                my $type = lc $token->{attributes}->{type}->{value};
5959                if ($type eq 'hidden') {                if ($type eq 'hidden') {
5960                  !!!cp ('t227.3');                  !!!cp ('t227.3');
5961                  !!!parse-error (type => 'in table:'.$token->{tag_name}, token => $token);                  !!!parse-error (type => 'in table',
5962                                    text => $token->{tag_name}, token => $token);
5963    
5964                  !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);                  !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5965                    $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5966    
5967                  ## TODO: form element pointer                  ## TODO: form element pointer
5968    
# Line 5460  sub _tree_construction_main ($) { Line 5988  sub _tree_construction_main ($) {
5988            #            #
5989          }          }
5990    
5991          !!!parse-error (type => 'in table:'.$token->{tag_name}, token => $token);          !!!parse-error (type => 'in table', text => $token->{tag_name},
5992                            token => $token);
5993    
5994          $insert = $insert_to_foster;          $insert = $insert_to_foster;
5995          #          #
# Line 5482  sub _tree_construction_main ($) { Line 6011  sub _tree_construction_main ($) {
6011                } # INSCOPE                } # INSCOPE
6012                unless (defined $i) {                unless (defined $i) {
6013                  !!!cp ('t230');                  !!!cp ('t230');
6014                  !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);                  !!!parse-error (type => 'unmatched end tag',
6015                                    text => $token->{tag_name}, token => $token);
6016                  ## Ignore the token                  ## Ignore the token
6017                  !!!nack ('t230.1');                  !!!nack ('t230.1');
6018                  !!!next-token;                  !!!next-token;
# Line 5523  sub _tree_construction_main ($) { Line 6053  sub _tree_construction_main ($) {
6053                  unless (defined $i) {                  unless (defined $i) {
6054                    !!!cp ('t235');                    !!!cp ('t235');
6055  ## TODO: The following is wrong.  ## TODO: The following is wrong.
6056                    !!!parse-error (type => 'unmatched end tag:'.$token->{type}, token => $token);                    !!!parse-error (type => 'unmatched end tag',
6057                                      text => $token->{type}, token => $token);
6058                    ## Ignore the token                    ## Ignore the token
6059                    !!!nack ('t236.1');                    !!!nack ('t236.1');
6060                    !!!next-token;                    !!!next-token;
# Line 5559  sub _tree_construction_main ($) { Line 6090  sub _tree_construction_main ($) {
6090                  } # INSCOPE                  } # INSCOPE
6091                  unless (defined $i) {                  unless (defined $i) {
6092                    !!!cp ('t239');                    !!!cp ('t239');
6093                    !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);                    !!!parse-error (type => 'unmatched end tag',
6094                                      text => $token->{tag_name}, token => $token);
6095                    ## Ignore the token                    ## Ignore the token
6096                    !!!nack ('t239.1');                    !!!nack ('t239.1');
6097                    !!!next-token;                    !!!next-token;
# Line 5605  sub _tree_construction_main ($) { Line 6137  sub _tree_construction_main ($) {
6137                } # INSCOPE                } # INSCOPE
6138                unless (defined $i) {                unless (defined $i) {
6139                  !!!cp ('t243');                  !!!cp ('t243');
6140                  !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);                  !!!parse-error (type => 'unmatched end tag',
6141                                    text => $token->{tag_name}, token => $token);
6142                  ## Ignore the token                  ## Ignore the token
6143                  !!!nack ('t243.1');                  !!!nack ('t243.1');
6144                  !!!next-token;                  !!!next-token;
# Line 5639  sub _tree_construction_main ($) { Line 6172  sub _tree_construction_main ($) {
6172                  } # INSCOPE                  } # INSCOPE
6173                    unless (defined $i) {                    unless (defined $i) {
6174                      !!!cp ('t249');                      !!!cp ('t249');
6175                      !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);                      !!!parse-error (type => 'unmatched end tag',
6176                                        text => $token->{tag_name}, token => $token);
6177                      ## Ignore the token                      ## Ignore the token
6178                      !!!nack ('t249.1');                      !!!nack ('t249.1');
6179                      !!!next-token;                      !!!next-token;
# Line 5662  sub _tree_construction_main ($) { Line 6196  sub _tree_construction_main ($) {
6196                  } # INSCOPE                  } # INSCOPE
6197                    unless (defined $i) {                    unless (defined $i) {
6198                      !!!cp ('t252');                      !!!cp ('t252');
6199                      !!!parse-error (type => 'unmatched end tag:tr', token => $token);                      !!!parse-error (type => 'unmatched end tag',
6200                                        text => 'tr', token => $token);
6201                      ## Ignore the token                      ## Ignore the token
6202                      !!!nack ('t252.1');                      !!!nack ('t252.1');
6203                      !!!next-token;                      !!!next-token;
# Line 5697  sub _tree_construction_main ($) { Line 6232  sub _tree_construction_main ($) {
6232                } # INSCOPE                } # INSCOPE
6233                unless (defined $i) {                unless (defined $i) {
6234                  !!!cp ('t256');                  !!!cp ('t256');
6235                  !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);                  !!!parse-error (type => 'unmatched end tag',
6236                                    text => $token->{tag_name}, token => $token);
6237                  ## Ignore the token                  ## Ignore the token
6238                  !!!nack ('t256.1');                  !!!nack ('t256.1');
6239                  !!!next-token;                  !!!next-token;
# Line 5724  sub _tree_construction_main ($) { Line 6260  sub _tree_construction_main ($) {
6260                        tbody => 1, tfoot => 1, thead => 1, # $self->{insertion_mode} == IN_TABLE_IM                        tbody => 1, tfoot => 1, thead => 1, # $self->{insertion_mode} == IN_TABLE_IM
6261                       }->{$token->{tag_name}}) {                       }->{$token->{tag_name}}) {
6262            !!!cp ('t258');            !!!cp ('t258');
6263            !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);            !!!parse-error (type => 'unmatched end tag',
6264                              text => $token->{tag_name}, token => $token);
6265            ## Ignore the token            ## Ignore the token
6266            !!!nack ('t258.1');            !!!nack ('t258.1');
6267             !!!next-token;             !!!next-token;
6268            next B;            next B;
6269          } else {          } else {
6270            !!!cp ('t259');            !!!cp ('t259');
6271            !!!parse-error (type => 'in table:/'.$token->{tag_name}, token => $token);            !!!parse-error (type => 'in table:/',
6272                              text => $token->{tag_name}, token => $token);
6273    
6274            $insert = $insert_to_foster;            $insert = $insert_to_foster;
6275            #            #
# Line 5754  sub _tree_construction_main ($) { Line 6292  sub _tree_construction_main ($) {
6292        }        }
6293      } elsif ($self->{insertion_mode} == IN_COLUMN_GROUP_IM) {      } elsif ($self->{insertion_mode} == IN_COLUMN_GROUP_IM) {
6294            if ($token->{type} == CHARACTER_TOKEN) {            if ($token->{type} == CHARACTER_TOKEN) {
6295              if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {              if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
6296                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
6297                unless (length $token->{data}) {                unless (length $token->{data}) {
6298                  !!!cp ('t260');                  !!!cp ('t260');
# Line 5781  sub _tree_construction_main ($) { Line 6319  sub _tree_construction_main ($) {
6319              if ($token->{tag_name} eq 'colgroup') {              if ($token->{tag_name} eq 'colgroup') {
6320                if ($self->{open_elements}->[-1]->[1] & HTML_EL) {                if ($self->{open_elements}->[-1]->[1] & HTML_EL) {
6321                  !!!cp ('t264');                  !!!cp ('t264');
6322                  !!!parse-error (type => 'unmatched end tag:colgroup', token => $token);                  !!!parse-error (type => 'unmatched end tag',
6323                                    text => 'colgroup', token => $token);
6324                  ## Ignore the token                  ## Ignore the token
6325                  !!!next-token;                  !!!next-token;
6326                  next B;                  next B;
# Line 5794  sub _tree_construction_main ($) { Line 6333  sub _tree_construction_main ($) {
6333                }                }
6334              } elsif ($token->{tag_name} eq 'col') {              } elsif ($token->{tag_name} eq 'col') {
6335                !!!cp ('t266');                !!!cp ('t266');
6336                !!!parse-error (type => 'unmatched end tag:col', token => $token);                !!!parse-error (type => 'unmatched end tag',
6337                                  text => 'col', token => $token);
6338                ## Ignore the token                ## Ignore the token
6339                !!!next-token;                !!!next-token;
6340                next B;                next B;
# Line 5824  sub _tree_construction_main ($) { Line 6364  sub _tree_construction_main ($) {
6364            if ($self->{open_elements}->[-1]->[1] & HTML_EL) {            if ($self->{open_elements}->[-1]->[1] & HTML_EL) {
6365              !!!cp ('t269');              !!!cp ('t269');
6366  ## TODO: Wrong error type?  ## TODO: Wrong error type?
6367              !!!parse-error (type => 'unmatched end tag:colgroup', token => $token);              !!!parse-error (type => 'unmatched end tag',
6368                                text => 'colgroup', token => $token);
6369              ## Ignore the token              ## Ignore the token
6370              !!!nack ('t269.1');              !!!nack ('t269.1');
6371              !!!next-token;              !!!next-token;
# Line 5878  sub _tree_construction_main ($) { Line 6419  sub _tree_construction_main ($) {
6419            !!!nack ('t277.1');            !!!nack ('t277.1');
6420            !!!next-token;            !!!next-token;
6421            next B;            next B;
6422          } elsif ($token->{tag_name} eq 'select' or          } elsif ({
6423                   $token->{tag_name} eq 'input' or                     select => 1, input => 1, textarea => 1,
6424                     }->{$token->{tag_name}} or
6425                   ($self->{insertion_mode} == IN_SELECT_IN_TABLE_IM and                   ($self->{insertion_mode} == IN_SELECT_IN_TABLE_IM and
6426                    {                    {
6427                     caption => 1, table => 1,                     caption => 1, table => 1,
# Line 5887  sub _tree_construction_main ($) { Line 6429  sub _tree_construction_main ($) {
6429                     tr => 1, td => 1, th => 1,                     tr => 1, td => 1, th => 1,
6430                    }->{$token->{tag_name}})) {                    }->{$token->{tag_name}})) {
6431            ## TODO: The type below is not good - <select> is replaced by </select>            ## TODO: The type below is not good - <select> is replaced by </select>
6432            !!!parse-error (type => 'not closed:select', token => $token);            !!!parse-error (type => 'not closed', text => 'select',
6433                              token => $token);
6434            ## NOTE: As if the token were </select> (<select> case) or            ## NOTE: As if the token were </select> (<select> case) or
6435            ## as if there were </select> (otherwise).            ## as if there were </select> (otherwise).
6436            ## have an element in table scope            ## have an element in table scope
# Line 5905  sub _tree_construction_main ($) { Line 6448  sub _tree_construction_main ($) {
6448            } # INSCOPE            } # INSCOPE
6449            unless (defined $i) {            unless (defined $i) {
6450              !!!cp ('t280');              !!!cp ('t280');
6451              !!!parse-error (type => 'unmatched end tag:select', token => $token);              !!!parse-error (type => 'unmatched end tag',
6452                                text => 'select', token => $token);
6453              ## Ignore the token              ## Ignore the token
6454              !!!nack ('t280.1');              !!!nack ('t280.1');
6455              !!!next-token;              !!!next-token;
# Line 5929  sub _tree_construction_main ($) { Line 6473  sub _tree_construction_main ($) {
6473            }            }
6474          } else {          } else {
6475            !!!cp ('t282');            !!!cp ('t282');
6476            !!!parse-error (type => 'in select:'.$token->{tag_name}, token => $token);            !!!parse-error (type => 'in select',
6477                              text => $token->{tag_name}, token => $token);
6478            ## Ignore the token            ## Ignore the token
6479            !!!nack ('t282.1');            !!!nack ('t282.1');
6480            !!!next-token;            !!!next-token;
# Line 5947  sub _tree_construction_main ($) { Line 6492  sub _tree_construction_main ($) {
6492              pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
6493            } else {            } else {
6494              !!!cp ('t285');              !!!cp ('t285');
6495              !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);              !!!parse-error (type => 'unmatched end tag',
6496                                text => $token->{tag_name}, token => $token);
6497              ## Ignore the token              ## Ignore the token
6498            }            }
6499            !!!nack ('t285.1');            !!!nack ('t285.1');
# Line 5959  sub _tree_construction_main ($) { Line 6505  sub _tree_construction_main ($) {
6505              pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
6506            } else {            } else {
6507              !!!cp ('t287');              !!!cp ('t287');
6508              !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);              !!!parse-error (type => 'unmatched end tag',
6509                                text => $token->{tag_name}, token => $token);
6510              ## Ignore the token              ## Ignore the token
6511            }            }
6512            !!!nack ('t287.1');            !!!nack ('t287.1');
# Line 5981  sub _tree_construction_main ($) { Line 6528  sub _tree_construction_main ($) {
6528            } # INSCOPE            } # INSCOPE
6529            unless (defined $i) {            unless (defined $i) {
6530              !!!cp ('t290');              !!!cp ('t290');
6531              !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);              !!!parse-error (type => 'unmatched end tag',
6532                                text => $token->{tag_name}, token => $token);
6533              ## Ignore the token              ## Ignore the token
6534              !!!nack ('t290.1');              !!!nack ('t290.1');
6535              !!!next-token;              !!!next-token;
# Line 6002  sub _tree_construction_main ($) { Line 6550  sub _tree_construction_main ($) {
6550                    tfoot => 1, thead => 1, tr => 1, td => 1, th => 1,                    tfoot => 1, thead => 1, tr => 1, td => 1, th => 1,
6551                   }->{$token->{tag_name}}) {                   }->{$token->{tag_name}}) {
6552  ## TODO: The following is wrong?  ## TODO: The following is wrong?
6553            !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);            !!!parse-error (type => 'unmatched end tag',
6554                              text => $token->{tag_name}, token => $token);
6555                                
6556            ## have an element in table scope            ## have an element in table scope
6557            my $i;            my $i;
# Line 6043  sub _tree_construction_main ($) { Line 6592  sub _tree_construction_main ($) {
6592            unless (defined $i) {            unless (defined $i) {
6593              !!!cp ('t297');              !!!cp ('t297');
6594  ## TODO: The following error type is correct?  ## TODO: The following error type is correct?
6595              !!!parse-error (type => 'unmatched end tag:select', token => $token);              !!!parse-error (type => 'unmatched end tag',
6596                                text => 'select', token => $token);
6597              ## Ignore the </select> token              ## Ignore the </select> token
6598              !!!nack ('t297.1');              !!!nack ('t297.1');
6599              !!!next-token; ## TODO: ok?              !!!next-token; ## TODO: ok?
# Line 6060  sub _tree_construction_main ($) { Line 6610  sub _tree_construction_main ($) {
6610            next B;            next B;
6611          } else {          } else {
6612            !!!cp ('t299');            !!!cp ('t299');
6613            !!!parse-error (type => 'in select:/'.$token->{tag_name}, token => $token);            !!!parse-error (type => 'in select:/',
6614                              text => $token->{tag_name}, token => $token);
6615            ## Ignore the token            ## Ignore the token
6616            !!!nack ('t299.3');            !!!nack ('t299.3');
6617            !!!next-token;            !!!next-token;
# Line 6082  sub _tree_construction_main ($) { Line 6633  sub _tree_construction_main ($) {
6633        }        }
6634      } elsif ($self->{insertion_mode} & BODY_AFTER_IMS) {      } elsif ($self->{insertion_mode} & BODY_AFTER_IMS) {
6635        if ($token->{type} == CHARACTER_TOKEN) {        if ($token->{type} == CHARACTER_TOKEN) {
6636          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {          if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
6637            my $data = $1;            my $data = $1;
6638            ## As if in body            ## As if in body
6639            $reconstruct_active_formatting_elements->($insert_to_current);            $reconstruct_active_formatting_elements->($insert_to_current);
# Line 6098  sub _tree_construction_main ($) { Line 6649  sub _tree_construction_main ($) {
6649                    
6650          if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {          if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
6651            !!!cp ('t301');            !!!cp ('t301');
6652            !!!parse-error (type => 'after html:#character', token => $token);            !!!parse-error (type => 'after html:#text', token => $token);
6653              #
           ## Reprocess in the "after body" insertion mode.  
6654          } else {          } else {
6655            !!!cp ('t302');            !!!cp ('t302');
6656              ## "after body" insertion mode
6657              !!!parse-error (type => 'after body:#text', token => $token);
6658              #
6659          }          }
           
         ## "after body" insertion mode  
         !!!parse-error (type => 'after body:#character', token => $token);  
6660    
6661          $self->{insertion_mode} = IN_BODY_IM;          $self->{insertion_mode} = IN_BODY_IM;
6662          ## reprocess          ## reprocess
# Line 6114  sub _tree_construction_main ($) { Line 6664  sub _tree_construction_main ($) {
6664        } elsif ($token->{type} == START_TAG_TOKEN) {        } elsif ($token->{type} == START_TAG_TOKEN) {
6665          if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {          if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
6666            !!!cp ('t303');            !!!cp ('t303');
6667            !!!parse-error (type => 'after html:'.$token->{tag_name}, token => $token);            !!!parse-error (type => 'after html',
6668                                        text => $token->{tag_name}, token => $token);
6669            ## Reprocess in the "after body" insertion mode.            #
6670          } else {          } else {
6671            !!!cp ('t304');            !!!cp ('t304');
6672              ## "after body" insertion mode
6673              !!!parse-error (type => 'after body',
6674                              text => $token->{tag_name}, token => $token);
6675              #
6676          }          }
6677    
         ## "after body" insertion mode  
         !!!parse-error (type => 'after body:'.$token->{tag_name}, token => $token);  
   
6678          $self->{insertion_mode} = IN_BODY_IM;          $self->{insertion_mode} = IN_BODY_IM;
6679          !!!ack-later;          !!!ack-later;
6680          ## reprocess          ## reprocess
# Line 6131  sub _tree_construction_main ($) { Line 6682  sub _tree_construction_main ($) {
6682        } elsif ($token->{type} == END_TAG_TOKEN) {        } elsif ($token->{type} == END_TAG_TOKEN) {
6683          if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {          if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
6684            !!!cp ('t305');            !!!cp ('t305');
6685            !!!parse-error (type => 'after html:/'.$token->{tag_name}, token => $token);            !!!parse-error (type => 'after html:/',
6686                              text => $token->{tag_name}, token => $token);
6687                        
6688            $self->{insertion_mode} = AFTER_BODY_IM;            $self->{insertion_mode} = IN_BODY_IM;
6689            ## Reprocess in the "after body" insertion mode.            ## Reprocess.
6690              next B;
6691          } else {          } else {
6692            !!!cp ('t306');            !!!cp ('t306');
6693          }          }
# Line 6143  sub _tree_construction_main ($) { Line 6696  sub _tree_construction_main ($) {
6696          if ($token->{tag_name} eq 'html') {          if ($token->{tag_name} eq 'html') {
6697            if (defined $self->{inner_html_node}) {            if (defined $self->{inner_html_node}) {
6698              !!!cp ('t307');              !!!cp ('t307');
6699              !!!parse-error (type => 'unmatched end tag:html', token => $token);              !!!parse-error (type => 'unmatched end tag',
6700                                text => 'html', token => $token);
6701              ## Ignore the token              ## Ignore the token
6702              !!!next-token;              !!!next-token;
6703              next B;              next B;
# Line 6155  sub _tree_construction_main ($) { Line 6709  sub _tree_construction_main ($) {
6709            }            }
6710          } else {          } else {
6711            !!!cp ('t309');            !!!cp ('t309');
6712            !!!parse-error (type => 'after body:/'.$token->{tag_name}, token => $token);            !!!parse-error (type => 'after body:/',
6713                              text => $token->{tag_name}, token => $token);
6714    
6715            $self->{insertion_mode} = IN_BODY_IM;            $self->{insertion_mode} = IN_BODY_IM;
6716            ## reprocess            ## reprocess
# Line 6170  sub _tree_construction_main ($) { Line 6725  sub _tree_construction_main ($) {
6725        }        }
6726      } elsif ($self->{insertion_mode} & FRAME_IMS) {      } elsif ($self->{insertion_mode} & FRAME_IMS) {
6727        if ($token->{type} == CHARACTER_TOKEN) {        if ($token->{type} == CHARACTER_TOKEN) {
6728          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {          if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
6729            $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);            $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
6730                        
6731            unless (length $token->{data}) {            unless (length $token->{data}) {
# Line 6180  sub _tree_construction_main ($) { Line 6735  sub _tree_construction_main ($) {
6735            }            }
6736          }          }
6737                    
6738          if ($token->{data} =~ s/^[^\x09\x0A\x0B\x0C\x20]+//) {          if ($token->{data} =~ s/^[^\x09\x0A\x0C\x20]+//) {
6739            if ($self->{insertion_mode} == IN_FRAMESET_IM) {            if ($self->{insertion_mode} == IN_FRAMESET_IM) {
6740              !!!cp ('t311');              !!!cp ('t311');
6741              !!!parse-error (type => 'in frameset:#character', token => $token);              !!!parse-error (type => 'in frameset:#text', token => $token);
6742            } elsif ($self->{insertion_mode} == AFTER_FRAMESET_IM) {            } elsif ($self->{insertion_mode} == AFTER_FRAMESET_IM) {
6743              !!!cp ('t312');              !!!cp ('t312');
6744              !!!parse-error (type => 'after frameset:#character', token => $token);              !!!parse-error (type => 'after frameset:#text', token => $token);
6745            } else { # "after html frameset"            } else { # "after after frameset"
6746              !!!cp ('t313');              !!!cp ('t313');
6747              !!!parse-error (type => 'after html:#character', token => $token);              !!!parse-error (type => 'after html:#text', token => $token);
   
             $self->{insertion_mode} = AFTER_FRAMESET_IM;  
             ## Reprocess in the "after frameset" insertion mode.  
             !!!parse-error (type => 'after frameset:#character', token => $token);  
6748            }            }
6749                        
6750            ## Ignore the token.            ## Ignore the token.
# Line 6209  sub _tree_construction_main ($) { Line 6760  sub _tree_construction_main ($) {
6760                    
6761          die qq[$0: Character "$token->{data}"];          die qq[$0: Character "$token->{data}"];
6762        } elsif ($token->{type} == START_TAG_TOKEN) {        } elsif ($token->{type} == START_TAG_TOKEN) {
         if ($self->{insertion_mode} == AFTER_HTML_FRAMESET_IM) {  
           !!!cp ('t316');  
           !!!parse-error (type => 'after html:'.$token->{tag_name}, token => $token);  
   
           $self->{insertion_mode} = AFTER_FRAMESET_IM;  
           ## Process in the "after frameset" insertion mode.  
         } else {  
           !!!cp ('t317');  
         }  
   
6763          if ($token->{tag_name} eq 'frameset' and          if ($token->{tag_name} eq 'frameset' and
6764              $self->{insertion_mode} == IN_FRAMESET_IM) {              $self->{insertion_mode} == IN_FRAMESET_IM) {
6765            !!!cp ('t318');            !!!cp ('t318');
# Line 6236  sub _tree_construction_main ($) { Line 6777  sub _tree_construction_main ($) {
6777            next B;            next B;
6778          } elsif ($token->{tag_name} eq 'noframes') {          } elsif ($token->{tag_name} eq 'noframes') {
6779            !!!cp ('t320');            !!!cp ('t320');
6780            ## NOTE: As if in body.            ## NOTE: As if in head.
6781            $parse_rcdata->(CDATA_CONTENT_MODEL);            $parse_rcdata->(CDATA_CONTENT_MODEL);
6782            next B;            next B;
6783    
6784              ## NOTE: |<!DOCTYPE HTML><frameset></frameset></html><noframes></noframes>|
6785              ## has no parse error.
6786          } else {          } else {
6787            if ($self->{insertion_mode} == IN_FRAMESET_IM) {            if ($self->{insertion_mode} == IN_FRAMESET_IM) {
6788              !!!cp ('t321');              !!!cp ('t321');
6789              !!!parse-error (type => 'in frameset:'.$token->{tag_name}, token => $token);              !!!parse-error (type => 'in frameset',
6790            } else {                              text => $token->{tag_name}, token => $token);
6791              } elsif ($self->{insertion_mode} == AFTER_FRAMESET_IM) {
6792              !!!cp ('t322');              !!!cp ('t322');
6793              !!!parse-error (type => 'after frameset:'.$token->{tag_name}, token => $token);              !!!parse-error (type => 'after frameset',
6794                                text => $token->{tag_name}, token => $token);
6795              } else { # "after after frameset"
6796                !!!cp ('t322.2');
6797                !!!parse-error (type => 'after after frameset',
6798                                text => $token->{tag_name}, token => $token);
6799            }            }
6800            ## Ignore the token            ## Ignore the token
6801            !!!nack ('t322.1');            !!!nack ('t322.1');
# Line 6253  sub _tree_construction_main ($) { Line 6803  sub _tree_construction_main ($) {
6803            next B;            next B;
6804          }          }
6805        } elsif ($token->{type} == END_TAG_TOKEN) {        } elsif ($token->{type} == END_TAG_TOKEN) {
         if ($self->{insertion_mode} == AFTER_HTML_FRAMESET_IM) {  
           !!!cp ('t323');  
           !!!parse-error (type => 'after html:/'.$token->{tag_name}, token => $token);  
   
           $self->{insertion_mode} = AFTER_FRAMESET_IM;  
           ## Process in the "after frameset" insertion mode.  
         } else {  
           !!!cp ('t324');  
         }  
   
6806          if ($token->{tag_name} eq 'frameset' and          if ($token->{tag_name} eq 'frameset' and
6807              $self->{insertion_mode} == IN_FRAMESET_IM) {              $self->{insertion_mode} == IN_FRAMESET_IM) {
6808            if ($self->{open_elements}->[-1]->[1] & HTML_EL and            if ($self->{open_elements}->[-1]->[1] & HTML_EL and
6809                @{$self->{open_elements}} == 1) {                @{$self->{open_elements}} == 1) {
6810              !!!cp ('t325');              !!!cp ('t325');
6811              !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);              !!!parse-error (type => 'unmatched end tag',
6812                                text => $token->{tag_name}, token => $token);
6813              ## Ignore the token              ## Ignore the token
6814              !!!next-token;              !!!next-token;
6815            } else {            } else {
# Line 6294  sub _tree_construction_main ($) { Line 6835  sub _tree_construction_main ($) {
6835          } else {          } else {
6836            if ($self->{insertion_mode} == IN_FRAMESET_IM) {            if ($self->{insertion_mode} == IN_FRAMESET_IM) {
6837              !!!cp ('t330');              !!!cp ('t330');
6838              !!!parse-error (type => 'in frameset:/'.$token->{tag_name}, token => $token);              !!!parse-error (type => 'in frameset:/',
6839            } else {                              text => $token->{tag_name}, token => $token);
6840              } elsif ($self->{insertion_mode} == AFTER_FRAMESET_IM) {
6841                !!!cp ('t330.1');
6842                !!!parse-error (type => 'after frameset:/',
6843                                text => $token->{tag_name}, token => $token);
6844              } else { # "after after html"
6845              !!!cp ('t331');              !!!cp ('t331');
6846              !!!parse-error (type => 'after frameset:/'.$token->{tag_name}, token => $token);              !!!parse-error (type => 'after after frameset:/',
6847                                text => $token->{tag_name}, token => $token);
6848            }            }
6849            ## Ignore the token            ## Ignore the token
6850            !!!next-token;            !!!next-token;
# Line 6317  sub _tree_construction_main ($) { Line 6864  sub _tree_construction_main ($) {
6864        } else {        } else {
6865          die "$0: $token->{type}: Unknown token type";          die "$0: $token->{type}: Unknown token type";
6866        }        }
   
       ## ISSUE: An issue in spec here  
6867      } else {      } else {
6868        die "$0: $self->{insertion_mode}: Unknown insertion mode";        die "$0: $self->{insertion_mode}: Unknown insertion mode";
6869      }      }
# Line 6336  sub _tree_construction_main ($) { Line 6881  sub _tree_construction_main ($) {
6881          $parse_rcdata->(CDATA_CONTENT_MODEL);          $parse_rcdata->(CDATA_CONTENT_MODEL);
6882          next B;          next B;
6883        } elsif ({        } elsif ({
6884                  base => 1, link => 1,                  base => 1, command => 1, eventsource => 1, link => 1,
6885                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
6886          !!!cp ('t334');          !!!cp ('t334');
6887          ## NOTE: This is an "as if in head" code clone, only "-t" differs          ## NOTE: This is an "as if in head" code clone, only "-t" differs
6888          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
6889          pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.          pop @{$self->{open_elements}};
6890          !!!ack ('t334.1');          !!!ack ('t334.1');
6891          !!!next-token;          !!!next-token;
6892          next B;          next B;
6893        } elsif ($token->{tag_name} eq 'meta') {        } elsif ($token->{tag_name} eq 'meta') {
6894          ## NOTE: This is an "as if in head" code clone, only "-t" differs          ## NOTE: This is an "as if in head" code clone, only "-t" differs
6895          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
6896          my $meta_el = pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.          my $meta_el = pop @{$self->{open_elements}};
6897    
6898          unless ($self->{confident}) {          unless ($self->{confident}) {
6899            if ($token->{attributes}->{charset}) {            if ($token->{attributes}->{charset}) {
# Line 6364  sub _tree_construction_main ($) { Line 6909  sub _tree_construction_main ($) {
6909                                           ->{has_reference});                                           ->{has_reference});
6910            } elsif ($token->{attributes}->{content}) {            } elsif ($token->{attributes}->{content}) {
6911              if ($token->{attributes}->{content}->{value}              if ($token->{attributes}->{content}->{value}
6912                  =~ /\A[^;]*;[\x09-\x0D\x20]*[Cc][Hh][Aa][Rr][Ss][Ee][Tt]                  =~ /[Cc][Hh][Aa][Rr][Ss][Ee][Tt]
6913                      [\x09-\x0D\x20]*=                      [\x09\x0A\x0C\x0D\x20]*=
6914                      [\x09-\x0D\x20]*(?>"([^"]*)"|'([^']*)'|                      [\x09\x0A\x0C\x0D\x20]*(?>"([^"]*)"|'([^']*)'|
6915                      ([^"'\x09-\x0D\x20][^\x09-\x0D\x20]*))/x) {                      ([^"'\x09\x0A\x0C\x0D\x20][^\x09\x0A\x0C\x0D\x20\x3B]*))
6916                       /x) {
6917                !!!cp ('t336');                !!!cp ('t336');
6918                ## NOTE: Whether the encoding is supported or not is handled                ## NOTE: Whether the encoding is supported or not is handled
6919                ## in the {change_encoding} callback.                ## in the {change_encoding} callback.
# Line 6405  sub _tree_construction_main ($) { Line 6951  sub _tree_construction_main ($) {
6951          $parse_rcdata->(RCDATA_CONTENT_MODEL);          $parse_rcdata->(RCDATA_CONTENT_MODEL);
6952          next B;          next B;
6953        } elsif ($token->{tag_name} eq 'body') {        } elsif ($token->{tag_name} eq 'body') {
6954          !!!parse-error (type => 'in body:body', token => $token);          !!!parse-error (type => 'in body', text => 'body', token => $token);
6955                                
6956          if (@{$self->{open_elements}} == 1 or          if (@{$self->{open_elements}} == 1 or
6957              not ($self->{open_elements}->[1]->[1] & BODY_EL)) {              not ($self->{open_elements}->[1]->[1] & BODY_EL)) {
# Line 6426  sub _tree_construction_main ($) { Line 6972  sub _tree_construction_main ($) {
6972          !!!next-token;          !!!next-token;
6973          next B;          next B;
6974        } elsif ({        } elsif ({
6975                  address => 1, blockquote => 1, center => 1, dir => 1,                  ## NOTE: Start tags for non-phrasing flow content elements
6976                  div => 1, dl => 1, fieldset => 1,  
6977                  h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,                  ## NOTE: The normal one
6978                  menu => 1, ol => 1, p => 1, ul => 1,                  address => 1, article => 1, aside => 1, blockquote => 1,
6979                    center => 1, datagrid => 1, details => 1, dialog => 1,
6980                    dir => 1, div => 1, dl => 1, fieldset => 1, figure => 1,
6981                    footer => 1, h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1,
6982                    h6 => 1, header => 1, menu => 1, nav => 1, ol => 1, p => 1,
6983                    section => 1, ul => 1,
6984                    ## NOTE: As normal, but drops leading newline
6985                  pre => 1, listing => 1,                  pre => 1, listing => 1,
6986                    ## NOTE: As normal, but interacts with the form element pointer
6987                  form => 1,                  form => 1,
6988                    
6989                  table => 1,                  table => 1,
6990                  hr => 1,                  hr => 1,
6991                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
# Line 6498  sub _tree_construction_main ($) { Line 7052  sub _tree_construction_main ($) {
7052            !!!next-token;            !!!next-token;
7053          }          }
7054          next B;          next B;
7055        } elsif ({li => 1, dt => 1, dd => 1}->{$token->{tag_name}}) {        } elsif ($token->{tag_name} eq 'li') {
7056          ## has a p element in scope          ## NOTE: As normal, but imply </li> when there's another <li> ...
7057    
7058            ## NOTE: Special, Scope (<li><foo><li> == <li><foo><li/></foo></li>)
7059              ## Interpreted as <li><foo/></li><li/> (non-conforming)
7060              ## blockquote (O9.27), center (O), dd (Fx3, O, S3.1.2, IE7),
7061              ## dt (Fx, O, S, IE), dl (O), fieldset (O, S, IE), form (Fx, O, S),
7062              ## hn (O), pre (O), applet (O, S), button (O, S), marquee (Fx, O, S),
7063              ## object (Fx)
7064              ## Generate non-tree (non-conforming)
7065              ## basefont (IE7 (where basefont is non-void)), center (IE),
7066              ## form (IE), hn (IE)
7067            ## address, div, p (<li><foo><li> == <li><foo/></li><li/>)
7068              ## Interpreted as <li><foo><li/></foo></li> (non-conforming)
7069              ## div (Fx, S)
7070    
7071            my $non_optional;
7072            my $i = -1;
7073    
7074            ## 1.
7075            for my $node (reverse @{$self->{open_elements}}) {
7076              if ($node->[1] & LI_EL) {
7077                ## 2. (a) As if </li>
7078                {
7079                  ## If no </li> - not applied
7080                  #
7081    
7082                  ## Otherwise
7083    
7084                  ## 1. generate implied end tags, except for </li>
7085                  #
7086    
7087                  ## 2. If current node != "li", parse error
7088                  if ($non_optional) {
7089                    !!!parse-error (type => 'not closed',
7090                                    text => $non_optional->[0]->manakai_local_name,
7091                                    token => $token);
7092                    !!!cp ('t355');
7093                  } else {
7094                    !!!cp ('t356');
7095                  }
7096    
7097                  ## 3. Pop
7098                  splice @{$self->{open_elements}}, $i;
7099                }
7100    
7101                last; ## 2. (b) goto 5.
7102              } elsif (
7103                       ## NOTE: not "formatting" and not "phrasing"
7104                       ($node->[1] & SPECIAL_EL or
7105                        $node->[1] & SCOPING_EL) and
7106                       ## NOTE: "li", "dt", and "dd" are in |SPECIAL_EL|.
7107    
7108                       (not $node->[1] & ADDRESS_EL) &
7109                       (not $node->[1] & DIV_EL) &
7110                       (not $node->[1] & P_EL)) {
7111                ## 3.
7112                !!!cp ('t357');
7113                last; ## goto 5.
7114              } elsif ($node->[1] & END_TAG_OPTIONAL_EL) {
7115                !!!cp ('t358');
7116                #
7117              } else {
7118                !!!cp ('t359');
7119                $non_optional ||= $node;
7120                #
7121              }
7122              ## 4.
7123              ## goto 2.
7124              $i--;
7125            }
7126    
7127            ## 5. (a) has a |p| element in scope
7128          INSCOPE: for (reverse @{$self->{open_elements}}) {          INSCOPE: for (reverse @{$self->{open_elements}}) {
7129            if ($_->[1] & P_EL) {            if ($_->[1] & P_EL) {
7130              !!!cp ('t353');              !!!cp ('t353');
7131    
7132                ## NOTE: |<p><li>|, for example.
7133    
7134              !!!back-token; # <x>              !!!back-token; # <x>
7135              $token = {type => END_TAG_TOKEN, tag_name => 'p',              $token = {type => END_TAG_TOKEN, tag_name => 'p',
7136                        line => $token->{line}, column => $token->{column}};                        line => $token->{line}, column => $token->{column}};
# Line 6512  sub _tree_construction_main ($) { Line 7140  sub _tree_construction_main ($) {
7140              last INSCOPE;              last INSCOPE;
7141            }            }
7142          } # INSCOPE          } # INSCOPE
7143              
7144          ## Step 1          ## 5. (b) insert
7145            !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7146            !!!nack ('t359.1');
7147            !!!next-token;
7148            next B;
7149          } elsif ($token->{tag_name} eq 'dt' or
7150                   $token->{tag_name} eq 'dd') {
7151            ## NOTE: As normal, but imply </dt> or </dd> when ...
7152    
7153            my $non_optional;
7154          my $i = -1;          my $i = -1;
7155          my $node = $self->{open_elements}->[$i];  
7156          my $li_or_dtdd = {li => {li => 1},          ## 1.
7157                            dt => {dt => 1, dd => 1},          for my $node (reverse @{$self->{open_elements}}) {
7158                            dd => {dt => 1, dd => 1}}->{$token->{tag_name}};            if ($node->[1] & DT_EL or $node->[1] & DD_EL) {
7159          LI: {              ## 2. (a) As if </li>
7160            ## Step 2              {
7161            if ($li_or_dtdd->{$node->[0]->manakai_local_name}) {                ## If no </li> - not applied
7162              if ($i != -1) {                #
7163                !!!cp ('t355');  
7164                !!!parse-error (type => 'not closed',                ## Otherwise
7165                                value => $self->{open_elements}->[-1]->[0]  
7166                                    ->manakai_local_name,                ## 1. generate implied end tags, except for </dt> or </dd>
7167                                token => $token);                #
7168              } else {  
7169                !!!cp ('t356');                ## 2. If current node != "dt"|"dd", parse error
7170                  if ($non_optional) {
7171                    !!!parse-error (type => 'not closed',
7172                                    text => $non_optional->[0]->manakai_local_name,
7173                                    token => $token);
7174                    !!!cp ('t355.1');
7175                  } else {
7176                    !!!cp ('t356.1');
7177                  }
7178    
7179                  ## 3. Pop
7180                  splice @{$self->{open_elements}}, $i;
7181              }              }
7182              splice @{$self->{open_elements}}, $i;  
7183              last LI;              last; ## 2. (b) goto 5.
7184              } elsif (
7185                       ## NOTE: not "formatting" and not "phrasing"
7186                       ($node->[1] & SPECIAL_EL or
7187                        $node->[1] & SCOPING_EL) and
7188                       ## NOTE: "li", "dt", and "dd" are in |SPECIAL_EL|.
7189    
7190                       (not $node->[1] & ADDRESS_EL) &
7191                       (not $node->[1] & DIV_EL) &
7192                       (not $node->[1] & P_EL)) {
7193                ## 3.
7194                !!!cp ('t357.1');
7195                last; ## goto 5.
7196              } elsif ($node->[1] & END_TAG_OPTIONAL_EL) {
7197                !!!cp ('t358.1');
7198                #
7199            } else {            } else {
7200              !!!cp ('t357');              !!!cp ('t359.1');
7201            }              $non_optional ||= $node;
7202                          #
           ## Step 3  
           if (not ($node->[1] & FORMATTING_EL) and  
               #not $phrasing_category->{$node->[1]} and  
               ($node->[1] & SPECIAL_EL or  
                $node->[1] & SCOPING_EL) and  
               not ($node->[1] & ADDRESS_EL) and  
               not ($node->[1] & DIV_EL)) {  
             !!!cp ('t358');  
             last LI;  
7203            }            }
7204                        ## 4.
7205            !!!cp ('t359');            ## goto 2.
           ## Step 4  
7206            $i--;            $i--;
7207            $node = $self->{open_elements}->[$i];          }
7208            redo LI;  
7209          } # LI          ## 5. (a) has a |p| element in scope
7210                      INSCOPE: for (reverse @{$self->{open_elements}}) {
7211              if ($_->[1] & P_EL) {
7212                !!!cp ('t353.1');
7213                !!!back-token; # <x>
7214                $token = {type => END_TAG_TOKEN, tag_name => 'p',
7215                          line => $token->{line}, column => $token->{column}};
7216                next B;
7217              } elsif ($_->[1] & SCOPING_EL) {
7218                !!!cp ('t354.1');
7219                last INSCOPE;
7220              }
7221            } # INSCOPE
7222    
7223            ## 5. (b) insert
7224          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7225          !!!nack ('t359.1');          !!!nack ('t359.2');
7226          !!!next-token;          !!!next-token;
7227          next B;          next B;
7228        } elsif ($token->{tag_name} eq 'plaintext') {        } elsif ($token->{tag_name} eq 'plaintext') {
7229            ## NOTE: As normal, but effectively ends parsing
7230    
7231          ## has a p element in scope          ## has a p element in scope
7232          INSCOPE: for (reverse @{$self->{open_elements}}) {          INSCOPE: for (reverse @{$self->{open_elements}}) {
7233            if ($_->[1] & P_EL) {            if ($_->[1] & P_EL) {
# Line 6679  sub _tree_construction_main ($) { Line 7347  sub _tree_construction_main ($) {
7347                  xmp => 1,                  xmp => 1,
7348                  iframe => 1,                  iframe => 1,
7349                  noembed => 1,                  noembed => 1,
7350                  noframes => 1,                  noframes => 1, ## NOTE: This is an "as if in head" code clone.
7351                  noscript => 0, ## TODO: 1 if scripting is enabled                  noscript => 0, ## TODO: 1 if scripting is enabled
7352                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
7353          if ($token->{tag_name} eq 'xmp') {          if ($token->{tag_name} eq 'xmp') {
# Line 6701  sub _tree_construction_main ($) { Line 7369  sub _tree_construction_main ($) {
7369            !!!next-token;            !!!next-token;
7370            next B;            next B;
7371          } else {          } else {
7372              !!!ack ('t391.1');
7373    
7374            my $at = $token->{attributes};            my $at = $token->{attributes};
7375            my $form_attrs;            my $form_attrs;
7376            $form_attrs->{action} = $at->{action} if $at->{action};            $form_attrs->{action} = $at->{action} if $at->{action};
# Line 6744  sub _tree_construction_main ($) { Line 7414  sub _tree_construction_main ($) {
7414                           line => $token->{line}, column => $token->{column}},                           line => $token->{line}, column => $token->{column}},
7415                          {type => END_TAG_TOKEN, tag_name => 'form',                          {type => END_TAG_TOKEN, tag_name => 'form',
7416                           line => $token->{line}, column => $token->{column}};                           line => $token->{line}, column => $token->{column}};
           !!!nack ('t391.1'); ## NOTE: Not acknowledged.  
7417            !!!back-token (@tokens);            !!!back-token (@tokens);
7418            !!!next-token;            !!!next-token;
7419            next B;            next B;
7420          }          }
7421        } elsif ($token->{tag_name} eq 'textarea') {        } elsif ($token->{tag_name} eq 'textarea') {
7422          my $tag_name = $token->{tag_name};          ## Step 1
7423          my $el;          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
         !!!create-element ($el, $HTML_NS, $token->{tag_name}, $token->{attributes}, $token);  
7424                    
7425            ## Step 2
7426          ## TODO: $self->{form_element} if defined          ## TODO: $self->{form_element} if defined
7427    
7428            ## Step 3
7429            $self->{ignore_newline} = 1;
7430    
7431            ## Step 4
7432            ## ISSUE: This step is wrong. (r2302 enbugged)
7433    
7434            ## Step 5
7435          $self->{content_model} = RCDATA_CONTENT_MODEL;          $self->{content_model} = RCDATA_CONTENT_MODEL;
7436          delete $self->{escape}; # MUST          delete $self->{escape}; # MUST
7437            
7438          $insert->($el);          ## Step 6-7
7439                    $self->{insertion_mode} |= IN_CDATA_RCDATA_IM;
7440          my $text = '';  
7441          !!!nack ('t392.1');          !!!nack ('t392.1');
7442          !!!next-token;          !!!next-token;
7443          if ($token->{type} == CHARACTER_TOKEN) {          next B;
7444            $token->{data} =~ s/^\x0A//;        } elsif ($token->{tag_name} eq 'optgroup' or
7445            unless (length $token->{data}) {                 $token->{tag_name} eq 'option') {
7446              !!!cp ('t392');          ## has an |option| element in scope
7447              !!!next-token;          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7448            } else {            my $node = $self->{open_elements}->[$_];
7449              !!!cp ('t393');            if ($node->[1] & OPTION_EL) {
7450                !!!cp ('t397.1');
7451                ## NOTE: As if </option>
7452                !!!back-token; # <option> or <optgroup>
7453                $token = {type => END_TAG_TOKEN, tag_name => 'option',
7454                          line => $token->{line}, column => $token->{column}};
7455                next B;
7456              } elsif ($node->[1] & SCOPING_EL) {
7457                !!!cp ('t397.2');
7458                last INSCOPE;
7459            }            }
7460          } else {          } # INSCOPE
7461            !!!cp ('t394');  
7462          }          $reconstruct_active_formatting_elements->($insert_to_current);
7463          while ($token->{type} == CHARACTER_TOKEN) {  
7464            !!!cp ('t395');          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7465            $text .= $token->{data};  
7466            !!!next-token;          !!!nack ('t397.3');
         }  
         if (length $text) {  
           !!!cp ('t396');  
           $el->manakai_append_text ($text);  
         }  
           
         $self->{content_model} = PCDATA_CONTENT_MODEL;  
           
         if ($token->{type} == END_TAG_TOKEN and  
             $token->{tag_name} eq $tag_name) {  
           !!!cp ('t397');  
           ## Ignore the token  
         } else {  
           !!!cp ('t398');  
           !!!parse-error (type => 'in RCDATA:#'.$token->{type}, token => $token);  
         }  
7467          !!!next-token;          !!!next-token;
7468          next B;          redo B;
7469          } elsif ($token->{tag_name} eq 'rt' or
7470                   $token->{tag_name} eq 'rp') {
7471            ## has a |ruby| element in scope
7472            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7473              my $node = $self->{open_elements}->[$_];
7474              if ($node->[1] & RUBY_EL) {
7475                !!!cp ('t398.1');
7476                ## generate implied end tags
7477                while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
7478                  !!!cp ('t398.2');
7479                  pop @{$self->{open_elements}};
7480                }
7481                unless ($self->{open_elements}->[-1]->[1] & RUBY_EL) {
7482                  !!!cp ('t398.3');
7483                  !!!parse-error (type => 'not closed',
7484                                  text => $self->{open_elements}->[-1]->[0]
7485                                      ->manakai_local_name,
7486                                  token => $token);
7487                  pop @{$self->{open_elements}}
7488                      while not $self->{open_elements}->[-1]->[1] & RUBY_EL;
7489                }
7490                last INSCOPE;
7491              } elsif ($node->[1] & SCOPING_EL) {
7492                !!!cp ('t398.4');
7493                last INSCOPE;
7494              }
7495            } # INSCOPE
7496    
7497            !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7498    
7499            !!!nack ('t398.5');
7500            !!!next-token;
7501            redo B;
7502        } elsif ($token->{tag_name} eq 'math' or        } elsif ($token->{tag_name} eq 'math' or
7503                 $token->{tag_name} eq 'svg') {                 $token->{tag_name} eq 'svg') {
7504          $reconstruct_active_formatting_elements->($insert_to_current);          $reconstruct_active_formatting_elements->($insert_to_current);
7505    
7506            ## "Adjust MathML attributes" ('math' only) - done in insert-element-f
7507    
7508          ## "adjust SVG attributes" ('svg' only) - done in insert-element-f          ## "adjust SVG attributes" ('svg' only) - done in insert-element-f
7509    
7510          ## "adjust foreign attributes" - done in insert-element-f          ## "adjust foreign attributes" - done in insert-element-f
# Line 6808  sub _tree_construction_main ($) { Line 7513  sub _tree_construction_main ($) {
7513                    
7514          if ($self->{self_closing}) {          if ($self->{self_closing}) {
7515            pop @{$self->{open_elements}};            pop @{$self->{open_elements}};
7516            !!!ack ('t398.1');            !!!ack ('t398.6');
7517          } else {          } else {
7518            !!!cp ('t398.2');            !!!cp ('t398.7');
7519            $self->{insertion_mode} |= IN_FOREIGN_CONTENT_IM;            $self->{insertion_mode} |= IN_FOREIGN_CONTENT_IM;
7520            ## NOTE: |<body><math><mi><svg>| -> "in foreign content" insertion            ## NOTE: |<body><math><mi><svg>| -> "in foreign content" insertion
7521            ## mode, "in body" (not "in foreign content") secondary insertion            ## mode, "in body" (not "in foreign content") secondary insertion
# Line 6821  sub _tree_construction_main ($) { Line 7526  sub _tree_construction_main ($) {
7526          next B;          next B;
7527        } elsif ({        } elsif ({
7528                  caption => 1, col => 1, colgroup => 1, frame => 1,                  caption => 1, col => 1, colgroup => 1, frame => 1,
7529                  frameset => 1, head => 1, option => 1, optgroup => 1,                  frameset => 1, head => 1,
7530                  tbody => 1, td => 1, tfoot => 1, th => 1,                  tbody => 1, td => 1, tfoot => 1, th => 1,
7531                  thead => 1, tr => 1,                  thead => 1, tr => 1,
7532                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
7533          !!!cp ('t401');          !!!cp ('t401');
7534          !!!parse-error (type => 'in body:'.$token->{tag_name}, token => $token);          !!!parse-error (type => 'in body',
7535                            text => $token->{tag_name}, token => $token);
7536          ## Ignore the token          ## Ignore the token
7537          !!!nack ('t401.1'); ## NOTE: |<col/>| or |<frame/>| here is an error.          !!!nack ('t401.1'); ## NOTE: |<col/>| or |<frame/>| here is an error.
7538          !!!next-token;          !!!next-token;
7539          next B;          next B;
7540                  } elsif ($token->{tag_name} eq 'param' or
7541          ## ISSUE: An issue on HTML5 new elements in the spec.                 $token->{tag_name} eq 'source') {
7542            !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7543            pop @{$self->{open_elements}};
7544    
7545            !!!ack ('t398.5');
7546            !!!next-token;
7547            redo B;
7548        } else {        } else {
7549          if ($token->{tag_name} eq 'image') {          if ($token->{tag_name} eq 'image') {
7550            !!!cp ('t384');            !!!cp ('t384');
# Line 6855  sub _tree_construction_main ($) { Line 7567  sub _tree_construction_main ($) {
7567            !!!nack ('t380.1');            !!!nack ('t380.1');
7568          } elsif ({          } elsif ({
7569                    b => 1, big => 1, em => 1, font => 1, i => 1,                    b => 1, big => 1, em => 1, font => 1, i => 1,
7570                    s => 1, small => 1, strile => 1,                    s => 1, small => 1, strike => 1,
7571                    strong => 1, tt => 1, u => 1,                    strong => 1, tt => 1, u => 1,
7572                   }->{$token->{tag_name}}) {                   }->{$token->{tag_name}}) {
7573            !!!cp ('t375');            !!!cp ('t375');
# Line 6868  sub _tree_construction_main ($) { Line 7580  sub _tree_construction_main ($) {
7580            !!!ack ('t388.2');            !!!ack ('t388.2');
7581          } elsif ({          } elsif ({
7582                    area => 1, basefont => 1, bgsound => 1, br => 1,                    area => 1, basefont => 1, bgsound => 1, br => 1,
7583                    embed => 1, img => 1, param => 1, spacer => 1, wbr => 1,                    embed => 1, img => 1, spacer => 1, wbr => 1,
                   #image => 1,  
7584                   }->{$token->{tag_name}}) {                   }->{$token->{tag_name}}) {
7585            !!!cp ('t388.1');            !!!cp ('t388.1');
7586            pop @{$self->{open_elements}};            pop @{$self->{open_elements}};
# Line 6910  sub _tree_construction_main ($) { Line 7621  sub _tree_construction_main ($) {
7621              }              }
7622            }            }
7623    
7624            !!!parse-error (type => 'start tag not allowed',            ## NOTE: |<marquee></body>|, |<svg><foreignobject></body>|
7625                            value => $token->{tag_name}, token => $token);  
7626              !!!parse-error (type => 'unmatched end tag',
7627                              text => $token->{tag_name}, token => $token);
7628            ## NOTE: Ignore the token.            ## NOTE: Ignore the token.
7629            !!!next-token;            !!!next-token;
7630            next B;            next B;
# Line 6921  sub _tree_construction_main ($) { Line 7634  sub _tree_construction_main ($) {
7634            unless ($_->[1] & ALL_END_TAG_OPTIONAL_EL) {            unless ($_->[1] & ALL_END_TAG_OPTIONAL_EL) {
7635              !!!cp ('t403');              !!!cp ('t403');
7636              !!!parse-error (type => 'not closed',              !!!parse-error (type => 'not closed',
7637                              value => $_->[0]->manakai_local_name,                              text => $_->[0]->manakai_local_name,
7638                              token => $token);                              token => $token);
7639              last;              last;
7640            } else {            } else {
# Line 6937  sub _tree_construction_main ($) { Line 7650  sub _tree_construction_main ($) {
7650          ## up-to-date, though it has same effect as speced.          ## up-to-date, though it has same effect as speced.
7651          if (@{$self->{open_elements}} > 1 and          if (@{$self->{open_elements}} > 1 and
7652              $self->{open_elements}->[1]->[1] & BODY_EL) {              $self->{open_elements}->[1]->[1] & BODY_EL) {
           ## ISSUE: There is an issue in the spec.  
7653            unless ($self->{open_elements}->[-1]->[1] & BODY_EL) {            unless ($self->{open_elements}->[-1]->[1] & BODY_EL) {
7654              !!!cp ('t406');              !!!cp ('t406');
7655              !!!parse-error (type => 'not closed',              !!!parse-error (type => 'not closed',
7656                              value => $self->{open_elements}->[1]->[0]                              text => $self->{open_elements}->[1]->[0]
7657                                  ->manakai_local_name,                                  ->manakai_local_name,
7658                              token => $token);                              token => $token);
7659            } else {            } else {
# Line 6952  sub _tree_construction_main ($) { Line 7664  sub _tree_construction_main ($) {
7664            next B;            next B;
7665          } else {          } else {
7666            !!!cp ('t408');            !!!cp ('t408');
7667            !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);            !!!parse-error (type => 'unmatched end tag',
7668                              text => $token->{tag_name}, token => $token);
7669            ## Ignore the token            ## Ignore the token
7670            !!!next-token;            !!!next-token;
7671            next B;            next B;
7672          }          }
7673        } elsif ({        } elsif ({
7674                  address => 1, blockquote => 1, center => 1, dir => 1,                  ## NOTE: End tags for non-phrasing flow content elements
7675                  div => 1, dl => 1, fieldset => 1, listing => 1,  
7676                  menu => 1, ol => 1, pre => 1, ul => 1,                  ## NOTE: The normal ones
7677                    address => 1, article => 1, aside => 1, blockquote => 1,
7678                    center => 1, datagrid => 1, details => 1, dialog => 1,
7679                    dir => 1, div => 1, dl => 1, fieldset => 1, figure => 1,
7680                    footer => 1, header => 1, listing => 1, menu => 1, nav => 1,
7681                    ol => 1, pre => 1, section => 1, ul => 1,
7682    
7683                    ## NOTE: As normal, but ... optional tags
7684                  dd => 1, dt => 1, li => 1,                  dd => 1, dt => 1, li => 1,
7685    
7686                  applet => 1, button => 1, marquee => 1, object => 1,                  applet => 1, button => 1, marquee => 1, object => 1,
7687                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
7688            ## NOTE: Code for <li> start tags includes "as if </li>" code.
7689            ## Code for <dt> or <dd> start tags includes "as if </dt> or
7690            ## </dd>" code.
7691    
7692          ## has an element in scope          ## has an element in scope
7693          my $i;          my $i;
7694          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
# Line 6980  sub _tree_construction_main ($) { Line 7705  sub _tree_construction_main ($) {
7705    
7706          unless (defined $i) { # has an element in scope          unless (defined $i) { # has an element in scope
7707            !!!cp ('t413');            !!!cp ('t413');
7708            !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);            !!!parse-error (type => 'unmatched end tag',
7709                              text => $token->{tag_name}, token => $token);
7710              ## NOTE: Ignore the token.
7711          } else {          } else {
7712            ## Step 1. generate implied end tags            ## Step 1. generate implied end tags
7713            while ({            while ({
7714                      ## END_TAG_OPTIONAL_EL
7715                    dd => ($token->{tag_name} ne 'dd'),                    dd => ($token->{tag_name} ne 'dd'),
7716                    dt => ($token->{tag_name} ne 'dt'),                    dt => ($token->{tag_name} ne 'dt'),
7717                    li => ($token->{tag_name} ne 'li'),                    li => ($token->{tag_name} ne 'li'),
7718                      option => 1,
7719                      optgroup => 1,
7720                    p => 1,                    p => 1,
7721                      rt => 1,
7722                      rp => 1,
7723                   }->{$self->{open_elements}->[-1]->[0]->manakai_local_name}) {                   }->{$self->{open_elements}->[-1]->[0]->manakai_local_name}) {
7724              !!!cp ('t409');              !!!cp ('t409');
7725              pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
# Line 6998  sub _tree_construction_main ($) { Line 7730  sub _tree_construction_main ($) {
7730                    ne $token->{tag_name}) {                    ne $token->{tag_name}) {
7731              !!!cp ('t412');              !!!cp ('t412');
7732              !!!parse-error (type => 'not closed',              !!!parse-error (type => 'not closed',
7733                              value => $self->{open_elements}->[-1]->[0]                              text => $self->{open_elements}->[-1]->[0]
7734                                  ->manakai_local_name,                                  ->manakai_local_name,
7735                              token => $token);                              token => $token);
7736            } else {            } else {
# Line 7017  sub _tree_construction_main ($) { Line 7749  sub _tree_construction_main ($) {
7749          !!!next-token;          !!!next-token;
7750          next B;          next B;
7751        } elsif ($token->{tag_name} eq 'form') {        } elsif ($token->{tag_name} eq 'form') {
7752            ## NOTE: As normal, but interacts with the form element pointer
7753    
7754          undef $self->{form_element};          undef $self->{form_element};
7755    
7756          ## has an element in scope          ## has an element in scope
# Line 7035  sub _tree_construction_main ($) { Line 7769  sub _tree_construction_main ($) {
7769    
7770          unless (defined $i) { # has an element in scope          unless (defined $i) { # has an element in scope
7771            !!!cp ('t421');            !!!cp ('t421');
7772            !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);            !!!parse-error (type => 'unmatched end tag',
7773                              text => $token->{tag_name}, token => $token);
7774              ## NOTE: Ignore the token.
7775          } else {          } else {
7776            ## Step 1. generate implied end tags            ## Step 1. generate implied end tags
7777            while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {            while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
# Line 7048  sub _tree_construction_main ($) { Line 7784  sub _tree_construction_main ($) {
7784                    ne $token->{tag_name}) {                    ne $token->{tag_name}) {
7785              !!!cp ('t417.1');              !!!cp ('t417.1');
7786              !!!parse-error (type => 'not closed',              !!!parse-error (type => 'not closed',
7787                              value => $self->{open_elements}->[-1]->[0]                              text => $self->{open_elements}->[-1]->[0]
7788                                  ->manakai_local_name,                                  ->manakai_local_name,
7789                              token => $token);                              token => $token);
7790            } else {            } else {
# Line 7062  sub _tree_construction_main ($) { Line 7798  sub _tree_construction_main ($) {
7798          !!!next-token;          !!!next-token;
7799          next B;          next B;
7800        } elsif ({        } elsif ({
7801                    ## NOTE: As normal, except acts as a closer for any ...
7802                  h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,                  h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,
7803                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
7804          ## has an element in scope          ## has an element in scope
# Line 7080  sub _tree_construction_main ($) { Line 7817  sub _tree_construction_main ($) {
7817    
7818          unless (defined $i) { # has an element in scope          unless (defined $i) { # has an element in scope
7819            !!!cp ('t425.1');            !!!cp ('t425.1');
7820            !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);            !!!parse-error (type => 'unmatched end tag',
7821                              text => $token->{tag_name}, token => $token);
7822              ## NOTE: Ignore the token.
7823          } else {          } else {
7824            ## Step 1. generate implied end tags            ## Step 1. generate implied end tags
7825            while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {            while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
# Line 7092  sub _tree_construction_main ($) { Line 7831  sub _tree_construction_main ($) {
7831            if ($self->{open_elements}->[-1]->[0]->manakai_local_name            if ($self->{open_elements}->[-1]->[0]->manakai_local_name
7832                    ne $token->{tag_name}) {                    ne $token->{tag_name}) {
7833              !!!cp ('t425');              !!!cp ('t425');
7834              !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);              !!!parse-error (type => 'unmatched end tag',
7835                                text => $token->{tag_name}, token => $token);
7836            } else {            } else {
7837              !!!cp ('t426');              !!!cp ('t426');
7838            }            }
# Line 7104  sub _tree_construction_main ($) { Line 7844  sub _tree_construction_main ($) {
7844          !!!next-token;          !!!next-token;
7845          next B;          next B;
7846        } elsif ($token->{tag_name} eq 'p') {        } elsif ($token->{tag_name} eq 'p') {
7847            ## NOTE: As normal, except </p> implies <p> and ...
7848    
7849          ## has an element in scope          ## has an element in scope
7850            my $non_optional;
7851          my $i;          my $i;
7852          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7853            my $node = $self->{open_elements}->[$_];            my $node = $self->{open_elements}->[$_];
# Line 7115  sub _tree_construction_main ($) { Line 7858  sub _tree_construction_main ($) {
7858            } elsif ($node->[1] & SCOPING_EL) {            } elsif ($node->[1] & SCOPING_EL) {
7859              !!!cp ('t411.1');              !!!cp ('t411.1');
7860              last INSCOPE;              last INSCOPE;
7861              } elsif ($node->[1] & END_TAG_OPTIONAL_EL) {
7862                ## NOTE: |END_TAG_OPTIONAL_EL| includes "p"
7863                !!!cp ('t411.2');
7864                #
7865              } else {
7866                !!!cp ('t411.3');
7867                $non_optional ||= $node;
7868                #
7869            }            }
7870          } # INSCOPE          } # INSCOPE
7871    
7872          if (defined $i) {          if (defined $i) {
7873            if ($self->{open_elements}->[-1]->[0]->manakai_local_name            ## 1. Generate implied end tags
7874                    ne $token->{tag_name}) {            #
7875    
7876              ## 2. If current node != "p", parse error
7877              if ($non_optional) {
7878              !!!cp ('t412.1');              !!!cp ('t412.1');
7879              !!!parse-error (type => 'not closed',              !!!parse-error (type => 'not closed',
7880                              value => $self->{open_elements}->[-1]->[0]                              text => $non_optional->[0]->manakai_local_name,
                                 ->manakai_local_name,  
7881                              token => $token);                              token => $token);
7882            } else {            } else {
7883              !!!cp ('t414.1');              !!!cp ('t414.1');
7884            }            }
7885    
7886              ## 3. Pop
7887            splice @{$self->{open_elements}}, $i;            splice @{$self->{open_elements}}, $i;
7888          } else {          } else {
7889            !!!cp ('t413.1');            !!!cp ('t413.1');
7890            !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);            !!!parse-error (type => 'unmatched end tag',
7891                              text => $token->{tag_name}, token => $token);
7892    
7893            !!!cp ('t415.1');            !!!cp ('t415.1');
7894            ## As if <p>, then reprocess the current token            ## As if <p>, then reprocess the current token
# Line 7148  sub _tree_construction_main ($) { Line 7903  sub _tree_construction_main ($) {
7903        } elsif ({        } elsif ({
7904                  a => 1,                  a => 1,
7905                  b => 1, big => 1, em => 1, font => 1, i => 1,                  b => 1, big => 1, em => 1, font => 1, i => 1,
7906                  nobr => 1, s => 1, small => 1, strile => 1,                  nobr => 1, s => 1, small => 1, strike => 1,
7907                  strong => 1, tt => 1, u => 1,                  strong => 1, tt => 1, u => 1,
7908                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
7909          !!!cp ('t427');          !!!cp ('t427');
# Line 7156  sub _tree_construction_main ($) { Line 7911  sub _tree_construction_main ($) {
7911          next B;          next B;
7912        } elsif ($token->{tag_name} eq 'br') {        } elsif ($token->{tag_name} eq 'br') {
7913          !!!cp ('t428');          !!!cp ('t428');
7914          !!!parse-error (type => 'unmatched end tag:br', token => $token);          !!!parse-error (type => 'unmatched end tag',
7915                            text => 'br', token => $token);
7916    
7917          ## As if <br>          ## As if <br>
7918          $reconstruct_active_formatting_elements->($insert_to_current);          $reconstruct_active_formatting_elements->($insert_to_current);
# Line 7168  sub _tree_construction_main ($) { Line 7924  sub _tree_construction_main ($) {
7924          ## Ignore the token.          ## Ignore the token.
7925          !!!next-token;          !!!next-token;
7926          next B;          next B;
       } elsif ({  
                 caption => 1, col => 1, colgroup => 1, frame => 1,  
                 frameset => 1, head => 1, option => 1, optgroup => 1,  
                 tbody => 1, td => 1, tfoot => 1, th => 1,  
                 thead => 1, tr => 1,  
                 area => 1, basefont => 1, bgsound => 1,  
                 embed => 1, hr => 1, iframe => 1, image => 1,  
                 img => 1, input => 1, isindex => 1, noembed => 1,  
                 noframes => 1, param => 1, select => 1, spacer => 1,  
                 table => 1, textarea => 1, wbr => 1,  
                 noscript => 0, ## TODO: if scripting is enabled  
                }->{$token->{tag_name}}) {  
         !!!cp ('t429');  
         !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);  
         ## Ignore the token  
         !!!next-token;  
         next B;  
           
         ## ISSUE: Issue on HTML5 new elements in spec  
           
7927        } else {        } else {
7928            if ($token->{tag_name} eq 'sarcasm') {
7929              sleep 0.001; # take a deep breath
7930            }
7931    
7932          ## Step 1          ## Step 1
7933          my $node_i = -1;          my $node_i = -1;
7934          my $node = $self->{open_elements}->[$node_i];          my $node = $self->{open_elements}->[$node_i];
7935    
7936          ## Step 2          ## Step 2
7937          S2: {          S2: {
7938            if ($node->[0]->manakai_local_name eq $token->{tag_name}) {            my $node_tag_name = $node->[0]->manakai_local_name;
7939              $node_tag_name =~ tr/A-Z/a-z/; # for SVG camelCase tag names
7940              if ($node_tag_name eq $token->{tag_name}) {
7941              ## Step 1              ## Step 1
7942              ## generate implied end tags              ## generate implied end tags
7943              while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {              while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
7944                !!!cp ('t430');                !!!cp ('t430');
7945                ## ISSUE: Can this case be reached?                ## NOTE: |<ruby><rt></ruby>|.
7946                  ## ISSUE: <ruby><rt></rt> will also take this code path,
7947                  ## which seems wrong.
7948                pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
7949                  $node_i++;
7950              }              }
7951                    
7952              ## Step 2              ## Step 2
7953              if ($self->{open_elements}->[-1]->[0]->manakai_local_name              my $current_tag_name
7954                      ne $token->{tag_name}) {                  = $self->{open_elements}->[-1]->[0]->manakai_local_name;
7955                $current_tag_name =~ tr/A-Z/a-z/;
7956                if ($current_tag_name ne $token->{tag_name}) {
7957                !!!cp ('t431');                !!!cp ('t431');
7958                ## NOTE: <x><y></x>                ## NOTE: <x><y></x>
7959                !!!parse-error (type => 'not closed',                !!!parse-error (type => 'not closed',
7960                                value => $self->{open_elements}->[-1]->[0]                                text => $self->{open_elements}->[-1]->[0]
7961                                    ->manakai_local_name,                                    ->manakai_local_name,
7962                                token => $token);                                token => $token);
7963              } else {              } else {
# Line 7218  sub _tree_construction_main ($) { Line 7965  sub _tree_construction_main ($) {
7965              }              }
7966                            
7967              ## Step 3              ## Step 3
7968              splice @{$self->{open_elements}}, $node_i;              splice @{$self->{open_elements}}, $node_i if $node_i < 0;
7969    
7970              !!!next-token;              !!!next-token;
7971              last S2;              last S2;
# Line 7229  sub _tree_construction_main ($) { Line 7976  sub _tree_construction_main ($) {
7976                  ($node->[1] & SPECIAL_EL or                  ($node->[1] & SPECIAL_EL or
7977                   $node->[1] & SCOPING_EL)) {                   $node->[1] & SCOPING_EL)) {
7978                !!!cp ('t433');                !!!cp ('t433');
7979                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);                !!!parse-error (type => 'unmatched end tag',
7980                                  text => $token->{tag_name}, token => $token);
7981                ## Ignore the token                ## Ignore the token
7982                !!!next-token;                !!!next-token;
7983                last S2;                last S2;
             }  
7984    
7985                  ## NOTE: |<span><dd></span>a|: In Safari 3.1.2 and Opera
7986                  ## 9.27, "a" is a child of <dd> (conforming).  In
7987                  ## Firefox 3.0.2, "a" is a child of <body>.  In WinIE 7,
7988                  ## "a" is a child of both <body> and <dd>.
7989                }
7990                
7991              !!!cp ('t434');              !!!cp ('t434');
7992            }            }
7993                        
# Line 7275  sub _tree_construction_main ($) { Line 8028  sub _tree_construction_main ($) {
8028    ## TODO: script stuffs    ## TODO: script stuffs
8029  } # _tree_construct_main  } # _tree_construct_main
8030    
8031  sub set_inner_html ($$$) {  sub set_inner_html ($$$$;$) {
8032    my $class = shift;    my $class = shift;
8033    my $node = shift;    my $node = shift;
8034    my $s = \$_[0];    #my $s = \$_[0];
8035    my $onerror = $_[1];    my $onerror = $_[1];
8036      my $get_wrapper = $_[2] || sub ($) { return $_[0] };
8037    
8038    ## ISSUE: Should {confident} be true?    ## ISSUE: Should {confident} be true?
8039    
# Line 7298  sub set_inner_html ($$$) { Line 8052  sub set_inner_html ($$$) {
8052      }      }
8053    
8054      ## Step 3, 4, 5 # MUST      ## Step 3, 4, 5 # MUST
8055      $class->parse_string ($$s => $node, $onerror);      $class->parse_char_string ($_[0] => $node, $onerror, $get_wrapper);
8056    } elsif ($nt == 1) {    } elsif ($nt == 1) {
8057      ## TODO: If non-html element      ## TODO: If non-html element
8058    
8059      ## NOTE: Most of this code is copied from |parse_string|      ## NOTE: Most of this code is copied from |parse_string|
8060    
8061    ## TODO: Support for $get_wrapper
8062    
8063      ## Step 1 # MUST      ## Step 1 # MUST
8064      my $this_doc = $node->owner_document;      my $this_doc = $node->owner_document;
8065      my $doc = $this_doc->implementation->create_document;      my $doc = $this_doc->implementation->create_document;
# Line 7315  sub set_inner_html ($$$) { Line 8071  sub set_inner_html ($$$) {
8071      my $i = 0;      my $i = 0;
8072      $p->{line_prev} = $p->{line} = 1;      $p->{line_prev} = $p->{line} = 1;
8073      $p->{column_prev} = $p->{column} = 0;      $p->{column_prev} = $p->{column} = 0;
8074      $p->{set_next_char} = sub {      require Whatpm::Charset::DecodeHandle;
8075        my $input = Whatpm::Charset::DecodeHandle::CharString->new (\($_[0]));
8076        $input = $get_wrapper->($input);
8077        $p->{set_nc} = sub {
8078        my $self = shift;        my $self = shift;
8079    
8080        pop @{$self->{prev_char}};        my $char = '';
8081        unshift @{$self->{prev_char}}, $self->{next_char};        if (defined $self->{next_nc}) {
8082            $char = $self->{next_nc};
8083        $self->{next_char} = -1 and return if $i >= length $$s;          delete $self->{next_nc};
8084        $self->{next_char} = ord substr $$s, $i++, 1;          $self->{nc} = ord $char;
8085          } else {
8086            $self->{char_buffer} = '';
8087            $self->{char_buffer_pos} = 0;
8088            
8089            my $count = $input->manakai_read_until
8090                ($self->{char_buffer}, qr/[^\x00\x0A\x0D]/,
8091                 $self->{char_buffer_pos});
8092            if ($count) {
8093              $self->{line_prev} = $self->{line};
8094              $self->{column_prev} = $self->{column};
8095              $self->{column}++;
8096              $self->{nc}
8097                  = ord substr ($self->{char_buffer},
8098                                $self->{char_buffer_pos}++, 1);
8099              return;
8100            }
8101            
8102            if ($input->read ($char, 1)) {
8103              $self->{nc} = ord $char;
8104            } else {
8105              $self->{nc} = -1;
8106              return;
8107            }
8108          }
8109    
8110        ($p->{line_prev}, $p->{column_prev}) = ($p->{line}, $p->{column});        ($p->{line_prev}, $p->{column_prev}) = ($p->{line}, $p->{column});
8111        $p->{column}++;        $p->{column}++;
8112    
8113        if ($self->{next_char} == 0x000A) { # LF        if ($self->{nc} == 0x000A) { # LF
8114          $p->{line}++;          $p->{line}++;
8115          $p->{column} = 0;          $p->{column} = 0;
8116          !!!cp ('i1');          !!!cp ('i1');
8117        } elsif ($self->{next_char} == 0x000D) { # CR        } elsif ($self->{nc} == 0x000D) { # CR
8118          $i++ if substr ($$s, $i, 1) eq "\x0A";  ## TODO: support for abort/streaming
8119          $self->{next_char} = 0x000A; # LF # MUST          my $next = '';
8120            if ($input->read ($next, 1) and $next ne "\x0A") {
8121              $self->{next_nc} = $next;
8122            }
8123            $self->{nc} = 0x000A; # LF # MUST
8124          $p->{line}++;          $p->{line}++;
8125          $p->{column} = 0;          $p->{column} = 0;
8126          !!!cp ('i2');          !!!cp ('i2');
8127        } elsif ($self->{next_char} > 0x10FFFF) {        } elsif ($self->{nc} == 0x0000) { # NULL
         $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST  
         !!!cp ('i3');  
       } elsif ($self->{next_char} == 0x0000) { # NULL  
8128          !!!cp ('i4');          !!!cp ('i4');
8129          !!!parse-error (type => 'NULL');          !!!parse-error (type => 'NULL');
8130          $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST          $self->{nc} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
       } elsif ($self->{next_char} <= 0x0008 or  
                (0x000E <= $self->{next_char} and  
                 $self->{next_char} <= 0x001F) or  
                (0x007F <= $self->{next_char} and  
                 $self->{next_char} <= 0x009F) or  
                (0xD800 <= $self->{next_char} and  
                 $self->{next_char} <= 0xDFFF) or  
                (0xFDD0 <= $self->{next_char} and  
                 $self->{next_char} <= 0xFDDF) or  
                {  
                 0xFFFE => 1, 0xFFFF => 1, 0x1FFFE => 1, 0x1FFFF => 1,  
                 0x2FFFE => 1, 0x2FFFF => 1, 0x3FFFE => 1, 0x3FFFF => 1,  
                 0x4FFFE => 1, 0x4FFFF => 1, 0x5FFFE => 1, 0x5FFFF => 1,  
                 0x6FFFE => 1, 0x6FFFF => 1, 0x7FFFE => 1, 0x7FFFF => 1,  
                 0x8FFFE => 1, 0x8FFFF => 1, 0x9FFFE => 1, 0x9FFFF => 1,  
                 0xAFFFE => 1, 0xAFFFF => 1, 0xBFFFE => 1, 0xBFFFF => 1,  
                 0xCFFFE => 1, 0xCFFFF => 1, 0xDFFFE => 1, 0xDFFFF => 1,  
                 0xEFFFE => 1, 0xEFFFF => 1, 0xFFFFE => 1, 0xFFFFF => 1,  
                 0x10FFFE => 1, 0x10FFFF => 1,  
                }->{$self->{next_char}}) {  
         !!!cp ('i4.1');  
         !!!parse-error (type => 'control char', level => $self->{must_level});  
 ## TODO: error type documentation  
8131        }        }
8132      };      };
8133      $p->{prev_char} = [-1, -1, -1];  
8134      $p->{next_char} = -1;      $p->{read_until} = sub {
8135              #my ($scalar, $specials_range, $offset) = @_;
8136          return 0 if defined $p->{next_nc};
8137    
8138          my $pattern = qr/[^$_[1]\x00\x0A\x0D]/;
8139          my $offset = $_[2] || 0;
8140          
8141          if ($p->{char_buffer_pos} < length $p->{char_buffer}) {
8142            pos ($p->{char_buffer}) = $p->{char_buffer_pos};
8143            if ($p->{char_buffer} =~ /\G(?>$pattern)+/) {
8144              substr ($_[0], $offset)
8145                  = substr ($p->{char_buffer}, $-[0], $+[0] - $-[0]);
8146              my $count = $+[0] - $-[0];
8147              if ($count) {
8148                $p->{column} += $count;
8149                $p->{char_buffer_pos} += $count;
8150                $p->{line_prev} = $p->{line};
8151                $p->{column_prev} = $p->{column} - 1;
8152                $p->{nc} = -1;
8153              }
8154              return $count;
8155            } else {
8156              return 0;
8157            }
8158          } else {
8159            my $count = $input->manakai_read_until ($_[0], $pattern, $_[2]);
8160            if ($count) {
8161              $p->{column} += $count;
8162              $p->{column_prev} += $count;
8163              $p->{nc} = -1;
8164            }
8165            return $count;
8166          }
8167        }; # $p->{read_until}
8168    
8169      my $ponerror = $onerror || sub {      my $ponerror = $onerror || sub {
8170        my (%opt) = @_;        my (%opt) = @_;
8171        my $line = $opt{line};        my $line = $opt{line};
# Line 7386  sub set_inner_html ($$$) { Line 8180  sub set_inner_html ($$$) {
8180        $ponerror->(line => $p->{line}, column => $p->{column}, @_);        $ponerror->(line => $p->{line}, column => $p->{column}, @_);
8181      };      };
8182            
8183        my $char_onerror = sub {
8184          my (undef, $type, %opt) = @_;
8185          $ponerror->(layer => 'encode',
8186                      line => $p->{line}, column => $p->{column} + 1,
8187                      %opt, type => $type);
8188        }; # $char_onerror
8189        $input->onerror ($char_onerror);
8190    
8191      $p->_initialize_tokenizer;      $p->_initialize_tokenizer;
8192      $p->_initialize_tree_constructor;      $p->_initialize_tree_constructor;
8193    
# Line 7421  sub set_inner_html ($$$) { Line 8223  sub set_inner_html ($$$) {
8223      push @{$p->{open_elements}}, [$root, $el_category->{html}];      push @{$p->{open_elements}}, [$root, $el_category->{html}];
8224    
8225      undef $p->{head_element};      undef $p->{head_element};
8226        undef $p->{head_element_inserted};
8227    
8228      ## Step 6 # MUST      ## Step 6 # MUST
8229      $p->_reset_insertion_mode;      $p->_reset_insertion_mode;

Legend:
Removed from v.1.141  
changed lines
  Added in v.1.205

[email protected]
ViewVC Help
Powered by ViewVC 1.1.24