/[suikacvs]/markup/html/whatpm/Whatpm/HTML.pm.src
Suika

Diff of /markup/html/whatpm/Whatpm/HTML.pm.src

Parent Directory Parent Directory | Revision Log Revision Log | View Patch Patch

revision 1.203 by wakaba, Sat Oct 4 17:16:02 2008 UTC revision 1.244 by wakaba, Sun Sep 6 23:32:06 2009 UTC
# Line 3  use strict; Line 3  use strict;
3  our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r};  our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r};
4  use Error qw(:try);  use Error qw(:try);
5    
6    use Whatpm::HTML::Tokenizer;
7    
8  ## NOTE: This module don't check all HTML5 parse errors; character  ## NOTE: This module don't check all HTML5 parse errors; character
9  ## encoding related parse errors are expected to be handled by relevant  ## encoding related parse errors are expected to be handled by relevant
10  ## modules.  ## modules.
# Line 21  use Error qw(:try); Line 23  use Error qw(:try);
23    
24  require IO::Handle;  require IO::Handle;
25    
26    ## Namespace URLs
27    
28  my $HTML_NS = q<http://www.w3.org/1999/xhtml>;  my $HTML_NS = q<http://www.w3.org/1999/xhtml>;
29  my $MML_NS = q<http://www.w3.org/1998/Math/MathML>;  my $MML_NS = q<http://www.w3.org/1998/Math/MathML>;
30  my $SVG_NS = q<http://www.w3.org/2000/svg>;  my $SVG_NS = q<http://www.w3.org/2000/svg>;
# Line 28  my $XLINK_NS = q<http://www.w3.org/1999/ Line 32  my $XLINK_NS = q<http://www.w3.org/1999/
32  my $XML_NS = q<http://www.w3.org/XML/1998/namespace>;  my $XML_NS = q<http://www.w3.org/XML/1998/namespace>;
33  my $XMLNS_NS = q<http://www.w3.org/2000/xmlns/>;  my $XMLNS_NS = q<http://www.w3.org/2000/xmlns/>;
34    
35  sub A_EL () { 0b1 }  ## Element categories
36  sub ADDRESS_EL () { 0b10 }  
37  sub BODY_EL () { 0b100 }  ## Bits 12-15
38  sub BUTTON_EL () { 0b1000 }  sub SPECIAL_EL () { 0b1_000000000000000 }
39  sub CAPTION_EL () { 0b10000 }  sub SCOPING_EL () { 0b1_00000000000000 }
40  sub DD_EL () { 0b100000 }  sub FORMATTING_EL () { 0b1_0000000000000 }
41  sub DIV_EL () { 0b1000000 }  sub PHRASING_EL () { 0b1_000000000000 }
42  sub DT_EL () { 0b10000000 }  
43  sub FORM_EL () { 0b100000000 }  ## Bits 10-11
44  sub FORMATTING_EL () { 0b1000000000 }  #sub FOREIGN_EL () { 0b1_00000000000 } # see Whatpm::HTML::Tokenizer
45  sub FRAMESET_EL () { 0b10000000000 }  sub FOREIGN_FLOW_CONTENT_EL () { 0b1_0000000000 }
46  sub HEADING_EL () { 0b100000000000 }  
47  sub HTML_EL () { 0b1000000000000 }  ## Bits 6-9
48  sub LI_EL () { 0b10000000000000 }  sub TABLE_SCOPING_EL () { 0b1_000000000 }
49  sub NOBR_EL () { 0b100000000000000 }  sub TABLE_ROWS_SCOPING_EL () { 0b1_00000000 }
50  sub OPTION_EL () { 0b1000000000000000 }  sub TABLE_ROW_SCOPING_EL () { 0b1_0000000 }
51  sub OPTGROUP_EL () { 0b10000000000000000 }  sub TABLE_ROWS_EL () { 0b1_000000 }
52  sub P_EL () { 0b100000000000000000 }  
53  sub SELECT_EL () { 0b1000000000000000000 }  ## Bit 5
54  sub TABLE_EL () { 0b10000000000000000000 }  sub ADDRESS_DIV_P_EL () { 0b1_00000 }
55  sub TABLE_CELL_EL () { 0b100000000000000000000 }  
56  sub TABLE_ROW_EL () { 0b1000000000000000000000 }  ## NOTE: Used in </body> and EOF algorithms.
57  sub TABLE_ROW_GROUP_EL () { 0b10000000000000000000000 }  ## Bit 4
58  sub MISC_SCOPING_EL () { 0b100000000000000000000000 }  sub ALL_END_TAG_OPTIONAL_EL () { 0b1_0000 }
 sub MISC_SPECIAL_EL () { 0b1000000000000000000000000 }  
 sub FOREIGN_EL () { 0b10000000000000000000000000 }  
 sub FOREIGN_FLOW_CONTENT_EL () { 0b100000000000000000000000000 }  
 sub MML_AXML_EL () { 0b1000000000000000000000000000 }  
 sub RUBY_EL () { 0b10000000000000000000000000000 }  
 sub RUBY_COMPONENT_EL () { 0b100000000000000000000000000000 }  
   
 sub TABLE_ROWS_EL () {  
   TABLE_EL |  
   TABLE_ROW_EL |  
   TABLE_ROW_GROUP_EL  
 }  
59    
60  ## NOTE: Used in "generate implied end tags" algorithm.  ## NOTE: Used in "generate implied end tags" algorithm.
61  ## NOTE: There is a code where a modified version of  ## NOTE: There is a code where a modified version of
62  ## END_TAG_OPTIONAL_EL is used in "generate implied end tags"  ## END_TAG_OPTIONAL_EL is used in "generate implied end tags"
63  ## implementation (search for the algorithm name).  ## implementation (search for the algorithm name).
64  sub END_TAG_OPTIONAL_EL () {  ## Bit 3
65    DD_EL |  sub END_TAG_OPTIONAL_EL () { 0b1_000 }
   DT_EL |  
   LI_EL |  
   OPTION_EL |  
   OPTGROUP_EL |  
   P_EL |  
   RUBY_COMPONENT_EL  
 }  
66    
67  ## NOTE: Used in </body> and EOF algorithms.  ## Bits 0-2
 sub ALL_END_TAG_OPTIONAL_EL () {  
   DD_EL |  
   DT_EL |  
   LI_EL |  
   P_EL |  
   
   ## ISSUE: option, optgroup, rt, rp?  
   
   BODY_EL |  
   HTML_EL |  
   TABLE_CELL_EL |  
   TABLE_ROW_EL |  
   TABLE_ROW_GROUP_EL  
 }  
68    
69  sub SCOPING_EL () {  sub MISC_SPECIAL_EL () { SPECIAL_EL | 0b000 }
70    BUTTON_EL |  sub FORM_EL () { SPECIAL_EL | 0b001 }
71    CAPTION_EL |  sub FRAMESET_EL () { SPECIAL_EL | 0b010 }
72    HTML_EL |  sub HEADING_EL () { SPECIAL_EL | 0b011 }
73    TABLE_EL |  sub SELECT_EL () { SPECIAL_EL | 0b100 }
74    TABLE_CELL_EL |  sub SCRIPT_EL () { SPECIAL_EL | 0b101 }
75    MISC_SCOPING_EL  
76    sub ADDRESS_DIV_EL () { SPECIAL_EL | ADDRESS_DIV_P_EL | 0b001 }
77    sub BODY_EL () { SPECIAL_EL | ALL_END_TAG_OPTIONAL_EL | 0b001 }
78    
79    sub DTDD_EL () {
80      SPECIAL_EL |
81      END_TAG_OPTIONAL_EL |
82      ALL_END_TAG_OPTIONAL_EL |
83      0b010
84  }  }
85    sub LI_EL () {
86  sub TABLE_SCOPING_EL () {    SPECIAL_EL |
87    HTML_EL |    END_TAG_OPTIONAL_EL |
88    TABLE_EL    ALL_END_TAG_OPTIONAL_EL |
89      0b100
90  }  }
91    sub P_EL () {
92  sub TABLE_ROWS_SCOPING_EL () {    SPECIAL_EL |
93    HTML_EL |    ADDRESS_DIV_P_EL |
94    TABLE_ROW_GROUP_EL    END_TAG_OPTIONAL_EL |
95      ALL_END_TAG_OPTIONAL_EL |
96      0b001
97  }  }
98    
99  sub TABLE_ROW_SCOPING_EL () {  sub TABLE_ROW_EL () {
100    HTML_EL |    SPECIAL_EL |
101    TABLE_ROW_EL    TABLE_ROWS_EL |
102      TABLE_ROW_SCOPING_EL |
103      ALL_END_TAG_OPTIONAL_EL |
104      0b001
105    }
106    sub TABLE_ROW_GROUP_EL () {
107      SPECIAL_EL |
108      TABLE_ROWS_EL |
109      TABLE_ROWS_SCOPING_EL |
110      ALL_END_TAG_OPTIONAL_EL |
111      0b001
112  }  }
113    
114  sub SPECIAL_EL () {  sub MISC_SCOPING_EL () { SCOPING_EL | 0b000 }
115    ADDRESS_EL |  sub BUTTON_EL () { SCOPING_EL | 0b001 }
116    BODY_EL |  sub CAPTION_EL () { SCOPING_EL | 0b010 }
117    DIV_EL |  sub HTML_EL () {
118      SCOPING_EL |
119    DD_EL |    TABLE_SCOPING_EL |
120    DT_EL |    TABLE_ROWS_SCOPING_EL |
121    LI_EL |    TABLE_ROW_SCOPING_EL |
122    P_EL |    ALL_END_TAG_OPTIONAL_EL |
123      0b001
124    FORM_EL |  }
125    FRAMESET_EL |  sub TABLE_EL () {
126    HEADING_EL |    SCOPING_EL |
127    SELECT_EL |    TABLE_ROWS_EL |
128    TABLE_ROW_EL |    TABLE_SCOPING_EL |
129    TABLE_ROW_GROUP_EL |    0b001
   MISC_SPECIAL_EL  
130  }  }
131    sub TABLE_CELL_EL () {
132      SCOPING_EL |
133      TABLE_ROW_SCOPING_EL |
134      ALL_END_TAG_OPTIONAL_EL |
135      0b001
136    }
137    
138    sub MISC_FORMATTING_EL () { FORMATTING_EL | 0b000 }
139    sub A_EL () { FORMATTING_EL | 0b001 }
140    sub NOBR_EL () { FORMATTING_EL | 0b010 }
141    
142    sub RUBY_EL () { PHRASING_EL | 0b001 }
143    
144    ## ISSUE: ALL_END_TAG_OPTIONAL_EL?
145    sub OPTGROUP_EL () { PHRASING_EL | END_TAG_OPTIONAL_EL | 0b001 }
146    sub OPTION_EL () { PHRASING_EL | END_TAG_OPTIONAL_EL | 0b010 }
147    sub RUBY_COMPONENT_EL () { PHRASING_EL | END_TAG_OPTIONAL_EL | 0b100 }
148    
149    sub MML_AXML_EL () { PHRASING_EL | FOREIGN_EL | 0b001 }
150    
151  my $el_category = {  my $el_category = {
152    a => A_EL | FORMATTING_EL,    a => A_EL,
153    address => ADDRESS_EL,    address => ADDRESS_DIV_EL,
154    applet => MISC_SCOPING_EL,    applet => MISC_SCOPING_EL,
155    area => MISC_SPECIAL_EL,    area => MISC_SPECIAL_EL,
156    article => MISC_SPECIAL_EL,    article => MISC_SPECIAL_EL,
# Line 160  my $el_category = { Line 170  my $el_category = {
170    colgroup => MISC_SPECIAL_EL,    colgroup => MISC_SPECIAL_EL,
171    command => MISC_SPECIAL_EL,    command => MISC_SPECIAL_EL,
172    datagrid => MISC_SPECIAL_EL,    datagrid => MISC_SPECIAL_EL,
173    dd => DD_EL,    dd => DTDD_EL,
174    details => MISC_SPECIAL_EL,    details => MISC_SPECIAL_EL,
175    dialog => MISC_SPECIAL_EL,    dialog => MISC_SPECIAL_EL,
176    dir => MISC_SPECIAL_EL,    dir => MISC_SPECIAL_EL,
177    div => DIV_EL,    div => ADDRESS_DIV_EL,
178    dl => MISC_SPECIAL_EL,    dl => MISC_SPECIAL_EL,
179    dt => DT_EL,    dt => DTDD_EL,
180    em => FORMATTING_EL,    em => FORMATTING_EL,
181    embed => MISC_SPECIAL_EL,    embed => MISC_SPECIAL_EL,
   eventsource => MISC_SPECIAL_EL,  
182    fieldset => MISC_SPECIAL_EL,    fieldset => MISC_SPECIAL_EL,
183    figure => MISC_SPECIAL_EL,    figure => MISC_SPECIAL_EL,
184    font => FORMATTING_EL,    font => FORMATTING_EL,
# Line 185  my $el_category = { Line 194  my $el_category = {
194    h6 => HEADING_EL,    h6 => HEADING_EL,
195    head => MISC_SPECIAL_EL,    head => MISC_SPECIAL_EL,
196    header => MISC_SPECIAL_EL,    header => MISC_SPECIAL_EL,
197      hgroup => MISC_SPECIAL_EL,
198    hr => MISC_SPECIAL_EL,    hr => MISC_SPECIAL_EL,
199    html => HTML_EL,    html => HTML_EL,
200    i => FORMATTING_EL,    i => FORMATTING_EL,
# Line 193  my $el_category = { Line 203  my $el_category = {
203    #image => MISC_SPECIAL_EL, ## NOTE: Commented out in the spec.    #image => MISC_SPECIAL_EL, ## NOTE: Commented out in the spec.
204    input => MISC_SPECIAL_EL,    input => MISC_SPECIAL_EL,
205    isindex => MISC_SPECIAL_EL,    isindex => MISC_SPECIAL_EL,
206      ## XXX keygen? (Whether a void element is in Special or not does not
207      ## affect to the processing, however.)
208    li => LI_EL,    li => LI_EL,
209    link => MISC_SPECIAL_EL,    link => MISC_SPECIAL_EL,
210    listing => MISC_SPECIAL_EL,    listing => MISC_SPECIAL_EL,
# Line 200  my $el_category = { Line 212  my $el_category = {
212    menu => MISC_SPECIAL_EL,    menu => MISC_SPECIAL_EL,
213    meta => MISC_SPECIAL_EL,    meta => MISC_SPECIAL_EL,
214    nav => MISC_SPECIAL_EL,    nav => MISC_SPECIAL_EL,
215    nobr => NOBR_EL | FORMATTING_EL,    nobr => NOBR_EL,
216    noembed => MISC_SPECIAL_EL,    noembed => MISC_SPECIAL_EL,
217    noframes => MISC_SPECIAL_EL,    noframes => MISC_SPECIAL_EL,
218    noscript => MISC_SPECIAL_EL,    noscript => MISC_SPECIAL_EL,
# Line 237  my $el_category = { Line 249  my $el_category = {
249    u => FORMATTING_EL,    u => FORMATTING_EL,
250    ul => MISC_SPECIAL_EL,    ul => MISC_SPECIAL_EL,
251    wbr => MISC_SPECIAL_EL,    wbr => MISC_SPECIAL_EL,
252      xmp => MISC_SPECIAL_EL,
253  };  };
254    
255  my $el_category_f = {  my $el_category_f = {
256    $MML_NS => {    $MML_NS => {
257      'annotation-xml' => MML_AXML_EL,      'annotation-xml' => MML_AXML_EL,
258      mi => FOREIGN_FLOW_CONTENT_EL,      mi => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
259      mo => FOREIGN_FLOW_CONTENT_EL,      mo => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
260      mn => FOREIGN_FLOW_CONTENT_EL,      mn => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
261      ms => FOREIGN_FLOW_CONTENT_EL,      ms => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
262      mtext => FOREIGN_FLOW_CONTENT_EL,      mtext => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
263    },    },
264    $SVG_NS => {    $SVG_NS => {
265      foreignObject => FOREIGN_FLOW_CONTENT_EL | MISC_SCOPING_EL,      foreignObject => SCOPING_EL | FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
266      desc => FOREIGN_FLOW_CONTENT_EL,      desc => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
267      title => FOREIGN_FLOW_CONTENT_EL,      title => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
268    },    },
269    ## NOTE: In addition, FOREIGN_EL is set to non-HTML elements.    ## NOTE: In addition, FOREIGN_EL is set to non-HTML elements.
270  };  };
# Line 338  my $foreign_attr_xname = { Line 351  my $foreign_attr_xname = {
351    
352  ## ISSUE: xmlns:xlink="non-xlink-ns" is not an error.  ## ISSUE: xmlns:xlink="non-xlink-ns" is not an error.
353    
 my $charref_map = {  
   0x0D => 0x000A,  
   0x80 => 0x20AC,  
   0x81 => 0xFFFD,  
   0x82 => 0x201A,  
   0x83 => 0x0192,  
   0x84 => 0x201E,  
   0x85 => 0x2026,  
   0x86 => 0x2020,  
   0x87 => 0x2021,  
   0x88 => 0x02C6,  
   0x89 => 0x2030,  
   0x8A => 0x0160,  
   0x8B => 0x2039,  
   0x8C => 0x0152,  
   0x8D => 0xFFFD,  
   0x8E => 0x017D,  
   0x8F => 0xFFFD,  
   0x90 => 0xFFFD,  
   0x91 => 0x2018,  
   0x92 => 0x2019,  
   0x93 => 0x201C,  
   0x94 => 0x201D,  
   0x95 => 0x2022,  
   0x96 => 0x2013,  
   0x97 => 0x2014,  
   0x98 => 0x02DC,  
   0x99 => 0x2122,  
   0x9A => 0x0161,  
   0x9B => 0x203A,  
   0x9C => 0x0153,  
   0x9D => 0xFFFD,  
   0x9E => 0x017E,  
   0x9F => 0x0178,  
 }; # $charref_map  
 $charref_map->{$_} = 0xFFFD  
     for 0x0000..0x0008, 0x000B, 0x000E..0x001F, 0x007F,  
         0xD800..0xDFFF, 0xFDD0..0xFDDF, ## ISSUE: 0xFDEF  
         0xFFFE, 0xFFFF, 0x1FFFE, 0x1FFFF, 0x2FFFE, 0x2FFFF, 0x3FFFE, 0x3FFFF,  
         0x4FFFE, 0x4FFFF, 0x5FFFE, 0x5FFFF, 0x6FFFE, 0x6FFFF, 0x7FFFE,  
         0x7FFFF, 0x8FFFE, 0x8FFFF, 0x9FFFE, 0x9FFFF, 0xAFFFE, 0xAFFFF,  
         0xBFFFE, 0xBFFFF, 0xCFFFE, 0xCFFFF, 0xDFFFE, 0xDFFFF, 0xEFFFE,  
         0xEFFFF, 0xFFFFE, 0xFFFFF, 0x10FFFE, 0x10FFFF;  
   
354  ## TODO: Invoke the reset algorithm when a resettable element is  ## TODO: Invoke the reset algorithm when a resettable element is
355  ## created (cf. HTML5 revision 2259).  ## created (cf. HTML5 revision 2259).
356    
# Line 559  sub parse_byte_stream ($$$$;$$) { Line 528  sub parse_byte_stream ($$$$;$$) {
528            
529      if ($char_stream) { # if supported      if ($char_stream) { # if supported
530        ## "Change the encoding" algorithm:        ## "Change the encoding" algorithm:
   
       ## Step 1      
       if ($charset->{category} &  
           Message::Charset::Info::CHARSET_CATEGORY_UTF16 ()) {  
         $charset = Message::Charset::Info->get_by_html_name ('utf-8');  
         ($char_stream, $e_status) = $charset->get_decode_handle  
             ($byte_stream,  
              byte_buffer => \ $buffer->{buffer});  
       }  
       $charset_name = $charset->get_iana_name;  
531                
532        ## Step 2        ## Step 1
533        if (defined $self->{input_encoding} and        if (defined $self->{input_encoding} and
534            $self->{input_encoding} eq $charset_name) {            $self->{input_encoding} eq $charset_name) {
535          !!!parse-error (type => 'charset label:matching',          !!!parse-error (type => 'charset label:matching',
# Line 580  sub parse_byte_stream ($$$$;$$) { Line 539  sub parse_byte_stream ($$$$;$$) {
539          return;          return;
540        }        }
541    
542          ## Step 2 (HTML5 revision 3205)
543          if (defined $self->{input_encoding} and
544              Message::Charset::Info->get_by_html_name ($self->{input_encoding})
545              ->{category} & Message::Charset::Info::CHARSET_CATEGORY_UTF16 ()) {
546            $self->{confident} = 1;
547            return;
548          }
549    
550          ## Step 3
551          if ($charset->{category} &
552              Message::Charset::Info::CHARSET_CATEGORY_UTF16 ()) {
553            $charset = Message::Charset::Info->get_by_html_name ('utf-8');
554            ($char_stream, $e_status) = $charset->get_decode_handle
555                ($byte_stream,
556                 byte_buffer => \ $buffer->{buffer});
557          }
558          $charset_name = $charset->get_iana_name;
559    
560        !!!parse-error (type => 'charset label detected',        !!!parse-error (type => 'charset label detected',
561                        text => $self->{input_encoding},                        text => $self->{input_encoding},
562                        value => $charset_name,                        value => $charset_name,
563                        level => $self->{level}->{warn},                        level => $self->{level}->{warn},
564                        token => $token);                        token => $token);
565                
566        ## Step 3        ## Step 4
567        # if (can) {        # if (can) {
568          ## change the encoding on the fly.          ## change the encoding on the fly.
569          #$self->{confident} = 1;          #$self->{confident} = 1;
570          #return;          #return;
571        # }        # }
572                
573        ## Step 4        ## Step 5
574        throw Whatpm::HTML::RestartParser ();        throw Whatpm::HTML::RestartParser ();
575      }      }
576    }; # $self->{change_encoding}    }; # $self->{change_encoding}
# Line 673  sub parse_char_stream ($$$;$$) { Line 650  sub parse_char_stream ($$$;$$) {
650    
651    ## NOTE: |set_inner_html| copies most of this method's code    ## NOTE: |set_inner_html| copies most of this method's code
652    
653      ## Confidence: irrelevant.
654    $self->{confident} = 1 unless exists $self->{confident};    $self->{confident} = 1 unless exists $self->{confident};
655    
656    $self->{document}->input_encoding ($self->{input_encoding})    $self->{document}->input_encoding ($self->{input_encoding})
657        if defined $self->{input_encoding};        if defined $self->{input_encoding};
658  ## TODO: |{input_encoding}| is needless?  ## TODO: |{input_encoding}| is needless?
# Line 810  sub parse_char_stream ($$$;$$) { Line 789  sub parse_char_stream ($$$;$$) {
789  sub new ($) {  sub new ($) {
790    my $class = shift;    my $class = shift;
791    my $self = bless {    my $self = bless {
792      level => {must => 'm',      level => {
793                should => 's',        must => 'm',
794                warn => 'w',        should => 's',
795                info => 'i',        obc => 's', ## Obsolete but conforming, # XXX distinguish from "should"
796                uncertain => 'u'},        warn => 'w',
797          info => 'i',
798          uncertain => 'u',
799        },
800    }, $class;    }, $class;
801    $self->{set_nc} = sub {    $self->{set_nc} = sub {
802      $self->{nc} = -1;      $self->{nc} = -1;
# Line 834  sub new ($) { Line 816  sub new ($) {
816    return $self;    return $self;
817  } # new  } # new
818    
819  sub CM_ENTITY () { 0b001 } # & markup in data  ## Insertion modes
 sub CM_LIMITED_MARKUP () { 0b010 } # < markup in data (limited)  
 sub CM_FULL_MARKUP () { 0b100 } # < markup in data (any)  
   
 sub PLAINTEXT_CONTENT_MODEL () { 0 }  
 sub CDATA_CONTENT_MODEL () { CM_LIMITED_MARKUP }  
 sub RCDATA_CONTENT_MODEL () { CM_ENTITY | CM_LIMITED_MARKUP }  
 sub PCDATA_CONTENT_MODEL () { CM_ENTITY | CM_FULL_MARKUP }  
   
 sub DATA_STATE () { 0 }  
 #sub ENTITY_DATA_STATE () { 1 }  
 sub TAG_OPEN_STATE () { 2 }  
 sub CLOSE_TAG_OPEN_STATE () { 3 }  
 sub TAG_NAME_STATE () { 4 }  
 sub BEFORE_ATTRIBUTE_NAME_STATE () { 5 }  
 sub ATTRIBUTE_NAME_STATE () { 6 }  
 sub AFTER_ATTRIBUTE_NAME_STATE () { 7 }  
 sub BEFORE_ATTRIBUTE_VALUE_STATE () { 8 }  
 sub ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE () { 9 }  
 sub ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE () { 10 }  
 sub ATTRIBUTE_VALUE_UNQUOTED_STATE () { 11 }  
 #sub ENTITY_IN_ATTRIBUTE_VALUE_STATE () { 12 }  
 sub MARKUP_DECLARATION_OPEN_STATE () { 13 }  
 sub COMMENT_START_STATE () { 14 }  
 sub COMMENT_START_DASH_STATE () { 15 }  
 sub COMMENT_STATE () { 16 }  
 sub COMMENT_END_STATE () { 17 }  
 sub COMMENT_END_DASH_STATE () { 18 }  
 sub BOGUS_COMMENT_STATE () { 19 }  
 sub DOCTYPE_STATE () { 20 }  
 sub BEFORE_DOCTYPE_NAME_STATE () { 21 }  
 sub DOCTYPE_NAME_STATE () { 22 }  
 sub AFTER_DOCTYPE_NAME_STATE () { 23 }  
 sub BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE () { 24 }  
 sub DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE () { 25 }  
 sub DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE () { 26 }  
 sub AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE () { 27 }  
 sub BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE () { 28 }  
 sub DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE () { 29 }  
 sub DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE () { 30 }  
 sub AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE () { 31 }  
 sub BOGUS_DOCTYPE_STATE () { 32 }  
 sub AFTER_ATTRIBUTE_VALUE_QUOTED_STATE () { 33 }  
 sub SELF_CLOSING_START_TAG_STATE () { 34 }  
 sub CDATA_SECTION_STATE () { 35 }  
 sub MD_HYPHEN_STATE () { 36 } # "markup declaration open state" in the spec  
 sub MD_DOCTYPE_STATE () { 37 } # "markup declaration open state" in the spec  
 sub MD_CDATA_STATE () { 38 } # "markup declaration open state" in the spec  
 sub CDATA_RCDATA_CLOSE_TAG_STATE () { 39 } # "close tag open state" in the spec  
 sub CDATA_SECTION_MSE1_STATE () { 40 } # "CDATA section state" in the spec  
 sub CDATA_SECTION_MSE2_STATE () { 41 } # "CDATA section state" in the spec  
 sub PUBLIC_STATE () { 42 } # "after DOCTYPE name state" in the spec  
 sub SYSTEM_STATE () { 43 } # "after DOCTYPE name state" in the spec  
 ## NOTE: "Entity data state", "entity in attribute value state", and  
 ## "consume a character reference" algorithm are jointly implemented  
 ## using the following six states:  
 sub ENTITY_STATE () { 44 }  
 sub ENTITY_HASH_STATE () { 45 }  
 sub NCR_NUM_STATE () { 46 }  
 sub HEXREF_X_STATE () { 47 }  
 sub HEXREF_HEX_STATE () { 48 }  
 sub ENTITY_NAME_STATE () { 49 }  
 sub PCDATA_STATE () { 50 } # "data state" in the spec  
   
 sub DOCTYPE_TOKEN () { 1 }  
 sub COMMENT_TOKEN () { 2 }  
 sub START_TAG_TOKEN () { 3 }  
 sub END_TAG_TOKEN () { 4 }  
 sub END_OF_FILE_TOKEN () { 5 }  
 sub CHARACTER_TOKEN () { 6 }  
820    
821  sub AFTER_HTML_IMS () { 0b100 }  sub AFTER_HTML_IMS () { 0b100 }
822  sub HEAD_IMS ()       { 0b1000 }  sub HEAD_IMS ()       { 0b1000 }
# Line 914  sub ROW_IMS ()        { 0b10000000 } Line 827  sub ROW_IMS ()        { 0b10000000 }
827  sub BODY_AFTER_IMS () { 0b100000000 }  sub BODY_AFTER_IMS () { 0b100000000 }
828  sub FRAME_IMS ()      { 0b1000000000 }  sub FRAME_IMS ()      { 0b1000000000 }
829  sub SELECT_IMS ()     { 0b10000000000 }  sub SELECT_IMS ()     { 0b10000000000 }
830  sub IN_FOREIGN_CONTENT_IM () { 0b100000000000 }  #sub IN_FOREIGN_CONTENT_IM () { 0b100000000000 } # see Whatpm::HTML::Tokenizer
831      ## NOTE: "in foreign content" insertion mode is special; it is combined      ## NOTE: "in foreign content" insertion mode is special; it is combined
832      ## with the secondary insertion mode.  In this parser, they are stored      ## with the secondary insertion mode.  In this parser, they are stored
833      ## together in the bit-or'ed form.      ## together in the bit-or'ed form.
834    sub IN_CDATA_RCDATA_IM () { 0b1000000000000 }
835        ## NOTE: "in CDATA/RCDATA" insertion mode is also special; it is
836        ## combined with the original insertion mode.  In thie parser,
837        ## they are stored together in the bit-or'ed form.
838    
839    sub IM_MASK () { 0b11111111111 }
840    
841  ## NOTE: "initial" and "before html" insertion modes have no constants.  ## NOTE: "initial" and "before html" insertion modes have no constants.
842    
# Line 944  sub IN_SELECT_IM () { SELECT_IMS | 0b01 Line 863  sub IN_SELECT_IM () { SELECT_IMS | 0b01
863  sub IN_SELECT_IN_TABLE_IM () { SELECT_IMS | 0b10 }  sub IN_SELECT_IN_TABLE_IM () { SELECT_IMS | 0b10 }
864  sub IN_COLUMN_GROUP_IM () { 0b10 }  sub IN_COLUMN_GROUP_IM () { 0b10 }
865    
 ## Implementations MUST act as if state machine in the spec  
   
 sub _initialize_tokenizer ($) {  
   my $self = shift;  
   $self->{state} = DATA_STATE; # MUST  
   #$self->{s_kwd}; # state keyword - initialized when used  
   #$self->{entity__value}; # initialized when used  
   #$self->{entity__match}; # initialized when used  
   $self->{content_model} = PCDATA_CONTENT_MODEL; # be  
   undef $self->{ct}; # current token  
   undef $self->{ca}; # current attribute  
   undef $self->{last_stag_name}; # last emitted start tag name  
   #$self->{prev_state}; # initialized when used  
   delete $self->{self_closing};  
   $self->{char_buffer} = '';  
   $self->{char_buffer_pos} = 0;  
   $self->{nc} = -1; # next input character  
   #$self->{next_nc}  
   !!!next-input-character;  
   $self->{token} = [];  
   # $self->{escape}  
 } # _initialize_tokenizer  
   
 ## A token has:  
 ##   ->{type} == DOCTYPE_TOKEN, START_TAG_TOKEN, END_TAG_TOKEN, COMMENT_TOKEN,  
 ##       CHARACTER_TOKEN, or END_OF_FILE_TOKEN  
 ##   ->{name} (DOCTYPE_TOKEN)  
 ##   ->{tag_name} (START_TAG_TOKEN, END_TAG_TOKEN)  
 ##   ->{pubid} (DOCTYPE_TOKEN)  
 ##   ->{sysid} (DOCTYPE_TOKEN)  
 ##   ->{quirks} == 1 or 0 (DOCTYPE_TOKEN): "force-quirks" flag  
 ##   ->{attributes} isa HASH (START_TAG_TOKEN, END_TAG_TOKEN)  
 ##        ->{name}  
 ##        ->{value}  
 ##        ->{has_reference} == 1 or 0  
 ##   ->{data} (COMMENT_TOKEN, CHARACTER_TOKEN)  
 ## NOTE: The "self-closing flag" is hold as |$self->{self_closing}|.  
 ##     |->{self_closing}| is used to save the value of |$self->{self_closing}|  
 ##     while the token is pushed back to the stack.  
   
 ## Emitted token MUST immediately be handled by the tree construction state.  
   
 ## Before each step, UA MAY check to see if either one of the scripts in  
 ## "list of scripts that will execute as soon as possible" or the first  
 ## script in the "list of scripts that will execute asynchronously",  
 ## has completed loading.  If one has, then it MUST be executed  
 ## and removed from the list.  
   
 ## TODO: Polytheistic slash SHOULD NOT be used. (Applied only to atheists.)  
 ## (This requirement was dropped from HTML5 spec, unfortunately.)  
   
 my $is_space = {  
   0x0009 => 1, # CHARACTER TABULATION (HT)  
   0x000A => 1, # LINE FEED (LF)  
   #0x000B => 0, # LINE TABULATION (VT)  
   0x000C => 1, # FORM FEED (FF)  
   #0x000D => 1, # CARRIAGE RETURN (CR)  
   0x0020 => 1, # SPACE (SP)  
 };  
   
 sub _get_next_token ($) {  
   my $self = shift;  
   
   if ($self->{self_closing}) {  
     !!!parse-error (type => 'nestc', token => $self->{ct});  
     ## NOTE: The |self_closing| flag is only set by start tag token.  
     ## In addition, when a start tag token is emitted, it is always set to  
     ## |ct|.  
     delete $self->{self_closing};  
   }  
   
   if (@{$self->{token}}) {  
     $self->{self_closing} = $self->{token}->[0]->{self_closing};  
     return shift @{$self->{token}};  
   }  
   
   A: {  
     if ($self->{state} == PCDATA_STATE) {  
       ## NOTE: Same as |DATA_STATE|, but only for |PCDATA| content model.  
   
       if ($self->{nc} == 0x0026) { # &  
         !!!cp (0.1);  
         ## NOTE: In the spec, the tokenizer is switched to the  
         ## "entity data state".  In this implementation, the tokenizer  
         ## is switched to the |ENTITY_STATE|, which is an implementation  
         ## of the "consume a character reference" algorithm.  
         $self->{entity_add} = -1;  
         $self->{prev_state} = DATA_STATE;  
         $self->{state} = ENTITY_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003C) { # <  
         !!!cp (0.2);  
         $self->{state} = TAG_OPEN_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (0.3);  
         !!!emit ({type => END_OF_FILE_TOKEN,  
                   line => $self->{line}, column => $self->{column}});  
         last A; ## TODO: ok?  
       } else {  
         !!!cp (0.4);  
         #  
       }  
   
       # Anything else  
       my $token = {type => CHARACTER_TOKEN,  
                    data => chr $self->{nc},  
                    line => $self->{line}, column => $self->{column},  
                   };  
       $self->{read_until}->($token->{data}, q[<&], length $token->{data});  
   
       ## Stay in the state.  
       !!!next-input-character;  
       !!!emit ($token);  
       redo A;  
     } elsif ($self->{state} == DATA_STATE) {  
       $self->{s_kwd} = '' unless defined $self->{s_kwd};  
       if ($self->{nc} == 0x0026) { # &  
         $self->{s_kwd} = '';  
         if ($self->{content_model} & CM_ENTITY and # PCDATA | RCDATA  
             not $self->{escape}) {  
           !!!cp (1);  
           ## NOTE: In the spec, the tokenizer is switched to the  
           ## "entity data state".  In this implementation, the tokenizer  
           ## is switched to the |ENTITY_STATE|, which is an implementation  
           ## of the "consume a character reference" algorithm.  
           $self->{entity_add} = -1;  
           $self->{prev_state} = DATA_STATE;  
           $self->{state} = ENTITY_STATE;  
           !!!next-input-character;  
           redo A;  
         } else {  
           !!!cp (2);  
           #  
         }  
       } elsif ($self->{nc} == 0x002D) { # -  
         if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA  
           $self->{s_kwd} .= '-';  
             
           if ($self->{s_kwd} eq '<!--') {  
             !!!cp (3);  
             $self->{escape} = 1; # unless $self->{escape};  
             $self->{s_kwd} = '--';  
             #  
           } elsif ($self->{s_kwd} eq '---') {  
             !!!cp (4);  
             $self->{s_kwd} = '--';  
             #  
           } else {  
             !!!cp (5);  
             #  
           }  
         }  
           
         #  
       } elsif ($self->{nc} == 0x0021) { # !  
         if (length $self->{s_kwd}) {  
           !!!cp (5.1);  
           $self->{s_kwd} .= '!';  
           #  
         } else {  
           !!!cp (5.2);  
           #$self->{s_kwd} = '';  
           #  
         }  
         #  
       } elsif ($self->{nc} == 0x003C) { # <  
         if ($self->{content_model} & CM_FULL_MARKUP or # PCDATA  
             (($self->{content_model} & CM_LIMITED_MARKUP) and # CDATA | RCDATA  
              not $self->{escape})) {  
           !!!cp (6);  
           $self->{state} = TAG_OPEN_STATE;  
           !!!next-input-character;  
           redo A;  
         } else {  
           !!!cp (7);  
           $self->{s_kwd} = '';  
           #  
         }  
       } elsif ($self->{nc} == 0x003E) { # >  
         if ($self->{escape} and  
             ($self->{content_model} & CM_LIMITED_MARKUP)) { # RCDATA | CDATA  
           if ($self->{s_kwd} eq '--') {  
             !!!cp (8);  
             delete $self->{escape};  
           } else {  
             !!!cp (9);  
           }  
         } else {  
           !!!cp (10);  
         }  
           
         $self->{s_kwd} = '';  
         #  
       } elsif ($self->{nc} == -1) {  
         !!!cp (11);  
         $self->{s_kwd} = '';  
         !!!emit ({type => END_OF_FILE_TOKEN,  
                   line => $self->{line}, column => $self->{column}});  
         last A; ## TODO: ok?  
       } else {  
         !!!cp (12);  
         $self->{s_kwd} = '';  
         #  
       }  
   
       # Anything else  
       my $token = {type => CHARACTER_TOKEN,  
                    data => chr $self->{nc},  
                    line => $self->{line}, column => $self->{column},  
                   };  
       if ($self->{read_until}->($token->{data}, q[-!<>&],  
                                 length $token->{data})) {  
         $self->{s_kwd} = '';  
       }  
   
       ## Stay in the data state.  
       if ($self->{content_model} == PCDATA_CONTENT_MODEL) {  
         !!!cp (13);  
         $self->{state} = PCDATA_STATE;  
       } else {  
         !!!cp (14);  
         ## Stay in the state.  
       }  
       !!!next-input-character;  
       !!!emit ($token);  
       redo A;  
     } elsif ($self->{state} == TAG_OPEN_STATE) {  
       if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA  
         if ($self->{nc} == 0x002F) { # /  
           !!!cp (15);  
           !!!next-input-character;  
           $self->{state} = CLOSE_TAG_OPEN_STATE;  
           redo A;  
         } elsif ($self->{nc} == 0x0021) { # !  
           !!!cp (15.1);  
           $self->{s_kwd} = '<' unless $self->{escape};  
           #  
         } else {  
           !!!cp (16);  
           #  
         }  
   
         ## reconsume  
         $self->{state} = DATA_STATE;  
         !!!emit ({type => CHARACTER_TOKEN, data => '<',  
                   line => $self->{line_prev},  
                   column => $self->{column_prev},  
                  });  
         redo A;  
       } elsif ($self->{content_model} & CM_FULL_MARKUP) { # PCDATA  
         if ($self->{nc} == 0x0021) { # !  
           !!!cp (17);  
           $self->{state} = MARKUP_DECLARATION_OPEN_STATE;  
           !!!next-input-character;  
           redo A;  
         } elsif ($self->{nc} == 0x002F) { # /  
           !!!cp (18);  
           $self->{state} = CLOSE_TAG_OPEN_STATE;  
           !!!next-input-character;  
           redo A;  
         } elsif (0x0041 <= $self->{nc} and  
                  $self->{nc} <= 0x005A) { # A..Z  
           !!!cp (19);  
           $self->{ct}  
             = {type => START_TAG_TOKEN,  
                tag_name => chr ($self->{nc} + 0x0020),  
                line => $self->{line_prev},  
                column => $self->{column_prev}};  
           $self->{state} = TAG_NAME_STATE;  
           !!!next-input-character;  
           redo A;  
         } elsif (0x0061 <= $self->{nc} and  
                  $self->{nc} <= 0x007A) { # a..z  
           !!!cp (20);  
           $self->{ct} = {type => START_TAG_TOKEN,  
                                     tag_name => chr ($self->{nc}),  
                                     line => $self->{line_prev},  
                                     column => $self->{column_prev}};  
           $self->{state} = TAG_NAME_STATE;  
           !!!next-input-character;  
           redo A;  
         } elsif ($self->{nc} == 0x003E) { # >  
           !!!cp (21);  
           !!!parse-error (type => 'empty start tag',  
                           line => $self->{line_prev},  
                           column => $self->{column_prev});  
           $self->{state} = DATA_STATE;  
           !!!next-input-character;  
   
           !!!emit ({type => CHARACTER_TOKEN, data => '<>',  
                     line => $self->{line_prev},  
                     column => $self->{column_prev},  
                    });  
   
           redo A;  
         } elsif ($self->{nc} == 0x003F) { # ?  
           !!!cp (22);  
           !!!parse-error (type => 'pio',  
                           line => $self->{line_prev},  
                           column => $self->{column_prev});  
           $self->{state} = BOGUS_COMMENT_STATE;  
           $self->{ct} = {type => COMMENT_TOKEN, data => '',  
                                     line => $self->{line_prev},  
                                     column => $self->{column_prev},  
                                    };  
           ## $self->{nc} is intentionally left as is  
           redo A;  
         } else {  
           !!!cp (23);  
           !!!parse-error (type => 'bare stago',  
                           line => $self->{line_prev},  
                           column => $self->{column_prev});  
           $self->{state} = DATA_STATE;  
           ## reconsume  
   
           !!!emit ({type => CHARACTER_TOKEN, data => '<',  
                     line => $self->{line_prev},  
                     column => $self->{column_prev},  
                    });  
   
           redo A;  
         }  
       } else {  
         die "$0: $self->{content_model} in tag open";  
       }  
     } elsif ($self->{state} == CLOSE_TAG_OPEN_STATE) {  
       ## NOTE: The "close tag open state" in the spec is implemented as  
       ## |CLOSE_TAG_OPEN_STATE| and |CDATA_RCDATA_CLOSE_TAG_STATE|.  
   
       my ($l, $c) = ($self->{line_prev}, $self->{column_prev} - 1); # "<"of"</"  
       if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA  
         if (defined $self->{last_stag_name}) {  
           $self->{state} = CDATA_RCDATA_CLOSE_TAG_STATE;  
           $self->{s_kwd} = '';  
           ## Reconsume.  
           redo A;  
         } else {  
           ## No start tag token has ever been emitted  
           ## NOTE: See <http://krijnhoetmer.nl/irc-logs/whatwg/20070626#l-564>.  
           !!!cp (28);  
           $self->{state} = DATA_STATE;  
           ## Reconsume.  
           !!!emit ({type => CHARACTER_TOKEN, data => '</',  
                     line => $l, column => $c,  
                    });  
           redo A;  
         }  
       }  
   
       if (0x0041 <= $self->{nc} and  
           $self->{nc} <= 0x005A) { # A..Z  
         !!!cp (29);  
         $self->{ct}  
             = {type => END_TAG_TOKEN,  
                tag_name => chr ($self->{nc} + 0x0020),  
                line => $l, column => $c};  
         $self->{state} = TAG_NAME_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif (0x0061 <= $self->{nc} and  
                $self->{nc} <= 0x007A) { # a..z  
         !!!cp (30);  
         $self->{ct} = {type => END_TAG_TOKEN,  
                                   tag_name => chr ($self->{nc}),  
                                   line => $l, column => $c};  
         $self->{state} = TAG_NAME_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         !!!cp (31);  
         !!!parse-error (type => 'empty end tag',  
                         line => $self->{line_prev}, ## "<" in "</>"  
                         column => $self->{column_prev} - 1);  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (32);  
         !!!parse-error (type => 'bare etago');  
         $self->{state} = DATA_STATE;  
         # reconsume  
   
         !!!emit ({type => CHARACTER_TOKEN, data => '</',  
                   line => $l, column => $c,  
                  });  
   
         redo A;  
       } else {  
         !!!cp (33);  
         !!!parse-error (type => 'bogus end tag');  
         $self->{state} = BOGUS_COMMENT_STATE;  
         $self->{ct} = {type => COMMENT_TOKEN, data => '',  
                                   line => $self->{line_prev}, # "<" of "</"  
                                   column => $self->{column_prev} - 1,  
                                  };  
         ## NOTE: $self->{nc} is intentionally left as is.  
         ## Although the "anything else" case of the spec not explicitly  
         ## states that the next input character is to be reconsumed,  
         ## it will be included to the |data| of the comment token  
         ## generated from the bogus end tag, as defined in the  
         ## "bogus comment state" entry.  
         redo A;  
       }  
     } elsif ($self->{state} == CDATA_RCDATA_CLOSE_TAG_STATE) {  
       my $ch = substr $self->{last_stag_name}, length $self->{s_kwd}, 1;  
       if (length $ch) {  
         my $CH = $ch;  
         $ch =~ tr/a-z/A-Z/;  
         my $nch = chr $self->{nc};  
         if ($nch eq $ch or $nch eq $CH) {  
           !!!cp (24);  
           ## Stay in the state.  
           $self->{s_kwd} .= $nch;  
           !!!next-input-character;  
           redo A;  
         } else {  
           !!!cp (25);  
           $self->{state} = DATA_STATE;  
           ## Reconsume.  
           !!!emit ({type => CHARACTER_TOKEN,  
                     data => '</' . $self->{s_kwd},  
                     line => $self->{line_prev},  
                     column => $self->{column_prev} - 1 - length $self->{s_kwd},  
                    });  
           redo A;  
         }  
       } else { # after "<{tag-name}"  
         unless ($is_space->{$self->{nc}} or  
                 {  
                  0x003E => 1, # >  
                  0x002F => 1, # /  
                  -1 => 1, # EOF  
                 }->{$self->{nc}}) {  
           !!!cp (26);  
           ## Reconsume.  
           $self->{state} = DATA_STATE;  
           !!!emit ({type => CHARACTER_TOKEN,  
                     data => '</' . $self->{s_kwd},  
                     line => $self->{line_prev},  
                     column => $self->{column_prev} - 1 - length $self->{s_kwd},  
                    });  
           redo A;  
         } else {  
           !!!cp (27);  
           $self->{ct}  
               = {type => END_TAG_TOKEN,  
                  tag_name => $self->{last_stag_name},  
                  line => $self->{line_prev},  
                  column => $self->{column_prev} - 1 - length $self->{s_kwd}};  
           $self->{state} = TAG_NAME_STATE;  
           ## Reconsume.  
           redo A;  
         }  
       }  
     } elsif ($self->{state} == TAG_NAME_STATE) {  
       if ($is_space->{$self->{nc}}) {  
         !!!cp (34);  
         $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (35);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           #if ($self->{ct}->{attributes}) {  
           #  ## NOTE: This should never be reached.  
           #  !!! cp (36);  
           #  !!! parse-error (type => 'end tag attribute');  
           #} else {  
             !!!cp (37);  
           #}  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } elsif (0x0041 <= $self->{nc} and  
                $self->{nc} <= 0x005A) { # A..Z  
         !!!cp (38);  
         $self->{ct}->{tag_name} .= chr ($self->{nc} + 0x0020);  
           # start tag or end tag  
         ## Stay in this state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!parse-error (type => 'unclosed tag');  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (39);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           #if ($self->{ct}->{attributes}) {  
           #  ## NOTE: This state should never be reached.  
           #  !!! cp (40);  
           #  !!! parse-error (type => 'end tag attribute');  
           #} else {  
             !!!cp (41);  
           #}  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         # reconsume  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } elsif ($self->{nc} == 0x002F) { # /  
         !!!cp (42);  
         $self->{state} = SELF_CLOSING_START_TAG_STATE;  
         !!!next-input-character;  
         redo A;  
       } else {  
         !!!cp (44);  
         $self->{ct}->{tag_name} .= chr $self->{nc};  
           # start tag or end tag  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == BEFORE_ATTRIBUTE_NAME_STATE) {  
       if ($is_space->{$self->{nc}}) {  
         !!!cp (45);  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (46);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           if ($self->{ct}->{attributes}) {  
             !!!cp (47);  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             !!!cp (48);  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } elsif (0x0041 <= $self->{nc} and  
                $self->{nc} <= 0x005A) { # A..Z  
         !!!cp (49);  
         $self->{ca}  
             = {name => chr ($self->{nc} + 0x0020),  
                value => '',  
                line => $self->{line}, column => $self->{column}};  
         $self->{state} = ATTRIBUTE_NAME_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x002F) { # /  
         !!!cp (50);  
         $self->{state} = SELF_CLOSING_START_TAG_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!parse-error (type => 'unclosed tag');  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (52);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           if ($self->{ct}->{attributes}) {  
             !!!cp (53);  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             !!!cp (54);  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         # reconsume  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } else {  
         if ({  
              0x0022 => 1, # "  
              0x0027 => 1, # '  
              0x003D => 1, # =  
             }->{$self->{nc}}) {  
           !!!cp (55);  
           !!!parse-error (type => 'bad attribute name');  
         } else {  
           !!!cp (56);  
         }  
         $self->{ca}  
             = {name => chr ($self->{nc}),  
                value => '',  
                line => $self->{line}, column => $self->{column}};  
         $self->{state} = ATTRIBUTE_NAME_STATE;  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == ATTRIBUTE_NAME_STATE) {  
       my $before_leave = sub {  
         if (exists $self->{ct}->{attributes} # start tag or end tag  
             ->{$self->{ca}->{name}}) { # MUST  
           !!!cp (57);  
           !!!parse-error (type => 'duplicate attribute', text => $self->{ca}->{name}, line => $self->{ca}->{line}, column => $self->{ca}->{column});  
           ## Discard $self->{ca} # MUST  
         } else {  
           !!!cp (58);  
           $self->{ct}->{attributes}->{$self->{ca}->{name}}  
             = $self->{ca};  
         }  
       }; # $before_leave  
   
       if ($is_space->{$self->{nc}}) {  
         !!!cp (59);  
         $before_leave->();  
         $self->{state} = AFTER_ATTRIBUTE_NAME_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003D) { # =  
         !!!cp (60);  
         $before_leave->();  
         $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         $before_leave->();  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (61);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           !!!cp (62);  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           if ($self->{ct}->{attributes}) {  
             !!!parse-error (type => 'end tag attribute');  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } elsif (0x0041 <= $self->{nc} and  
                $self->{nc} <= 0x005A) { # A..Z  
         !!!cp (63);  
         $self->{ca}->{name} .= chr ($self->{nc} + 0x0020);  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x002F) { # /  
         !!!cp (64);  
         $before_leave->();  
         $self->{state} = SELF_CLOSING_START_TAG_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!parse-error (type => 'unclosed tag');  
         $before_leave->();  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (66);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           if ($self->{ct}->{attributes}) {  
             !!!cp (67);  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             ## NOTE: This state should never be reached.  
             !!!cp (68);  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         # reconsume  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } else {  
         if ($self->{nc} == 0x0022 or # "  
             $self->{nc} == 0x0027) { # '  
           !!!cp (69);  
           !!!parse-error (type => 'bad attribute name');  
         } else {  
           !!!cp (70);  
         }  
         $self->{ca}->{name} .= chr ($self->{nc});  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == AFTER_ATTRIBUTE_NAME_STATE) {  
       if ($is_space->{$self->{nc}}) {  
         !!!cp (71);  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003D) { # =  
         !!!cp (72);  
         $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (73);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           if ($self->{ct}->{attributes}) {  
             !!!cp (74);  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             ## NOTE: This state should never be reached.  
             !!!cp (75);  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } elsif (0x0041 <= $self->{nc} and  
                $self->{nc} <= 0x005A) { # A..Z  
         !!!cp (76);  
         $self->{ca}  
             = {name => chr ($self->{nc} + 0x0020),  
                value => '',  
                line => $self->{line}, column => $self->{column}};  
         $self->{state} = ATTRIBUTE_NAME_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x002F) { # /  
         !!!cp (77);  
         $self->{state} = SELF_CLOSING_START_TAG_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!parse-error (type => 'unclosed tag');  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (79);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           if ($self->{ct}->{attributes}) {  
             !!!cp (80);  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             ## NOTE: This state should never be reached.  
             !!!cp (81);  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         # reconsume  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } else {  
         if ($self->{nc} == 0x0022 or # "  
             $self->{nc} == 0x0027) { # '  
           !!!cp (78);  
           !!!parse-error (type => 'bad attribute name');  
         } else {  
           !!!cp (82);  
         }  
         $self->{ca}  
             = {name => chr ($self->{nc}),  
                value => '',  
                line => $self->{line}, column => $self->{column}};  
         $self->{state} = ATTRIBUTE_NAME_STATE;  
         !!!next-input-character;  
         redo A;          
       }  
     } elsif ($self->{state} == BEFORE_ATTRIBUTE_VALUE_STATE) {  
       if ($is_space->{$self->{nc}}) {  
         !!!cp (83);  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x0022) { # "  
         !!!cp (84);  
         $self->{state} = ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x0026) { # &  
         !!!cp (85);  
         $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;  
         ## reconsume  
         redo A;  
       } elsif ($self->{nc} == 0x0027) { # '  
         !!!cp (86);  
         $self->{state} = ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         !!!parse-error (type => 'empty unquoted attribute value');  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (87);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           if ($self->{ct}->{attributes}) {  
             !!!cp (88);  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             ## NOTE: This state should never be reached.  
             !!!cp (89);  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!parse-error (type => 'unclosed tag');  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (90);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           if ($self->{ct}->{attributes}) {  
             !!!cp (91);  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             ## NOTE: This state should never be reached.  
             !!!cp (92);  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } else {  
         if ($self->{nc} == 0x003D) { # =  
           !!!cp (93);  
           !!!parse-error (type => 'bad attribute value');  
         } else {  
           !!!cp (94);  
         }  
         $self->{ca}->{value} .= chr ($self->{nc});  
         $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE) {  
       if ($self->{nc} == 0x0022) { # "  
         !!!cp (95);  
         $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x0026) { # &  
         !!!cp (96);  
         ## NOTE: In the spec, the tokenizer is switched to the  
         ## "entity in attribute value state".  In this implementation, the  
         ## tokenizer is switched to the |ENTITY_STATE|, which is an  
         ## implementation of the "consume a character reference" algorithm.  
         $self->{prev_state} = $self->{state};  
         $self->{entity_add} = 0x0022; # "  
         $self->{state} = ENTITY_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!parse-error (type => 'unclosed attribute value');  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (97);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           if ($self->{ct}->{attributes}) {  
             !!!cp (98);  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             ## NOTE: This state should never be reached.  
             !!!cp (99);  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } else {  
         !!!cp (100);  
         $self->{ca}->{value} .= chr ($self->{nc});  
         $self->{read_until}->($self->{ca}->{value},  
                               q["&],  
                               length $self->{ca}->{value});  
   
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE) {  
       if ($self->{nc} == 0x0027) { # '  
         !!!cp (101);  
         $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x0026) { # &  
         !!!cp (102);  
         ## NOTE: In the spec, the tokenizer is switched to the  
         ## "entity in attribute value state".  In this implementation, the  
         ## tokenizer is switched to the |ENTITY_STATE|, which is an  
         ## implementation of the "consume a character reference" algorithm.  
         $self->{entity_add} = 0x0027; # '  
         $self->{prev_state} = $self->{state};  
         $self->{state} = ENTITY_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!parse-error (type => 'unclosed attribute value');  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (103);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           if ($self->{ct}->{attributes}) {  
             !!!cp (104);  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             ## NOTE: This state should never be reached.  
             !!!cp (105);  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } else {  
         !!!cp (106);  
         $self->{ca}->{value} .= chr ($self->{nc});  
         $self->{read_until}->($self->{ca}->{value},  
                               q['&],  
                               length $self->{ca}->{value});  
   
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == ATTRIBUTE_VALUE_UNQUOTED_STATE) {  
       if ($is_space->{$self->{nc}}) {  
         !!!cp (107);  
         $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x0026) { # &  
         !!!cp (108);  
         ## NOTE: In the spec, the tokenizer is switched to the  
         ## "entity in attribute value state".  In this implementation, the  
         ## tokenizer is switched to the |ENTITY_STATE|, which is an  
         ## implementation of the "consume a character reference" algorithm.  
         $self->{entity_add} = -1;  
         $self->{prev_state} = $self->{state};  
         $self->{state} = ENTITY_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (109);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           if ($self->{ct}->{attributes}) {  
             !!!cp (110);  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             ## NOTE: This state should never be reached.  
             !!!cp (111);  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!parse-error (type => 'unclosed tag');  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (112);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           if ($self->{ct}->{attributes}) {  
             !!!cp (113);  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             ## NOTE: This state should never be reached.  
             !!!cp (114);  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } else {  
         if ({  
              0x0022 => 1, # "  
              0x0027 => 1, # '  
              0x003D => 1, # =  
             }->{$self->{nc}}) {  
           !!!cp (115);  
           !!!parse-error (type => 'bad attribute value');  
         } else {  
           !!!cp (116);  
         }  
         $self->{ca}->{value} .= chr ($self->{nc});  
         $self->{read_until}->($self->{ca}->{value},  
                               q["'=& >],  
                               length $self->{ca}->{value});  
   
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == AFTER_ATTRIBUTE_VALUE_QUOTED_STATE) {  
       if ($is_space->{$self->{nc}}) {  
         !!!cp (118);  
         $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (119);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           if ($self->{ct}->{attributes}) {  
             !!!cp (120);  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             ## NOTE: This state should never be reached.  
             !!!cp (121);  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } elsif ($self->{nc} == 0x002F) { # /  
         !!!cp (122);  
         $self->{state} = SELF_CLOSING_START_TAG_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!parse-error (type => 'unclosed tag');  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (122.3);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           if ($self->{ct}->{attributes}) {  
             !!!cp (122.1);  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             ## NOTE: This state should never be reached.  
             !!!cp (122.2);  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         ## Reconsume.  
         !!!emit ($self->{ct}); # start tag or end tag  
         redo A;  
       } else {  
         !!!cp ('124.1');  
         !!!parse-error (type => 'no space between attributes');  
         $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;  
         ## reconsume  
         redo A;  
       }  
     } elsif ($self->{state} == SELF_CLOSING_START_TAG_STATE) {  
       if ($self->{nc} == 0x003E) { # >  
         if ($self->{ct}->{type} == END_TAG_TOKEN) {  
           !!!cp ('124.2');  
           !!!parse-error (type => 'nestc', token => $self->{ct});  
           ## TODO: Different type than slash in start tag  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           if ($self->{ct}->{attributes}) {  
             !!!cp ('124.4');  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             !!!cp ('124.5');  
           }  
           ## TODO: Test |<title></title/>|  
         } else {  
           !!!cp ('124.3');  
           $self->{self_closing} = 1;  
         }  
   
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!parse-error (type => 'unclosed tag');  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (124.7);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           if ($self->{ct}->{attributes}) {  
             !!!cp (124.5);  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             ## NOTE: This state should never be reached.  
             !!!cp (124.6);  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         ## Reconsume.  
         !!!emit ($self->{ct}); # start tag or end tag  
         redo A;  
       } else {  
         !!!cp ('124.4');  
         !!!parse-error (type => 'nestc');  
         ## TODO: This error type is wrong.  
         $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;  
         ## Reconsume.  
         redo A;  
       }  
     } elsif ($self->{state} == BOGUS_COMMENT_STATE) {  
       ## (only happen if PCDATA state)  
   
       ## NOTE: Unlike spec's "bogus comment state", this implementation  
       ## consumes characters one-by-one basis.  
         
       if ($self->{nc} == 0x003E) { # >  
         !!!cp (124);  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # comment  
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (125);  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         !!!emit ($self->{ct}); # comment  
         redo A;  
       } else {  
         !!!cp (126);  
         $self->{ct}->{data} .= chr ($self->{nc}); # comment  
         $self->{read_until}->($self->{ct}->{data},  
                               q[>],  
                               length $self->{ct}->{data});  
   
         ## Stay in the state.  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == MARKUP_DECLARATION_OPEN_STATE) {  
       ## (only happen if PCDATA state)  
         
       if ($self->{nc} == 0x002D) { # -  
         !!!cp (133);  
         $self->{state} = MD_HYPHEN_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x0044 or # D  
                $self->{nc} == 0x0064) { # d  
         ## ASCII case-insensitive.  
         !!!cp (130);  
         $self->{state} = MD_DOCTYPE_STATE;  
         $self->{s_kwd} = chr $self->{nc};  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM and  
                $self->{open_elements}->[-1]->[1] & FOREIGN_EL and  
                $self->{nc} == 0x005B) { # [  
         !!!cp (135.4);                  
         $self->{state} = MD_CDATA_STATE;  
         $self->{s_kwd} = '[';  
         !!!next-input-character;  
         redo A;  
       } else {  
         !!!cp (136);  
       }  
   
       !!!parse-error (type => 'bogus comment',  
                       line => $self->{line_prev},  
                       column => $self->{column_prev} - 1);  
       ## Reconsume.  
       $self->{state} = BOGUS_COMMENT_STATE;  
       $self->{ct} = {type => COMMENT_TOKEN, data => '',  
                                 line => $self->{line_prev},  
                                 column => $self->{column_prev} - 1,  
                                };  
       redo A;  
     } elsif ($self->{state} == MD_HYPHEN_STATE) {  
       if ($self->{nc} == 0x002D) { # -  
         !!!cp (127);  
         $self->{ct} = {type => COMMENT_TOKEN, data => '',  
                                   line => $self->{line_prev},  
                                   column => $self->{column_prev} - 2,  
                                  };  
         $self->{state} = COMMENT_START_STATE;  
         !!!next-input-character;  
         redo A;  
       } else {  
         !!!cp (128);  
         !!!parse-error (type => 'bogus comment',  
                         line => $self->{line_prev},  
                         column => $self->{column_prev} - 2);  
         $self->{state} = BOGUS_COMMENT_STATE;  
         ## Reconsume.  
         $self->{ct} = {type => COMMENT_TOKEN,  
                                   data => '-',  
                                   line => $self->{line_prev},  
                                   column => $self->{column_prev} - 2,  
                                  };  
         redo A;  
       }  
     } elsif ($self->{state} == MD_DOCTYPE_STATE) {  
       ## ASCII case-insensitive.  
       if ($self->{nc} == [  
             undef,  
             0x004F, # O  
             0x0043, # C  
             0x0054, # T  
             0x0059, # Y  
             0x0050, # P  
           ]->[length $self->{s_kwd}] or  
           $self->{nc} == [  
             undef,  
             0x006F, # o  
             0x0063, # c  
             0x0074, # t  
             0x0079, # y  
             0x0070, # p  
           ]->[length $self->{s_kwd}]) {  
         !!!cp (131);  
         ## Stay in the state.  
         $self->{s_kwd} .= chr $self->{nc};  
         !!!next-input-character;  
         redo A;  
       } elsif ((length $self->{s_kwd}) == 6 and  
                ($self->{nc} == 0x0045 or # E  
                 $self->{nc} == 0x0065)) { # e  
         !!!cp (129);  
         $self->{state} = DOCTYPE_STATE;  
         $self->{ct} = {type => DOCTYPE_TOKEN,  
                                   quirks => 1,  
                                   line => $self->{line_prev},  
                                   column => $self->{column_prev} - 7,  
                                  };  
         !!!next-input-character;  
         redo A;  
       } else {  
         !!!cp (132);          
         !!!parse-error (type => 'bogus comment',  
                         line => $self->{line_prev},  
                         column => $self->{column_prev} - 1 - length $self->{s_kwd});  
         $self->{state} = BOGUS_COMMENT_STATE;  
         ## Reconsume.  
         $self->{ct} = {type => COMMENT_TOKEN,  
                                   data => $self->{s_kwd},  
                                   line => $self->{line_prev},  
                                   column => $self->{column_prev} - 1 - length $self->{s_kwd},  
                                  };  
         redo A;  
       }  
     } elsif ($self->{state} == MD_CDATA_STATE) {  
       if ($self->{nc} == {  
             '[' => 0x0043, # C  
             '[C' => 0x0044, # D  
             '[CD' => 0x0041, # A  
             '[CDA' => 0x0054, # T  
             '[CDAT' => 0x0041, # A  
           }->{$self->{s_kwd}}) {  
         !!!cp (135.1);  
         ## Stay in the state.  
         $self->{s_kwd} .= chr $self->{nc};  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{s_kwd} eq '[CDATA' and  
                $self->{nc} == 0x005B) { # [  
         !!!cp (135.2);  
         $self->{ct} = {type => CHARACTER_TOKEN,  
                                   data => '',  
                                   line => $self->{line_prev},  
                                   column => $self->{column_prev} - 7};  
         $self->{state} = CDATA_SECTION_STATE;  
         !!!next-input-character;  
         redo A;  
       } else {  
         !!!cp (135.3);  
         !!!parse-error (type => 'bogus comment',  
                         line => $self->{line_prev},  
                         column => $self->{column_prev} - 1 - length $self->{s_kwd});  
         $self->{state} = BOGUS_COMMENT_STATE;  
         ## Reconsume.  
         $self->{ct} = {type => COMMENT_TOKEN,  
                                   data => $self->{s_kwd},  
                                   line => $self->{line_prev},  
                                   column => $self->{column_prev} - 1 - length $self->{s_kwd},  
                                  };  
         redo A;  
       }  
     } elsif ($self->{state} == COMMENT_START_STATE) {  
       if ($self->{nc} == 0x002D) { # -  
         !!!cp (137);  
         $self->{state} = COMMENT_START_DASH_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         !!!cp (138);  
         !!!parse-error (type => 'bogus comment');  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # comment  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (139);  
         !!!parse-error (type => 'unclosed comment');  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         !!!emit ($self->{ct}); # comment  
   
         redo A;  
       } else {  
         !!!cp (140);  
         $self->{ct}->{data} # comment  
             .= chr ($self->{nc});  
         $self->{state} = COMMENT_STATE;  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == COMMENT_START_DASH_STATE) {  
       if ($self->{nc} == 0x002D) { # -  
         !!!cp (141);  
         $self->{state} = COMMENT_END_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         !!!cp (142);  
         !!!parse-error (type => 'bogus comment');  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # comment  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (143);  
         !!!parse-error (type => 'unclosed comment');  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         !!!emit ($self->{ct}); # comment  
   
         redo A;  
       } else {  
         !!!cp (144);  
         $self->{ct}->{data} # comment  
             .= '-' . chr ($self->{nc});  
         $self->{state} = COMMENT_STATE;  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == COMMENT_STATE) {  
       if ($self->{nc} == 0x002D) { # -  
         !!!cp (145);  
         $self->{state} = COMMENT_END_DASH_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (146);  
         !!!parse-error (type => 'unclosed comment');  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         !!!emit ($self->{ct}); # comment  
   
         redo A;  
       } else {  
         !!!cp (147);  
         $self->{ct}->{data} .= chr ($self->{nc}); # comment  
         $self->{read_until}->($self->{ct}->{data},  
                               q[-],  
                               length $self->{ct}->{data});  
   
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == COMMENT_END_DASH_STATE) {  
       if ($self->{nc} == 0x002D) { # -  
         !!!cp (148);  
         $self->{state} = COMMENT_END_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (149);  
         !!!parse-error (type => 'unclosed comment');  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         !!!emit ($self->{ct}); # comment  
   
         redo A;  
       } else {  
         !!!cp (150);  
         $self->{ct}->{data} .= '-' . chr ($self->{nc}); # comment  
         $self->{state} = COMMENT_STATE;  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == COMMENT_END_STATE) {  
       if ($self->{nc} == 0x003E) { # >  
         !!!cp (151);  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # comment  
   
         redo A;  
       } elsif ($self->{nc} == 0x002D) { # -  
         !!!cp (152);  
         !!!parse-error (type => 'dash in comment',  
                         line => $self->{line_prev},  
                         column => $self->{column_prev});  
         $self->{ct}->{data} .= '-'; # comment  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (153);  
         !!!parse-error (type => 'unclosed comment');  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         !!!emit ($self->{ct}); # comment  
   
         redo A;  
       } else {  
         !!!cp (154);  
         !!!parse-error (type => 'dash in comment',  
                         line => $self->{line_prev},  
                         column => $self->{column_prev});  
         $self->{ct}->{data} .= '--' . chr ($self->{nc}); # comment  
         $self->{state} = COMMENT_STATE;  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == DOCTYPE_STATE) {  
       if ($is_space->{$self->{nc}}) {  
         !!!cp (155);  
         $self->{state} = BEFORE_DOCTYPE_NAME_STATE;  
         !!!next-input-character;  
         redo A;  
       } else {  
         !!!cp (156);  
         !!!parse-error (type => 'no space before DOCTYPE name');  
         $self->{state} = BEFORE_DOCTYPE_NAME_STATE;  
         ## reconsume  
         redo A;  
       }  
     } elsif ($self->{state} == BEFORE_DOCTYPE_NAME_STATE) {  
       if ($is_space->{$self->{nc}}) {  
         !!!cp (157);  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         !!!cp (158);  
         !!!parse-error (type => 'no DOCTYPE name');  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # DOCTYPE (quirks)  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (159);  
         !!!parse-error (type => 'no DOCTYPE name');  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         !!!emit ($self->{ct}); # DOCTYPE (quirks)  
   
         redo A;  
       } else {  
         !!!cp (160);  
         $self->{ct}->{name} = chr $self->{nc};  
         delete $self->{ct}->{quirks};  
         $self->{state} = DOCTYPE_NAME_STATE;  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == DOCTYPE_NAME_STATE) {  
 ## ISSUE: Redundant "First," in the spec.  
       if ($is_space->{$self->{nc}}) {  
         !!!cp (161);  
         $self->{state} = AFTER_DOCTYPE_NAME_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         !!!cp (162);  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (163);  
         !!!parse-error (type => 'unclosed DOCTYPE');  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } else {  
         !!!cp (164);  
         $self->{ct}->{name}  
           .= chr ($self->{nc}); # DOCTYPE  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == AFTER_DOCTYPE_NAME_STATE) {  
       if ($is_space->{$self->{nc}}) {  
         !!!cp (165);  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         !!!cp (166);  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (167);  
         !!!parse-error (type => 'unclosed DOCTYPE');  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } elsif ($self->{nc} == 0x0050 or # P  
                $self->{nc} == 0x0070) { # p  
         $self->{state} = PUBLIC_STATE;  
         $self->{s_kwd} = chr $self->{nc};  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x0053 or # S  
                $self->{nc} == 0x0073) { # s  
         $self->{state} = SYSTEM_STATE;  
         $self->{s_kwd} = chr $self->{nc};  
         !!!next-input-character;  
         redo A;  
       } else {  
         !!!cp (180);  
         !!!parse-error (type => 'string after DOCTYPE name');  
         $self->{ct}->{quirks} = 1;  
   
         $self->{state} = BOGUS_DOCTYPE_STATE;  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == PUBLIC_STATE) {  
       ## ASCII case-insensitive  
       if ($self->{nc} == [  
             undef,  
             0x0055, # U  
             0x0042, # B  
             0x004C, # L  
             0x0049, # I  
           ]->[length $self->{s_kwd}] or  
           $self->{nc} == [  
             undef,  
             0x0075, # u  
             0x0062, # b  
             0x006C, # l  
             0x0069, # i  
           ]->[length $self->{s_kwd}]) {  
         !!!cp (175);  
         ## Stay in the state.  
         $self->{s_kwd} .= chr $self->{nc};  
         !!!next-input-character;  
         redo A;  
       } elsif ((length $self->{s_kwd}) == 5 and  
                ($self->{nc} == 0x0043 or # C  
                 $self->{nc} == 0x0063)) { # c  
         !!!cp (168);  
         $self->{state} = BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE;  
         !!!next-input-character;  
         redo A;  
       } else {  
         !!!cp (169);  
         !!!parse-error (type => 'string after DOCTYPE name',  
                         line => $self->{line_prev},  
                         column => $self->{column_prev} + 1 - length $self->{s_kwd});  
         $self->{ct}->{quirks} = 1;  
   
         $self->{state} = BOGUS_DOCTYPE_STATE;  
         ## Reconsume.  
         redo A;  
       }  
     } elsif ($self->{state} == SYSTEM_STATE) {  
       ## ASCII case-insensitive  
       if ($self->{nc} == [  
             undef,  
             0x0059, # Y  
             0x0053, # S  
             0x0054, # T  
             0x0045, # E  
           ]->[length $self->{s_kwd}] or  
           $self->{nc} == [  
             undef,  
             0x0079, # y  
             0x0073, # s  
             0x0074, # t  
             0x0065, # e  
           ]->[length $self->{s_kwd}]) {  
         !!!cp (170);  
         ## Stay in the state.  
         $self->{s_kwd} .= chr $self->{nc};  
         !!!next-input-character;  
         redo A;  
       } elsif ((length $self->{s_kwd}) == 5 and  
                ($self->{nc} == 0x004D or # M  
                 $self->{nc} == 0x006D)) { # m  
         !!!cp (171);  
         $self->{state} = BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE;  
         !!!next-input-character;  
         redo A;  
       } else {  
         !!!cp (172);  
         !!!parse-error (type => 'string after DOCTYPE name',  
                         line => $self->{line_prev},  
                         column => $self->{column_prev} + 1 - length $self->{s_kwd});  
         $self->{ct}->{quirks} = 1;  
   
         $self->{state} = BOGUS_DOCTYPE_STATE;  
         ## Reconsume.  
         redo A;  
       }  
     } elsif ($self->{state} == BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {  
       if ($is_space->{$self->{nc}}) {  
         !!!cp (181);  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} eq 0x0022) { # "  
         !!!cp (182);  
         $self->{ct}->{pubid} = ''; # DOCTYPE  
         $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} eq 0x0027) { # '  
         !!!cp (183);  
         $self->{ct}->{pubid} = ''; # DOCTYPE  
         $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} eq 0x003E) { # >  
         !!!cp (184);  
         !!!parse-error (type => 'no PUBLIC literal');  
   
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (185);  
         !!!parse-error (type => 'unclosed DOCTYPE');  
   
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } else {  
         !!!cp (186);  
         !!!parse-error (type => 'string after PUBLIC');  
         $self->{ct}->{quirks} = 1;  
   
         $self->{state} = BOGUS_DOCTYPE_STATE;  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE) {  
       if ($self->{nc} == 0x0022) { # "  
         !!!cp (187);  
         $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         !!!cp (188);  
         !!!parse-error (type => 'unclosed PUBLIC literal');  
   
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (189);  
         !!!parse-error (type => 'unclosed PUBLIC literal');  
   
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } else {  
         !!!cp (190);  
         $self->{ct}->{pubid} # DOCTYPE  
             .= chr $self->{nc};  
         $self->{read_until}->($self->{ct}->{pubid}, q[">],  
                               length $self->{ct}->{pubid});  
   
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE) {  
       if ($self->{nc} == 0x0027) { # '  
         !!!cp (191);  
         $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         !!!cp (192);  
         !!!parse-error (type => 'unclosed PUBLIC literal');  
   
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (193);  
         !!!parse-error (type => 'unclosed PUBLIC literal');  
   
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } else {  
         !!!cp (194);  
         $self->{ct}->{pubid} # DOCTYPE  
             .= chr $self->{nc};  
         $self->{read_until}->($self->{ct}->{pubid}, q['>],  
                               length $self->{ct}->{pubid});  
   
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {  
       if ($is_space->{$self->{nc}}) {  
         !!!cp (195);  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x0022) { # "  
         !!!cp (196);  
         $self->{ct}->{sysid} = ''; # DOCTYPE  
         $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x0027) { # '  
         !!!cp (197);  
         $self->{ct}->{sysid} = ''; # DOCTYPE  
         $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         !!!cp (198);  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (199);  
         !!!parse-error (type => 'unclosed DOCTYPE');  
   
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } else {  
         !!!cp (200);  
         !!!parse-error (type => 'string after PUBLIC literal');  
         $self->{ct}->{quirks} = 1;  
   
         $self->{state} = BOGUS_DOCTYPE_STATE;  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {  
       if ($is_space->{$self->{nc}}) {  
         !!!cp (201);  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x0022) { # "  
         !!!cp (202);  
         $self->{ct}->{sysid} = ''; # DOCTYPE  
         $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x0027) { # '  
         !!!cp (203);  
         $self->{ct}->{sysid} = ''; # DOCTYPE  
         $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         !!!cp (204);  
         !!!parse-error (type => 'no SYSTEM literal');  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (205);  
         !!!parse-error (type => 'unclosed DOCTYPE');  
   
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } else {  
         !!!cp (206);  
         !!!parse-error (type => 'string after SYSTEM');  
         $self->{ct}->{quirks} = 1;  
   
         $self->{state} = BOGUS_DOCTYPE_STATE;  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE) {  
       if ($self->{nc} == 0x0022) { # "  
         !!!cp (207);  
         $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         !!!cp (208);  
         !!!parse-error (type => 'unclosed SYSTEM literal');  
   
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (209);  
         !!!parse-error (type => 'unclosed SYSTEM literal');  
   
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } else {  
         !!!cp (210);  
         $self->{ct}->{sysid} # DOCTYPE  
             .= chr $self->{nc};  
         $self->{read_until}->($self->{ct}->{sysid}, q[">],  
                               length $self->{ct}->{sysid});  
   
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE) {  
       if ($self->{nc} == 0x0027) { # '  
         !!!cp (211);  
         $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         !!!cp (212);  
         !!!parse-error (type => 'unclosed SYSTEM literal');  
   
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (213);  
         !!!parse-error (type => 'unclosed SYSTEM literal');  
   
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } else {  
         !!!cp (214);  
         $self->{ct}->{sysid} # DOCTYPE  
             .= chr $self->{nc};  
         $self->{read_until}->($self->{ct}->{sysid}, q['>],  
                               length $self->{ct}->{sysid});  
   
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {  
       if ($is_space->{$self->{nc}}) {  
         !!!cp (215);  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         !!!cp (216);  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (217);  
         !!!parse-error (type => 'unclosed DOCTYPE');  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } else {  
         !!!cp (218);  
         !!!parse-error (type => 'string after SYSTEM literal');  
         #$self->{ct}->{quirks} = 1;  
   
         $self->{state} = BOGUS_DOCTYPE_STATE;  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == BOGUS_DOCTYPE_STATE) {  
       if ($self->{nc} == 0x003E) { # >  
         !!!cp (219);  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (220);  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } else {  
         !!!cp (221);  
         my $s = '';  
         $self->{read_until}->($s, q[>], 0);  
   
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == CDATA_SECTION_STATE) {  
       ## NOTE: "CDATA section state" in the state is jointly implemented  
       ## by three states, |CDATA_SECTION_STATE|, |CDATA_SECTION_MSE1_STATE|,  
       ## and |CDATA_SECTION_MSE2_STATE|.  
         
       if ($self->{nc} == 0x005D) { # ]  
         !!!cp (221.1);  
         $self->{state} = CDATA_SECTION_MSE1_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == -1) {  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
         if (length $self->{ct}->{data}) { # character  
           !!!cp (221.2);  
           !!!emit ($self->{ct}); # character  
         } else {  
           !!!cp (221.3);  
           ## No token to emit. $self->{ct} is discarded.  
         }          
         redo A;  
       } else {  
         !!!cp (221.4);  
         $self->{ct}->{data} .= chr $self->{nc};  
         $self->{read_until}->($self->{ct}->{data},  
                               q<]>,  
                               length $self->{ct}->{data});  
   
         ## Stay in the state.  
         !!!next-input-character;  
         redo A;  
       }  
   
       ## ISSUE: "text tokens" in spec.  
     } elsif ($self->{state} == CDATA_SECTION_MSE1_STATE) {  
       if ($self->{nc} == 0x005D) { # ]  
         !!!cp (221.5);  
         $self->{state} = CDATA_SECTION_MSE2_STATE;  
         !!!next-input-character;  
         redo A;  
       } else {  
         !!!cp (221.6);  
         $self->{ct}->{data} .= ']';  
         $self->{state} = CDATA_SECTION_STATE;  
         ## Reconsume.  
         redo A;  
       }  
     } elsif ($self->{state} == CDATA_SECTION_MSE2_STATE) {  
       if ($self->{nc} == 0x003E) { # >  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
         if (length $self->{ct}->{data}) { # character  
           !!!cp (221.7);  
           !!!emit ($self->{ct}); # character  
         } else {  
           !!!cp (221.8);  
           ## No token to emit. $self->{ct} is discarded.  
         }  
         redo A;  
       } elsif ($self->{nc} == 0x005D) { # ]  
         !!!cp (221.9); # character  
         $self->{ct}->{data} .= ']'; ## Add first "]" of "]]]".  
         ## Stay in the state.  
         !!!next-input-character;  
         redo A;  
       } else {  
         !!!cp (221.11);  
         $self->{ct}->{data} .= ']]'; # character  
         $self->{state} = CDATA_SECTION_STATE;  
         ## Reconsume.  
         redo A;  
       }  
     } elsif ($self->{state} == ENTITY_STATE) {  
       if ($is_space->{$self->{nc}} or  
           {  
             0x003C => 1, 0x0026 => 1, -1 => 1, # <, &  
             $self->{entity_add} => 1,  
           }->{$self->{nc}}) {  
         !!!cp (1001);  
         ## Don't consume  
         ## No error  
         ## Return nothing.  
         #  
       } elsif ($self->{nc} == 0x0023) { # #  
         !!!cp (999);  
         $self->{state} = ENTITY_HASH_STATE;  
         $self->{s_kwd} = '#';  
         !!!next-input-character;  
         redo A;  
       } elsif ((0x0041 <= $self->{nc} and  
                 $self->{nc} <= 0x005A) or # A..Z  
                (0x0061 <= $self->{nc} and  
                 $self->{nc} <= 0x007A)) { # a..z  
         !!!cp (998);  
         require Whatpm::_NamedEntityList;  
         $self->{state} = ENTITY_NAME_STATE;  
         $self->{s_kwd} = chr $self->{nc};  
         $self->{entity__value} = $self->{s_kwd};  
         $self->{entity__match} = 0;  
         !!!next-input-character;  
         redo A;  
       } else {  
         !!!cp (1027);  
         !!!parse-error (type => 'bare ero');  
         ## Return nothing.  
         #  
       }  
   
       ## NOTE: No character is consumed by the "consume a character  
       ## reference" algorithm.  In other word, there is an "&" character  
       ## that does not introduce a character reference, which would be  
       ## appended to the parent element or the attribute value in later  
       ## process of the tokenizer.  
   
       if ($self->{prev_state} == DATA_STATE) {  
         !!!cp (997);  
         $self->{state} = $self->{prev_state};  
         ## Reconsume.  
         !!!emit ({type => CHARACTER_TOKEN, data => '&',  
                   line => $self->{line_prev},  
                   column => $self->{column_prev},  
                  });  
         redo A;  
       } else {  
         !!!cp (996);  
         $self->{ca}->{value} .= '&';  
         $self->{state} = $self->{prev_state};  
         ## Reconsume.  
         redo A;  
       }  
     } elsif ($self->{state} == ENTITY_HASH_STATE) {  
       if ($self->{nc} == 0x0078 or # x  
           $self->{nc} == 0x0058) { # X  
         !!!cp (995);  
         $self->{state} = HEXREF_X_STATE;  
         $self->{s_kwd} .= chr $self->{nc};  
         !!!next-input-character;  
         redo A;  
       } elsif (0x0030 <= $self->{nc} and  
                $self->{nc} <= 0x0039) { # 0..9  
         !!!cp (994);  
         $self->{state} = NCR_NUM_STATE;  
         $self->{s_kwd} = $self->{nc} - 0x0030;  
         !!!next-input-character;  
         redo A;  
       } else {  
         !!!parse-error (type => 'bare nero',  
                         line => $self->{line_prev},  
                         column => $self->{column_prev} - 1);  
   
         ## NOTE: According to the spec algorithm, nothing is returned,  
         ## and then "&#" is appended to the parent element or the attribute  
         ## value in the later processing.  
   
         if ($self->{prev_state} == DATA_STATE) {  
           !!!cp (1019);  
           $self->{state} = $self->{prev_state};  
           ## Reconsume.  
           !!!emit ({type => CHARACTER_TOKEN,  
                     data => '&#',  
                     line => $self->{line_prev},  
                     column => $self->{column_prev} - 1,  
                    });  
           redo A;  
         } else {  
           !!!cp (993);  
           $self->{ca}->{value} .= '&#';  
           $self->{state} = $self->{prev_state};  
           ## Reconsume.  
           redo A;  
         }  
       }  
     } elsif ($self->{state} == NCR_NUM_STATE) {  
       if (0x0030 <= $self->{nc} and  
           $self->{nc} <= 0x0039) { # 0..9  
         !!!cp (1012);  
         $self->{s_kwd} *= 10;  
         $self->{s_kwd} += $self->{nc} - 0x0030;  
           
         ## Stay in the state.  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003B) { # ;  
         !!!cp (1013);  
         !!!next-input-character;  
         #  
       } else {  
         !!!cp (1014);  
         !!!parse-error (type => 'no refc');  
         ## Reconsume.  
         #  
       }  
   
       my $code = $self->{s_kwd};  
       my $l = $self->{line_prev};  
       my $c = $self->{column_prev};  
       if ($charref_map->{$code}) {  
         !!!cp (1015);  
         !!!parse-error (type => 'invalid character reference',  
                         text => (sprintf 'U+%04X', $code),  
                         line => $l, column => $c);  
         $code = $charref_map->{$code};  
       } elsif ($code > 0x10FFFF) {  
         !!!cp (1016);  
         !!!parse-error (type => 'invalid character reference',  
                         text => (sprintf 'U-%08X', $code),  
                         line => $l, column => $c);  
         $code = 0xFFFD;  
       }  
   
       if ($self->{prev_state} == DATA_STATE) {  
         !!!cp (992);  
         $self->{state} = $self->{prev_state};  
         ## Reconsume.  
         !!!emit ({type => CHARACTER_TOKEN, data => chr $code,  
                   line => $l, column => $c,  
                  });  
         redo A;  
       } else {  
         !!!cp (991);  
         $self->{ca}->{value} .= chr $code;  
         $self->{ca}->{has_reference} = 1;  
         $self->{state} = $self->{prev_state};  
         ## Reconsume.  
         redo A;  
       }  
     } elsif ($self->{state} == HEXREF_X_STATE) {  
       if ((0x0030 <= $self->{nc} and $self->{nc} <= 0x0039) or  
           (0x0041 <= $self->{nc} and $self->{nc} <= 0x0046) or  
           (0x0061 <= $self->{nc} and $self->{nc} <= 0x0066)) {  
         # 0..9, A..F, a..f  
         !!!cp (990);  
         $self->{state} = HEXREF_HEX_STATE;  
         $self->{s_kwd} = 0;  
         ## Reconsume.  
         redo A;  
       } else {  
         !!!parse-error (type => 'bare hcro',  
                         line => $self->{line_prev},  
                         column => $self->{column_prev} - 2);  
   
         ## NOTE: According to the spec algorithm, nothing is returned,  
         ## and then "&#" followed by "X" or "x" is appended to the parent  
         ## element or the attribute value in the later processing.  
   
         if ($self->{prev_state} == DATA_STATE) {  
           !!!cp (1005);  
           $self->{state} = $self->{prev_state};  
           ## Reconsume.  
           !!!emit ({type => CHARACTER_TOKEN,  
                     data => '&' . $self->{s_kwd},  
                     line => $self->{line_prev},  
                     column => $self->{column_prev} - length $self->{s_kwd},  
                    });  
           redo A;  
         } else {  
           !!!cp (989);  
           $self->{ca}->{value} .= '&' . $self->{s_kwd};  
           $self->{state} = $self->{prev_state};  
           ## Reconsume.  
           redo A;  
         }  
       }  
     } elsif ($self->{state} == HEXREF_HEX_STATE) {  
       if (0x0030 <= $self->{nc} and $self->{nc} <= 0x0039) {  
         # 0..9  
         !!!cp (1002);  
         $self->{s_kwd} *= 0x10;  
         $self->{s_kwd} += $self->{nc} - 0x0030;  
         ## Stay in the state.  
         !!!next-input-character;  
         redo A;  
       } elsif (0x0061 <= $self->{nc} and  
                $self->{nc} <= 0x0066) { # a..f  
         !!!cp (1003);  
         $self->{s_kwd} *= 0x10;  
         $self->{s_kwd} += $self->{nc} - 0x0060 + 9;  
         ## Stay in the state.  
         !!!next-input-character;  
         redo A;  
       } elsif (0x0041 <= $self->{nc} and  
                $self->{nc} <= 0x0046) { # A..F  
         !!!cp (1004);  
         $self->{s_kwd} *= 0x10;  
         $self->{s_kwd} += $self->{nc} - 0x0040 + 9;  
         ## Stay in the state.  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003B) { # ;  
         !!!cp (1006);  
         !!!next-input-character;  
         #  
       } else {  
         !!!cp (1007);  
         !!!parse-error (type => 'no refc',  
                         line => $self->{line},  
                         column => $self->{column});  
         ## Reconsume.  
         #  
       }  
   
       my $code = $self->{s_kwd};  
       my $l = $self->{line_prev};  
       my $c = $self->{column_prev};  
       if ($charref_map->{$code}) {  
         !!!cp (1008);  
         !!!parse-error (type => 'invalid character reference',  
                         text => (sprintf 'U+%04X', $code),  
                         line => $l, column => $c);  
         $code = $charref_map->{$code};  
       } elsif ($code > 0x10FFFF) {  
         !!!cp (1009);  
         !!!parse-error (type => 'invalid character reference',  
                         text => (sprintf 'U-%08X', $code),  
                         line => $l, column => $c);  
         $code = 0xFFFD;  
       }  
   
       if ($self->{prev_state} == DATA_STATE) {  
         !!!cp (988);  
         $self->{state} = $self->{prev_state};  
         ## Reconsume.  
         !!!emit ({type => CHARACTER_TOKEN, data => chr $code,  
                   line => $l, column => $c,  
                  });  
         redo A;  
       } else {  
         !!!cp (987);  
         $self->{ca}->{value} .= chr $code;  
         $self->{ca}->{has_reference} = 1;  
         $self->{state} = $self->{prev_state};  
         ## Reconsume.  
         redo A;  
       }  
     } elsif ($self->{state} == ENTITY_NAME_STATE) {  
       if (length $self->{s_kwd} < 30 and  
           ## NOTE: Some number greater than the maximum length of entity name  
           ((0x0041 <= $self->{nc} and # a  
             $self->{nc} <= 0x005A) or # x  
            (0x0061 <= $self->{nc} and # a  
             $self->{nc} <= 0x007A) or # z  
            (0x0030 <= $self->{nc} and # 0  
             $self->{nc} <= 0x0039) or # 9  
            $self->{nc} == 0x003B)) { # ;  
         our $EntityChar;  
         $self->{s_kwd} .= chr $self->{nc};  
         if (defined $EntityChar->{$self->{s_kwd}}) {  
           if ($self->{nc} == 0x003B) { # ;  
             !!!cp (1020);  
             $self->{entity__value} = $EntityChar->{$self->{s_kwd}};  
             $self->{entity__match} = 1;  
             !!!next-input-character;  
             #  
           } else {  
             !!!cp (1021);  
             $self->{entity__value} = $EntityChar->{$self->{s_kwd}};  
             $self->{entity__match} = -1;  
             ## Stay in the state.  
             !!!next-input-character;  
             redo A;  
           }  
         } else {  
           !!!cp (1022);  
           $self->{entity__value} .= chr $self->{nc};  
           $self->{entity__match} *= 2;  
           ## Stay in the state.  
           !!!next-input-character;  
           redo A;  
         }  
       }  
   
       my $data;  
       my $has_ref;  
       if ($self->{entity__match} > 0) {  
         !!!cp (1023);  
         $data = $self->{entity__value};  
         $has_ref = 1;  
         #  
       } elsif ($self->{entity__match} < 0) {  
         !!!parse-error (type => 'no refc');  
         if ($self->{prev_state} != DATA_STATE and # in attribute  
             $self->{entity__match} < -1) {  
           !!!cp (1024);  
           $data = '&' . $self->{s_kwd};  
           #  
         } else {  
           !!!cp (1025);  
           $data = $self->{entity__value};  
           $has_ref = 1;  
           #  
         }  
       } else {  
         !!!cp (1026);  
         !!!parse-error (type => 'bare ero',  
                         line => $self->{line_prev},  
                         column => $self->{column_prev} - length $self->{s_kwd});  
         $data = '&' . $self->{s_kwd};  
         #  
       }  
     
       ## NOTE: In these cases, when a character reference is found,  
       ## it is consumed and a character token is returned, or, otherwise,  
       ## nothing is consumed and returned, according to the spec algorithm.  
       ## In this implementation, anything that has been examined by the  
       ## tokenizer is appended to the parent element or the attribute value  
       ## as string, either literal string when no character reference or  
       ## entity-replaced string otherwise, in this stage, since any characters  
       ## that would not be consumed are appended in the data state or in an  
       ## appropriate attribute value state anyway.  
   
       if ($self->{prev_state} == DATA_STATE) {  
         !!!cp (986);  
         $self->{state} = $self->{prev_state};  
         ## Reconsume.  
         !!!emit ({type => CHARACTER_TOKEN,  
                   data => $data,  
                   line => $self->{line_prev},  
                   column => $self->{column_prev} + 1 - length $self->{s_kwd},  
                  });  
         redo A;  
       } else {  
         !!!cp (985);  
         $self->{ca}->{value} .= $data;  
         $self->{ca}->{has_reference} = 1 if $has_ref;  
         $self->{state} = $self->{prev_state};  
         ## Reconsume.  
         redo A;  
       }  
     } else {  
       die "$0: $self->{state}: Unknown state";  
     }  
   } # A    
   
   die "$0: _get_next_token: unexpected case";  
 } # _get_next_token  
   
866  sub _initialize_tree_constructor ($) {  sub _initialize_tree_constructor ($) {
867    my $self = shift;    my $self = shift;
868    ## NOTE: $self->{document} MUST be specified before this method is called    ## NOTE: $self->{document} MUST be specified before this method is called
# Line 3447  sub _initialize_tree_constructor ($) { Line 872  sub _initialize_tree_constructor ($) {
872    $self->{document}->manakai_is_html (1); # MUST    $self->{document}->manakai_is_html (1); # MUST
873    $self->{document}->set_user_data (manakai_source_line => 1);    $self->{document}->set_user_data (manakai_source_line => 1);
874    $self->{document}->set_user_data (manakai_source_column => 1);    $self->{document}->set_user_data (manakai_source_column => 1);
875    
876      $self->{frameset_ok} = 1;
877  } # _initialize_tree_constructor  } # _initialize_tree_constructor
878    
879  sub _terminate_tree_constructor ($) {  sub _terminate_tree_constructor ($) {
# Line 3466  sub _construct_tree ($) { Line 893  sub _construct_tree ($) {
893    ## When an interactive UA render the $self->{document} available    ## When an interactive UA render the $self->{document} available
894    ## to the user, or when it begin accepting user input, are    ## to the user, or when it begin accepting user input, are
895    ## not defined.    ## not defined.
   
   ## Append a character: collect it and all subsequent consecutive  
   ## characters and insert one Text node whose data is concatenation  
   ## of all those characters. # MUST  
896        
897    !!!next-token;    !!!next-token;
898    
# Line 3478  sub _construct_tree ($) { Line 901  sub _construct_tree ($) {
901    undef $self->{head_element_inserted};    undef $self->{head_element_inserted};
902    $self->{open_elements} = [];    $self->{open_elements} = [];
903    undef $self->{inner_html_node};    undef $self->{inner_html_node};
904      undef $self->{ignore_newline};
905    
906    ## NOTE: The "initial" insertion mode.    ## NOTE: The "initial" insertion mode.
907    $self->_tree_construction_initial; # MUST    $self->_tree_construction_initial; # MUST
# Line 3497  sub _tree_construction_initial ($) { Line 921  sub _tree_construction_initial ($) {
921    
922    INITIAL: {    INITIAL: {
923      if ($token->{type} == DOCTYPE_TOKEN) {      if ($token->{type} == DOCTYPE_TOKEN) {
924        ## NOTE: Conformance checkers MAY, instead of reporting "not HTML5"        ## NOTE: Conformance checkers MAY, instead of reporting "not
925        ## error, switch to a conformance checking mode for another        ## HTML5" error, switch to a conformance checking mode for
926        ## language.        ## another language.  (We don't support such mode switchings; it
927          ## is nonsense to do anything different from what browsers do.)
928        my $doctype_name = $token->{name};        my $doctype_name = $token->{name};
929        $doctype_name = '' unless defined $doctype_name;        $doctype_name = '' unless defined $doctype_name;
930        $doctype_name =~ tr/a-z/A-Z/; # ASCII case-insensitive        my $doctype = $self->{document}->create_document_type_definition
931        if (not defined $token->{name} or # <!DOCTYPE>            ($doctype_name);
932            defined $token->{sysid}) {  
933          $doctype_name =~ tr/A-Z/a-z/; # ASCII case-insensitive.
934          if ($doctype_name ne 'html') {
935          !!!cp ('t1');          !!!cp ('t1');
936          !!!parse-error (type => 'not HTML5', token => $token);          !!!parse-error (type => 'not HTML5', token => $token);
       } elsif ($doctype_name ne 'HTML') {  
         !!!cp ('t2');  
         !!!parse-error (type => 'not HTML5', token => $token);  
937        } elsif (defined $token->{pubid}) {        } elsif (defined $token->{pubid}) {
938          if ($token->{pubid} eq 'XSLT-compat') {          ## Obsolete permitted DOCTYPEs (case-sensitive)
939            !!!cp ('t1.2');          my $xsysid = {
940              '-//W3C//DTD HTML 4.0//EN' => 'http://www.w3.org/TR/REC-html40/strict.dtd',
941              '-//W3C//DTD HTML 4.01//EN' => 'http://www.w3.org/TR/html4/strict.dtd',
942              '-//W3C//DTD XHTML 1.0 Strict//EN' => 'http://www.w3.org/TR/xhtml1/DTD/xhtml1-strict.dtd',
943              '-//W3C//DTD XHTML 1.1//EN' => 'http://www.w3.org/TR/xhtml11/DTD/xhtml11.dtd',
944            }->{$token->{pubid}};
945            if (defined $xsysid and
946                (not defined $token->{sysid} or $token->{sysid} eq $xsysid)) {
947              !!!cp ('t2');
948              !!!parse-error (type => 'obs DOCTYPE', token => $token,
949                              level => $self->{level}->{obc}); ## XXX error type
950            } else {
951              !!!cp ('t2.1');
952              !!!parse-error (type => 'not HTML5', token => $token);
953            }
954          } elsif (defined $token->{sysid}) {
955            if ($token->{sysid} eq 'about:legacy-compat') {
956              !!!cp ('t1.2'); ## <!DOCTYPE HTML SYSTEM "about:legacy-compat">
957            !!!parse-error (type => 'XSLT-compat', token => $token,            !!!parse-error (type => 'XSLT-compat', token => $token,
958                            level => $self->{level}->{should});                            level => $self->{level}->{should});
959          } else {          } else {
960            !!!parse-error (type => 'not HTML5', token => $token);            !!!parse-error (type => 'not HTML5', token => $token);
961          }          }
962        } else {        } else { ## <!DOCTYPE HTML>
963          !!!cp ('t3');          !!!cp ('t3');
964          #          #
965        }        }
966                
       my $doctype = $self->{document}->create_document_type_definition  
         ($token->{name}); ## ISSUE: If name is missing (e.g. <!DOCTYPE>)?  
967        ## NOTE: Default value for both |public_id| and |system_id| attributes        ## NOTE: Default value for both |public_id| and |system_id| attributes
968        ## are empty strings, so that we don't set any value in missing cases.        ## are empty strings, so that we don't set any value in missing cases.
969        $doctype->public_id ($token->{pubid}) if defined $token->{pubid};        $doctype->public_id ($token->{pubid}) if defined $token->{pubid};
970        $doctype->system_id ($token->{sysid}) if defined $token->{sysid};        $doctype->system_id ($token->{sysid}) if defined $token->{sysid};
971    
972        ## NOTE: Other DocumentType attributes are null or empty lists.        ## NOTE: Other DocumentType attributes are null or empty lists.
973        ## ISSUE: internalSubset = null??        ## In Firefox3, |internalSubset| attribute is set to the empty
974          ## string, while |null| is an allowed value for the attribute
975          ## according to DOM3 Core.
976        $self->{document}->append_child ($doctype);        $self->{document}->append_child ($doctype);
977                
978        if ($token->{quirks} or $doctype_name ne 'HTML') {        if ($token->{quirks} or $doctype_name ne 'html') {
979          !!!cp ('t4');          !!!cp ('t4');
980          $self->{document}->manakai_compat_mode ('quirks');          $self->{document}->manakai_compat_mode ('quirks');
981        } elsif (defined $token->{pubid}) {        } elsif (defined $token->{pubid}) {
982          my $pubid = $token->{pubid};          my $pubid = $token->{pubid};
983          $pubid =~ tr/a-z/A-z/;          $pubid =~ tr/a-z/A-Z/; ## ASCII case-insensitive.
984          my $prefix = [          my $prefix = [
985            "+//SILMARIL//DTD HTML PRO V0R11 19970101//",            "+//SILMARIL//DTD HTML PRO V0R11 19970101//",
986            "-//ADVASOFT LTD//DTD HTML 3.0 ASWEDIT + EXTENSIONS//",            "-//ADVASOFT LTD//DTD HTML 3.0 ASWEDIT + EXTENSIONS//",
# Line 3630  sub _tree_construction_initial ($) { Line 1072  sub _tree_construction_initial ($) {
1072        }        }
1073        if (defined $token->{sysid}) {        if (defined $token->{sysid}) {
1074          my $sysid = $token->{sysid};          my $sysid = $token->{sysid};
1075          $sysid =~ tr/A-Z/a-z/;          $sysid =~ tr/A-Z/a-z/; ## ASCII case-insensitive.
1076          if ($sysid eq "http://www.ibm.com/data/dtd/v11/ibmxhtml1-transitional.dtd") {          if ($sysid eq "http://www.ibm.com/data/dtd/v11/ibmxhtml1-transitional.dtd") {
1077            ## NOTE: Ensure that |PUBLIC "(limited quirks)" "(quirks)"| is            ## NOTE: Ensure that |PUBLIC "(limited quirks)" "(quirks)"|
1078            ## marked as quirks.            ## is signaled as in quirks mode!
1079            $self->{document}->manakai_compat_mode ('quirks');            $self->{document}->manakai_compat_mode ('quirks');
1080            !!!cp ('t11');            !!!cp ('t11');
1081          } else {          } else {
# Line 3818  sub _reset_insertion_mode ($) { Line 1260  sub _reset_insertion_mode ($) {
1260          ## SVG elements.  Currently the HTML syntax supports only MathML and          ## SVG elements.  Currently the HTML syntax supports only MathML and
1261          ## SVG elements as foreigners.          ## SVG elements as foreigners.
1262          $new_mode = IN_BODY_IM | IN_FOREIGN_CONTENT_IM;          $new_mode = IN_BODY_IM | IN_FOREIGN_CONTENT_IM;
1263        } elsif ($node->[1] & TABLE_CELL_EL) {        } elsif ($node->[1] == TABLE_CELL_EL) {
1264          if ($last) {          if ($last) {
1265            !!!cp ('t28.2');            !!!cp ('t28.2');
1266            #            #
# Line 3847  sub _reset_insertion_mode ($) { Line 1289  sub _reset_insertion_mode ($) {
1289        $self->{insertion_mode} = $new_mode and return if defined $new_mode;        $self->{insertion_mode} = $new_mode and return if defined $new_mode;
1290                
1291        ## Step 15        ## Step 15
1292        if ($node->[1] & HTML_EL) {        if ($node->[1] == HTML_EL) {
1293          unless (defined $self->{head_element}) {          unless (defined $self->{head_element}) {
1294            !!!cp ('t29');            !!!cp ('t29');
1295            $self->{insertion_mode} = BEFORE_HEAD_IM;            $self->{insertion_mode} = BEFORE_HEAD_IM;
# Line 3979  sub _tree_construction_main ($) { Line 1421  sub _tree_construction_main ($) {
1421    
1422      ## Step 1      ## Step 1
1423      my $start_tag_name = $token->{tag_name};      my $start_tag_name = $token->{tag_name};
1424      my $el;      !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
     !!!create-element ($el, $HTML_NS, $start_tag_name, $token->{attributes}, $token);  
1425    
1426      ## Step 2      ## Step 2
     $insert->($el);  
   
     ## Step 3  
1427      $self->{content_model} = $content_model_flag; # CDATA or RCDATA      $self->{content_model} = $content_model_flag; # CDATA or RCDATA
1428      delete $self->{escape}; # MUST      delete $self->{escape}; # MUST
1429    
1430      ## Step 4      ## Step 3, 4
1431      my $text = '';      $self->{insertion_mode} |= IN_CDATA_RCDATA_IM;
     !!!nack ('t40.1');  
     !!!next-token;  
     while ($token->{type} == CHARACTER_TOKEN) { # or until stop tokenizing  
       !!!cp ('t40');  
       $text .= $token->{data};  
       !!!next-token;  
     }  
   
     ## Step 5  
     if (length $text) {  
       !!!cp ('t41');  
       my $text = $self->{document}->create_text_node ($text);  
       $el->append_child ($text);  
     }  
   
     ## Step 6  
     $self->{content_model} = PCDATA_CONTENT_MODEL;  
1432    
1433      ## Step 7      !!!nack ('t40.1');
     if ($token->{type} == END_TAG_TOKEN and  
         $token->{tag_name} eq $start_tag_name) {  
       !!!cp ('t42');  
       ## Ignore the token  
     } else {  
       ## NOTE: An end-of-file token.  
       if ($content_model_flag == CDATA_CONTENT_MODEL) {  
         !!!cp ('t43');  
         !!!parse-error (type => 'in CDATA:#eof', token => $token);  
       } elsif ($content_model_flag == RCDATA_CONTENT_MODEL) {  
         !!!cp ('t44');  
         !!!parse-error (type => 'in RCDATA:#eof', token => $token);  
       } else {  
         die "$0: $content_model_flag in parse_rcdata";  
       }  
     }  
1434      !!!next-token;      !!!next-token;
1435    }; # $parse_rcdata    }; # $parse_rcdata
1436    
1437    my $script_start_tag = sub () {    my $script_start_tag = sub () {
1438        ## Step 1
1439      my $script_el;      my $script_el;
1440      !!!create-element ($script_el, $HTML_NS, 'script', $token->{attributes}, $token);      !!!create-element ($script_el, $HTML_NS, 'script', $token->{attributes}, $token);
1441    
1442        ## Step 2
1443      ## TODO: mark as "parser-inserted"      ## TODO: mark as "parser-inserted"
1444    
1445        ## Step 3
1446        ## TODO: Mark as "already executed", if ...
1447    
1448        ## Step 4 (HTML5 revision 2702)
1449        $insert->($script_el);
1450        push @{$self->{open_elements}}, [$script_el, $el_category->{script}];
1451    
1452        ## Step 5
1453      $self->{content_model} = CDATA_CONTENT_MODEL;      $self->{content_model} = CDATA_CONTENT_MODEL;
1454      delete $self->{escape}; # MUST      delete $self->{escape}; # MUST
       
     my $text = '';  
     !!!nack ('t45.1');  
     !!!next-token;  
     while ($token->{type} == CHARACTER_TOKEN) {  
       !!!cp ('t45');  
       $text .= $token->{data};  
       !!!next-token;  
     } # stop if non-character token or tokenizer stops tokenising  
     if (length $text) {  
       !!!cp ('t46');  
       $script_el->manakai_append_text ($text);  
     }  
                 
     $self->{content_model} = PCDATA_CONTENT_MODEL;  
1455    
1456      if ($token->{type} == END_TAG_TOKEN and      ## Step 6-7
1457          $token->{tag_name} eq 'script') {      $self->{insertion_mode} |= IN_CDATA_RCDATA_IM;
       !!!cp ('t47');  
       ## Ignore the token  
     } else {  
       !!!cp ('t48');  
       !!!parse-error (type => 'in CDATA:#eof', token => $token);  
       ## ISSUE: And ignore?  
       ## TODO: mark as "already executed"  
     }  
       
     if (defined $self->{inner_html_node}) {  
       !!!cp ('t49');  
       ## TODO: mark as "already executed"  
     } else {  
       !!!cp ('t50');  
       ## TODO: $old_insertion_point = current insertion point  
       ## TODO: insertion point = just before the next input character  
1458    
1459        $insert->($script_el);      !!!nack ('t40.2');
         
       ## TODO: insertion point = $old_insertion_point (might be "undefined")  
         
       ## TODO: if there is a script that will execute as soon as the parser resume, then...  
     }  
       
1460      !!!next-token;      !!!next-token;
1461    }; # $script_start_tag    }; # $script_start_tag
1462    
1463    ## NOTE: $open_tables->[-1]->[0] is the "current table" element node.    ## NOTE: $open_tables->[-1]->[0] is the "current table" element node.
1464    ## NOTE: $open_tables->[-1]->[1] is the "tainted" flag.    ## NOTE: $open_tables->[-1]->[1] is the "tainted" flag (OBSOLETE; unused).
1465    ## NOTE: $open_tables->[-1]->[2] is set false when non-Text node inserted.    ## NOTE: $open_tables->[-1]->[2] is set false when non-Text node inserted.
1466    my $open_tables = [[$self->{open_elements}->[0]->[0]]];    my $open_tables = [[$self->{open_elements}->[0]->[0]]];
1467    
# Line 4252  sub _tree_construction_main ($) { Line 1631  sub _tree_construction_main ($) {
1631                
1632        ## Step 8        ## Step 8
1633        if ($common_ancestor_node->[1] & TABLE_ROWS_EL) {        if ($common_ancestor_node->[1] & TABLE_ROWS_EL) {
1634            ## Foster parenting.
1635          my $foster_parent_element;          my $foster_parent_element;
1636          my $next_sibling;          my $next_sibling;
1637          OE: for (reverse 0..$#{$self->{open_elements}}) {          OE: for (reverse 0..$#{$self->{open_elements}}) {
1638            if ($self->{open_elements}->[$_]->[1] & TABLE_EL) {            if ($self->{open_elements}->[$_]->[1] == TABLE_EL) {
1639                               my $parent = $self->{open_elements}->[$_]->[0]->parent_node;              !!!cp ('t65.2');
1640                               if (defined $parent and $parent->node_type == 1) {              $foster_parent_element = $self->{open_elements}->[$_ - 1]->[0];
1641                                 !!!cp ('t65.1');              $next_sibling = $self->{open_elements}->[$_]->[0];
1642                                 $foster_parent_element = $parent;              undef $next_sibling
1643                                 $next_sibling = $self->{open_elements}->[$_]->[0];                  unless $next_sibling->parent_node eq $foster_parent_element;
1644                               } else {              last OE;
1645                                 !!!cp ('t65.2');            }
1646                                 $foster_parent_element          } # OE
1647                                   = $self->{open_elements}->[$_ - 1]->[0];          $foster_parent_element ||= $self->{open_elements}->[0]->[0];
1648                               }  
                              last OE;  
                            }  
                          } # OE  
                          $foster_parent_element = $self->{open_elements}->[0]->[0]  
                            unless defined $foster_parent_element;  
1649          $foster_parent_element->insert_before ($last_node->[0], $next_sibling);          $foster_parent_element->insert_before ($last_node->[0], $next_sibling);
1650          $open_tables->[-1]->[1] = 1; # tainted          $open_tables->[-1]->[1] = 1; # tainted
1651        } else {        } else {
# Line 4326  sub _tree_construction_main ($) { Line 1701  sub _tree_construction_main ($) {
1701      $self->{open_elements}->[-1]->[0]->append_child ($_[0]);      $self->{open_elements}->[-1]->[0]->append_child ($_[0]);
1702    }; # $insert_to_current    }; # $insert_to_current
1703    
1704      ## Foster parenting.  Note that there are three "foster parenting"
1705      ## code in the parser: for elements (this one), for texts, and for
1706      ## elements in the AAA code.
1707    my $insert_to_foster = sub {    my $insert_to_foster = sub {
1708      my $child = shift;      my $child = shift;
1709      if ($self->{open_elements}->[-1]->[1] & TABLE_ROWS_EL) {      if ($self->{open_elements}->[-1]->[1] & TABLE_ROWS_EL) {
# Line 4333  sub _tree_construction_main ($) { Line 1711  sub _tree_construction_main ($) {
1711        my $foster_parent_element;        my $foster_parent_element;
1712        my $next_sibling;        my $next_sibling;
1713        OE: for (reverse 0..$#{$self->{open_elements}}) {        OE: for (reverse 0..$#{$self->{open_elements}}) {
1714          if ($self->{open_elements}->[$_]->[1] & TABLE_EL) {          if ($self->{open_elements}->[$_]->[1] == TABLE_EL) {
1715                               my $parent = $self->{open_elements}->[$_]->[0]->parent_node;            !!!cp ('t71');
1716                               if (defined $parent and $parent->node_type == 1) {            $foster_parent_element = $self->{open_elements}->[$_ - 1]->[0];
1717                                 !!!cp ('t70');            $next_sibling = $self->{open_elements}->[$_]->[0];
1718                                 $foster_parent_element = $parent;            undef $next_sibling
1719                                 $next_sibling = $self->{open_elements}->[$_]->[0];                unless $next_sibling->parent_node eq $foster_parent_element;
1720                               } else {            last OE;
1721                                 !!!cp ('t71');          }
1722                                 $foster_parent_element        } # OE
1723                                   = $self->{open_elements}->[$_ - 1]->[0];        $foster_parent_element ||= $self->{open_elements}->[0]->[0];
1724                               }  
1725                               last OE;        $foster_parent_element->insert_before ($child, $next_sibling);
                            }  
                          } # OE  
                          $foster_parent_element = $self->{open_elements}->[0]->[0]  
                            unless defined $foster_parent_element;  
                          $foster_parent_element->insert_before  
                            ($child, $next_sibling);  
1726        $open_tables->[-1]->[1] = 1; # tainted        $open_tables->[-1]->[1] = 1; # tainted
1727      } else {      } else {
1728        !!!cp ('t72');        !!!cp ('t72');
# Line 4358  sub _tree_construction_main ($) { Line 1730  sub _tree_construction_main ($) {
1730      }      }
1731    }; # $insert_to_foster    }; # $insert_to_foster
1732    
1733    ## NOTE: When a character is inserted, if the last node that was    ## NOTE: Insert a character (MUST): When a character is inserted, if
1734    ## inserted by the parser is a Text node and the character has to be    ## the last node that was inserted by the parser is a Text node and
1735    ## inserted after that node, then the character is appended to the    ## the character has to be inserted after that node, then the
1736    ## Text node.  However, if any other node is inserted by the parser,    ## character is appended to the Text node.  However, if any other
1737    ## then a new Text node is created and the character is appended as    ## node is inserted by the parser, then a new Text node is created
1738    ## that Text node.  If I'm not wrong, there are only two cases where    ## and the character is appended as that Text node.  If I'm not
1739    ## this occurs.  One is the case where an element node is inserted    ## wrong, for a parser with scripting disabled, there are only two
1740    ## to the |head| element.  This is covered by using the    ## cases where this occurs.  One is the case where an element node
1741      ## is inserted to the |head| element.  This is covered by using the
1742    ## |$self->{head_element_inserted}| flag.  Another is the case where    ## |$self->{head_element_inserted}| flag.  Another is the case where
1743    ## an element or comment is inserted into the |table| subtree while    ## an element or comment is inserted into the |table| subtree while
1744    ## foster parenting happens.  This is covered by using the [2] flag    ## foster parenting happens.  This is covered by using the [2] flag
1745    ## of the |$open_tables| structure.  All other cases are handled    ## of the |$open_tables| structure.  All other cases are handled
1746    ## simply by calling |manakai_append_text| method.    ## simply by calling |manakai_append_text| method.
1747    
1748      ## TODO: |<body><script>document.write("a<br>");
1749      ## document.body.removeChild (document.body.lastChild);
1750      ## document.write ("b")</script>|
1751    
1752    B: while (1) {    B: while (1) {
1753    
1754        ## The "in table text" insertion mode.
1755        if ($self->{insertion_mode} & TABLE_IMS and
1756            not $self->{insertion_mode} & IN_FOREIGN_CONTENT_IM and
1757            not $self->{insertion_mode} & IN_CDATA_RCDATA_IM) {
1758          C: {
1759            my $s;
1760            if ($token->{type} == CHARACTER_TOKEN) {
1761              !!!cp ('t194');
1762              $self->{pending_chars} ||= [];
1763              push @{$self->{pending_chars}}, $token;
1764              !!!next-token;
1765              next B;
1766            } else {
1767              if ($self->{pending_chars}) {
1768                $s = join '', map { $_->{data} } @{$self->{pending_chars}};
1769                delete $self->{pending_chars};
1770                if ($s =~ /[^\x09\x0A\x0C\x0D\x20]/) {
1771                  !!!cp ('t195');
1772                  #
1773                } else {
1774                  !!!cp ('t195.1');
1775                  #$self->{open_elements}->[-1]->[0]->manakai_append_text ($s);
1776                  $self->{open_elements}->[-1]->[0]->append_child
1777                      ($self->{document}->create_text_node ($s));
1778                  last C;
1779                }
1780              } else {
1781                !!!cp ('t195.2');
1782                last C;
1783              }
1784            }
1785    
1786            ## Foster parenting.
1787            !!!parse-error (type => 'in table:#text', token => $token);
1788    
1789            ## NOTE: As if in body, but insert into the foster parent element.
1790            $reconstruct_active_formatting_elements->($insert_to_foster);
1791                
1792            if ($self->{open_elements}->[-1]->[1] & TABLE_ROWS_EL) {
1793              # MUST
1794              my $foster_parent_element;
1795              my $next_sibling;
1796              OE: for (reverse 0..$#{$self->{open_elements}}) {
1797                if ($self->{open_elements}->[$_]->[1] == TABLE_EL) {
1798                  !!!cp ('t197');
1799                  $foster_parent_element = $self->{open_elements}->[$_ - 1]->[0];
1800                  $next_sibling = $self->{open_elements}->[$_]->[0];
1801                  undef $next_sibling
1802                    unless $next_sibling->parent_node eq $foster_parent_element;
1803                  last OE;
1804                }
1805              } # OE
1806              $foster_parent_element ||= $self->{open_elements}->[0]->[0];
1807    
1808              !!!cp ('t199');
1809              $foster_parent_element->insert_before
1810                  ($self->{document}->create_text_node ($s), $next_sibling);
1811    
1812              $open_tables->[-1]->[1] = 1; # tainted
1813              $open_tables->[-1]->[2] = 1; # ~node inserted
1814            } else {
1815              ## NOTE: Fragment case or in a foster parent'ed element
1816              ## (e.g. |<table><span>a|).  In fragment case, whether the
1817              ## character is appended to existing node or a new node is
1818              ## created is irrelevant, since the foster parent'ed nodes
1819              ## are discarded and fragment parsing does not invoke any
1820              ## script.
1821              !!!cp ('t200');
1822              $self->{open_elements}->[-1]->[0]->manakai_append_text ($s);
1823            }
1824          } # C
1825        } # TABLE_IMS
1826    
1827      if ($token->{type} == DOCTYPE_TOKEN) {      if ($token->{type} == DOCTYPE_TOKEN) {
1828        !!!cp ('t73');        !!!cp ('t73');
1829        !!!parse-error (type => 'in html:#DOCTYPE', token => $token);        !!!parse-error (type => 'in html:#DOCTYPE', token => $token);
# Line 4423  sub _tree_construction_main ($) { Line 1874  sub _tree_construction_main ($) {
1874        }        }
1875        !!!next-token;        !!!next-token;
1876        next B;        next B;
1877        } elsif ($self->{insertion_mode} & IN_CDATA_RCDATA_IM) {
1878          if ($token->{type} == CHARACTER_TOKEN) {
1879            $token->{data} =~ s/^\x0A// if $self->{ignore_newline};
1880            delete $self->{ignore_newline};
1881    
1882            if (length $token->{data}) {
1883              !!!cp ('t43');
1884              $self->{open_elements}->[-1]->[0]->manakai_append_text
1885                  ($token->{data});
1886            } else {
1887              !!!cp ('t43.1');
1888            }
1889            !!!next-token;
1890            next B;
1891          } elsif ($token->{type} == END_TAG_TOKEN) {
1892            delete $self->{ignore_newline};
1893    
1894            if ($token->{tag_name} eq 'script') {
1895              !!!cp ('t50');
1896              
1897              ## Para 1-2
1898              my $script = pop @{$self->{open_elements}};
1899              
1900              ## Para 3
1901              $self->{insertion_mode} &= ~ IN_CDATA_RCDATA_IM;
1902    
1903              ## Para 4
1904              ## TODO: $old_insertion_point = $current_insertion_point;
1905              ## TODO: $current_insertion_point = just before $self->{nc};
1906    
1907              ## Para 5
1908              ## TODO: Run the $script->[0].
1909    
1910              ## Para 6
1911              ## TODO: $current_insertion_point = $old_insertion_point;
1912    
1913              ## Para 7
1914              ## TODO: if ($pending_external_script) {
1915                ## TODO: ...
1916              ## TODO: }
1917    
1918              !!!next-token;
1919              next B;
1920            } else {
1921              !!!cp ('t42');
1922    
1923              pop @{$self->{open_elements}};
1924    
1925              $self->{insertion_mode} &= ~ IN_CDATA_RCDATA_IM;
1926              !!!next-token;
1927              next B;
1928            }
1929          } elsif ($token->{type} == END_OF_FILE_TOKEN) {
1930            delete $self->{ignore_newline};
1931    
1932            !!!cp ('t44');
1933            !!!parse-error (type => 'not closed',
1934                            text => $self->{open_elements}->[-1]->[0]
1935                                ->manakai_local_name,
1936                            token => $token);
1937    
1938            #if ($self->{open_elements}->[-1]->[1] == SCRIPT_EL) {
1939            #  ## TODO: Mark as "already executed"
1940            #}
1941    
1942            pop @{$self->{open_elements}};
1943    
1944            $self->{insertion_mode} &= ~ IN_CDATA_RCDATA_IM;
1945            ## Reprocess.
1946            next B;
1947          } else {
1948            die "$0: $token->{type}: In CDATA/RCDATA: Unknown token type";        
1949          }
1950      } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM) {      } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM) {
1951        if ($token->{type} == CHARACTER_TOKEN) {        if ($token->{type} == CHARACTER_TOKEN) {
1952          !!!cp ('t87.1');          !!!cp ('t87.1');
1953    
1954          $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});          $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});
1955    
1956            if ($token->{data} =~ /[^\x09\x0A\x0C\x0D\x20]/) {
1957              delete $self->{frameset_ok};
1958            }
1959    
1960          !!!next-token;          !!!next-token;
1961          next B;          next B;
1962        } elsif ($token->{type} == START_TAG_TOKEN) {        } elsif ($token->{type} == START_TAG_TOKEN) {
# Line 4434  sub _tree_construction_main ($) { Line 1964  sub _tree_construction_main ($) {
1964               $self->{open_elements}->[-1]->[1] & FOREIGN_FLOW_CONTENT_EL) or               $self->{open_elements}->[-1]->[1] & FOREIGN_FLOW_CONTENT_EL) or
1965              not ($self->{open_elements}->[-1]->[1] & FOREIGN_EL) or              not ($self->{open_elements}->[-1]->[1] & FOREIGN_EL) or
1966              ($token->{tag_name} eq 'svg' and              ($token->{tag_name} eq 'svg' and
1967               $self->{open_elements}->[-1]->[1] & MML_AXML_EL)) {               $self->{open_elements}->[-1]->[1] == MML_AXML_EL)) {
1968            ## NOTE: "using the rules for secondary insertion mode"then"continue"            ## NOTE: "using the rules for secondary insertion mode"then"continue"
1969            !!!cp ('t87.2');            !!!cp ('t87.2');
1970            #            #
1971          } elsif ({          } elsif ({
1972                    b => 1, big => 1, blockquote => 1, body => 1, br => 1,                    b => 1, big => 1, blockquote => 1, body => 1, br => 1,
1973                    center => 1, code => 1, dd => 1, div => 1, dl => 1, dt => 1,                    center => 1, code => 1, dd => 1, div => 1, dl => 1, dt => 1,
1974                    em => 1, embed => 1, font => 1, h1 => 1, h2 => 1, h3 => 1,                    em => 1, embed => 1, h1 => 1, h2 => 1, h3 => 1,
1975                    h4 => 1, h5 => 1, h6 => 1, head => 1, hr => 1, i => 1,                    h4 => 1, h5 => 1, h6 => 1, head => 1, hr => 1, i => 1,
1976                    img => 1, li => 1, listing => 1, menu => 1, meta => 1,                    img => 1, li => 1, listing => 1, menu => 1, meta => 1,
1977                    nobr => 1, ol => 1, p => 1, pre => 1, ruby => 1, s => 1,                    nobr => 1, ol => 1, p => 1, pre => 1, ruby => 1, s => 1,
1978                    small => 1, span => 1, strong => 1, strike => 1, sub => 1,                    small => 1, span => 1, strong => 1, strike => 1, sub => 1,
1979                    sup => 1, table => 1, tt => 1, u => 1, ul => 1, var => 1,                    sup => 1, table => 1, tt => 1, u => 1, ul => 1, var => 1,
1980                   }->{$token->{tag_name}}) {                   }->{$token->{tag_name}} or
1981                     ($token->{tag_name} eq 'font' and
1982                      ($token->{attributes}->{color} or
1983                       $token->{attributes}->{face} or
1984                       $token->{attributes}->{size}))) {
1985            !!!cp ('t87.2');            !!!cp ('t87.2');
1986            !!!parse-error (type => 'not closed',            !!!parse-error (type => 'not closed',
1987                            text => $self->{open_elements}->[-1]->[0]                            text => $self->{open_elements}->[-1]->[0]
# Line 4523  sub _tree_construction_main ($) { Line 2057  sub _tree_construction_main ($) {
2057          }          }
2058        } elsif ($token->{type} == END_TAG_TOKEN) {        } elsif ($token->{type} == END_TAG_TOKEN) {
2059          ## NOTE: "using the rules for secondary insertion mode" then "continue"          ## NOTE: "using the rules for secondary insertion mode" then "continue"
2060          !!!cp ('t87.5');          if ($token->{tag_name} eq 'script') {
2061          #            !!!cp ('t87.41');
2062              #
2063              ## XXXscript: Execute script here.
2064            } else {
2065              !!!cp ('t87.5');
2066              #
2067            }
2068        } elsif ($token->{type} == END_OF_FILE_TOKEN) {        } elsif ($token->{type} == END_OF_FILE_TOKEN) {
2069          !!!cp ('t87.6');          !!!cp ('t87.6');
2070          !!!parse-error (type => 'not closed',          !!!parse-error (type => 'not closed',
# Line 4611  sub _tree_construction_main ($) { Line 2151  sub _tree_construction_main ($) {
2151          ## As if <body>          ## As if <body>
2152          !!!insert-element ('body',, $token);          !!!insert-element ('body',, $token);
2153          $self->{insertion_mode} = IN_BODY_IM;          $self->{insertion_mode} = IN_BODY_IM;
2154          ## reprocess          ## The "frameset-ok" flag is left unchanged in this case.
2155            ## Reporcess the token.
2156          next B;          next B;
2157        } elsif ($token->{type} == START_TAG_TOKEN) {        } elsif ($token->{type} == START_TAG_TOKEN) {
2158          if ($token->{tag_name} eq 'head') {          if ($token->{tag_name} eq 'head') {
# Line 4708  sub _tree_construction_main ($) { Line 2249  sub _tree_construction_main ($) {
2249            !!!ack ('t103.1');            !!!ack ('t103.1');
2250            !!!next-token;            !!!next-token;
2251            next B;            next B;
2252          } elsif ($token->{tag_name} eq 'command' or          } elsif ($token->{tag_name} eq 'command') {
                  $token->{tag_name} eq 'eventsource') {  
2253            if ($self->{insertion_mode} == IN_HEAD_IM) {            if ($self->{insertion_mode} == IN_HEAD_IM) {
2254              ## NOTE: If the insertion mode at the time of the emission              ## NOTE: If the insertion mode at the time of the emission
2255              ## of the token was "before head", $self->{insertion_mode}              ## of the token was "before head", $self->{insertion_mode}
# Line 4824  sub _tree_construction_main ($) { Line 2364  sub _tree_construction_main ($) {
2364    
2365            ## NOTE: There is a "as if in head" code clone.            ## NOTE: There is a "as if in head" code clone.
2366            $parse_rcdata->(RCDATA_CONTENT_MODEL);            $parse_rcdata->(RCDATA_CONTENT_MODEL);
2367            pop @{$self->{open_elements}} # <head>  
2368                if $self->{insertion_mode} == AFTER_HEAD_IM;            ## NOTE: At this point the stack of open elements contain
2369              ## the |head| element (index == -2) and the |script| element
2370              ## (index == -1).  In the "after head" insertion mode the
2371              ## |head| element is inserted only for the purpose of
2372              ## providing the context for the |script| element, and
2373              ## therefore we can now and have to remove the element from
2374              ## the stack.
2375              splice @{$self->{open_elements}}, -2, 1, () # <head>
2376                  if ($self->{insertion_mode} & IM_MASK) == AFTER_HEAD_IM;
2377            next B;            next B;
2378          } elsif ($token->{tag_name} eq 'style' or          } elsif ($token->{tag_name} eq 'style' or
2379                   $token->{tag_name} eq 'noframes') {                   $token->{tag_name} eq 'noframes') {
# Line 4843  sub _tree_construction_main ($) { Line 2391  sub _tree_construction_main ($) {
2391              !!!cp ('t115');              !!!cp ('t115');
2392            }            }
2393            $parse_rcdata->(CDATA_CONTENT_MODEL);            $parse_rcdata->(CDATA_CONTENT_MODEL);
2394            pop @{$self->{open_elements}} # <head>            ## ISSUE: A spec bug [Bug 6038]
2395                if $self->{insertion_mode} == AFTER_HEAD_IM;            splice @{$self->{open_elements}}, -2, 1, () # <head>
2396                  if ($self->{insertion_mode} & IM_MASK) == AFTER_HEAD_IM;
2397            next B;            next B;
2398              } elsif ($token->{tag_name} eq 'noscript') {          } elsif ($token->{tag_name} eq 'noscript') {
2399                if ($self->{insertion_mode} == IN_HEAD_IM) {                if ($self->{insertion_mode} == IN_HEAD_IM) {
2400                  !!!cp ('t116');                  !!!cp ('t116');
2401                  ## NOTE: and scripting is disalbed                  ## NOTE: and scripting is disalbed
# Line 4890  sub _tree_construction_main ($) { Line 2439  sub _tree_construction_main ($) {
2439    
2440            ## NOTE: There is a "as if in head" code clone.            ## NOTE: There is a "as if in head" code clone.
2441            $script_start_tag->();            $script_start_tag->();
2442            pop @{$self->{open_elements}} # <head>            ## ISSUE: A spec bug  [Bug 6038]
2443                if $self->{insertion_mode} == AFTER_HEAD_IM;            splice @{$self->{open_elements}}, -2, 1 # <head>
2444                  if ($self->{insertion_mode} & IM_MASK) == AFTER_HEAD_IM;
2445            next B;            next B;
2446          } elsif ($token->{tag_name} eq 'body' or          } elsif ($token->{tag_name} eq 'body' or
2447                   $token->{tag_name} eq 'frameset') {                   $token->{tag_name} eq 'frameset') {
2448                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {            if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
2449                  !!!cp ('t122');              !!!cp ('t122');
2450                  ## As if </noscript>              ## As if </noscript>
2451                  pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
2452                  !!!parse-error (type => 'in noscript',              !!!parse-error (type => 'in noscript',
2453                                  text => $token->{tag_name}, token => $token);                              text => $token->{tag_name}, token => $token);
2454                                
2455                  ## Reprocess in the "in head" insertion mode...              ## Reprocess in the "in head" insertion mode...
2456                  ## As if </head>              ## As if </head>
2457                  pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
2458                                
2459                  ## Reprocess in the "after head" insertion mode...              ## Reprocess in the "after head" insertion mode...
2460                } elsif ($self->{insertion_mode} == IN_HEAD_IM) {            } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
2461                  !!!cp ('t124');              !!!cp ('t124');
2462                  pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
2463                                
2464                  ## Reprocess in the "after head" insertion mode...              ## Reprocess in the "after head" insertion mode...
2465                } else {            } else {
2466                  !!!cp ('t125');              !!!cp ('t125');
2467                }            }
2468    
2469                ## "after head" insertion mode            ## "after head" insertion mode
2470                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);            !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
2471                if ($token->{tag_name} eq 'body') {            if ($token->{tag_name} eq 'body') {
2472                  !!!cp ('t126');              !!!cp ('t126');
2473                  $self->{insertion_mode} = IN_BODY_IM;              delete $self->{frameset_ok};
2474                } elsif ($token->{tag_name} eq 'frameset') {              $self->{insertion_mode} = IN_BODY_IM;
2475                  !!!cp ('t127');            } elsif ($token->{tag_name} eq 'frameset') {
2476                  $self->{insertion_mode} = IN_FRAMESET_IM;              !!!cp ('t127');
2477                } else {              $self->{insertion_mode} = IN_FRAMESET_IM;
2478                  die "$0: tag name: $self->{tag_name}";            } else {
2479                }              die "$0: tag name: $self->{tag_name}";
2480                !!!nack ('t127.1');            }
2481                !!!next-token;            !!!nack ('t127.1');
2482                next B;            !!!next-token;
2483              } else {            next B;
2484                !!!cp ('t128');          } else {
2485                #            !!!cp ('t128');
2486              }            #
2487            }
2488    
2489              if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {              if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
2490                !!!cp ('t129');                !!!cp ('t129');
# Line 4957  sub _tree_construction_main ($) { Line 2508  sub _tree_construction_main ($) {
2508                !!!cp ('t131');                !!!cp ('t131');
2509              }              }
2510    
2511              ## "after head" insertion mode          ## "after head" insertion mode
2512              ## As if <body>          ## As if <body>
2513              !!!insert-element ('body',, $token);          !!!insert-element ('body',, $token);
2514              $self->{insertion_mode} = IN_BODY_IM;          $self->{insertion_mode} = IN_BODY_IM;
2515              ## reprocess          ## The "frameset-ok" flag is not changed in this case.
2516              !!!ack-later;          ## Reprocess the token.
2517              next B;          !!!ack-later;
2518            } elsif ($token->{type} == END_TAG_TOKEN) {          next B;
2519              if ($token->{tag_name} eq 'head') {        } elsif ($token->{type} == END_TAG_TOKEN) {
2520                if ($self->{insertion_mode} == BEFORE_HEAD_IM) {          ## "Before head", "in head", and "after head" insertion modes
2521                  !!!cp ('t132');          ## ignore most of end tags.  Exceptions are "body", "html",
2522                  ## As if <head>          ## and "br" end tags.  "Before head" and "in head" insertion
2523                  !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);          ## modes also recognize "head" end tag.  "In head noscript"
2524                  $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});          ## insertion modes ignore end tags except for "noscript" and
2525                  push @{$self->{open_elements}},          ## "br".
                     [$self->{head_element}, $el_category->{head}];  
2526    
2527                  ## Reprocess in the "in head" insertion mode...          if ($token->{tag_name} eq 'head') {
2528                  pop @{$self->{open_elements}};            if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
2529                  $self->{insertion_mode} = AFTER_HEAD_IM;              !!!cp ('t132');
2530                  !!!next-token;              ## As if <head>
2531                  next B;              !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);
2532                } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {              $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
2533                  !!!cp ('t133');              push @{$self->{open_elements}},
2534                  ## As if </noscript>                  [$self->{head_element}, $el_category->{head}];
2535                  pop @{$self->{open_elements}};  
2536                  !!!parse-error (type => 'in noscript:/',              ## Reprocess in the "in head" insertion mode...
2537                                  text => 'head', token => $token);              pop @{$self->{open_elements}};
2538                                $self->{insertion_mode} = AFTER_HEAD_IM;
2539                  ## Reprocess in the "in head" insertion mode...              !!!next-token;
2540                  pop @{$self->{open_elements}};              next B;
2541                  $self->{insertion_mode} = AFTER_HEAD_IM;            } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
2542                  !!!next-token;              !!!cp ('t133');
2543                  next B;              #
2544                } elsif ($self->{insertion_mode} == IN_HEAD_IM) {            } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
2545                  !!!cp ('t134');              !!!cp ('t134');
2546                  pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
2547                  $self->{insertion_mode} = AFTER_HEAD_IM;              $self->{insertion_mode} = AFTER_HEAD_IM;
2548                  !!!next-token;              !!!next-token;
2549                  next B;              next B;
2550                } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {            } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
2551                  !!!cp ('t134.1');              !!!cp ('t134.1');
2552                  !!!parse-error (type => 'unmatched end tag', text => 'head',              #
2553                                  token => $token);            } else {
2554                  ## Ignore the token              die "$0: $self->{insertion_mode}: Unknown insertion mode";
2555                  !!!next-token;            }
2556                  next B;          } elsif ($token->{tag_name} eq 'noscript') {
2557                } else {            if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
2558                  die "$0: $self->{insertion_mode}: Unknown insertion mode";              !!!cp ('t136');
2559                }              pop @{$self->{open_elements}};
2560              } elsif ($token->{tag_name} eq 'noscript') {              $self->{insertion_mode} = IN_HEAD_IM;
2561                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {              !!!next-token;
2562                  !!!cp ('t136');              next B;
2563                  pop @{$self->{open_elements}};            } else {
2564                  $self->{insertion_mode} = IN_HEAD_IM;              !!!cp ('t138');
2565                  !!!next-token;              #
2566                  next B;            }
2567                } elsif ($self->{insertion_mode} == BEFORE_HEAD_IM or          } elsif ({
2568                         $self->{insertion_mode} == AFTER_HEAD_IM) {              body => ($self->{insertion_mode} != IN_HEAD_NOSCRIPT_IM),
2569                  !!!cp ('t137');              html => ($self->{insertion_mode} != IN_HEAD_NOSCRIPT_IM),
2570                  !!!parse-error (type => 'unmatched end tag',              br => 1,
2571                                  text => 'noscript', token => $token);          }->{$token->{tag_name}}) {
2572                  ## Ignore the token ## ISSUE: An issue in the spec.            if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
2573                  !!!next-token;              !!!cp ('t142.2');
2574                  next B;              ## (before head) as if <head>, (in head) as if </head>
2575                } else {              !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);
2576                  !!!cp ('t138');              $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
2577                  #              $self->{insertion_mode} = AFTER_HEAD_IM;
               }  
             } elsif ({  
                       body => 1, html => 1,  
                      }->{$token->{tag_name}}) {  
               ## TODO: This branch is entirely redundant.  
               if ($self->{insertion_mode} == BEFORE_HEAD_IM or  
                   $self->{insertion_mode} == IN_HEAD_IM or  
                   $self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {  
                 !!!cp ('t140');  
                 !!!parse-error (type => 'unmatched end tag',  
                                 text => $token->{tag_name}, token => $token);  
                 ## Ignore the token  
                 !!!next-token;  
                 next B;  
               } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {  
                 !!!cp ('t140.1');  
                 !!!parse-error (type => 'unmatched end tag',  
                                 text => $token->{tag_name}, token => $token);  
                 ## Ignore the token  
                 !!!next-token;  
                 next B;  
               } else {  
                 die "$0: $self->{insertion_mode}: Unknown insertion mode";  
               }  
             } elsif ($token->{tag_name} eq 'p') {  
               !!!cp ('t142');  
               !!!parse-error (type => 'unmatched end tag',  
                               text => $token->{tag_name}, token => $token);  
               ## Ignore the token  
               !!!next-token;  
               next B;  
             } elsif ($token->{tag_name} eq 'br') {  
               if ($self->{insertion_mode} == BEFORE_HEAD_IM) {  
                 !!!cp ('t142.2');  
                 ## (before head) as if <head>, (in head) as if </head>  
                 !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);  
                 $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});  
                 $self->{insertion_mode} = AFTER_HEAD_IM;  
2578        
2579                  ## Reprocess in the "after head" insertion mode...              ## Reprocess in the "after head" insertion mode...
2580                } elsif ($self->{insertion_mode} == IN_HEAD_IM) {            } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
2581                  !!!cp ('t143.2');              !!!cp ('t143.2');
2582                  ## As if </head>              ## As if </head>
2583                  pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
2584                  $self->{insertion_mode} = AFTER_HEAD_IM;              $self->{insertion_mode} = AFTER_HEAD_IM;
2585        
2586                  ## Reprocess in the "after head" insertion mode...              ## Reprocess in the "after head" insertion mode...
2587                } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {            } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
2588                  !!!cp ('t143.3');              !!!cp ('t143.3');
2589                  ## ISSUE: Two parse errors for <head><noscript></br>              ## NOTE: Two parse errors for <head><noscript></br>
2590                  !!!parse-error (type => 'unmatched end tag',              !!!parse-error (type => 'unmatched end tag',
2591                                  text => 'br', token => $token);                              text => $token->{tag_name}, token => $token);
2592                  ## As if </noscript>              ## As if </noscript>
2593                  pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
2594                  $self->{insertion_mode} = IN_HEAD_IM;              $self->{insertion_mode} = IN_HEAD_IM;
   
                 ## Reprocess in the "in head" insertion mode...  
                 ## As if </head>  
                 pop @{$self->{open_elements}};  
                 $self->{insertion_mode} = AFTER_HEAD_IM;  
   
                 ## Reprocess in the "after head" insertion mode...  
               } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {  
                 !!!cp ('t143.4');  
                 #  
               } else {  
                 die "$0: $self->{insertion_mode}: Unknown insertion mode";  
               }  
   
               ## ISSUE: does not agree with IE7 - it doesn't ignore </br>.  
               !!!parse-error (type => 'unmatched end tag',  
                               text => 'br', token => $token);  
               ## Ignore the token  
               !!!next-token;  
               next B;  
             } else {  
               !!!cp ('t145');  
               !!!parse-error (type => 'unmatched end tag',  
                               text => $token->{tag_name}, token => $token);  
               ## Ignore the token  
               !!!next-token;  
               next B;  
             }  
2595    
2596              if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {              ## Reprocess in the "in head" insertion mode...
2597                !!!cp ('t146');              ## As if </head>
2598                ## As if </noscript>              pop @{$self->{open_elements}};
2599                pop @{$self->{open_elements}};              $self->{insertion_mode} = AFTER_HEAD_IM;
               !!!parse-error (type => 'in noscript:/',  
                               text => $token->{tag_name}, token => $token);  
                 
               ## Reprocess in the "in head" insertion mode...  
               ## As if </head>  
               pop @{$self->{open_elements}};  
2600    
2601                ## Reprocess in the "after head" insertion mode...              ## Reprocess in the "after head" insertion mode...
2602              } elsif ($self->{insertion_mode} == IN_HEAD_IM) {            } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
2603                !!!cp ('t147');              !!!cp ('t143.4');
2604                ## As if </head>              #
2605                pop @{$self->{open_elements}};            } else {
2606                die "$0: $self->{insertion_mode}: Unknown insertion mode";
2607              }
2608    
2609                ## Reprocess in the "after head" insertion mode...            ## "after head" insertion mode
2610              } elsif ($self->{insertion_mode} == BEFORE_HEAD_IM) {            ## As if <body>
2611  ## ISSUE: This case cannot be reached?            !!!insert-element ('body',, $token);
2612                !!!cp ('t148');            $self->{insertion_mode} = IN_BODY_IM;
2613                !!!parse-error (type => 'unmatched end tag',            ## The "frameset-ok" flag is left unchanged in this case.
2614                                text => $token->{tag_name}, token => $token);            ## Reprocess the token.
2615                ## Ignore the token ## ISSUE: An issue in the spec.            next B;
2616                !!!next-token;          }
               next B;  
             } else {  
               !!!cp ('t149');  
             }  
2617    
2618              ## "after head" insertion mode          ## End tags are ignored by default.
2619              ## As if <body>          !!!cp ('t145');
2620              !!!insert-element ('body',, $token);          !!!parse-error (type => 'unmatched end tag',
2621              $self->{insertion_mode} = IN_BODY_IM;                          text => $token->{tag_name}, token => $token);
2622              ## reprocess          ## Ignore the token.
2623              next B;          !!!next-token;
2624            next B;
2625        } elsif ($token->{type} == END_OF_FILE_TOKEN) {        } elsif ($token->{type} == END_OF_FILE_TOKEN) {
2626          if ($self->{insertion_mode} == BEFORE_HEAD_IM) {          if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
2627            !!!cp ('t149.1');            !!!cp ('t149.1');
# Line 5197  sub _tree_construction_main ($) { Line 2674  sub _tree_construction_main ($) {
2674          ## NOTE: As if <body>          ## NOTE: As if <body>
2675          !!!insert-element ('body',, $token);          !!!insert-element ('body',, $token);
2676          $self->{insertion_mode} = IN_BODY_IM;          $self->{insertion_mode} = IN_BODY_IM;
2677          ## NOTE: Reprocess.          ## The "frameset-ok" flag is left unchanged in this case.
2678            ## Reprocess the token.
2679          next B;          next B;
2680        } else {        } else {
2681          die "$0: $token->{type}: Unknown token type";          die "$0: $token->{type}: Unknown token type";
2682        }        }
2683      } elsif ($self->{insertion_mode} & BODY_IMS) {      } elsif ($self->{insertion_mode} & BODY_IMS) {
2684            if ($token->{type} == CHARACTER_TOKEN) {        if ($token->{type} == CHARACTER_TOKEN) {
2685              !!!cp ('t150');          !!!cp ('t150');
2686              ## NOTE: There is a code clone of "character in body".          $reconstruct_active_formatting_elements->($insert_to_current);
2687              $reconstruct_active_formatting_elements->($insert_to_current);          
2688                        $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});
             $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});  
2689    
2690              !!!next-token;          if ($token->{data} =~ /[^\x09\x0A\x0C\x0D\x20]/) {
2691              next B;            delete $self->{frameset_ok};
2692            } elsif ($token->{type} == START_TAG_TOKEN) {          }
2693    
2694            !!!next-token;
2695            next B;
2696          } elsif ($token->{type} == START_TAG_TOKEN) {
2697              if ({              if ({
2698                   caption => 1, col => 1, colgroup => 1, tbody => 1,                   caption => 1, col => 1, colgroup => 1, tbody => 1,
2699                   td => 1, tfoot => 1, th => 1, thead => 1, tr => 1,                   td => 1, tfoot => 1, th => 1, thead => 1, tr => 1,
2700                  }->{$token->{tag_name}}) {                  }->{$token->{tag_name}}) {
2701                if ($self->{insertion_mode} == IN_CELL_IM) {                if (($self->{insertion_mode} & IM_MASK) == IN_CELL_IM) {
2702                  ## have an element in table scope                  ## have an element in table scope
2703                  for (reverse 0..$#{$self->{open_elements}}) {                  for (reverse 0..$#{$self->{open_elements}}) {
2704                    my $node = $self->{open_elements}->[$_];                    my $node = $self->{open_elements}->[$_];
2705                    if ($node->[1] & TABLE_CELL_EL) {                    if ($node->[1] == TABLE_CELL_EL) {
2706                      !!!cp ('t151');                      !!!cp ('t151');
2707    
2708                      ## Close the cell                      ## Close the cell
# Line 5245  sub _tree_construction_main ($) { Line 2726  sub _tree_construction_main ($) {
2726                  !!!nack ('t153.1');                  !!!nack ('t153.1');
2727                  !!!next-token;                  !!!next-token;
2728                  next B;                  next B;
2729                } elsif ($self->{insertion_mode} == IN_CAPTION_IM) {                } elsif (($self->{insertion_mode} & IM_MASK) == IN_CAPTION_IM) {
2730                  !!!parse-error (type => 'not closed', text => 'caption',                  !!!parse-error (type => 'not closed', text => 'caption',
2731                                  token => $token);                                  token => $token);
2732                                    
# Line 5255  sub _tree_construction_main ($) { Line 2736  sub _tree_construction_main ($) {
2736                  INSCOPE: {                  INSCOPE: {
2737                    for (reverse 0..$#{$self->{open_elements}}) {                    for (reverse 0..$#{$self->{open_elements}}) {
2738                      my $node = $self->{open_elements}->[$_];                      my $node = $self->{open_elements}->[$_];
2739                      if ($node->[1] & CAPTION_EL) {                      if ($node->[1] == CAPTION_EL) {
2740                        !!!cp ('t155');                        !!!cp ('t155');
2741                        $i = $_;                        $i = $_;
2742                        last INSCOPE;                        last INSCOPE;
# Line 5281  sub _tree_construction_main ($) { Line 2762  sub _tree_construction_main ($) {
2762                    pop @{$self->{open_elements}};                    pop @{$self->{open_elements}};
2763                  }                  }
2764    
2765                  unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {                  unless ($self->{open_elements}->[-1]->[1] == CAPTION_EL) {
2766                    !!!cp ('t159');                    !!!cp ('t159');
2767                    !!!parse-error (type => 'not closed',                    !!!parse-error (type => 'not closed',
2768                                    text => $self->{open_elements}->[-1]->[0]                                    text => $self->{open_elements}->[-1]->[0]
# Line 5310  sub _tree_construction_main ($) { Line 2791  sub _tree_construction_main ($) {
2791              }              }
2792            } elsif ($token->{type} == END_TAG_TOKEN) {            } elsif ($token->{type} == END_TAG_TOKEN) {
2793              if ($token->{tag_name} eq 'td' or $token->{tag_name} eq 'th') {              if ($token->{tag_name} eq 'td' or $token->{tag_name} eq 'th') {
2794                if ($self->{insertion_mode} == IN_CELL_IM) {                if (($self->{insertion_mode} & IM_MASK) == IN_CELL_IM) {
2795                  ## have an element in table scope                  ## have an element in table scope
2796                  my $i;                  my $i;
2797                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
# Line 5360  sub _tree_construction_main ($) { Line 2841  sub _tree_construction_main ($) {
2841                                    
2842                  !!!next-token;                  !!!next-token;
2843                  next B;                  next B;
2844                } elsif ($self->{insertion_mode} == IN_CAPTION_IM) {                } elsif (($self->{insertion_mode} & IM_MASK) == IN_CAPTION_IM) {
2845                  !!!cp ('t169');                  !!!cp ('t169');
2846                  !!!parse-error (type => 'unmatched end tag',                  !!!parse-error (type => 'unmatched end tag',
2847                                  text => $token->{tag_name}, token => $token);                                  text => $token->{tag_name}, token => $token);
# Line 5372  sub _tree_construction_main ($) { Line 2853  sub _tree_construction_main ($) {
2853                  #                  #
2854                }                }
2855              } elsif ($token->{tag_name} eq 'caption') {              } elsif ($token->{tag_name} eq 'caption') {
2856                if ($self->{insertion_mode} == IN_CAPTION_IM) {                if (($self->{insertion_mode} & IM_MASK) == IN_CAPTION_IM) {
2857                  ## have a table element in table scope                  ## have a table element in table scope
2858                  my $i;                  my $i;
2859                  INSCOPE: {                  INSCOPE: {
2860                    for (reverse 0..$#{$self->{open_elements}}) {                    for (reverse 0..$#{$self->{open_elements}}) {
2861                      my $node = $self->{open_elements}->[$_];                      my $node = $self->{open_elements}->[$_];
2862                      if ($node->[1] & CAPTION_EL) {                      if ($node->[1] == CAPTION_EL) {
2863                        !!!cp ('t171');                        !!!cp ('t171');
2864                        $i = $_;                        $i = $_;
2865                        last INSCOPE;                        last INSCOPE;
# Line 5403  sub _tree_construction_main ($) { Line 2884  sub _tree_construction_main ($) {
2884                    pop @{$self->{open_elements}};                    pop @{$self->{open_elements}};
2885                  }                  }
2886                                    
2887                  unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {                  unless ($self->{open_elements}->[-1]->[1] == CAPTION_EL) {
2888                    !!!cp ('t175');                    !!!cp ('t175');
2889                    !!!parse-error (type => 'not closed',                    !!!parse-error (type => 'not closed',
2890                                    text => $self->{open_elements}->[-1]->[0]                                    text => $self->{open_elements}->[-1]->[0]
# Line 5421  sub _tree_construction_main ($) { Line 2902  sub _tree_construction_main ($) {
2902                                    
2903                  !!!next-token;                  !!!next-token;
2904                  next B;                  next B;
2905                } elsif ($self->{insertion_mode} == IN_CELL_IM) {                } elsif (($self->{insertion_mode} & IM_MASK) == IN_CELL_IM) {
2906                  !!!cp ('t177');                  !!!cp ('t177');
2907                  !!!parse-error (type => 'unmatched end tag',                  !!!parse-error (type => 'unmatched end tag',
2908                                  text => $token->{tag_name}, token => $token);                                  text => $token->{tag_name}, token => $token);
# Line 5436  sub _tree_construction_main ($) { Line 2917  sub _tree_construction_main ($) {
2917                        table => 1, tbody => 1, tfoot => 1,                        table => 1, tbody => 1, tfoot => 1,
2918                        thead => 1, tr => 1,                        thead => 1, tr => 1,
2919                       }->{$token->{tag_name}} and                       }->{$token->{tag_name}} and
2920                       $self->{insertion_mode} == IN_CELL_IM) {                       ($self->{insertion_mode} & IM_MASK) == IN_CELL_IM) {
2921                ## have an element in table scope                ## have an element in table scope
2922                my $i;                my $i;
2923                my $tn;                my $tn;
# Line 5453  sub _tree_construction_main ($) { Line 2934  sub _tree_construction_main ($) {
2934                                line => $token->{line},                                line => $token->{line},
2935                                column => $token->{column}};                                column => $token->{column}};
2936                      next B;                      next B;
2937                    } elsif ($node->[1] & TABLE_CELL_EL) {                    } elsif ($node->[1] == TABLE_CELL_EL) {
2938                      !!!cp ('t180');                      !!!cp ('t180');
2939                      $tn = $node->[0]->manakai_local_name;                      $tn = $node->[0]->manakai_local_name;
2940                      ## NOTE: There is exactly one |td| or |th| element                      ## NOTE: There is exactly one |td| or |th| element
# Line 5473  sub _tree_construction_main ($) { Line 2954  sub _tree_construction_main ($) {
2954                  next B;                  next B;
2955                } # INSCOPE                } # INSCOPE
2956              } elsif ($token->{tag_name} eq 'table' and              } elsif ($token->{tag_name} eq 'table' and
2957                       $self->{insertion_mode} == IN_CAPTION_IM) {                       ($self->{insertion_mode} & IM_MASK) == IN_CAPTION_IM) {
2958                !!!parse-error (type => 'not closed', text => 'caption',                !!!parse-error (type => 'not closed', text => 'caption',
2959                                token => $token);                                token => $token);
2960    
# Line 5482  sub _tree_construction_main ($) { Line 2963  sub _tree_construction_main ($) {
2963                my $i;                my $i;
2964                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
2965                  my $node = $self->{open_elements}->[$_];                  my $node = $self->{open_elements}->[$_];
2966                  if ($node->[1] & CAPTION_EL) {                  if ($node->[1] == CAPTION_EL) {
2967                    !!!cp ('t184');                    !!!cp ('t184');
2968                    $i = $_;                    $i = $_;
2969                    last INSCOPE;                    last INSCOPE;
# Line 5493  sub _tree_construction_main ($) { Line 2974  sub _tree_construction_main ($) {
2974                } # INSCOPE                } # INSCOPE
2975                unless (defined $i) {                unless (defined $i) {
2976                  !!!cp ('t186');                  !!!cp ('t186');
2977            ## TODO: Wrong error type?
2978                  !!!parse-error (type => 'unmatched end tag',                  !!!parse-error (type => 'unmatched end tag',
2979                                  text => 'caption', token => $token);                                  text => 'caption', token => $token);
2980                  ## Ignore the token                  ## Ignore the token
# Line 5506  sub _tree_construction_main ($) { Line 2988  sub _tree_construction_main ($) {
2988                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
2989                }                }
2990    
2991                unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {                unless ($self->{open_elements}->[-1]->[1] == CAPTION_EL) {
2992                  !!!cp ('t188');                  !!!cp ('t188');
2993                  !!!parse-error (type => 'not closed',                  !!!parse-error (type => 'not closed',
2994                                  text => $self->{open_elements}->[-1]->[0]                                  text => $self->{open_elements}->[-1]->[0]
# Line 5538  sub _tree_construction_main ($) { Line 3020  sub _tree_construction_main ($) {
3020                  !!!cp ('t191');                  !!!cp ('t191');
3021                  #                  #
3022                }                }
3023              } elsif ({          } elsif ({
3024                        tbody => 1, tfoot => 1,                    tbody => 1, tfoot => 1,
3025                        thead => 1, tr => 1,                    thead => 1, tr => 1,
3026                       }->{$token->{tag_name}} and                   }->{$token->{tag_name}} and
3027                       $self->{insertion_mode} == IN_CAPTION_IM) {                   ($self->{insertion_mode} & IM_MASK) == IN_CAPTION_IM) {
3028                !!!cp ('t192');            !!!cp ('t192');
3029                !!!parse-error (type => 'unmatched end tag',            !!!parse-error (type => 'unmatched end tag',
3030                                text => $token->{tag_name}, token => $token);                            text => $token->{tag_name}, token => $token);
3031                ## Ignore the token            ## Ignore the token
3032                !!!next-token;            !!!next-token;
3033                next B;            next B;
3034              } else {          } else {
3035                !!!cp ('t193');            !!!cp ('t193');
3036                #            #
3037              }          }
3038        } elsif ($token->{type} == END_OF_FILE_TOKEN) {        } elsif ($token->{type} == END_OF_FILE_TOKEN) {
3039          for my $entry (@{$self->{open_elements}}) {          for my $entry (@{$self->{open_elements}}) {
3040            unless ($entry->[1] & ALL_END_TAG_OPTIONAL_EL) {            unless ($entry->[1] & ALL_END_TAG_OPTIONAL_EL) {
# Line 5571  sub _tree_construction_main ($) { Line 3053  sub _tree_construction_main ($) {
3053        $insert = $insert_to_current;        $insert = $insert_to_current;
3054        #        #
3055      } elsif ($self->{insertion_mode} & TABLE_IMS) {      } elsif ($self->{insertion_mode} & TABLE_IMS) {
3056        if ($token->{type} == CHARACTER_TOKEN) {        if ($token->{type} == START_TAG_TOKEN) {
         if (not $open_tables->[-1]->[1] and # tainted  
             $token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {  
           $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);  
                 
           unless (length $token->{data}) {  
             !!!cp ('t194');  
             !!!next-token;  
             next B;  
           } else {  
             !!!cp ('t195');  
           }  
         }  
   
         !!!parse-error (type => 'in table:#text', token => $token);  
   
         ## NOTE: As if in body, but insert into the foster parent element.  
         $reconstruct_active_formatting_elements->($insert_to_foster);  
               
         if ($self->{open_elements}->[-1]->[1] & TABLE_ROWS_EL) {  
           # MUST  
           my $foster_parent_element;  
           my $next_sibling;  
           my $prev_sibling;  
           OE: for (reverse 0..$#{$self->{open_elements}}) {  
             if ($self->{open_elements}->[$_]->[1] & TABLE_EL) {  
               my $parent = $self->{open_elements}->[$_]->[0]->parent_node;  
               if (defined $parent and $parent->node_type == 1) {  
                 $foster_parent_element = $parent;  
                 !!!cp ('t196');  
                 $next_sibling = $self->{open_elements}->[$_]->[0];  
                 $prev_sibling = $next_sibling->previous_sibling;  
                 #  
               } else {  
                 !!!cp ('t197');  
                 $foster_parent_element = $self->{open_elements}->[$_ - 1]->[0];  
                 $prev_sibling = $foster_parent_element->last_child;  
                 #  
               }  
               last OE;  
             }  
           } # OE  
           $foster_parent_element = $self->{open_elements}->[0]->[0] and  
           $prev_sibling = $foster_parent_element->last_child  
               unless defined $foster_parent_element;  
           undef $prev_sibling unless $open_tables->[-1]->[2]; # ~node inserted  
           if (defined $prev_sibling and  
               $prev_sibling->node_type == 3) {  
             !!!cp ('t198');  
             $prev_sibling->manakai_append_text ($token->{data});  
           } else {  
             !!!cp ('t199');  
             $foster_parent_element->insert_before  
                 ($self->{document}->create_text_node ($token->{data}),  
                  $next_sibling);  
           }  
           $open_tables->[-1]->[1] = 1; # tainted  
           $open_tables->[-1]->[2] = 1; # ~node inserted  
         } else {  
           ## NOTE: Fragment case or in a foster parent'ed element  
           ## (e.g. |<table><span>a|).  In fragment case, whether the  
           ## character is appended to existing node or a new node is  
           ## created is irrelevant, since the foster parent'ed nodes  
           ## are discarded and fragment parsing does not invoke any  
           ## script.  
           !!!cp ('t200');  
           $self->{open_elements}->[-1]->[0]->manakai_append_text  
               ($token->{data});  
         }  
               
         !!!next-token;  
         next B;  
       } elsif ($token->{type} == START_TAG_TOKEN) {  
3057          if ({          if ({
3058               tr => ($self->{insertion_mode} != IN_ROW_IM),               tr => (($self->{insertion_mode} & IM_MASK) != IN_ROW_IM),
3059               th => 1, td => 1,               th => 1, td => 1,
3060              }->{$token->{tag_name}}) {              }->{$token->{tag_name}}) {
3061            if ($self->{insertion_mode} == IN_TABLE_IM) {            if (($self->{insertion_mode} & IM_MASK) == IN_TABLE_IM) {
3062              ## Clear back to table context              ## Clear back to table context
3063              while (not ($self->{open_elements}->[-1]->[1]              while (not ($self->{open_elements}->[-1]->[1]
3064                              & TABLE_SCOPING_EL)) {                              & TABLE_SCOPING_EL)) {
# Line 5661  sub _tree_construction_main ($) { Line 3071  sub _tree_construction_main ($) {
3071              ## reprocess in the "in table body" insertion mode...              ## reprocess in the "in table body" insertion mode...
3072            }            }
3073                        
3074            if ($self->{insertion_mode} == IN_TABLE_BODY_IM) {            if (($self->{insertion_mode} & IM_MASK) == IN_TABLE_BODY_IM) {
3075              unless ($token->{tag_name} eq 'tr') {              unless ($token->{tag_name} eq 'tr') {
3076                !!!cp ('t202');                !!!cp ('t202');
3077                !!!parse-error (type => 'missing start tag:tr', token => $token);                !!!parse-error (type => 'missing start tag:tr', token => $token);
# Line 5713  sub _tree_construction_main ($) { Line 3123  sub _tree_construction_main ($) {
3123                    tbody => 1, tfoot => 1, thead => 1,                    tbody => 1, tfoot => 1, thead => 1,
3124                    tr => 1, # $self->{insertion_mode} == IN_ROW_IM                    tr => 1, # $self->{insertion_mode} == IN_ROW_IM
3125                   }->{$token->{tag_name}}) {                   }->{$token->{tag_name}}) {
3126            if ($self->{insertion_mode} == IN_ROW_IM) {            if (($self->{insertion_mode} & IM_MASK) == IN_ROW_IM) {
3127              ## As if </tr>              ## As if </tr>
3128              ## have an element in table scope              ## have an element in table scope
3129              my $i;              my $i;
3130              INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {              INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3131                my $node = $self->{open_elements}->[$_];                my $node = $self->{open_elements}->[$_];
3132                if ($node->[1] & TABLE_ROW_EL) {                if ($node->[1] == TABLE_ROW_EL) {
3133                  !!!cp ('t208');                  !!!cp ('t208');
3134                  $i = $_;                  $i = $_;
3135                  last INSCOPE;                  last INSCOPE;
# Line 5760  sub _tree_construction_main ($) { Line 3170  sub _tree_construction_main ($) {
3170                  }                  }
3171                }                }
3172    
3173                if ($self->{insertion_mode} == IN_TABLE_BODY_IM) {                if (($self->{insertion_mode} & IM_MASK) == IN_TABLE_BODY_IM) {
3174                  ## have an element in table scope                  ## have an element in table scope
3175                  my $i;                  my $i;
3176                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3177                    my $node = $self->{open_elements}->[$_];                    my $node = $self->{open_elements}->[$_];
3178                    if ($node->[1] & TABLE_ROW_GROUP_EL) {                    if ($node->[1] == TABLE_ROW_GROUP_EL) {
3179                      !!!cp ('t214');                      !!!cp ('t214');
3180                      $i = $_;                      $i = $_;
3181                      last INSCOPE;                      last INSCOPE;
# Line 5864  sub _tree_construction_main ($) { Line 3274  sub _tree_construction_main ($) {
3274                my $i;                my $i;
3275                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3276                  my $node = $self->{open_elements}->[$_];                  my $node = $self->{open_elements}->[$_];
3277                  if ($node->[1] & TABLE_EL) {                  if ($node->[1] == TABLE_EL) {
3278                    !!!cp ('t221');                    !!!cp ('t221');
3279                    $i = $_;                    $i = $_;
3280                    last INSCOPE;                    last INSCOPE;
# Line 5891  sub _tree_construction_main ($) { Line 3301  sub _tree_construction_main ($) {
3301                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
3302                }                }
3303    
3304                unless ($self->{open_elements}->[-1]->[1] & TABLE_EL) {                unless ($self->{open_elements}->[-1]->[1] == TABLE_EL) {
3305                  !!!cp ('t225');                  !!!cp ('t225');
3306                  ## NOTE: |<table><tr><table>|                  ## NOTE: |<table><tr><table>|
3307                  !!!parse-error (type => 'not closed',                  !!!parse-error (type => 'not closed',
# Line 5911  sub _tree_construction_main ($) { Line 3321  sub _tree_construction_main ($) {
3321            !!!ack-later;            !!!ack-later;
3322            next B;            next B;
3323          } elsif ($token->{tag_name} eq 'style') {          } elsif ($token->{tag_name} eq 'style') {
3324            if (not $open_tables->[-1]->[1]) { # tainted            !!!cp ('t227.8');
3325              !!!cp ('t227.8');            ## NOTE: This is a "as if in head" code clone.
3326              ## NOTE: This is a "as if in head" code clone.            $parse_rcdata->(CDATA_CONTENT_MODEL);
3327              $parse_rcdata->(CDATA_CONTENT_MODEL);            $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
3328              $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted            next B;
             next B;  
           } else {  
             !!!cp ('t227.7');  
             #  
           }  
3329          } elsif ($token->{tag_name} eq 'script') {          } elsif ($token->{tag_name} eq 'script') {
3330            if (not $open_tables->[-1]->[1]) { # tainted            !!!cp ('t227.6');
3331              !!!cp ('t227.6');            ## NOTE: This is a "as if in head" code clone.
3332              ## NOTE: This is a "as if in head" code clone.            $script_start_tag->();
3333              $script_start_tag->();            $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
3334              $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted            next B;
             next B;  
           } else {  
             !!!cp ('t227.5');  
             #  
           }  
3335          } elsif ($token->{tag_name} eq 'input') {          } elsif ($token->{tag_name} eq 'input') {
3336            if (not $open_tables->[-1]->[1]) { # tainted            if ($token->{attributes}->{type}) {
3337              if ($token->{attributes}->{type}) { ## TODO: case              my $type = $token->{attributes}->{type}->{value};
3338                my $type = lc $token->{attributes}->{type}->{value};              $type =~ tr/A-Z/a-z/; ## ASCII case-insensitive.
3339                if ($type eq 'hidden') {              if ($type eq 'hidden') {
3340                  !!!cp ('t227.3');                !!!cp ('t227.3');
3341                  !!!parse-error (type => 'in table',                !!!parse-error (type => 'in table',
3342                                  text => $token->{tag_name}, token => $token);                                text => $token->{tag_name}, token => $token);
3343    
3344                  !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
3345                  $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted                $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
3346    
3347                  ## TODO: form element pointer                ## TODO: form element pointer
3348    
3349                  pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
3350    
3351                  !!!next-token;                !!!next-token;
3352                  !!!ack ('t227.2.1');                !!!ack ('t227.2.1');
3353                  next B;                next B;
               } else {  
                 !!!cp ('t227.2');  
                 #  
               }  
3354              } else {              } else {
3355                !!!cp ('t227.1');                !!!cp ('t227.1');
3356                #                #
# Line 5974  sub _tree_construction_main ($) { Line 3370  sub _tree_construction_main ($) {
3370          $insert = $insert_to_foster;          $insert = $insert_to_foster;
3371          #          #
3372        } elsif ($token->{type} == END_TAG_TOKEN) {        } elsif ($token->{type} == END_TAG_TOKEN) {
3373              if ($token->{tag_name} eq 'tr' and          if ($token->{tag_name} eq 'tr' and
3374                  $self->{insertion_mode} == IN_ROW_IM) {              ($self->{insertion_mode} & IM_MASK) == IN_ROW_IM) {
3375                ## have an element in table scope            ## have an element in table scope
3376                my $i;                my $i;
3377                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3378                  my $node = $self->{open_elements}->[$_];                  my $node = $self->{open_elements}->[$_];
3379                  if ($node->[1] & TABLE_ROW_EL) {                  if ($node->[1] == TABLE_ROW_EL) {
3380                    !!!cp ('t228');                    !!!cp ('t228');
3381                    $i = $_;                    $i = $_;
3382                    last INSCOPE;                    last INSCOPE;
# Line 6015  sub _tree_construction_main ($) { Line 3411  sub _tree_construction_main ($) {
3411                !!!nack ('t231.1');                !!!nack ('t231.1');
3412                next B;                next B;
3413              } elsif ($token->{tag_name} eq 'table') {              } elsif ($token->{tag_name} eq 'table') {
3414                if ($self->{insertion_mode} == IN_ROW_IM) {                if (($self->{insertion_mode} & IM_MASK) == IN_ROW_IM) {
3415                  ## As if </tr>                  ## As if </tr>
3416                  ## have an element in table scope                  ## have an element in table scope
3417                  my $i;                  my $i;
3418                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3419                    my $node = $self->{open_elements}->[$_];                    my $node = $self->{open_elements}->[$_];
3420                    if ($node->[1] & TABLE_ROW_EL) {                    if ($node->[1] == TABLE_ROW_EL) {
3421                      !!!cp ('t233');                      !!!cp ('t233');
3422                      $i = $_;                      $i = $_;
3423                      last INSCOPE;                      last INSCOPE;
# Line 6054  sub _tree_construction_main ($) { Line 3450  sub _tree_construction_main ($) {
3450                  ## reprocess in the "in table body" insertion mode...                  ## reprocess in the "in table body" insertion mode...
3451                }                }
3452    
3453                if ($self->{insertion_mode} == IN_TABLE_BODY_IM) {                if (($self->{insertion_mode} & IM_MASK) == IN_TABLE_BODY_IM) {
3454                  ## have an element in table scope                  ## have an element in table scope
3455                  my $i;                  my $i;
3456                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3457                    my $node = $self->{open_elements}->[$_];                    my $node = $self->{open_elements}->[$_];
3458                    if ($node->[1] & TABLE_ROW_GROUP_EL) {                    if ($node->[1] == TABLE_ROW_GROUP_EL) {
3459                      !!!cp ('t237');                      !!!cp ('t237');
3460                      $i = $_;                      $i = $_;
3461                      last INSCOPE;                      last INSCOPE;
# Line 6106  sub _tree_construction_main ($) { Line 3502  sub _tree_construction_main ($) {
3502                my $i;                my $i;
3503                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3504                  my $node = $self->{open_elements}->[$_];                  my $node = $self->{open_elements}->[$_];
3505                  if ($node->[1] & TABLE_EL) {                  if ($node->[1] == TABLE_EL) {
3506                    !!!cp ('t241');                    !!!cp ('t241');
3507                    $i = $_;                    $i = $_;
3508                    last INSCOPE;                    last INSCOPE;
# Line 6136  sub _tree_construction_main ($) { Line 3532  sub _tree_construction_main ($) {
3532                        tbody => 1, tfoot => 1, thead => 1,                        tbody => 1, tfoot => 1, thead => 1,
3533                       }->{$token->{tag_name}} and                       }->{$token->{tag_name}} and
3534                       $self->{insertion_mode} & ROW_IMS) {                       $self->{insertion_mode} & ROW_IMS) {
3535                if ($self->{insertion_mode} == IN_ROW_IM) {                if (($self->{insertion_mode} & IM_MASK) == IN_ROW_IM) {
3536                  ## have an element in table scope                  ## have an element in table scope
3537                  my $i;                  my $i;
3538                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
# Line 6165  sub _tree_construction_main ($) { Line 3561  sub _tree_construction_main ($) {
3561                  my $i;                  my $i;
3562                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3563                    my $node = $self->{open_elements}->[$_];                    my $node = $self->{open_elements}->[$_];
3564                    if ($node->[1] & TABLE_ROW_EL) {                    if ($node->[1] == TABLE_ROW_EL) {
3565                      !!!cp ('t250');                      !!!cp ('t250');
3566                      $i = $_;                      $i = $_;
3567                      last INSCOPE;                      last INSCOPE;
# Line 6255  sub _tree_construction_main ($) { Line 3651  sub _tree_construction_main ($) {
3651            #            #
3652          }          }
3653        } elsif ($token->{type} == END_OF_FILE_TOKEN) {        } elsif ($token->{type} == END_OF_FILE_TOKEN) {
3654          unless ($self->{open_elements}->[-1]->[1] & HTML_EL and          unless ($self->{open_elements}->[-1]->[1] == HTML_EL and
3655                  @{$self->{open_elements}} == 1) { # redundant, maybe                  @{$self->{open_elements}} == 1) { # redundant, maybe
3656            !!!parse-error (type => 'in body:#eof', token => $token);            !!!parse-error (type => 'in body:#eof', token => $token);
3657            !!!cp ('t259.1');            !!!cp ('t259.1');
# Line 6270  sub _tree_construction_main ($) { Line 3666  sub _tree_construction_main ($) {
3666        } else {        } else {
3667          die "$0: $token->{type}: Unknown token type";          die "$0: $token->{type}: Unknown token type";
3668        }        }
3669      } elsif ($self->{insertion_mode} == IN_COLUMN_GROUP_IM) {      } elsif (($self->{insertion_mode} & IM_MASK) == IN_COLUMN_GROUP_IM) {
3670            if ($token->{type} == CHARACTER_TOKEN) {            if ($token->{type} == CHARACTER_TOKEN) {
3671              if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {              if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
3672                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
# Line 6297  sub _tree_construction_main ($) { Line 3693  sub _tree_construction_main ($) {
3693              }              }
3694            } elsif ($token->{type} == END_TAG_TOKEN) {            } elsif ($token->{type} == END_TAG_TOKEN) {
3695              if ($token->{tag_name} eq 'colgroup') {              if ($token->{tag_name} eq 'colgroup') {
3696                if ($self->{open_elements}->[-1]->[1] & HTML_EL) {                if ($self->{open_elements}->[-1]->[1] == HTML_EL) {
3697                  !!!cp ('t264');                  !!!cp ('t264');
3698                  !!!parse-error (type => 'unmatched end tag',                  !!!parse-error (type => 'unmatched end tag',
3699                                  text => 'colgroup', token => $token);                                  text => 'colgroup', token => $token);
# Line 6323  sub _tree_construction_main ($) { Line 3719  sub _tree_construction_main ($) {
3719                #                #
3720              }              }
3721        } elsif ($token->{type} == END_OF_FILE_TOKEN) {        } elsif ($token->{type} == END_OF_FILE_TOKEN) {
3722          if ($self->{open_elements}->[-1]->[1] & HTML_EL and          if ($self->{open_elements}->[-1]->[1] == HTML_EL and
3723              @{$self->{open_elements}} == 1) { # redundant, maybe              @{$self->{open_elements}} == 1) { # redundant, maybe
3724            !!!cp ('t270.2');            !!!cp ('t270.2');
3725            ## Stop parsing.            ## Stop parsing.
# Line 6341  sub _tree_construction_main ($) { Line 3737  sub _tree_construction_main ($) {
3737        }        }
3738    
3739            ## As if </colgroup>            ## As if </colgroup>
3740            if ($self->{open_elements}->[-1]->[1] & HTML_EL) {            if ($self->{open_elements}->[-1]->[1] == HTML_EL) {
3741              !!!cp ('t269');              !!!cp ('t269');
3742  ## TODO: Wrong error type?  ## TODO: Wrong error type?
3743              !!!parse-error (type => 'unmatched end tag',              !!!parse-error (type => 'unmatched end tag',
# Line 6366  sub _tree_construction_main ($) { Line 3762  sub _tree_construction_main ($) {
3762          next B;          next B;
3763        } elsif ($token->{type} == START_TAG_TOKEN) {        } elsif ($token->{type} == START_TAG_TOKEN) {
3764          if ($token->{tag_name} eq 'option') {          if ($token->{tag_name} eq 'option') {
3765            if ($self->{open_elements}->[-1]->[1] & OPTION_EL) {            if ($self->{open_elements}->[-1]->[1] == OPTION_EL) {
3766              !!!cp ('t272');              !!!cp ('t272');
3767              ## As if </option>              ## As if </option>
3768              pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
# Line 6379  sub _tree_construction_main ($) { Line 3775  sub _tree_construction_main ($) {
3775            !!!next-token;            !!!next-token;
3776            next B;            next B;
3777          } elsif ($token->{tag_name} eq 'optgroup') {          } elsif ($token->{tag_name} eq 'optgroup') {
3778            if ($self->{open_elements}->[-1]->[1] & OPTION_EL) {            if ($self->{open_elements}->[-1]->[1] == OPTION_EL) {
3779              !!!cp ('t274');              !!!cp ('t274');
3780              ## As if </option>              ## As if </option>
3781              pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
# Line 6387  sub _tree_construction_main ($) { Line 3783  sub _tree_construction_main ($) {
3783              !!!cp ('t275');              !!!cp ('t275');
3784            }            }
3785    
3786            if ($self->{open_elements}->[-1]->[1] & OPTGROUP_EL) {            if ($self->{open_elements}->[-1]->[1] == OPTGROUP_EL) {
3787              !!!cp ('t276');              !!!cp ('t276');
3788              ## As if </optgroup>              ## As if </optgroup>
3789              pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
# Line 6400  sub _tree_construction_main ($) { Line 3796  sub _tree_construction_main ($) {
3796            !!!next-token;            !!!next-token;
3797            next B;            next B;
3798          } elsif ({          } elsif ({
3799                     select => 1, input => 1, textarea => 1,                     select => 1, input => 1, textarea => 1, keygen => 1,
3800                   }->{$token->{tag_name}} or                   }->{$token->{tag_name}} or
3801                   ($self->{insertion_mode} == IN_SELECT_IN_TABLE_IM and                   (($self->{insertion_mode} & IM_MASK)
3802                          == IN_SELECT_IN_TABLE_IM and
3803                    {                    {
3804                     caption => 1, table => 1,                     caption => 1, table => 1,
3805                     tbody => 1, tfoot => 1, thead => 1,                     tbody => 1, tfoot => 1, thead => 1,
3806                     tr => 1, td => 1, th => 1,                     tr => 1, td => 1, th => 1,
3807                    }->{$token->{tag_name}})) {                    }->{$token->{tag_name}})) {
3808            ## TODO: The type below is not good - <select> is replaced by </select>  
3809            !!!parse-error (type => 'not closed', text => 'select',            ## 1. Parse error.
3810                            token => $token);            if ($token->{tag_name} eq 'select') {
3811            ## NOTE: As if the token were </select> (<select> case) or                !!!parse-error (type => 'select in select', ## XXX: documentation
3812            ## as if there were </select> (otherwise).                                token => $token);
3813            ## have an element in table scope            } else {
3814                !!!parse-error (type => 'not closed', text => 'select',
3815                                token => $token);
3816              }
3817    
3818              ## 2./<select>-1. Unless "have an element in table scope" (select):
3819            my $i;            my $i;
3820            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3821              my $node = $self->{open_elements}->[$_];              my $node = $self->{open_elements}->[$_];
3822              if ($node->[1] & SELECT_EL) {              if ($node->[1] == SELECT_EL) {
3823                !!!cp ('t278');                !!!cp ('t278');
3824                $i = $_;                $i = $_;
3825                last INSCOPE;                last INSCOPE;
# Line 6428  sub _tree_construction_main ($) { Line 3830  sub _tree_construction_main ($) {
3830            } # INSCOPE            } # INSCOPE
3831            unless (defined $i) {            unless (defined $i) {
3832              !!!cp ('t280');              !!!cp ('t280');
3833              !!!parse-error (type => 'unmatched end tag',              if ($token->{tag_name} eq 'select') {
3834                              text => 'select', token => $token);                ## NOTE: This error would be raised when
3835              ## Ignore the token                ## |select.innerHTML = '<select>'| is executed; in this
3836                  ## case two errors, "select in select" and "unmatched
3837                  ## end tags" are reported to the user, the latter might
3838                  ## be confusing but this is what the spec requires.
3839                  !!!parse-error (type => 'unmatched end tag',
3840                                  text => 'select',
3841                                  token => $token);
3842                }
3843                ## Ignore the token.
3844              !!!nack ('t280.1');              !!!nack ('t280.1');
3845              !!!next-token;              !!!next-token;
3846              next B;              next B;
3847            }            }
3848    
3849              ## 3. Otherwise, as if there were <select>:
3850                                
3851            !!!cp ('t281');            !!!cp ('t281');
3852            splice @{$self->{open_elements}}, $i;            splice @{$self->{open_elements}}, $i;
# Line 6451  sub _tree_construction_main ($) { Line 3863  sub _tree_construction_main ($) {
3863              ## Reprocess the token.              ## Reprocess the token.
3864              next B;              next B;
3865            }            }
3866            } elsif ($token->{tag_name} eq 'script') {
3867              !!!cp ('t281.3');
3868              ## NOTE: This is an "as if in head" code clone
3869              $script_start_tag->();
3870              next B;
3871          } else {          } else {
3872            !!!cp ('t282');            !!!cp ('t282');
3873            !!!parse-error (type => 'in select',            !!!parse-error (type => 'in select',
# Line 6462  sub _tree_construction_main ($) { Line 3879  sub _tree_construction_main ($) {
3879          }          }
3880        } elsif ($token->{type} == END_TAG_TOKEN) {        } elsif ($token->{type} == END_TAG_TOKEN) {
3881          if ($token->{tag_name} eq 'optgroup') {          if ($token->{tag_name} eq 'optgroup') {
3882            if ($self->{open_elements}->[-1]->[1] & OPTION_EL and            if ($self->{open_elements}->[-1]->[1] == OPTION_EL and
3883                $self->{open_elements}->[-2]->[1] & OPTGROUP_EL) {                $self->{open_elements}->[-2]->[1] == OPTGROUP_EL) {
3884              !!!cp ('t283');              !!!cp ('t283');
3885              ## As if </option>              ## As if </option>
3886              splice @{$self->{open_elements}}, -2;              splice @{$self->{open_elements}}, -2;
3887            } elsif ($self->{open_elements}->[-1]->[1] & OPTGROUP_EL) {            } elsif ($self->{open_elements}->[-1]->[1] == OPTGROUP_EL) {
3888              !!!cp ('t284');              !!!cp ('t284');
3889              pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
3890            } else {            } else {
# Line 6480  sub _tree_construction_main ($) { Line 3897  sub _tree_construction_main ($) {
3897            !!!next-token;            !!!next-token;
3898            next B;            next B;
3899          } elsif ($token->{tag_name} eq 'option') {          } elsif ($token->{tag_name} eq 'option') {
3900            if ($self->{open_elements}->[-1]->[1] & OPTION_EL) {            if ($self->{open_elements}->[-1]->[1] == OPTION_EL) {
3901              !!!cp ('t286');              !!!cp ('t286');
3902              pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
3903            } else {            } else {
# Line 6497  sub _tree_construction_main ($) { Line 3914  sub _tree_construction_main ($) {
3914            my $i;            my $i;
3915            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3916              my $node = $self->{open_elements}->[$_];              my $node = $self->{open_elements}->[$_];
3917              if ($node->[1] & SELECT_EL) {              if ($node->[1] == SELECT_EL) {
3918                !!!cp ('t288');                !!!cp ('t288');
3919                $i = $_;                $i = $_;
3920                last INSCOPE;                last INSCOPE;
# Line 6524  sub _tree_construction_main ($) { Line 3941  sub _tree_construction_main ($) {
3941            !!!nack ('t291.1');            !!!nack ('t291.1');
3942            !!!next-token;            !!!next-token;
3943            next B;            next B;
3944          } elsif ($self->{insertion_mode} == IN_SELECT_IN_TABLE_IM and          } elsif (($self->{insertion_mode} & IM_MASK)
3945                         == IN_SELECT_IN_TABLE_IM and
3946                   {                   {
3947                    caption => 1, table => 1, tbody => 1,                    caption => 1, table => 1, tbody => 1,
3948                    tfoot => 1, thead => 1, tr => 1, td => 1, th => 1,                    tfoot => 1, thead => 1, tr => 1, td => 1, th => 1,
# Line 6559  sub _tree_construction_main ($) { Line 3977  sub _tree_construction_main ($) {
3977            undef $i;            undef $i;
3978            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3979              my $node = $self->{open_elements}->[$_];              my $node = $self->{open_elements}->[$_];
3980              if ($node->[1] & SELECT_EL) {              if ($node->[1] == SELECT_EL) {
3981                !!!cp ('t295');                !!!cp ('t295');
3982                $i = $_;                $i = $_;
3983                last INSCOPE;                last INSCOPE;
# Line 6598  sub _tree_construction_main ($) { Line 4016  sub _tree_construction_main ($) {
4016            next B;            next B;
4017          }          }
4018        } elsif ($token->{type} == END_OF_FILE_TOKEN) {        } elsif ($token->{type} == END_OF_FILE_TOKEN) {
4019          unless ($self->{open_elements}->[-1]->[1] & HTML_EL and          unless ($self->{open_elements}->[-1]->[1] == HTML_EL and
4020                  @{$self->{open_elements}} == 1) { # redundant, maybe                  @{$self->{open_elements}} == 1) { # redundant, maybe
4021            !!!cp ('t299.1');            !!!cp ('t299.1');
4022            !!!parse-error (type => 'in body:#eof', token => $token);            !!!parse-error (type => 'in body:#eof', token => $token);
# Line 6785  sub _tree_construction_main ($) { Line 4203  sub _tree_construction_main ($) {
4203        } elsif ($token->{type} == END_TAG_TOKEN) {        } elsif ($token->{type} == END_TAG_TOKEN) {
4204          if ($token->{tag_name} eq 'frameset' and          if ($token->{tag_name} eq 'frameset' and
4205              $self->{insertion_mode} == IN_FRAMESET_IM) {              $self->{insertion_mode} == IN_FRAMESET_IM) {
4206            if ($self->{open_elements}->[-1]->[1] & HTML_EL and            if ($self->{open_elements}->[-1]->[1] == HTML_EL and
4207                @{$self->{open_elements}} == 1) {                @{$self->{open_elements}} == 1) {
4208              !!!cp ('t325');              !!!cp ('t325');
4209              !!!parse-error (type => 'unmatched end tag',              !!!parse-error (type => 'unmatched end tag',
# Line 6799  sub _tree_construction_main ($) { Line 4217  sub _tree_construction_main ($) {
4217            }            }
4218    
4219            if (not defined $self->{inner_html_node} and            if (not defined $self->{inner_html_node} and
4220                not ($self->{open_elements}->[-1]->[1] & FRAMESET_EL)) {                not ($self->{open_elements}->[-1]->[1] == FRAMESET_EL)) {
4221              !!!cp ('t327');              !!!cp ('t327');
4222              $self->{insertion_mode} = AFTER_FRAMESET_IM;              $self->{insertion_mode} = AFTER_FRAMESET_IM;
4223            } else {            } else {
# Line 6831  sub _tree_construction_main ($) { Line 4249  sub _tree_construction_main ($) {
4249            next B;            next B;
4250          }          }
4251        } elsif ($token->{type} == END_OF_FILE_TOKEN) {        } elsif ($token->{type} == END_OF_FILE_TOKEN) {
4252          unless ($self->{open_elements}->[-1]->[1] & HTML_EL and          unless ($self->{open_elements}->[-1]->[1] == HTML_EL and
4253                  @{$self->{open_elements}} == 1) { # redundant, maybe                  @{$self->{open_elements}} == 1) { # redundant, maybe
4254            !!!cp ('t331.1');            !!!cp ('t331.1');
4255            !!!parse-error (type => 'in body:#eof', token => $token);            !!!parse-error (type => 'in body:#eof', token => $token);
# Line 6861  sub _tree_construction_main ($) { Line 4279  sub _tree_construction_main ($) {
4279          $parse_rcdata->(CDATA_CONTENT_MODEL);          $parse_rcdata->(CDATA_CONTENT_MODEL);
4280          next B;          next B;
4281        } elsif ({        } elsif ({
4282                  base => 1, command => 1, eventsource => 1, link => 1,                  base => 1, command => 1, link => 1,
4283                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
4284          !!!cp ('t334');          !!!cp ('t334');
4285          ## NOTE: This is an "as if in head" code clone, only "-t" differs          ## NOTE: This is an "as if in head" code clone, only "-t" differs
# Line 6934  sub _tree_construction_main ($) { Line 4352  sub _tree_construction_main ($) {
4352          !!!parse-error (type => 'in body', text => 'body', token => $token);          !!!parse-error (type => 'in body', text => 'body', token => $token);
4353                                
4354          if (@{$self->{open_elements}} == 1 or          if (@{$self->{open_elements}} == 1 or
4355              not ($self->{open_elements}->[1]->[1] & BODY_EL)) {              not ($self->{open_elements}->[1]->[1] == BODY_EL)) {
4356            !!!cp ('t342');            !!!cp ('t342');
4357            ## Ignore the token            ## Ignore the token
4358          } else {          } else {
# Line 6951  sub _tree_construction_main ($) { Line 4369  sub _tree_construction_main ($) {
4369          !!!nack ('t343.1');          !!!nack ('t343.1');
4370          !!!next-token;          !!!next-token;
4371          next B;          next B;
4372          } elsif ($token->{tag_name} eq 'frameset') {
4373            !!!parse-error (type => 'in body', text => $token->{tag_name},
4374                            token => $token);
4375    
4376            if (@{$self->{open_elements}} == 1 or
4377                not ($self->{open_elements}->[1]->[1] == BODY_EL)) {
4378              !!!cp ('t343.2');
4379              ## Ignore the token.
4380            } elsif (not $self->{frameset_ok}) {
4381              !!!cp ('t343.3');
4382              ## Ignore the token.
4383            } else {
4384              !!!cp ('t343.4');
4385              
4386              ## 1. Remove the second element.
4387              my $body = $self->{open_elements}->[1]->[0];
4388              my $body_parent = $body->parent_node;
4389              $body_parent->remove_child ($body) if $body_parent;
4390    
4391              ## 2. Pop nodes.
4392              splice @{$self->{open_elements}}, 1;
4393    
4394              ## 3. Insert.
4395              !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
4396    
4397              ## 4. Switch.
4398              $self->{insertion_mode} = IN_FRAMESET_IM;
4399            }
4400    
4401            !!!nack ('t343.5');
4402            !!!next-token;
4403            next B;
4404        } elsif ({        } elsif ({
4405                  ## NOTE: Start tags for non-phrasing flow content elements                  ## NOTE: Start tags for non-phrasing flow content elements
4406    
# Line 6959  sub _tree_construction_main ($) { Line 4409  sub _tree_construction_main ($) {
4409                  center => 1, datagrid => 1, details => 1, dialog => 1,                  center => 1, datagrid => 1, details => 1, dialog => 1,
4410                  dir => 1, div => 1, dl => 1, fieldset => 1, figure => 1,                  dir => 1, div => 1, dl => 1, fieldset => 1, figure => 1,
4411                  footer => 1, h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1,                  footer => 1, h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1,
4412                  h6 => 1, header => 1, menu => 1, nav => 1, ol => 1, p => 1,                  h6 => 1, header => 1, hgroup => 1,
4413                    menu => 1, nav => 1, ol => 1, p => 1,
4414                  section => 1, ul => 1,                  section => 1, ul => 1,
4415                  ## NOTE: As normal, but drops leading newline                  ## NOTE: As normal, but drops leading newline
4416                  pre => 1, listing => 1,                  pre => 1, listing => 1,
# Line 6969  sub _tree_construction_main ($) { Line 4420  sub _tree_construction_main ($) {
4420                  table => 1,                  table => 1,
4421                  hr => 1,                  hr => 1,
4422                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
4423    
4424            ## 1. When there is an opening |form| element:
4425          if ($token->{tag_name} eq 'form' and defined $self->{form_element}) {          if ($token->{tag_name} eq 'form' and defined $self->{form_element}) {
4426            !!!cp ('t350');            !!!cp ('t350');
4427            !!!parse-error (type => 'in form:form', token => $token);            !!!parse-error (type => 'in form:form', token => $token);
# Line 6978  sub _tree_construction_main ($) { Line 4431  sub _tree_construction_main ($) {
4431            next B;            next B;
4432          }          }
4433    
4434          ## has a p element in scope          ## 2. Close the |p| element, if any.
4435          INSCOPE: for (reverse @{$self->{open_elements}}) {          if ($token->{tag_name} ne 'table' or # The Hixie Quirk
4436            if ($_->[1] & P_EL) {              $self->{document}->manakai_compat_mode ne 'quirks') {
4437              !!!cp ('t344');            ## has a p element in scope
4438              !!!back-token; # <form>            INSCOPE: for (reverse @{$self->{open_elements}}) {
4439              $token = {type => END_TAG_TOKEN, tag_name => 'p',              if ($_->[1] == P_EL) {
4440                        line => $token->{line}, column => $token->{column}};                !!!cp ('t344');
4441              next B;                !!!back-token; # <form>
4442            } elsif ($_->[1] & SCOPING_EL) {                $token = {type => END_TAG_TOKEN, tag_name => 'p',
4443              !!!cp ('t345');                          line => $token->{line}, column => $token->{column}};
4444              last INSCOPE;                next B;
4445                } elsif ($_->[1] & SCOPING_EL) {
4446                  !!!cp ('t345');
4447                  last INSCOPE;
4448                }
4449              } # INSCOPE
4450            }
4451    
4452            ## 3. Close the opening <hn> element, if any.
4453            if ({h1 => 1, h2 => 1, h3 => 1,
4454                 h4 => 1, h5 => 1, h6 => 1}->{$token->{tag_name}}) {
4455              if ($self->{open_elements}->[-1]->[1] == HEADING_EL) {
4456                !!!parse-error (type => 'not closed',
4457                                text => $self->{open_elements}->[-1]->[0]->manakai_local_name,
4458                                token => $token);
4459                pop @{$self->{open_elements}};
4460            }            }
4461          } # INSCOPE          }
4462              
4463            ## 4. Insertion.
4464          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
4465          if ($token->{tag_name} eq 'pre' or $token->{tag_name} eq 'listing') {          if ($token->{tag_name} eq 'pre' or $token->{tag_name} eq 'listing') {
4466            !!!nack ('t346.1');            !!!nack ('t346.1');
# Line 7007  sub _tree_construction_main ($) { Line 4476  sub _tree_construction_main ($) {
4476            } else {            } else {
4477              !!!cp ('t348');              !!!cp ('t348');
4478            }            }
4479    
4480              delete $self->{frameset_ok};
4481          } elsif ($token->{tag_name} eq 'form') {          } elsif ($token->{tag_name} eq 'form') {
4482            !!!cp ('t347.1');            !!!cp ('t347.1');
4483            $self->{form_element} = $self->{open_elements}->[-1]->[0];            $self->{form_element} = $self->{open_elements}->[-1]->[0];
# Line 7016  sub _tree_construction_main ($) { Line 4487  sub _tree_construction_main ($) {
4487          } elsif ($token->{tag_name} eq 'table') {          } elsif ($token->{tag_name} eq 'table') {
4488            !!!cp ('t382');            !!!cp ('t382');
4489            push @{$open_tables}, [$self->{open_elements}->[-1]->[0]];            push @{$open_tables}, [$self->{open_elements}->[-1]->[0]];
4490    
4491              delete $self->{frameset_ok};
4492                        
4493            $self->{insertion_mode} = IN_TABLE_IM;            $self->{insertion_mode} = IN_TABLE_IM;
4494    
# Line 7024  sub _tree_construction_main ($) { Line 4497  sub _tree_construction_main ($) {
4497          } elsif ($token->{tag_name} eq 'hr') {          } elsif ($token->{tag_name} eq 'hr') {
4498            !!!cp ('t386');            !!!cp ('t386');
4499            pop @{$self->{open_elements}};            pop @{$self->{open_elements}};
4500                      
4501            !!!nack ('t386.1');            !!!ack ('t386.1');
4502    
4503              delete $self->{frameset_ok};
4504    
4505            !!!next-token;            !!!next-token;
4506          } else {          } else {
4507            !!!nack ('t347.1');            !!!nack ('t347.1');
# Line 7035  sub _tree_construction_main ($) { Line 4511  sub _tree_construction_main ($) {
4511        } elsif ($token->{tag_name} eq 'li') {        } elsif ($token->{tag_name} eq 'li') {
4512          ## NOTE: As normal, but imply </li> when there's another <li> ...          ## NOTE: As normal, but imply </li> when there's another <li> ...
4513    
4514          ## NOTE: Special, Scope (<li><foo><li> == <li><foo><li/></foo></li>)          ## NOTE: Special, Scope (<li><foo><li> == <li><foo><li/></foo></li>)::
4515            ## Interpreted as <li><foo/></li><li/> (non-conforming)            ## Interpreted as <li><foo/></li><li/> (non-conforming):
4516            ## blockquote (O9.27), center (O), dd (Fx3, O, S3.1.2, IE7),            ## blockquote (O9.27), center (O), dd (Fx3, O, S3.1.2, IE7),
4517            ## dt (Fx, O, S, IE), dl (O), fieldset (O, S, IE), form (Fx, O, S),            ## dt (Fx, O, S, IE), dl (O), fieldset (O, S, IE), form (Fx, O, S),
4518            ## hn (O), pre (O), applet (O, S), button (O, S), marquee (Fx, O, S),            ## hn (O), pre (O), applet (O, S), button (O, S), marquee (Fx, O, S),
4519            ## object (Fx)            ## object (Fx)
4520            ## Generate non-tree (non-conforming)            ## Generate non-tree (non-conforming):
4521            ## basefont (IE7 (where basefont is non-void)), center (IE),            ## basefont (IE7 (where basefont is non-void)), center (IE),
4522            ## form (IE), hn (IE)            ## form (IE), hn (IE)
4523          ## address, div, p (<li><foo><li> == <li><foo/></li><li/>)          ## address, div, p (<li><foo><li> == <li><foo/></li><li/>)::
4524            ## Interpreted as <li><foo><li/></foo></li> (non-conforming)            ## Interpreted as <li><foo><li/></foo></li> (non-conforming):
4525            ## div (Fx, S)            ## div (Fx, S)
4526    
4527            ## 1. Frameset-ng
4528            delete $self->{frameset_ok};
4529    
4530          my $non_optional;          my $non_optional;
4531          my $i = -1;          my $i = -1;
4532    
4533          ## 1.          ## 2.
4534          for my $node (reverse @{$self->{open_elements}}) {          for my $node (reverse @{$self->{open_elements}}) {
4535            if ($node->[1] & LI_EL) {            if ($node->[1] == LI_EL) {
4536              ## 2. (a) As if </li>              ## 3. (a) As if </li>
4537              {              {
4538                ## If no </li> - not applied                ## If no </li> - not applied
4539                #                #
# Line 7078  sub _tree_construction_main ($) { Line 4557  sub _tree_construction_main ($) {
4557                splice @{$self->{open_elements}}, $i;                splice @{$self->{open_elements}}, $i;
4558              }              }
4559    
4560              last; ## 2. (b) goto 5.              last; ## 3. (b) goto 5.
4561            } elsif (            } elsif (
4562                     ## NOTE: not "formatting" and not "phrasing"                     ## NOTE: not "formatting" and not "phrasing"
4563                     ($node->[1] & SPECIAL_EL or                     ($node->[1] & SPECIAL_EL or
4564                      $node->[1] & SCOPING_EL) and                      $node->[1] & SCOPING_EL) and
4565                     ## NOTE: "li", "dt", and "dd" are in |SPECIAL_EL|.                     ## NOTE: "li", "dt", and "dd" are in |SPECIAL_EL|.
4566                       (not $node->[1] & ADDRESS_DIV_P_EL)
4567                     (not $node->[1] & ADDRESS_EL) &                    ) {
4568                     (not $node->[1] & DIV_EL) &              ## 4.
                    (not $node->[1] & P_EL)) {  
             ## 3.  
4569              !!!cp ('t357');              !!!cp ('t357');
4570              last; ## goto 5.              last; ## goto 6.
4571            } elsif ($node->[1] & END_TAG_OPTIONAL_EL) {            } elsif ($node->[1] & END_TAG_OPTIONAL_EL) {
4572              !!!cp ('t358');              !!!cp ('t358');
4573              #              #
# Line 7099  sub _tree_construction_main ($) { Line 4576  sub _tree_construction_main ($) {
4576              $non_optional ||= $node;              $non_optional ||= $node;
4577              #              #
4578            }            }
4579            ## 4.            ## 5.
4580            ## goto 2.            ## goto 3.
4581            $i--;            $i--;
4582          }          }
4583    
4584          ## 5. (a) has a |p| element in scope          ## 6. (a) has a |p| element in scope
4585          INSCOPE: for (reverse @{$self->{open_elements}}) {          INSCOPE: for (reverse @{$self->{open_elements}}) {
4586            if ($_->[1] & P_EL) {            if ($_->[1] == P_EL) {
4587              !!!cp ('t353');              !!!cp ('t353');
4588    
4589              ## NOTE: |<p><li>|, for example.              ## NOTE: |<p><li>|, for example.
# Line 7121  sub _tree_construction_main ($) { Line 4598  sub _tree_construction_main ($) {
4598            }            }
4599          } # INSCOPE          } # INSCOPE
4600    
4601          ## 5. (b) insert          ## 6. (b) insert
4602          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
4603          !!!nack ('t359.1');          !!!nack ('t359.1');
4604          !!!next-token;          !!!next-token;
# Line 7130  sub _tree_construction_main ($) { Line 4607  sub _tree_construction_main ($) {
4607                 $token->{tag_name} eq 'dd') {                 $token->{tag_name} eq 'dd') {
4608          ## NOTE: As normal, but imply </dt> or </dd> when ...          ## NOTE: As normal, but imply </dt> or </dd> when ...
4609    
4610            ## 1. Frameset-ng
4611            delete $self->{frameset_ok};
4612    
4613          my $non_optional;          my $non_optional;
4614          my $i = -1;          my $i = -1;
4615    
4616          ## 1.          ## 2.
4617          for my $node (reverse @{$self->{open_elements}}) {          for my $node (reverse @{$self->{open_elements}}) {
4618            if ($node->[1] & DT_EL or $node->[1] & DD_EL) {            if ($node->[1] == DTDD_EL) {
4619              ## 2. (a) As if </li>              ## 3. (a) As if </li>
4620              {              {
4621                ## If no </li> - not applied                ## If no </li> - not applied
4622                #                #
# Line 7160  sub _tree_construction_main ($) { Line 4640  sub _tree_construction_main ($) {
4640                splice @{$self->{open_elements}}, $i;                splice @{$self->{open_elements}}, $i;
4641              }              }
4642    
4643              last; ## 2. (b) goto 5.              last; ## 3. (b) goto 5.
4644            } elsif (            } elsif (
4645                     ## NOTE: not "formatting" and not "phrasing"                     ## NOTE: not "formatting" and not "phrasing"
4646                     ($node->[1] & SPECIAL_EL or                     ($node->[1] & SPECIAL_EL or
4647                      $node->[1] & SCOPING_EL) and                      $node->[1] & SCOPING_EL) and
4648                     ## NOTE: "li", "dt", and "dd" are in |SPECIAL_EL|.                     ## NOTE: "li", "dt", and "dd" are in |SPECIAL_EL|.
4649    
4650                     (not $node->[1] & ADDRESS_EL) &                     (not $node->[1] & ADDRESS_DIV_P_EL)
4651                     (not $node->[1] & DIV_EL) &                    ) {
4652                     (not $node->[1] & P_EL)) {              ## 4.
             ## 3.  
4653              !!!cp ('t357.1');              !!!cp ('t357.1');
4654              last; ## goto 5.              last; ## goto 5.
4655            } elsif ($node->[1] & END_TAG_OPTIONAL_EL) {            } elsif ($node->[1] & END_TAG_OPTIONAL_EL) {
# Line 7181  sub _tree_construction_main ($) { Line 4660  sub _tree_construction_main ($) {
4660              $non_optional ||= $node;              $non_optional ||= $node;
4661              #              #
4662            }            }
4663            ## 4.            ## 5.
4664            ## goto 2.            ## goto 3.
4665            $i--;            $i--;
4666          }          }
4667    
4668          ## 5. (a) has a |p| element in scope          ## 6. (a) has a |p| element in scope
4669          INSCOPE: for (reverse @{$self->{open_elements}}) {          INSCOPE: for (reverse @{$self->{open_elements}}) {
4670            if ($_->[1] & P_EL) {            if ($_->[1] == P_EL) {
4671              !!!cp ('t353.1');              !!!cp ('t353.1');
4672              !!!back-token; # <x>              !!!back-token; # <x>
4673              $token = {type => END_TAG_TOKEN, tag_name => 'p',              $token = {type => END_TAG_TOKEN, tag_name => 'p',
# Line 7200  sub _tree_construction_main ($) { Line 4679  sub _tree_construction_main ($) {
4679            }            }
4680          } # INSCOPE          } # INSCOPE
4681    
4682          ## 5. (b) insert          ## 6. (b) insert
4683          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
4684          !!!nack ('t359.2');          !!!nack ('t359.2');
4685          !!!next-token;          !!!next-token;
# Line 7210  sub _tree_construction_main ($) { Line 4689  sub _tree_construction_main ($) {
4689    
4690          ## has a p element in scope          ## has a p element in scope
4691          INSCOPE: for (reverse @{$self->{open_elements}}) {          INSCOPE: for (reverse @{$self->{open_elements}}) {
4692            if ($_->[1] & P_EL) {            if ($_->[1] == P_EL) {
4693              !!!cp ('t367');              !!!cp ('t367');
4694              !!!back-token; # <plaintext>              !!!back-token; # <plaintext>
4695              $token = {type => END_TAG_TOKEN, tag_name => 'p',              $token = {type => END_TAG_TOKEN, tag_name => 'p',
# Line 7232  sub _tree_construction_main ($) { Line 4711  sub _tree_construction_main ($) {
4711        } elsif ($token->{tag_name} eq 'a') {        } elsif ($token->{tag_name} eq 'a') {
4712          AFE: for my $i (reverse 0..$#$active_formatting_elements) {          AFE: for my $i (reverse 0..$#$active_formatting_elements) {
4713            my $node = $active_formatting_elements->[$i];            my $node = $active_formatting_elements->[$i];
4714            if ($node->[1] & A_EL) {            if ($node->[1] == A_EL) {
4715              !!!cp ('t371');              !!!cp ('t371');
4716              !!!parse-error (type => 'in a:a', token => $token);              !!!parse-error (type => 'in a:a', token => $token);
4717                            
# Line 7276  sub _tree_construction_main ($) { Line 4755  sub _tree_construction_main ($) {
4755          ## has a |nobr| element in scope          ## has a |nobr| element in scope
4756          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
4757            my $node = $self->{open_elements}->[$_];            my $node = $self->{open_elements}->[$_];
4758            if ($node->[1] & NOBR_EL) {            if ($node->[1] == NOBR_EL) {
4759              !!!cp ('t376');              !!!cp ('t376');
4760              !!!parse-error (type => 'in nobr:nobr', token => $token);              !!!parse-error (type => 'in nobr:nobr', token => $token);
4761              !!!back-token; # <nobr>              !!!back-token; # <nobr>
# Line 7299  sub _tree_construction_main ($) { Line 4778  sub _tree_construction_main ($) {
4778          ## has a button element in scope          ## has a button element in scope
4779          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
4780            my $node = $self->{open_elements}->[$_];            my $node = $self->{open_elements}->[$_];
4781            if ($node->[1] & BUTTON_EL) {            if ($node->[1] == BUTTON_EL) {
4782              !!!cp ('t378');              !!!cp ('t378');
4783              !!!parse-error (type => 'in button:button', token => $token);              !!!parse-error (type => 'in button:button', token => $token);
4784              !!!back-token; # <button>              !!!back-token; # <button>
# Line 7320  sub _tree_construction_main ($) { Line 4799  sub _tree_construction_main ($) {
4799    
4800          push @$active_formatting_elements, ['#marker', ''];          push @$active_formatting_elements, ['#marker', ''];
4801    
4802            delete $self->{frameset_ok};
4803    
4804          !!!nack ('t379.1');          !!!nack ('t379.1');
4805          !!!next-token;          !!!next-token;
4806          next B;          next B;
# Line 7333  sub _tree_construction_main ($) { Line 4814  sub _tree_construction_main ($) {
4814          if ($token->{tag_name} eq 'xmp') {          if ($token->{tag_name} eq 'xmp') {
4815            !!!cp ('t381');            !!!cp ('t381');
4816            $reconstruct_active_formatting_elements->($insert_to_current);            $reconstruct_active_formatting_elements->($insert_to_current);
4817    
4818              delete $self->{frameset_ok};
4819            } elsif ($token->{tag_name} eq 'iframe') {
4820              !!!cp ('t381.1');
4821              delete $self->{frameset_ok};
4822          } else {          } else {
4823            !!!cp ('t399');            !!!cp ('t399');
4824          }          }
# Line 7364  sub _tree_construction_main ($) { Line 4850  sub _tree_construction_main ($) {
4850                           line => $token->{line}, column => $token->{column}},                           line => $token->{line}, column => $token->{column}},
4851                          {type => START_TAG_TOKEN, tag_name => 'hr',                          {type => START_TAG_TOKEN, tag_name => 'hr',
4852                           line => $token->{line}, column => $token->{column}},                           line => $token->{line}, column => $token->{column}},
                         {type => START_TAG_TOKEN, tag_name => 'p',  
                          line => $token->{line}, column => $token->{column}},  
4853                          {type => START_TAG_TOKEN, tag_name => 'label',                          {type => START_TAG_TOKEN, tag_name => 'label',
4854                           line => $token->{line}, column => $token->{column}},                           line => $token->{line}, column => $token->{column}},
4855                         );                         );
# Line 7388  sub _tree_construction_main ($) { Line 4872  sub _tree_construction_main ($) {
4872                          #{type => CHARACTER_TOKEN, data => ''}, # SHOULD                          #{type => CHARACTER_TOKEN, data => ''}, # SHOULD
4873                          {type => END_TAG_TOKEN, tag_name => 'label',                          {type => END_TAG_TOKEN, tag_name => 'label',
4874                           line => $token->{line}, column => $token->{column}},                           line => $token->{line}, column => $token->{column}},
                         {type => END_TAG_TOKEN, tag_name => 'p',  
                          line => $token->{line}, column => $token->{column}},  
4875                          {type => START_TAG_TOKEN, tag_name => 'hr',                          {type => START_TAG_TOKEN, tag_name => 'hr',
4876                           line => $token->{line}, column => $token->{column}},                           line => $token->{line}, column => $token->{column}},
4877                          {type => END_TAG_TOKEN, tag_name => 'form',                          {type => END_TAG_TOKEN, tag_name => 'form',
# Line 7399  sub _tree_construction_main ($) { Line 4881  sub _tree_construction_main ($) {
4881            next B;            next B;
4882          }          }
4883        } elsif ($token->{tag_name} eq 'textarea') {        } elsif ($token->{tag_name} eq 'textarea') {
4884          my $tag_name = $token->{tag_name};          ## 1. Insert
4885          my $el;          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
         !!!create-element ($el, $HTML_NS, $token->{tag_name}, $token->{attributes}, $token);  
4886                    
4887            ## Step 2 # XXX
4888          ## TODO: $self->{form_element} if defined          ## TODO: $self->{form_element} if defined
4889    
4890            ## 2. Drop U+000A LINE FEED
4891            $self->{ignore_newline} = 1;
4892    
4893            ## 3. RCDATA
4894          $self->{content_model} = RCDATA_CONTENT_MODEL;          $self->{content_model} = RCDATA_CONTENT_MODEL;
4895          delete $self->{escape}; # MUST          delete $self->{escape}; # MUST
4896            
4897          $insert->($el);          ## 4., 6. Insertion mode
4898                    $self->{insertion_mode} |= IN_CDATA_RCDATA_IM;
4899          my $text = '';  
4900            ## 5. Frameset-ng.
4901            delete $self->{frameset_ok};
4902    
4903          !!!nack ('t392.1');          !!!nack ('t392.1');
4904          !!!next-token;          !!!next-token;
         if ($token->{type} == CHARACTER_TOKEN) {  
           $token->{data} =~ s/^\x0A//;  
           unless (length $token->{data}) {  
             !!!cp ('t392');  
             !!!next-token;  
           } else {  
             !!!cp ('t393');  
           }  
         } else {  
           !!!cp ('t394');  
         }  
         while ($token->{type} == CHARACTER_TOKEN) {  
           !!!cp ('t395');  
           $text .= $token->{data};  
           !!!next-token;  
         }  
         if (length $text) {  
           !!!cp ('t396');  
           $el->manakai_append_text ($text);  
         }  
           
         $self->{content_model} = PCDATA_CONTENT_MODEL;  
           
         if ($token->{type} == END_TAG_TOKEN and  
             $token->{tag_name} eq $tag_name) {  
           !!!cp ('t397');  
           ## Ignore the token  
         } else {  
           !!!cp ('t398');  
           !!!parse-error (type => 'in RCDATA:#eof', token => $token);  
         }  
         !!!next-token;  
4905          next B;          next B;
4906        } elsif ($token->{tag_name} eq 'optgroup' or        } elsif ($token->{tag_name} eq 'optgroup' or
4907                 $token->{tag_name} eq 'option') {                 $token->{tag_name} eq 'option') {
4908          ## has an |option| element in scope          ## has an |option| element in scope
4909          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
4910            my $node = $self->{open_elements}->[$_];            my $node = $self->{open_elements}->[$_];
4911            if ($node->[1] & OPTION_EL) {            if ($node->[1] == OPTION_EL) {
4912              !!!cp ('t397.1');              !!!cp ('t397.1');
4913              ## NOTE: As if </option>              ## NOTE: As if </option>
4914              !!!back-token; # <option> or <optgroup>              !!!back-token; # <option> or <optgroup>
# Line 7475  sub _tree_construction_main ($) { Line 4933  sub _tree_construction_main ($) {
4933          ## has a |ruby| element in scope          ## has a |ruby| element in scope
4934          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
4935            my $node = $self->{open_elements}->[$_];            my $node = $self->{open_elements}->[$_];
4936            if ($node->[1] & RUBY_EL) {            if ($node->[1] == RUBY_EL) {
4937              !!!cp ('t398.1');              !!!cp ('t398.1');
4938              ## generate implied end tags              ## generate implied end tags
4939              while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {              while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
4940                !!!cp ('t398.2');                !!!cp ('t398.2');
4941                pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
4942              }              }
4943              unless ($self->{open_elements}->[-1]->[1] & RUBY_EL) {              unless ($self->{open_elements}->[-1]->[1] == RUBY_EL) {
4944                !!!cp ('t398.3');                !!!cp ('t398.3');
4945                !!!parse-error (type => 'not closed',                !!!parse-error (type => 'not closed',
4946                                text => $self->{open_elements}->[-1]->[0]                                text => $self->{open_elements}->[-1]->[0]
4947                                    ->manakai_local_name,                                    ->manakai_local_name,
4948                                token => $token);                                token => $token);
4949                pop @{$self->{open_elements}}                pop @{$self->{open_elements}}
4950                    while not $self->{open_elements}->[-1]->[1] & RUBY_EL;                    while not $self->{open_elements}->[-1]->[1] == RUBY_EL;
4951              }              }
4952              last INSCOPE;              last INSCOPE;
4953            } elsif ($node->[1] & SCOPING_EL) {            } elsif ($node->[1] & SCOPING_EL) {
# Line 7497  sub _tree_construction_main ($) { Line 4955  sub _tree_construction_main ($) {
4955              last INSCOPE;              last INSCOPE;
4956            }            }
4957          } # INSCOPE          } # INSCOPE
4958              
4959            ## TODO: <non-ruby><rt> is not allowed.
4960    
4961          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
4962    
# Line 7530  sub _tree_construction_main ($) { Line 4990  sub _tree_construction_main ($) {
4990          next B;          next B;
4991        } elsif ({        } elsif ({
4992                  caption => 1, col => 1, colgroup => 1, frame => 1,                  caption => 1, col => 1, colgroup => 1, frame => 1,
4993                  frameset => 1, head => 1,                  head => 1,
4994                  tbody => 1, td => 1, tfoot => 1, th => 1,                  tbody => 1, td => 1, tfoot => 1, th => 1,
4995                  thead => 1, tr => 1,                  thead => 1, tr => 1,
4996                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
# Line 7567  sub _tree_construction_main ($) { Line 5027  sub _tree_construction_main ($) {
5027               applet => 1, marquee => 1, object => 1,               applet => 1, marquee => 1, object => 1,
5028              }->{$token->{tag_name}}) {              }->{$token->{tag_name}}) {
5029            !!!cp ('t380');            !!!cp ('t380');
5030    
5031            push @$active_formatting_elements, ['#marker', ''];            push @$active_formatting_elements, ['#marker', ''];
5032    
5033              delete $self->{frameset_ok};
5034    
5035            !!!nack ('t380.1');            !!!nack ('t380.1');
5036          } elsif ({          } elsif ({
5037                    b => 1, big => 1, em => 1, font => 1, i => 1,                    b => 1, big => 1, em => 1, font => 1, i => 1,
# Line 7585  sub _tree_construction_main ($) { Line 5049  sub _tree_construction_main ($) {
5049          } elsif ({          } elsif ({
5050                    area => 1, basefont => 1, bgsound => 1, br => 1,                    area => 1, basefont => 1, bgsound => 1, br => 1,
5051                    embed => 1, img => 1, spacer => 1, wbr => 1,                    embed => 1, img => 1, spacer => 1, wbr => 1,
5052                      keygen => 1,
5053                   }->{$token->{tag_name}}) {                   }->{$token->{tag_name}}) {
5054            !!!cp ('t388.1');            !!!cp ('t388.1');
5055    
5056            pop @{$self->{open_elements}};            pop @{$self->{open_elements}};
5057    
5058              delete $self->{frameset_ok};
5059    
5060            !!!ack ('t388.3');            !!!ack ('t388.3');
5061          } elsif ($token->{tag_name} eq 'select') {          } elsif ($token->{tag_name} eq 'select') {
5062            ## TODO: associate with $self->{form_element} if defined            ## TODO: associate with $self->{form_element} if defined
5063            
5064              delete $self->{frameset_ok};
5065              
5066            if ($self->{insertion_mode} & TABLE_IMS or            if ($self->{insertion_mode} & TABLE_IMS or
5067                $self->{insertion_mode} & BODY_TABLE_IMS or                $self->{insertion_mode} & BODY_TABLE_IMS or
5068                $self->{insertion_mode} == IN_COLUMN_GROUP_IM) {                ($self->{insertion_mode} & IM_MASK) == IN_COLUMN_GROUP_IM) {
5069              !!!cp ('t400.1');              !!!cp ('t400.1');
5070              $self->{insertion_mode} = IN_SELECT_IN_TABLE_IM;              $self->{insertion_mode} = IN_SELECT_IN_TABLE_IM;
5071            } else {            } else {
# Line 7610  sub _tree_construction_main ($) { Line 5081  sub _tree_construction_main ($) {
5081          next B;          next B;
5082        }        }
5083      } elsif ($token->{type} == END_TAG_TOKEN) {      } elsif ($token->{type} == END_TAG_TOKEN) {
5084        if ($token->{tag_name} eq 'body') {        if ($token->{tag_name} eq 'body' or $token->{tag_name} eq 'html') {
5085          ## has a |body| element in scope  
5086            ## 1. If not "have an element in scope":
5087            ## "has a |body| element in scope"
5088          my $i;          my $i;
5089          INSCOPE: {          INSCOPE: {
5090            for (reverse @{$self->{open_elements}}) {            for (reverse @{$self->{open_elements}}) {
5091              if ($_->[1] & BODY_EL) {              if ($_->[1] == BODY_EL) {
5092                !!!cp ('t405');                !!!cp ('t405');
5093                $i = $_;                $i = $_;
5094                last INSCOPE;                last INSCOPE;
# Line 7625  sub _tree_construction_main ($) { Line 5098  sub _tree_construction_main ($) {
5098              }              }
5099            }            }
5100    
5101            ## NOTE: |<marquee></body>|, |<svg><foreignobject></body>|            ## NOTE: |<marquee></body>|, |<svg><foreignobject></body>|,
5102              ## and fragment cases.
5103    
5104            !!!parse-error (type => 'unmatched end tag',            !!!parse-error (type => 'unmatched end tag',
5105                            text => $token->{tag_name}, token => $token);                            text => $token->{tag_name}, token => $token);
5106            ## NOTE: Ignore the token.            ## Ignore the token.  (</body> or </html>)
5107            !!!next-token;            !!!next-token;
5108            next B;            next B;
5109          } # INSCOPE          } # INSCOPE
5110    
5111            ## 2. If unclosed elements:
5112          for (@{$self->{open_elements}}) {          for (@{$self->{open_elements}}) {
5113            unless ($_->[1] & ALL_END_TAG_OPTIONAL_EL) {            unless ($_->[1] & ALL_END_TAG_OPTIONAL_EL ||
5114                      $_->[1] == OPTGROUP_EL ||
5115                      $_->[1] == OPTION_EL ||
5116                      $_->[1] == RUBY_COMPONENT_EL) {
5117              !!!cp ('t403');              !!!cp ('t403');
5118              !!!parse-error (type => 'not closed',              !!!parse-error (type => 'not closed',
5119                              text => $_->[0]->manakai_local_name,                              text => $_->[0]->manakai_local_name,
# Line 7646  sub _tree_construction_main ($) { Line 5124  sub _tree_construction_main ($) {
5124            }            }
5125          }          }
5126    
5127            ## 3. Switch the insertion mode.
5128          $self->{insertion_mode} = AFTER_BODY_IM;          $self->{insertion_mode} = AFTER_BODY_IM;
5129          !!!next-token;          if ($token->{tag_name} eq 'body') {
         next B;  
       } elsif ($token->{tag_name} eq 'html') {  
         ## TODO: Update this code.  It seems that the code below is not  
         ## up-to-date, though it has same effect as speced.  
         if (@{$self->{open_elements}} > 1 and  
             $self->{open_elements}->[1]->[1] & BODY_EL) {  
           unless ($self->{open_elements}->[-1]->[1] & BODY_EL) {  
             !!!cp ('t406');  
             !!!parse-error (type => 'not closed',  
                             text => $self->{open_elements}->[1]->[0]  
                                 ->manakai_local_name,  
                             token => $token);  
           } else {  
             !!!cp ('t407');  
           }  
           $self->{insertion_mode} = AFTER_BODY_IM;  
           ## reprocess  
           next B;  
         } else {  
           !!!cp ('t408');  
           !!!parse-error (type => 'unmatched end tag',  
                           text => $token->{tag_name}, token => $token);  
           ## Ignore the token  
5130            !!!next-token;            !!!next-token;
5131            next B;          } else { # html
5132              ## Reprocess.
5133          }          }
5134            next B;
5135        } elsif ({        } elsif ({
5136                  ## NOTE: End tags for non-phrasing flow content elements                  ## NOTE: End tags for non-phrasing flow content elements
5137    
# Line 7681  sub _tree_construction_main ($) { Line 5139  sub _tree_construction_main ($) {
5139                  address => 1, article => 1, aside => 1, blockquote => 1,                  address => 1, article => 1, aside => 1, blockquote => 1,
5140                  center => 1, datagrid => 1, details => 1, dialog => 1,                  center => 1, datagrid => 1, details => 1, dialog => 1,
5141                  dir => 1, div => 1, dl => 1, fieldset => 1, figure => 1,                  dir => 1, div => 1, dl => 1, fieldset => 1, figure => 1,
5142                  footer => 1, header => 1, listing => 1, menu => 1, nav => 1,                  footer => 1, header => 1, hgroup => 1,
5143                    listing => 1, menu => 1, nav => 1,
5144                  ol => 1, pre => 1, section => 1, ul => 1,                  ol => 1, pre => 1, section => 1, ul => 1,
5145    
5146                  ## NOTE: As normal, but ... optional tags                  ## NOTE: As normal, but ... optional tags
# Line 7761  sub _tree_construction_main ($) { Line 5220  sub _tree_construction_main ($) {
5220          my $i;          my $i;
5221          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5222            my $node = $self->{open_elements}->[$_];            my $node = $self->{open_elements}->[$_];
5223            if ($node->[1] & FORM_EL) {            if ($node->[1] == FORM_EL) {
5224              !!!cp ('t418');              !!!cp ('t418');
5225              $i = $_;              $i = $_;
5226              last INSCOPE;              last INSCOPE;
# Line 7809  sub _tree_construction_main ($) { Line 5268  sub _tree_construction_main ($) {
5268          my $i;          my $i;
5269          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5270            my $node = $self->{open_elements}->[$_];            my $node = $self->{open_elements}->[$_];
5271            if ($node->[1] & HEADING_EL) {            if ($node->[1] == HEADING_EL) {
5272              !!!cp ('t423');              !!!cp ('t423');
5273              $i = $_;              $i = $_;
5274              last INSCOPE;              last INSCOPE;
# Line 7855  sub _tree_construction_main ($) { Line 5314  sub _tree_construction_main ($) {
5314          my $i;          my $i;
5315          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5316            my $node = $self->{open_elements}->[$_];            my $node = $self->{open_elements}->[$_];
5317            if ($node->[1] & P_EL) {            if ($node->[1] == P_EL) {
5318              !!!cp ('t410.1');              !!!cp ('t410.1');
5319              $i = $_;              $i = $_;
5320              last INSCOPE;              last INSCOPE;
# Line 8032  sub _tree_construction_main ($) { Line 5491  sub _tree_construction_main ($) {
5491    ## TODO: script stuffs    ## TODO: script stuffs
5492  } # _tree_construct_main  } # _tree_construct_main
5493    
5494    ## XXX: How this method is organized is somewhat out of date, although
5495    ## it still does what the current spec documents.
5496  sub set_inner_html ($$$$;$) {  sub set_inner_html ($$$$;$) {
5497    my $class = shift;    my $class = shift;
5498    my $node = shift;    my $node = shift; # /context/
5499    #my $s = \$_[0];    #my $s = \$_[0];
5500    my $onerror = $_[1];    my $onerror = $_[1];
5501    my $get_wrapper = $_[2] || sub ($) { return $_[0] };    my $get_wrapper = $_[2] || sub ($) { return $_[0] };
5502    
   ## ISSUE: Should {confident} be true?  
   
5503    my $nt = $node->node_type;    my $nt = $node->node_type;
5504    if ($nt == 9) {    if ($nt == 9) { # Document (invoke the algorithm with no /context/ element)
5505      # MUST      # MUST
5506            
5507      ## Step 1 # MUST      ## Step 1 # MUST
# Line 8057  sub set_inner_html ($$$$;$) { Line 5516  sub set_inner_html ($$$$;$) {
5516    
5517      ## Step 3, 4, 5 # MUST      ## Step 3, 4, 5 # MUST
5518      $class->parse_char_string ($_[0] => $node, $onerror, $get_wrapper);      $class->parse_char_string ($_[0] => $node, $onerror, $get_wrapper);
5519    } elsif ($nt == 1) {    } elsif ($nt == 1) { # Element (invoke the algorithm with /context/ element)
5520      ## TODO: If non-html element      ## TODO: If non-html element
5521    
5522      ## NOTE: Most of this code is copied from |parse_string|      ## NOTE: Most of this code is copied from |parse_string|
5523    
5524  ## TODO: Support for $get_wrapper  ## TODO: Support for $get_wrapper
5525    
5526      ## Step 1 # MUST      ## F1. Create an HTML document.
5527      my $this_doc = $node->owner_document;      my $this_doc = $node->owner_document;
5528      my $doc = $this_doc->implementation->create_document;      my $doc = $this_doc->implementation->create_document;
5529      $doc->manakai_is_html (1);      $doc->manakai_is_html (1);
5530    
5531        ## F2. Propagate quirkness flag
5532        my $node_doc = $node->owner_document;
5533        $doc->manakai_compat_mode ($node_doc->manakai_compat_mode);
5534    
5535        ## F3. Create an HTML parser
5536      my $p = $class->new;      my $p = $class->new;
5537      $p->{document} = $doc;      $p->{document} = $doc;
5538    
# Line 8195  sub set_inner_html ($$$$;$) { Line 5660  sub set_inner_html ($$$$;$) {
5660      $p->_initialize_tokenizer;      $p->_initialize_tokenizer;
5661      $p->_initialize_tree_constructor;      $p->_initialize_tree_constructor;
5662    
5663      ## Step 2      ## F4. If /context/ is not undef...
5664    
5665        ## F4.1. content model flag
5666      my $node_ln = $node->manakai_local_name;      my $node_ln = $node->manakai_local_name;
5667      $p->{content_model} = {      $p->{content_model} = {
5668        title => RCDATA_CONTENT_MODEL,        title => RCDATA_CONTENT_MODEL,
# Line 8211  sub set_inner_html ($$$$;$) { Line 5678  sub set_inner_html ($$$$;$) {
5678      }->{$node_ln};      }->{$node_ln};
5679      $p->{content_model} = PCDATA_CONTENT_MODEL      $p->{content_model} = PCDATA_CONTENT_MODEL
5680          unless defined $p->{content_model};          unless defined $p->{content_model};
         ## ISSUE: What is "the name of the element"? local name?  
5681    
5682      $p->{inner_html_node} = [$node, $el_category->{$node_ln}];      $p->{inner_html_node} = [$node, $el_category->{$node_ln}];
5683        ## TODO: Foreign element OK?        ## TODO: Foreign element OK?
5684    
5685      ## Step 3      ## F4.2. Root |html| element
5686      my $root = $doc->create_element_ns      my $root = $doc->create_element_ns
5687        ('http://www.w3.org/1999/xhtml', [undef, 'html']);        ('http://www.w3.org/1999/xhtml', [undef, 'html']);
5688    
5689      ## Step 4 # MUST      ## F4.3.
5690      $doc->append_child ($root);      $doc->append_child ($root);
5691    
5692      ## Step 5 # MUST      ## F4.4.
5693      push @{$p->{open_elements}}, [$root, $el_category->{html}];      push @{$p->{open_elements}}, [$root, $el_category->{html}];
5694    
5695      undef $p->{head_element};      undef $p->{head_element};
5696      undef $p->{head_element_inserted};      undef $p->{head_element_inserted};
5697    
5698      ## Step 6 # MUST      ## F4.5.
5699      $p->_reset_insertion_mode;      $p->_reset_insertion_mode;
5700    
5701      ## Step 7 # MUST      ## F4.6.
5702      my $anode = $node;      my $anode = $node;
5703      AN: while (defined $anode) {      AN: while (defined $anode) {
5704        if ($anode->node_type == 1) {        if ($anode->node_type == 1) {
# Line 8247  sub set_inner_html ($$$$;$) { Line 5713  sub set_inner_html ($$$$;$) {
5713        }        }
5714        $anode = $anode->parent_node;        $anode = $anode->parent_node;
5715      } # AN      } # AN
5716        
5717      ## Step 9 # MUST      ## F.5. Set the input stream.
5718        $p->{confident} = 1; ## Confident: irrelevant.
5719    
5720        ## F.6. Start the parser.
5721      {      {
5722        my $self = $p;        my $self = $p;
5723        !!!next-token;        !!!next-token;
5724      }      }
5725      $p->_tree_construction_main;      $p->_tree_construction_main;
5726    
5727      ## Step 10 # MUST      ## F.7.
5728      my @cn = @{$node->child_nodes};      my @cn = @{$node->child_nodes};
5729      for (@cn) {      for (@cn) {
5730        $node->remove_child ($_);        $node->remove_child ($_);

Legend:
Removed from v.1.203  
changed lines
  Added in v.1.244

[email protected]
ViewVC Help
Powered by ViewVC 1.1.24