/[suikacvs]/markup/html/whatpm/Whatpm/HTML.pm.src
Suika

Diff of /markup/html/whatpm/Whatpm/HTML.pm.src

Parent Directory Parent Directory | Revision Log Revision Log | View Patch Patch

revision 1.193 by wakaba, Sat Oct 4 04:06:33 2008 UTC revision 1.211 by wakaba, Mon Oct 27 05:44:47 2008 UTC
# Line 3  use strict; Line 3  use strict;
3  our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r};  our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r};
4  use Error qw(:try);  use Error qw(:try);
5    
6    use Whatpm::HTML::Tokenizer;
7    
8  ## NOTE: This module don't check all HTML5 parse errors; character  ## NOTE: This module don't check all HTML5 parse errors; character
9  ## encoding related parse errors are expected to be handled by relevant  ## encoding related parse errors are expected to be handled by relevant
10  ## modules.  ## modules.
# Line 21  use Error qw(:try); Line 23  use Error qw(:try);
23    
24  require IO::Handle;  require IO::Handle;
25    
26    ## Namespace URLs
27    
28  my $HTML_NS = q<http://www.w3.org/1999/xhtml>;  my $HTML_NS = q<http://www.w3.org/1999/xhtml>;
29  my $MML_NS = q<http://www.w3.org/1998/Math/MathML>;  my $MML_NS = q<http://www.w3.org/1998/Math/MathML>;
30  my $SVG_NS = q<http://www.w3.org/2000/svg>;  my $SVG_NS = q<http://www.w3.org/2000/svg>;
# Line 28  my $XLINK_NS = q<http://www.w3.org/1999/ Line 32  my $XLINK_NS = q<http://www.w3.org/1999/
32  my $XML_NS = q<http://www.w3.org/XML/1998/namespace>;  my $XML_NS = q<http://www.w3.org/XML/1998/namespace>;
33  my $XMLNS_NS = q<http://www.w3.org/2000/xmlns/>;  my $XMLNS_NS = q<http://www.w3.org/2000/xmlns/>;
34    
35  sub A_EL () { 0b1 }  ## Element categories
 sub ADDRESS_EL () { 0b10 }  
 sub BODY_EL () { 0b100 }  
 sub BUTTON_EL () { 0b1000 }  
 sub CAPTION_EL () { 0b10000 }  
 sub DD_EL () { 0b100000 }  
 sub DIV_EL () { 0b1000000 }  
 sub DT_EL () { 0b10000000 }  
 sub FORM_EL () { 0b100000000 }  
 sub FORMATTING_EL () { 0b1000000000 }  
 sub FRAMESET_EL () { 0b10000000000 }  
 sub HEADING_EL () { 0b100000000000 }  
 sub HTML_EL () { 0b1000000000000 }  
 sub LI_EL () { 0b10000000000000 }  
 sub NOBR_EL () { 0b100000000000000 }  
 sub OPTION_EL () { 0b1000000000000000 }  
 sub OPTGROUP_EL () { 0b10000000000000000 }  
 sub P_EL () { 0b100000000000000000 }  
 sub SELECT_EL () { 0b1000000000000000000 }  
 sub TABLE_EL () { 0b10000000000000000000 }  
 sub TABLE_CELL_EL () { 0b100000000000000000000 }  
 sub TABLE_ROW_EL () { 0b1000000000000000000000 }  
 sub TABLE_ROW_GROUP_EL () { 0b10000000000000000000000 }  
 sub MISC_SCOPING_EL () { 0b100000000000000000000000 }  
 sub MISC_SPECIAL_EL () { 0b1000000000000000000000000 }  
 sub FOREIGN_EL () { 0b10000000000000000000000000 }  
 sub FOREIGN_FLOW_CONTENT_EL () { 0b100000000000000000000000000 }  
 sub MML_AXML_EL () { 0b1000000000000000000000000000 }  
 sub RUBY_EL () { 0b10000000000000000000000000000 }  
 sub RUBY_COMPONENT_EL () { 0b100000000000000000000000000000 }  
   
 sub TABLE_ROWS_EL () {  
   TABLE_EL |  
   TABLE_ROW_EL |  
   TABLE_ROW_GROUP_EL  
 }  
36    
37  ## NOTE: Used in "generate implied end tags" algorithm.  ## Bits 12-15
38  ## NOTE: There is a code where a modified version of END_TAG_OPTIONAL_EL  sub SPECIAL_EL () { 0b1_000000000000000 }
39  ## is used in "generate implied end tags" implementation (search for the  sub SCOPING_EL () { 0b1_00000000000000 }
40  ## function mae).  sub FORMATTING_EL () { 0b1_0000000000000 }
41  sub END_TAG_OPTIONAL_EL () {  sub PHRASING_EL () { 0b1_000000000000 }
42    DD_EL |  
43    DT_EL |  ## Bits 10-11
44    LI_EL |  #sub FOREIGN_EL () { 0b1_00000000000 } # see Whatpm::HTML::Tokenizer
45    P_EL |  sub FOREIGN_FLOW_CONTENT_EL () { 0b1_0000000000 }
46    RUBY_COMPONENT_EL  
47  }  ## Bits 6-9
48    sub TABLE_SCOPING_EL () { 0b1_000000000 }
49    sub TABLE_ROWS_SCOPING_EL () { 0b1_00000000 }
50    sub TABLE_ROW_SCOPING_EL () { 0b1_0000000 }
51    sub TABLE_ROWS_EL () { 0b1_000000 }
52    
53    ## Bit 5
54    sub ADDRESS_DIV_P_EL () { 0b1_00000 }
55    
56  ## NOTE: Used in </body> and EOF algorithms.  ## NOTE: Used in </body> and EOF algorithms.
57  sub ALL_END_TAG_OPTIONAL_EL () {  ## Bit 4
58    DD_EL |  sub ALL_END_TAG_OPTIONAL_EL () { 0b1_0000 }
   DT_EL |  
   LI_EL |  
   P_EL |  
   
   BODY_EL |  
   HTML_EL |  
   TABLE_CELL_EL |  
   TABLE_ROW_EL |  
   TABLE_ROW_GROUP_EL  
 }  
59    
60  sub SCOPING_EL () {  ## NOTE: Used in "generate implied end tags" algorithm.
61    BUTTON_EL |  ## NOTE: There is a code where a modified version of
62    CAPTION_EL |  ## END_TAG_OPTIONAL_EL is used in "generate implied end tags"
63    HTML_EL |  ## implementation (search for the algorithm name).
64    TABLE_EL |  ## Bit 3
65    TABLE_CELL_EL |  sub END_TAG_OPTIONAL_EL () { 0b1_000 }
66    MISC_SCOPING_EL  
67    ## Bits 0-2
68    
69    sub MISC_SPECIAL_EL () { SPECIAL_EL | 0b000 }
70    sub FORM_EL () { SPECIAL_EL | 0b001 }
71    sub FRAMESET_EL () { SPECIAL_EL | 0b010 }
72    sub HEADING_EL () { SPECIAL_EL | 0b011 }
73    sub SELECT_EL () { SPECIAL_EL | 0b100 }
74    sub SCRIPT_EL () { SPECIAL_EL | 0b101 }
75    
76    sub ADDRESS_DIV_EL () { SPECIAL_EL | ADDRESS_DIV_P_EL | 0b001 }
77    sub BODY_EL () { SPECIAL_EL | ALL_END_TAG_OPTIONAL_EL | 0b001 }
78    
79    sub DTDD_EL () {
80      SPECIAL_EL |
81      END_TAG_OPTIONAL_EL |
82      ALL_END_TAG_OPTIONAL_EL |
83      0b010
84  }  }
85    sub LI_EL () {
86  sub TABLE_SCOPING_EL () {    SPECIAL_EL |
87    HTML_EL |    END_TAG_OPTIONAL_EL |
88    TABLE_EL    ALL_END_TAG_OPTIONAL_EL |
89      0b100
90  }  }
91    sub P_EL () {
92  sub TABLE_ROWS_SCOPING_EL () {    SPECIAL_EL |
93    HTML_EL |    ADDRESS_DIV_P_EL |
94    TABLE_ROW_GROUP_EL    END_TAG_OPTIONAL_EL |
95      ALL_END_TAG_OPTIONAL_EL |
96      0b001
97  }  }
98    
99  sub TABLE_ROW_SCOPING_EL () {  sub TABLE_ROW_EL () {
100    HTML_EL |    SPECIAL_EL |
101    TABLE_ROW_EL    TABLE_ROWS_EL |
102      TABLE_ROW_SCOPING_EL |
103      ALL_END_TAG_OPTIONAL_EL |
104      0b001
105    }
106    sub TABLE_ROW_GROUP_EL () {
107      SPECIAL_EL |
108      TABLE_ROWS_EL |
109      TABLE_ROWS_SCOPING_EL |
110      ALL_END_TAG_OPTIONAL_EL |
111      0b001
112  }  }
113    
114  sub SPECIAL_EL () {  sub MISC_SCOPING_EL () { SCOPING_EL | 0b000 }
115    ADDRESS_EL |  sub BUTTON_EL () { SCOPING_EL | 0b001 }
116    BODY_EL |  sub CAPTION_EL () { SCOPING_EL | 0b010 }
117    DIV_EL |  sub HTML_EL () {
118      SCOPING_EL |
119    DD_EL |    TABLE_SCOPING_EL |
120    DT_EL |    TABLE_ROWS_SCOPING_EL |
121    LI_EL |    TABLE_ROW_SCOPING_EL |
122    P_EL |    ALL_END_TAG_OPTIONAL_EL |
123      0b001
124    FORM_EL |  }
125    FRAMESET_EL |  sub TABLE_EL () {
126    HEADING_EL |    SCOPING_EL |
127    OPTION_EL |    TABLE_ROWS_EL |
128    OPTGROUP_EL |    TABLE_SCOPING_EL |
129    SELECT_EL |    0b001
   TABLE_ROW_EL |  
   TABLE_ROW_GROUP_EL |  
   MISC_SPECIAL_EL  
130  }  }
131    sub TABLE_CELL_EL () {
132      SCOPING_EL |
133      TABLE_ROW_SCOPING_EL |
134      ALL_END_TAG_OPTIONAL_EL |
135      0b001
136    }
137    
138    sub MISC_FORMATTING_EL () { FORMATTING_EL | 0b000 }
139    sub A_EL () { FORMATTING_EL | 0b001 }
140    sub NOBR_EL () { FORMATTING_EL | 0b010 }
141    
142    sub RUBY_EL () { PHRASING_EL | 0b001 }
143    
144    ## ISSUE: ALL_END_TAG_OPTIONAL_EL?
145    sub OPTGROUP_EL () { PHRASING_EL | END_TAG_OPTIONAL_EL | 0b001 }
146    sub OPTION_EL () { PHRASING_EL | END_TAG_OPTIONAL_EL | 0b010 }
147    sub RUBY_COMPONENT_EL () { PHRASING_EL | END_TAG_OPTIONAL_EL | 0b100 }
148    
149    sub MML_AXML_EL () { PHRASING_EL | FOREIGN_EL | 0b001 }
150    
151  my $el_category = {  my $el_category = {
152    a => A_EL | FORMATTING_EL,    a => A_EL,
153    address => ADDRESS_EL,    address => ADDRESS_DIV_EL,
154    applet => MISC_SCOPING_EL,    applet => MISC_SCOPING_EL,
155    area => MISC_SPECIAL_EL,    area => MISC_SPECIAL_EL,
156    article => MISC_SPECIAL_EL,    article => MISC_SPECIAL_EL,
# Line 158  my $el_category = { Line 170  my $el_category = {
170    colgroup => MISC_SPECIAL_EL,    colgroup => MISC_SPECIAL_EL,
171    command => MISC_SPECIAL_EL,    command => MISC_SPECIAL_EL,
172    datagrid => MISC_SPECIAL_EL,    datagrid => MISC_SPECIAL_EL,
173    dd => DD_EL,    dd => DTDD_EL,
174    details => MISC_SPECIAL_EL,    details => MISC_SPECIAL_EL,
175    dialog => MISC_SPECIAL_EL,    dialog => MISC_SPECIAL_EL,
176    dir => MISC_SPECIAL_EL,    dir => MISC_SPECIAL_EL,
177    div => DIV_EL,    div => ADDRESS_DIV_EL,
178    dl => MISC_SPECIAL_EL,    dl => MISC_SPECIAL_EL,
179    dt => DT_EL,    dt => DTDD_EL,
180    em => FORMATTING_EL,    em => FORMATTING_EL,
181    embed => MISC_SPECIAL_EL,    embed => MISC_SPECIAL_EL,
182    eventsource => MISC_SPECIAL_EL,    eventsource => MISC_SPECIAL_EL,
# Line 198  my $el_category = { Line 210  my $el_category = {
210    menu => MISC_SPECIAL_EL,    menu => MISC_SPECIAL_EL,
211    meta => MISC_SPECIAL_EL,    meta => MISC_SPECIAL_EL,
212    nav => MISC_SPECIAL_EL,    nav => MISC_SPECIAL_EL,
213    nobr => NOBR_EL | FORMATTING_EL,    nobr => NOBR_EL,
214    noembed => MISC_SPECIAL_EL,    noembed => MISC_SPECIAL_EL,
215    noframes => MISC_SPECIAL_EL,    noframes => MISC_SPECIAL_EL,
216    noscript => MISC_SPECIAL_EL,    noscript => MISC_SPECIAL_EL,
# Line 240  my $el_category = { Line 252  my $el_category = {
252  my $el_category_f = {  my $el_category_f = {
253    $MML_NS => {    $MML_NS => {
254      'annotation-xml' => MML_AXML_EL,      'annotation-xml' => MML_AXML_EL,
255      mi => FOREIGN_FLOW_CONTENT_EL,      mi => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
256      mo => FOREIGN_FLOW_CONTENT_EL,      mo => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
257      mn => FOREIGN_FLOW_CONTENT_EL,      mn => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
258      ms => FOREIGN_FLOW_CONTENT_EL,      ms => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
259      mtext => FOREIGN_FLOW_CONTENT_EL,      mtext => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
260    },    },
261    $SVG_NS => {    $SVG_NS => {
262      foreignObject => FOREIGN_FLOW_CONTENT_EL,      foreignObject => SCOPING_EL | FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
263      desc => FOREIGN_FLOW_CONTENT_EL,      desc => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
264      title => FOREIGN_FLOW_CONTENT_EL,      title => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
265    },    },
266    ## NOTE: In addition, FOREIGN_EL is set to non-HTML elements.    ## NOTE: In addition, FOREIGN_EL is set to non-HTML elements.
267  };  };
# Line 336  my $foreign_attr_xname = { Line 348  my $foreign_attr_xname = {
348    
349  ## ISSUE: xmlns:xlink="non-xlink-ns" is not an error.  ## ISSUE: xmlns:xlink="non-xlink-ns" is not an error.
350    
 my $charref_map = {  
   0x0D => 0x000A,  
   0x80 => 0x20AC,  
   0x81 => 0xFFFD,  
   0x82 => 0x201A,  
   0x83 => 0x0192,  
   0x84 => 0x201E,  
   0x85 => 0x2026,  
   0x86 => 0x2020,  
   0x87 => 0x2021,  
   0x88 => 0x02C6,  
   0x89 => 0x2030,  
   0x8A => 0x0160,  
   0x8B => 0x2039,  
   0x8C => 0x0152,  
   0x8D => 0xFFFD,  
   0x8E => 0x017D,  
   0x8F => 0xFFFD,  
   0x90 => 0xFFFD,  
   0x91 => 0x2018,  
   0x92 => 0x2019,  
   0x93 => 0x201C,  
   0x94 => 0x201D,  
   0x95 => 0x2022,  
   0x96 => 0x2013,  
   0x97 => 0x2014,  
   0x98 => 0x02DC,  
   0x99 => 0x2122,  
   0x9A => 0x0161,  
   0x9B => 0x203A,  
   0x9C => 0x0153,  
   0x9D => 0xFFFD,  
   0x9E => 0x017E,  
   0x9F => 0x0178,  
 }; # $charref_map  
 $charref_map->{$_} = 0xFFFD  
     for 0x0000..0x0008, 0x000B, 0x000E..0x001F, 0x007F,  
         0xD800..0xDFFF, 0xFDD0..0xFDDF, ## ISSUE: 0xFDEF  
         0xFFFE, 0xFFFF, 0x1FFFE, 0x1FFFF, 0x2FFFE, 0x2FFFF, 0x3FFFE, 0x3FFFF,  
         0x4FFFE, 0x4FFFF, 0x5FFFE, 0x5FFFF, 0x6FFFE, 0x6FFFF, 0x7FFFE,  
         0x7FFFF, 0x8FFFE, 0x8FFFF, 0x9FFFE, 0x9FFFF, 0xAFFFE, 0xAFFFF,  
         0xBFFFE, 0xBFFFF, 0xCFFFE, 0xCFFFF, 0xDFFFE, 0xDFFFF, 0xEFFFE,  
         0xEFFFF, 0xFFFFE, 0xFFFFF, 0x10FFFE, 0x10FFFF;  
   
351  ## TODO: Invoke the reset algorithm when a resettable element is  ## TODO: Invoke the reset algorithm when a resettable element is
352  ## created (cf. HTML5 revision 2259).  ## created (cf. HTML5 revision 2259).
353    
# Line 486  sub parse_byte_stream ($$$$;$$) { Line 454  sub parse_byte_stream ($$$$;$$) {
454      if (defined $charset_name) {      if (defined $charset_name) {
455        $charset = Message::Charset::Info->get_by_html_name ($charset_name);        $charset = Message::Charset::Info->get_by_html_name ($charset_name);
456    
       ## ISSUE: Unsupported encoding is not ignored according to the spec.  
457        require Whatpm::Charset::DecodeHandle;        require Whatpm::Charset::DecodeHandle;
458        $buffer = Whatpm::Charset::DecodeHandle::ByteBuffer->new        $buffer = Whatpm::Charset::DecodeHandle::ByteBuffer->new
459            ($byte_stream);            ($byte_stream);
# Line 833  sub new ($) { Line 800  sub new ($) {
800    return $self;    return $self;
801  } # new  } # new
802    
803  sub CM_ENTITY () { 0b001 } # & markup in data  ## Insertion modes
 sub CM_LIMITED_MARKUP () { 0b010 } # < markup in data (limited)  
 sub CM_FULL_MARKUP () { 0b100 } # < markup in data (any)  
   
 sub PLAINTEXT_CONTENT_MODEL () { 0 }  
 sub CDATA_CONTENT_MODEL () { CM_LIMITED_MARKUP }  
 sub RCDATA_CONTENT_MODEL () { CM_ENTITY | CM_LIMITED_MARKUP }  
 sub PCDATA_CONTENT_MODEL () { CM_ENTITY | CM_FULL_MARKUP }  
   
 sub DATA_STATE () { 0 }  
 #sub ENTITY_DATA_STATE () { 1 }  
 sub TAG_OPEN_STATE () { 2 }  
 sub CLOSE_TAG_OPEN_STATE () { 3 }  
 sub TAG_NAME_STATE () { 4 }  
 sub BEFORE_ATTRIBUTE_NAME_STATE () { 5 }  
 sub ATTRIBUTE_NAME_STATE () { 6 }  
 sub AFTER_ATTRIBUTE_NAME_STATE () { 7 }  
 sub BEFORE_ATTRIBUTE_VALUE_STATE () { 8 }  
 sub ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE () { 9 }  
 sub ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE () { 10 }  
 sub ATTRIBUTE_VALUE_UNQUOTED_STATE () { 11 }  
 #sub ENTITY_IN_ATTRIBUTE_VALUE_STATE () { 12 }  
 sub MARKUP_DECLARATION_OPEN_STATE () { 13 }  
 sub COMMENT_START_STATE () { 14 }  
 sub COMMENT_START_DASH_STATE () { 15 }  
 sub COMMENT_STATE () { 16 }  
 sub COMMENT_END_STATE () { 17 }  
 sub COMMENT_END_DASH_STATE () { 18 }  
 sub BOGUS_COMMENT_STATE () { 19 }  
 sub DOCTYPE_STATE () { 20 }  
 sub BEFORE_DOCTYPE_NAME_STATE () { 21 }  
 sub DOCTYPE_NAME_STATE () { 22 }  
 sub AFTER_DOCTYPE_NAME_STATE () { 23 }  
 sub BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE () { 24 }  
 sub DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE () { 25 }  
 sub DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE () { 26 }  
 sub AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE () { 27 }  
 sub BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE () { 28 }  
 sub DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE () { 29 }  
 sub DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE () { 30 }  
 sub AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE () { 31 }  
 sub BOGUS_DOCTYPE_STATE () { 32 }  
 sub AFTER_ATTRIBUTE_VALUE_QUOTED_STATE () { 33 }  
 sub SELF_CLOSING_START_TAG_STATE () { 34 }  
 sub CDATA_SECTION_STATE () { 35 }  
 sub MD_HYPHEN_STATE () { 36 } # "markup declaration open state" in the spec  
 sub MD_DOCTYPE_STATE () { 37 } # "markup declaration open state" in the spec  
 sub MD_CDATA_STATE () { 38 } # "markup declaration open state" in the spec  
 sub CDATA_RCDATA_CLOSE_TAG_STATE () { 39 } # "close tag open state" in the spec  
 sub CDATA_SECTION_MSE1_STATE () { 40 } # "CDATA section state" in the spec  
 sub CDATA_SECTION_MSE2_STATE () { 41 } # "CDATA section state" in the spec  
 sub PUBLIC_STATE () { 42 } # "after DOCTYPE name state" in the spec  
 sub SYSTEM_STATE () { 43 } # "after DOCTYPE name state" in the spec  
 ## NOTE: "Entity data state", "entity in attribute value state", and  
 ## "consume a character reference" algorithm are jointly implemented  
 ## using the following six states:  
 sub ENTITY_STATE () { 44 }  
 sub ENTITY_HASH_STATE () { 45 }  
 sub NCR_NUM_STATE () { 46 }  
 sub HEXREF_X_STATE () { 47 }  
 sub HEXREF_HEX_STATE () { 48 }  
 sub ENTITY_NAME_STATE () { 49 }  
 sub PCDATA_STATE () { 50 } # "data state" in the spec  
   
 sub DOCTYPE_TOKEN () { 1 }  
 sub COMMENT_TOKEN () { 2 }  
 sub START_TAG_TOKEN () { 3 }  
 sub END_TAG_TOKEN () { 4 }  
 sub END_OF_FILE_TOKEN () { 5 }  
 sub CHARACTER_TOKEN () { 6 }  
804    
805  sub AFTER_HTML_IMS () { 0b100 }  sub AFTER_HTML_IMS () { 0b100 }
806  sub HEAD_IMS ()       { 0b1000 }  sub HEAD_IMS ()       { 0b1000 }
# Line 913  sub ROW_IMS ()        { 0b10000000 } Line 811  sub ROW_IMS ()        { 0b10000000 }
811  sub BODY_AFTER_IMS () { 0b100000000 }  sub BODY_AFTER_IMS () { 0b100000000 }
812  sub FRAME_IMS ()      { 0b1000000000 }  sub FRAME_IMS ()      { 0b1000000000 }
813  sub SELECT_IMS ()     { 0b10000000000 }  sub SELECT_IMS ()     { 0b10000000000 }
814  sub IN_FOREIGN_CONTENT_IM () { 0b100000000000 }  #sub IN_FOREIGN_CONTENT_IM () { 0b100000000000 } # see Whatpm::HTML::Tokenizer
815      ## NOTE: "in foreign content" insertion mode is special; it is combined      ## NOTE: "in foreign content" insertion mode is special; it is combined
816      ## with the secondary insertion mode.  In this parser, they are stored      ## with the secondary insertion mode.  In this parser, they are stored
817      ## together in the bit-or'ed form.      ## together in the bit-or'ed form.
818    sub IN_CDATA_RCDATA_IM () { 0b1000000000000 }
819        ## NOTE: "in CDATA/RCDATA" insertion mode is also special; it is
820        ## combined with the original insertion mode.  In thie parser,
821        ## they are stored together in the bit-or'ed form.
822    
823    sub IM_MASK () { 0b11111111111 }
824    
825  ## NOTE: "initial" and "before html" insertion modes have no constants.  ## NOTE: "initial" and "before html" insertion modes have no constants.
826    
# Line 943  sub IN_SELECT_IM () { SELECT_IMS | 0b01 Line 847  sub IN_SELECT_IM () { SELECT_IMS | 0b01
847  sub IN_SELECT_IN_TABLE_IM () { SELECT_IMS | 0b10 }  sub IN_SELECT_IN_TABLE_IM () { SELECT_IMS | 0b10 }
848  sub IN_COLUMN_GROUP_IM () { 0b10 }  sub IN_COLUMN_GROUP_IM () { 0b10 }
849    
 ## Implementations MUST act as if state machine in the spec  
   
 sub _initialize_tokenizer ($) {  
   my $self = shift;  
   $self->{state} = DATA_STATE; # MUST  
   #$self->{s_kwd}; # state keyword - initialized when used  
   #$self->{entity__value}; # initialized when used  
   #$self->{entity__match}; # initialized when used  
   $self->{content_model} = PCDATA_CONTENT_MODEL; # be  
   undef $self->{ct}; # current token  
   undef $self->{ca}; # current attribute  
   undef $self->{last_stag_name}; # last emitted start tag name  
   #$self->{prev_state}; # initialized when used  
   delete $self->{self_closing};  
   $self->{char_buffer} = '';  
   $self->{char_buffer_pos} = 0;  
   $self->{nc} = -1; # next input character  
   #$self->{next_nc}  
   !!!next-input-character;  
   $self->{token} = [];  
   # $self->{escape}  
 } # _initialize_tokenizer  
   
 ## A token has:  
 ##   ->{type} == DOCTYPE_TOKEN, START_TAG_TOKEN, END_TAG_TOKEN, COMMENT_TOKEN,  
 ##       CHARACTER_TOKEN, or END_OF_FILE_TOKEN  
 ##   ->{name} (DOCTYPE_TOKEN)  
 ##   ->{tag_name} (START_TAG_TOKEN, END_TAG_TOKEN)  
 ##   ->{pubid} (DOCTYPE_TOKEN)  
 ##   ->{sysid} (DOCTYPE_TOKEN)  
 ##   ->{quirks} == 1 or 0 (DOCTYPE_TOKEN): "force-quirks" flag  
 ##   ->{attributes} isa HASH (START_TAG_TOKEN, END_TAG_TOKEN)  
 ##        ->{name}  
 ##        ->{value}  
 ##        ->{has_reference} == 1 or 0  
 ##   ->{data} (COMMENT_TOKEN, CHARACTER_TOKEN)  
 ## NOTE: The "self-closing flag" is hold as |$self->{self_closing}|.  
 ##     |->{self_closing}| is used to save the value of |$self->{self_closing}|  
 ##     while the token is pushed back to the stack.  
   
 ## Emitted token MUST immediately be handled by the tree construction state.  
   
 ## Before each step, UA MAY check to see if either one of the scripts in  
 ## "list of scripts that will execute as soon as possible" or the first  
 ## script in the "list of scripts that will execute asynchronously",  
 ## has completed loading.  If one has, then it MUST be executed  
 ## and removed from the list.  
   
 ## TODO: Polytheistic slash SHOULD NOT be used. (Applied only to atheists.)  
 ## (This requirement was dropped from HTML5 spec, unfortunately.)  
   
 my $is_space = {  
   0x0009 => 1, # CHARACTER TABULATION (HT)  
   0x000A => 1, # LINE FEED (LF)  
   #0x000B => 0, # LINE TABULATION (VT)  
   0x000C => 1, # FORM FEED (FF)  
   #0x000D => 1, # CARRIAGE RETURN (CR)  
   0x0020 => 1, # SPACE (SP)  
 };  
   
 sub _get_next_token ($) {  
   my $self = shift;  
   
   if ($self->{self_closing}) {  
     !!!parse-error (type => 'nestc', token => $self->{ct});  
     ## NOTE: The |self_closing| flag is only set by start tag token.  
     ## In addition, when a start tag token is emitted, it is always set to  
     ## |ct|.  
     delete $self->{self_closing};  
   }  
   
   if (@{$self->{token}}) {  
     $self->{self_closing} = $self->{token}->[0]->{self_closing};  
     return shift @{$self->{token}};  
   }  
   
   A: {  
     if ($self->{state} == PCDATA_STATE) {  
       ## NOTE: Same as |DATA_STATE|, but only for |PCDATA| content model.  
   
       if ($self->{nc} == 0x0026) { # &  
         !!!cp (0.1);  
         ## NOTE: In the spec, the tokenizer is switched to the  
         ## "entity data state".  In this implementation, the tokenizer  
         ## is switched to the |ENTITY_STATE|, which is an implementation  
         ## of the "consume a character reference" algorithm.  
         $self->{entity_add} = -1;  
         $self->{prev_state} = DATA_STATE;  
         $self->{state} = ENTITY_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003C) { # <  
         !!!cp (0.2);  
         $self->{state} = TAG_OPEN_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (0.3);  
         !!!emit ({type => END_OF_FILE_TOKEN,  
                   line => $self->{line}, column => $self->{column}});  
         last A; ## TODO: ok?  
       } else {  
         !!!cp (0.4);  
         #  
       }  
   
       # Anything else  
       my $token = {type => CHARACTER_TOKEN,  
                    data => chr $self->{nc},  
                    line => $self->{line}, column => $self->{column},  
                   };  
       $self->{read_until}->($token->{data}, q[<&], length $token->{data});  
   
       ## Stay in the state.  
       !!!next-input-character;  
       !!!emit ($token);  
       redo A;  
     } elsif ($self->{state} == DATA_STATE) {  
       $self->{s_kwd} = '' unless defined $self->{s_kwd};  
       if ($self->{nc} == 0x0026) { # &  
         $self->{s_kwd} = '';  
         if ($self->{content_model} & CM_ENTITY and # PCDATA | RCDATA  
             not $self->{escape}) {  
           !!!cp (1);  
           ## NOTE: In the spec, the tokenizer is switched to the  
           ## "entity data state".  In this implementation, the tokenizer  
           ## is switched to the |ENTITY_STATE|, which is an implementation  
           ## of the "consume a character reference" algorithm.  
           $self->{entity_add} = -1;  
           $self->{prev_state} = DATA_STATE;  
           $self->{state} = ENTITY_STATE;  
           !!!next-input-character;  
           redo A;  
         } else {  
           !!!cp (2);  
           #  
         }  
       } elsif ($self->{nc} == 0x002D) { # -  
         if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA  
           $self->{s_kwd} .= '-';  
             
           if ($self->{s_kwd} eq '<!--') {  
             !!!cp (3);  
             $self->{escape} = 1; # unless $self->{escape};  
             $self->{s_kwd} = '--';  
             #  
           } elsif ($self->{s_kwd} eq '---') {  
             !!!cp (4);  
             $self->{s_kwd} = '--';  
             #  
           } else {  
             !!!cp (5);  
             #  
           }  
         }  
           
         #  
       } elsif ($self->{nc} == 0x0021) { # !  
         if (length $self->{s_kwd}) {  
           !!!cp (5.1);  
           $self->{s_kwd} .= '!';  
           #  
         } else {  
           !!!cp (5.2);  
           #$self->{s_kwd} = '';  
           #  
         }  
         #  
       } elsif ($self->{nc} == 0x003C) { # <  
         if ($self->{content_model} & CM_FULL_MARKUP or # PCDATA  
             (($self->{content_model} & CM_LIMITED_MARKUP) and # CDATA | RCDATA  
              not $self->{escape})) {  
           !!!cp (6);  
           $self->{state} = TAG_OPEN_STATE;  
           !!!next-input-character;  
           redo A;  
         } else {  
           !!!cp (7);  
           $self->{s_kwd} = '';  
           #  
         }  
       } elsif ($self->{nc} == 0x003E) { # >  
         if ($self->{escape} and  
             ($self->{content_model} & CM_LIMITED_MARKUP)) { # RCDATA | CDATA  
           if ($self->{s_kwd} eq '--') {  
             !!!cp (8);  
             delete $self->{escape};  
           } else {  
             !!!cp (9);  
           }  
         } else {  
           !!!cp (10);  
         }  
           
         $self->{s_kwd} = '';  
         #  
       } elsif ($self->{nc} == -1) {  
         !!!cp (11);  
         $self->{s_kwd} = '';  
         !!!emit ({type => END_OF_FILE_TOKEN,  
                   line => $self->{line}, column => $self->{column}});  
         last A; ## TODO: ok?  
       } else {  
         !!!cp (12);  
         $self->{s_kwd} = '';  
         #  
       }  
   
       # Anything else  
       my $token = {type => CHARACTER_TOKEN,  
                    data => chr $self->{nc},  
                    line => $self->{line}, column => $self->{column},  
                   };  
       if ($self->{read_until}->($token->{data}, q[-!<>&],  
                                 length $token->{data})) {  
         $self->{s_kwd} = '';  
       }  
   
       ## Stay in the data state.  
       if ($self->{content_model} == PCDATA_CONTENT_MODEL) {  
         !!!cp (13);  
         $self->{state} = PCDATA_STATE;  
       } else {  
         !!!cp (14);  
         ## Stay in the state.  
       }  
       !!!next-input-character;  
       !!!emit ($token);  
       redo A;  
     } elsif ($self->{state} == TAG_OPEN_STATE) {  
       if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA  
         if ($self->{nc} == 0x002F) { # /  
           !!!cp (15);  
           !!!next-input-character;  
           $self->{state} = CLOSE_TAG_OPEN_STATE;  
           redo A;  
         } elsif ($self->{nc} == 0x0021) { # !  
           !!!cp (15.1);  
           $self->{s_kwd} = '<' unless $self->{escape};  
           #  
         } else {  
           !!!cp (16);  
           #  
         }  
   
         ## reconsume  
         $self->{state} = DATA_STATE;  
         !!!emit ({type => CHARACTER_TOKEN, data => '<',  
                   line => $self->{line_prev},  
                   column => $self->{column_prev},  
                  });  
         redo A;  
       } elsif ($self->{content_model} & CM_FULL_MARKUP) { # PCDATA  
         if ($self->{nc} == 0x0021) { # !  
           !!!cp (17);  
           $self->{state} = MARKUP_DECLARATION_OPEN_STATE;  
           !!!next-input-character;  
           redo A;  
         } elsif ($self->{nc} == 0x002F) { # /  
           !!!cp (18);  
           $self->{state} = CLOSE_TAG_OPEN_STATE;  
           !!!next-input-character;  
           redo A;  
         } elsif (0x0041 <= $self->{nc} and  
                  $self->{nc} <= 0x005A) { # A..Z  
           !!!cp (19);  
           $self->{ct}  
             = {type => START_TAG_TOKEN,  
                tag_name => chr ($self->{nc} + 0x0020),  
                line => $self->{line_prev},  
                column => $self->{column_prev}};  
           $self->{state} = TAG_NAME_STATE;  
           !!!next-input-character;  
           redo A;  
         } elsif (0x0061 <= $self->{nc} and  
                  $self->{nc} <= 0x007A) { # a..z  
           !!!cp (20);  
           $self->{ct} = {type => START_TAG_TOKEN,  
                                     tag_name => chr ($self->{nc}),  
                                     line => $self->{line_prev},  
                                     column => $self->{column_prev}};  
           $self->{state} = TAG_NAME_STATE;  
           !!!next-input-character;  
           redo A;  
         } elsif ($self->{nc} == 0x003E) { # >  
           !!!cp (21);  
           !!!parse-error (type => 'empty start tag',  
                           line => $self->{line_prev},  
                           column => $self->{column_prev});  
           $self->{state} = DATA_STATE;  
           !!!next-input-character;  
   
           !!!emit ({type => CHARACTER_TOKEN, data => '<>',  
                     line => $self->{line_prev},  
                     column => $self->{column_prev},  
                    });  
   
           redo A;  
         } elsif ($self->{nc} == 0x003F) { # ?  
           !!!cp (22);  
           !!!parse-error (type => 'pio',  
                           line => $self->{line_prev},  
                           column => $self->{column_prev});  
           $self->{state} = BOGUS_COMMENT_STATE;  
           $self->{ct} = {type => COMMENT_TOKEN, data => '',  
                                     line => $self->{line_prev},  
                                     column => $self->{column_prev},  
                                    };  
           ## $self->{nc} is intentionally left as is  
           redo A;  
         } else {  
           !!!cp (23);  
           !!!parse-error (type => 'bare stago',  
                           line => $self->{line_prev},  
                           column => $self->{column_prev});  
           $self->{state} = DATA_STATE;  
           ## reconsume  
   
           !!!emit ({type => CHARACTER_TOKEN, data => '<',  
                     line => $self->{line_prev},  
                     column => $self->{column_prev},  
                    });  
   
           redo A;  
         }  
       } else {  
         die "$0: $self->{content_model} in tag open";  
       }  
     } elsif ($self->{state} == CLOSE_TAG_OPEN_STATE) {  
       ## NOTE: The "close tag open state" in the spec is implemented as  
       ## |CLOSE_TAG_OPEN_STATE| and |CDATA_RCDATA_CLOSE_TAG_STATE|.  
   
       my ($l, $c) = ($self->{line_prev}, $self->{column_prev} - 1); # "<"of"</"  
       if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA  
         if (defined $self->{last_stag_name}) {  
           $self->{state} = CDATA_RCDATA_CLOSE_TAG_STATE;  
           $self->{s_kwd} = '';  
           ## Reconsume.  
           redo A;  
         } else {  
           ## No start tag token has ever been emitted  
           ## NOTE: See <http://krijnhoetmer.nl/irc-logs/whatwg/20070626#l-564>.  
           !!!cp (28);  
           $self->{state} = DATA_STATE;  
           ## Reconsume.  
           !!!emit ({type => CHARACTER_TOKEN, data => '</',  
                     line => $l, column => $c,  
                    });  
           redo A;  
         }  
       }  
   
       if (0x0041 <= $self->{nc} and  
           $self->{nc} <= 0x005A) { # A..Z  
         !!!cp (29);  
         $self->{ct}  
             = {type => END_TAG_TOKEN,  
                tag_name => chr ($self->{nc} + 0x0020),  
                line => $l, column => $c};  
         $self->{state} = TAG_NAME_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif (0x0061 <= $self->{nc} and  
                $self->{nc} <= 0x007A) { # a..z  
         !!!cp (30);  
         $self->{ct} = {type => END_TAG_TOKEN,  
                                   tag_name => chr ($self->{nc}),  
                                   line => $l, column => $c};  
         $self->{state} = TAG_NAME_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         !!!cp (31);  
         !!!parse-error (type => 'empty end tag',  
                         line => $self->{line_prev}, ## "<" in "</>"  
                         column => $self->{column_prev} - 1);  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (32);  
         !!!parse-error (type => 'bare etago');  
         $self->{state} = DATA_STATE;  
         # reconsume  
   
         !!!emit ({type => CHARACTER_TOKEN, data => '</',  
                   line => $l, column => $c,  
                  });  
   
         redo A;  
       } else {  
         !!!cp (33);  
         !!!parse-error (type => 'bogus end tag');  
         $self->{state} = BOGUS_COMMENT_STATE;  
         $self->{ct} = {type => COMMENT_TOKEN, data => '',  
                                   line => $self->{line_prev}, # "<" of "</"  
                                   column => $self->{column_prev} - 1,  
                                  };  
         ## NOTE: $self->{nc} is intentionally left as is.  
         ## Although the "anything else" case of the spec not explicitly  
         ## states that the next input character is to be reconsumed,  
         ## it will be included to the |data| of the comment token  
         ## generated from the bogus end tag, as defined in the  
         ## "bogus comment state" entry.  
         redo A;  
       }  
     } elsif ($self->{state} == CDATA_RCDATA_CLOSE_TAG_STATE) {  
       my $ch = substr $self->{last_stag_name}, length $self->{s_kwd}, 1;  
       if (length $ch) {  
         my $CH = $ch;  
         $ch =~ tr/a-z/A-Z/;  
         my $nch = chr $self->{nc};  
         if ($nch eq $ch or $nch eq $CH) {  
           !!!cp (24);  
           ## Stay in the state.  
           $self->{s_kwd} .= $nch;  
           !!!next-input-character;  
           redo A;  
         } else {  
           !!!cp (25);  
           $self->{state} = DATA_STATE;  
           ## Reconsume.  
           !!!emit ({type => CHARACTER_TOKEN,  
                     data => '</' . $self->{s_kwd},  
                     line => $self->{line_prev},  
                     column => $self->{column_prev} - 1 - length $self->{s_kwd},  
                    });  
           redo A;  
         }  
       } else { # after "<{tag-name}"  
         unless ($is_space->{$self->{nc}} or  
                 {  
                  0x003E => 1, # >  
                  0x002F => 1, # /  
                  -1 => 1, # EOF  
                 }->{$self->{nc}}) {  
           !!!cp (26);  
           ## Reconsume.  
           $self->{state} = DATA_STATE;  
           !!!emit ({type => CHARACTER_TOKEN,  
                     data => '</' . $self->{s_kwd},  
                     line => $self->{line_prev},  
                     column => $self->{column_prev} - 1 - length $self->{s_kwd},  
                    });  
           redo A;  
         } else {  
           !!!cp (27);  
           $self->{ct}  
               = {type => END_TAG_TOKEN,  
                  tag_name => $self->{last_stag_name},  
                  line => $self->{line_prev},  
                  column => $self->{column_prev} - 1 - length $self->{s_kwd}};  
           $self->{state} = TAG_NAME_STATE;  
           ## Reconsume.  
           redo A;  
         }  
       }  
     } elsif ($self->{state} == TAG_NAME_STATE) {  
       if ($is_space->{$self->{nc}}) {  
         !!!cp (34);  
         $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (35);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           #if ($self->{ct}->{attributes}) {  
           #  ## NOTE: This should never be reached.  
           #  !!! cp (36);  
           #  !!! parse-error (type => 'end tag attribute');  
           #} else {  
             !!!cp (37);  
           #}  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } elsif (0x0041 <= $self->{nc} and  
                $self->{nc} <= 0x005A) { # A..Z  
         !!!cp (38);  
         $self->{ct}->{tag_name} .= chr ($self->{nc} + 0x0020);  
           # start tag or end tag  
         ## Stay in this state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!parse-error (type => 'unclosed tag');  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (39);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           #if ($self->{ct}->{attributes}) {  
           #  ## NOTE: This state should never be reached.  
           #  !!! cp (40);  
           #  !!! parse-error (type => 'end tag attribute');  
           #} else {  
             !!!cp (41);  
           #}  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         # reconsume  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } elsif ($self->{nc} == 0x002F) { # /  
         !!!cp (42);  
         $self->{state} = SELF_CLOSING_START_TAG_STATE;  
         !!!next-input-character;  
         redo A;  
       } else {  
         !!!cp (44);  
         $self->{ct}->{tag_name} .= chr $self->{nc};  
           # start tag or end tag  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == BEFORE_ATTRIBUTE_NAME_STATE) {  
       if ($is_space->{$self->{nc}}) {  
         !!!cp (45);  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (46);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           if ($self->{ct}->{attributes}) {  
             !!!cp (47);  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             !!!cp (48);  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } elsif (0x0041 <= $self->{nc} and  
                $self->{nc} <= 0x005A) { # A..Z  
         !!!cp (49);  
         $self->{ca}  
             = {name => chr ($self->{nc} + 0x0020),  
                value => '',  
                line => $self->{line}, column => $self->{column}};  
         $self->{state} = ATTRIBUTE_NAME_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x002F) { # /  
         !!!cp (50);  
         $self->{state} = SELF_CLOSING_START_TAG_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!parse-error (type => 'unclosed tag');  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (52);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           if ($self->{ct}->{attributes}) {  
             !!!cp (53);  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             !!!cp (54);  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         # reconsume  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } else {  
         if ({  
              0x0022 => 1, # "  
              0x0027 => 1, # '  
              0x003D => 1, # =  
             }->{$self->{nc}}) {  
           !!!cp (55);  
           !!!parse-error (type => 'bad attribute name');  
         } else {  
           !!!cp (56);  
         }  
         $self->{ca}  
             = {name => chr ($self->{nc}),  
                value => '',  
                line => $self->{line}, column => $self->{column}};  
         $self->{state} = ATTRIBUTE_NAME_STATE;  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == ATTRIBUTE_NAME_STATE) {  
       my $before_leave = sub {  
         if (exists $self->{ct}->{attributes} # start tag or end tag  
             ->{$self->{ca}->{name}}) { # MUST  
           !!!cp (57);  
           !!!parse-error (type => 'duplicate attribute', text => $self->{ca}->{name}, line => $self->{ca}->{line}, column => $self->{ca}->{column});  
           ## Discard $self->{ca} # MUST  
         } else {  
           !!!cp (58);  
           $self->{ct}->{attributes}->{$self->{ca}->{name}}  
             = $self->{ca};  
         }  
       }; # $before_leave  
   
       if ($is_space->{$self->{nc}}) {  
         !!!cp (59);  
         $before_leave->();  
         $self->{state} = AFTER_ATTRIBUTE_NAME_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003D) { # =  
         !!!cp (60);  
         $before_leave->();  
         $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         $before_leave->();  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (61);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           !!!cp (62);  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           if ($self->{ct}->{attributes}) {  
             !!!parse-error (type => 'end tag attribute');  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } elsif (0x0041 <= $self->{nc} and  
                $self->{nc} <= 0x005A) { # A..Z  
         !!!cp (63);  
         $self->{ca}->{name} .= chr ($self->{nc} + 0x0020);  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x002F) { # /  
         !!!cp (64);  
         $before_leave->();  
         $self->{state} = SELF_CLOSING_START_TAG_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!parse-error (type => 'unclosed tag');  
         $before_leave->();  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (66);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           if ($self->{ct}->{attributes}) {  
             !!!cp (67);  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             ## NOTE: This state should never be reached.  
             !!!cp (68);  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         # reconsume  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } else {  
         if ($self->{nc} == 0x0022 or # "  
             $self->{nc} == 0x0027) { # '  
           !!!cp (69);  
           !!!parse-error (type => 'bad attribute name');  
         } else {  
           !!!cp (70);  
         }  
         $self->{ca}->{name} .= chr ($self->{nc});  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == AFTER_ATTRIBUTE_NAME_STATE) {  
       if ($is_space->{$self->{nc}}) {  
         !!!cp (71);  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003D) { # =  
         !!!cp (72);  
         $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (73);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           if ($self->{ct}->{attributes}) {  
             !!!cp (74);  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             ## NOTE: This state should never be reached.  
             !!!cp (75);  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } elsif (0x0041 <= $self->{nc} and  
                $self->{nc} <= 0x005A) { # A..Z  
         !!!cp (76);  
         $self->{ca}  
             = {name => chr ($self->{nc} + 0x0020),  
                value => '',  
                line => $self->{line}, column => $self->{column}};  
         $self->{state} = ATTRIBUTE_NAME_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x002F) { # /  
         !!!cp (77);  
         $self->{state} = SELF_CLOSING_START_TAG_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!parse-error (type => 'unclosed tag');  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (79);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           if ($self->{ct}->{attributes}) {  
             !!!cp (80);  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             ## NOTE: This state should never be reached.  
             !!!cp (81);  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         # reconsume  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } else {  
         if ($self->{nc} == 0x0022 or # "  
             $self->{nc} == 0x0027) { # '  
           !!!cp (78);  
           !!!parse-error (type => 'bad attribute name');  
         } else {  
           !!!cp (82);  
         }  
         $self->{ca}  
             = {name => chr ($self->{nc}),  
                value => '',  
                line => $self->{line}, column => $self->{column}};  
         $self->{state} = ATTRIBUTE_NAME_STATE;  
         !!!next-input-character;  
         redo A;          
       }  
     } elsif ($self->{state} == BEFORE_ATTRIBUTE_VALUE_STATE) {  
       if ($is_space->{$self->{nc}}) {  
         !!!cp (83);  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x0022) { # "  
         !!!cp (84);  
         $self->{state} = ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x0026) { # &  
         !!!cp (85);  
         $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;  
         ## reconsume  
         redo A;  
       } elsif ($self->{nc} == 0x0027) { # '  
         !!!cp (86);  
         $self->{state} = ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         !!!parse-error (type => 'empty unquoted attribute value');  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (87);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           if ($self->{ct}->{attributes}) {  
             !!!cp (88);  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             ## NOTE: This state should never be reached.  
             !!!cp (89);  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!parse-error (type => 'unclosed tag');  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (90);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           if ($self->{ct}->{attributes}) {  
             !!!cp (91);  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             ## NOTE: This state should never be reached.  
             !!!cp (92);  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } else {  
         if ($self->{nc} == 0x003D) { # =  
           !!!cp (93);  
           !!!parse-error (type => 'bad attribute value');  
         } else {  
           !!!cp (94);  
         }  
         $self->{ca}->{value} .= chr ($self->{nc});  
         $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE) {  
       if ($self->{nc} == 0x0022) { # "  
         !!!cp (95);  
         $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x0026) { # &  
         !!!cp (96);  
         ## NOTE: In the spec, the tokenizer is switched to the  
         ## "entity in attribute value state".  In this implementation, the  
         ## tokenizer is switched to the |ENTITY_STATE|, which is an  
         ## implementation of the "consume a character reference" algorithm.  
         $self->{prev_state} = $self->{state};  
         $self->{entity_add} = 0x0022; # "  
         $self->{state} = ENTITY_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!parse-error (type => 'unclosed attribute value');  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (97);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           if ($self->{ct}->{attributes}) {  
             !!!cp (98);  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             ## NOTE: This state should never be reached.  
             !!!cp (99);  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } else {  
         !!!cp (100);  
         $self->{ca}->{value} .= chr ($self->{nc});  
         $self->{read_until}->($self->{ca}->{value},  
                               q["&],  
                               length $self->{ca}->{value});  
   
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE) {  
       if ($self->{nc} == 0x0027) { # '  
         !!!cp (101);  
         $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x0026) { # &  
         !!!cp (102);  
         ## NOTE: In the spec, the tokenizer is switched to the  
         ## "entity in attribute value state".  In this implementation, the  
         ## tokenizer is switched to the |ENTITY_STATE|, which is an  
         ## implementation of the "consume a character reference" algorithm.  
         $self->{entity_add} = 0x0027; # '  
         $self->{prev_state} = $self->{state};  
         $self->{state} = ENTITY_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!parse-error (type => 'unclosed attribute value');  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (103);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           if ($self->{ct}->{attributes}) {  
             !!!cp (104);  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             ## NOTE: This state should never be reached.  
             !!!cp (105);  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } else {  
         !!!cp (106);  
         $self->{ca}->{value} .= chr ($self->{nc});  
         $self->{read_until}->($self->{ca}->{value},  
                               q['&],  
                               length $self->{ca}->{value});  
   
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == ATTRIBUTE_VALUE_UNQUOTED_STATE) {  
       if ($is_space->{$self->{nc}}) {  
         !!!cp (107);  
         $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x0026) { # &  
         !!!cp (108);  
         ## NOTE: In the spec, the tokenizer is switched to the  
         ## "entity in attribute value state".  In this implementation, the  
         ## tokenizer is switched to the |ENTITY_STATE|, which is an  
         ## implementation of the "consume a character reference" algorithm.  
         $self->{entity_add} = -1;  
         $self->{prev_state} = $self->{state};  
         $self->{state} = ENTITY_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (109);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           if ($self->{ct}->{attributes}) {  
             !!!cp (110);  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             ## NOTE: This state should never be reached.  
             !!!cp (111);  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!parse-error (type => 'unclosed tag');  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (112);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           if ($self->{ct}->{attributes}) {  
             !!!cp (113);  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             ## NOTE: This state should never be reached.  
             !!!cp (114);  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } else {  
         if ({  
              0x0022 => 1, # "  
              0x0027 => 1, # '  
              0x003D => 1, # =  
             }->{$self->{nc}}) {  
           !!!cp (115);  
           !!!parse-error (type => 'bad attribute value');  
         } else {  
           !!!cp (116);  
         }  
         $self->{ca}->{value} .= chr ($self->{nc});  
         $self->{read_until}->($self->{ca}->{value},  
                               q["'=& >],  
                               length $self->{ca}->{value});  
   
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == AFTER_ATTRIBUTE_VALUE_QUOTED_STATE) {  
       if ($is_space->{$self->{nc}}) {  
         !!!cp (118);  
         $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (119);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           if ($self->{ct}->{attributes}) {  
             !!!cp (120);  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             ## NOTE: This state should never be reached.  
             !!!cp (121);  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } elsif ($self->{nc} == 0x002F) { # /  
         !!!cp (122);  
         $self->{state} = SELF_CLOSING_START_TAG_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!parse-error (type => 'unclosed tag');  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (122.3);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           if ($self->{ct}->{attributes}) {  
             !!!cp (122.1);  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             ## NOTE: This state should never be reached.  
             !!!cp (122.2);  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         ## Reconsume.  
         !!!emit ($self->{ct}); # start tag or end tag  
         redo A;  
       } else {  
         !!!cp ('124.1');  
         !!!parse-error (type => 'no space between attributes');  
         $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;  
         ## reconsume  
         redo A;  
       }  
     } elsif ($self->{state} == SELF_CLOSING_START_TAG_STATE) {  
       if ($self->{nc} == 0x003E) { # >  
         if ($self->{ct}->{type} == END_TAG_TOKEN) {  
           !!!cp ('124.2');  
           !!!parse-error (type => 'nestc', token => $self->{ct});  
           ## TODO: Different type than slash in start tag  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           if ($self->{ct}->{attributes}) {  
             !!!cp ('124.4');  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             !!!cp ('124.5');  
           }  
           ## TODO: Test |<title></title/>|  
         } else {  
           !!!cp ('124.3');  
           $self->{self_closing} = 1;  
         }  
   
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!parse-error (type => 'unclosed tag');  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (124.7);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           if ($self->{ct}->{attributes}) {  
             !!!cp (124.5);  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             ## NOTE: This state should never be reached.  
             !!!cp (124.6);  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         ## Reconsume.  
         !!!emit ($self->{ct}); # start tag or end tag  
         redo A;  
       } else {  
         !!!cp ('124.4');  
         !!!parse-error (type => 'nestc');  
         ## TODO: This error type is wrong.  
         $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;  
         ## Reconsume.  
         redo A;  
       }  
     } elsif ($self->{state} == BOGUS_COMMENT_STATE) {  
       ## (only happen if PCDATA state)  
   
       ## NOTE: Unlike spec's "bogus comment state", this implementation  
       ## consumes characters one-by-one basis.  
         
       if ($self->{nc} == 0x003E) { # >  
         !!!cp (124);  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # comment  
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (125);  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         !!!emit ($self->{ct}); # comment  
         redo A;  
       } else {  
         !!!cp (126);  
         $self->{ct}->{data} .= chr ($self->{nc}); # comment  
         $self->{read_until}->($self->{ct}->{data},  
                               q[>],  
                               length $self->{ct}->{data});  
   
         ## Stay in the state.  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == MARKUP_DECLARATION_OPEN_STATE) {  
       ## (only happen if PCDATA state)  
         
       if ($self->{nc} == 0x002D) { # -  
         !!!cp (133);  
         $self->{state} = MD_HYPHEN_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x0044 or # D  
                $self->{nc} == 0x0064) { # d  
         ## ASCII case-insensitive.  
         !!!cp (130);  
         $self->{state} = MD_DOCTYPE_STATE;  
         $self->{s_kwd} = chr $self->{nc};  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM and  
                $self->{open_elements}->[-1]->[1] & FOREIGN_EL and  
                $self->{nc} == 0x005B) { # [  
         !!!cp (135.4);                  
         $self->{state} = MD_CDATA_STATE;  
         $self->{s_kwd} = '[';  
         !!!next-input-character;  
         redo A;  
       } else {  
         !!!cp (136);  
       }  
   
       !!!parse-error (type => 'bogus comment',  
                       line => $self->{line_prev},  
                       column => $self->{column_prev} - 1);  
       ## Reconsume.  
       $self->{state} = BOGUS_COMMENT_STATE;  
       $self->{ct} = {type => COMMENT_TOKEN, data => '',  
                                 line => $self->{line_prev},  
                                 column => $self->{column_prev} - 1,  
                                };  
       redo A;  
     } elsif ($self->{state} == MD_HYPHEN_STATE) {  
       if ($self->{nc} == 0x002D) { # -  
         !!!cp (127);  
         $self->{ct} = {type => COMMENT_TOKEN, data => '',  
                                   line => $self->{line_prev},  
                                   column => $self->{column_prev} - 2,  
                                  };  
         $self->{state} = COMMENT_START_STATE;  
         !!!next-input-character;  
         redo A;  
       } else {  
         !!!cp (128);  
         !!!parse-error (type => 'bogus comment',  
                         line => $self->{line_prev},  
                         column => $self->{column_prev} - 2);  
         $self->{state} = BOGUS_COMMENT_STATE;  
         ## Reconsume.  
         $self->{ct} = {type => COMMENT_TOKEN,  
                                   data => '-',  
                                   line => $self->{line_prev},  
                                   column => $self->{column_prev} - 2,  
                                  };  
         redo A;  
       }  
     } elsif ($self->{state} == MD_DOCTYPE_STATE) {  
       ## ASCII case-insensitive.  
       if ($self->{nc} == [  
             undef,  
             0x004F, # O  
             0x0043, # C  
             0x0054, # T  
             0x0059, # Y  
             0x0050, # P  
           ]->[length $self->{s_kwd}] or  
           $self->{nc} == [  
             undef,  
             0x006F, # o  
             0x0063, # c  
             0x0074, # t  
             0x0079, # y  
             0x0070, # p  
           ]->[length $self->{s_kwd}]) {  
         !!!cp (131);  
         ## Stay in the state.  
         $self->{s_kwd} .= chr $self->{nc};  
         !!!next-input-character;  
         redo A;  
       } elsif ((length $self->{s_kwd}) == 6 and  
                ($self->{nc} == 0x0045 or # E  
                 $self->{nc} == 0x0065)) { # e  
         !!!cp (129);  
         $self->{state} = DOCTYPE_STATE;  
         $self->{ct} = {type => DOCTYPE_TOKEN,  
                                   quirks => 1,  
                                   line => $self->{line_prev},  
                                   column => $self->{column_prev} - 7,  
                                  };  
         !!!next-input-character;  
         redo A;  
       } else {  
         !!!cp (132);          
         !!!parse-error (type => 'bogus comment',  
                         line => $self->{line_prev},  
                         column => $self->{column_prev} - 1 - length $self->{s_kwd});  
         $self->{state} = BOGUS_COMMENT_STATE;  
         ## Reconsume.  
         $self->{ct} = {type => COMMENT_TOKEN,  
                                   data => $self->{s_kwd},  
                                   line => $self->{line_prev},  
                                   column => $self->{column_prev} - 1 - length $self->{s_kwd},  
                                  };  
         redo A;  
       }  
     } elsif ($self->{state} == MD_CDATA_STATE) {  
       if ($self->{nc} == {  
             '[' => 0x0043, # C  
             '[C' => 0x0044, # D  
             '[CD' => 0x0041, # A  
             '[CDA' => 0x0054, # T  
             '[CDAT' => 0x0041, # A  
           }->{$self->{s_kwd}}) {  
         !!!cp (135.1);  
         ## Stay in the state.  
         $self->{s_kwd} .= chr $self->{nc};  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{s_kwd} eq '[CDATA' and  
                $self->{nc} == 0x005B) { # [  
         !!!cp (135.2);  
         $self->{ct} = {type => CHARACTER_TOKEN,  
                                   data => '',  
                                   line => $self->{line_prev},  
                                   column => $self->{column_prev} - 7};  
         $self->{state} = CDATA_SECTION_STATE;  
         !!!next-input-character;  
         redo A;  
       } else {  
         !!!cp (135.3);  
         !!!parse-error (type => 'bogus comment',  
                         line => $self->{line_prev},  
                         column => $self->{column_prev} - 1 - length $self->{s_kwd});  
         $self->{state} = BOGUS_COMMENT_STATE;  
         ## Reconsume.  
         $self->{ct} = {type => COMMENT_TOKEN,  
                                   data => $self->{s_kwd},  
                                   line => $self->{line_prev},  
                                   column => $self->{column_prev} - 1 - length $self->{s_kwd},  
                                  };  
         redo A;  
       }  
     } elsif ($self->{state} == COMMENT_START_STATE) {  
       if ($self->{nc} == 0x002D) { # -  
         !!!cp (137);  
         $self->{state} = COMMENT_START_DASH_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         !!!cp (138);  
         !!!parse-error (type => 'bogus comment');  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # comment  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (139);  
         !!!parse-error (type => 'unclosed comment');  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         !!!emit ($self->{ct}); # comment  
   
         redo A;  
       } else {  
         !!!cp (140);  
         $self->{ct}->{data} # comment  
             .= chr ($self->{nc});  
         $self->{state} = COMMENT_STATE;  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == COMMENT_START_DASH_STATE) {  
       if ($self->{nc} == 0x002D) { # -  
         !!!cp (141);  
         $self->{state} = COMMENT_END_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         !!!cp (142);  
         !!!parse-error (type => 'bogus comment');  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # comment  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (143);  
         !!!parse-error (type => 'unclosed comment');  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         !!!emit ($self->{ct}); # comment  
   
         redo A;  
       } else {  
         !!!cp (144);  
         $self->{ct}->{data} # comment  
             .= '-' . chr ($self->{nc});  
         $self->{state} = COMMENT_STATE;  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == COMMENT_STATE) {  
       if ($self->{nc} == 0x002D) { # -  
         !!!cp (145);  
         $self->{state} = COMMENT_END_DASH_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (146);  
         !!!parse-error (type => 'unclosed comment');  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         !!!emit ($self->{ct}); # comment  
   
         redo A;  
       } else {  
         !!!cp (147);  
         $self->{ct}->{data} .= chr ($self->{nc}); # comment  
         $self->{read_until}->($self->{ct}->{data},  
                               q[-],  
                               length $self->{ct}->{data});  
   
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == COMMENT_END_DASH_STATE) {  
       if ($self->{nc} == 0x002D) { # -  
         !!!cp (148);  
         $self->{state} = COMMENT_END_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (149);  
         !!!parse-error (type => 'unclosed comment');  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         !!!emit ($self->{ct}); # comment  
   
         redo A;  
       } else {  
         !!!cp (150);  
         $self->{ct}->{data} .= '-' . chr ($self->{nc}); # comment  
         $self->{state} = COMMENT_STATE;  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == COMMENT_END_STATE) {  
       if ($self->{nc} == 0x003E) { # >  
         !!!cp (151);  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # comment  
   
         redo A;  
       } elsif ($self->{nc} == 0x002D) { # -  
         !!!cp (152);  
         !!!parse-error (type => 'dash in comment',  
                         line => $self->{line_prev},  
                         column => $self->{column_prev});  
         $self->{ct}->{data} .= '-'; # comment  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (153);  
         !!!parse-error (type => 'unclosed comment');  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         !!!emit ($self->{ct}); # comment  
   
         redo A;  
       } else {  
         !!!cp (154);  
         !!!parse-error (type => 'dash in comment',  
                         line => $self->{line_prev},  
                         column => $self->{column_prev});  
         $self->{ct}->{data} .= '--' . chr ($self->{nc}); # comment  
         $self->{state} = COMMENT_STATE;  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == DOCTYPE_STATE) {  
       if ($is_space->{$self->{nc}}) {  
         !!!cp (155);  
         $self->{state} = BEFORE_DOCTYPE_NAME_STATE;  
         !!!next-input-character;  
         redo A;  
       } else {  
         !!!cp (156);  
         !!!parse-error (type => 'no space before DOCTYPE name');  
         $self->{state} = BEFORE_DOCTYPE_NAME_STATE;  
         ## reconsume  
         redo A;  
       }  
     } elsif ($self->{state} == BEFORE_DOCTYPE_NAME_STATE) {  
       if ($is_space->{$self->{nc}}) {  
         !!!cp (157);  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         !!!cp (158);  
         !!!parse-error (type => 'no DOCTYPE name');  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # DOCTYPE (quirks)  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (159);  
         !!!parse-error (type => 'no DOCTYPE name');  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         !!!emit ($self->{ct}); # DOCTYPE (quirks)  
   
         redo A;  
       } else {  
         !!!cp (160);  
         $self->{ct}->{name} = chr $self->{nc};  
         delete $self->{ct}->{quirks};  
 ## ISSUE: "Set the token's name name to the" in the spec  
         $self->{state} = DOCTYPE_NAME_STATE;  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == DOCTYPE_NAME_STATE) {  
 ## ISSUE: Redundant "First," in the spec.  
       if ($is_space->{$self->{nc}}) {  
         !!!cp (161);  
         $self->{state} = AFTER_DOCTYPE_NAME_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         !!!cp (162);  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (163);  
         !!!parse-error (type => 'unclosed DOCTYPE');  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } else {  
         !!!cp (164);  
         $self->{ct}->{name}  
           .= chr ($self->{nc}); # DOCTYPE  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == AFTER_DOCTYPE_NAME_STATE) {  
       if ($is_space->{$self->{nc}}) {  
         !!!cp (165);  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         !!!cp (166);  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (167);  
         !!!parse-error (type => 'unclosed DOCTYPE');  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } elsif ($self->{nc} == 0x0050 or # P  
                $self->{nc} == 0x0070) { # p  
         $self->{state} = PUBLIC_STATE;  
         $self->{s_kwd} = chr $self->{nc};  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x0053 or # S  
                $self->{nc} == 0x0073) { # s  
         $self->{state} = SYSTEM_STATE;  
         $self->{s_kwd} = chr $self->{nc};  
         !!!next-input-character;  
         redo A;  
       } else {  
         !!!cp (180);  
         !!!parse-error (type => 'string after DOCTYPE name');  
         $self->{ct}->{quirks} = 1;  
   
         $self->{state} = BOGUS_DOCTYPE_STATE;  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == PUBLIC_STATE) {  
       ## ASCII case-insensitive  
       if ($self->{nc} == [  
             undef,  
             0x0055, # U  
             0x0042, # B  
             0x004C, # L  
             0x0049, # I  
           ]->[length $self->{s_kwd}] or  
           $self->{nc} == [  
             undef,  
             0x0075, # u  
             0x0062, # b  
             0x006C, # l  
             0x0069, # i  
           ]->[length $self->{s_kwd}]) {  
         !!!cp (175);  
         ## Stay in the state.  
         $self->{s_kwd} .= chr $self->{nc};  
         !!!next-input-character;  
         redo A;  
       } elsif ((length $self->{s_kwd}) == 5 and  
                ($self->{nc} == 0x0043 or # C  
                 $self->{nc} == 0x0063)) { # c  
         !!!cp (168);  
         $self->{state} = BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE;  
         !!!next-input-character;  
         redo A;  
       } else {  
         !!!cp (169);  
         !!!parse-error (type => 'string after DOCTYPE name',  
                         line => $self->{line_prev},  
                         column => $self->{column_prev} + 1 - length $self->{s_kwd});  
         $self->{ct}->{quirks} = 1;  
   
         $self->{state} = BOGUS_DOCTYPE_STATE;  
         ## Reconsume.  
         redo A;  
       }  
     } elsif ($self->{state} == SYSTEM_STATE) {  
       ## ASCII case-insensitive  
       if ($self->{nc} == [  
             undef,  
             0x0059, # Y  
             0x0053, # S  
             0x0054, # T  
             0x0045, # E  
           ]->[length $self->{s_kwd}] or  
           $self->{nc} == [  
             undef,  
             0x0079, # y  
             0x0073, # s  
             0x0074, # t  
             0x0065, # e  
           ]->[length $self->{s_kwd}]) {  
         !!!cp (170);  
         ## Stay in the state.  
         $self->{s_kwd} .= chr $self->{nc};  
         !!!next-input-character;  
         redo A;  
       } elsif ((length $self->{s_kwd}) == 5 and  
                ($self->{nc} == 0x004D or # M  
                 $self->{nc} == 0x006D)) { # m  
         !!!cp (171);  
         $self->{state} = BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE;  
         !!!next-input-character;  
         redo A;  
       } else {  
         !!!cp (172);  
         !!!parse-error (type => 'string after DOCTYPE name',  
                         line => $self->{line_prev},  
                         column => $self->{column_prev} + 1 - length $self->{s_kwd});  
         $self->{ct}->{quirks} = 1;  
   
         $self->{state} = BOGUS_DOCTYPE_STATE;  
         ## Reconsume.  
         redo A;  
       }  
     } elsif ($self->{state} == BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {  
       if ($is_space->{$self->{nc}}) {  
         !!!cp (181);  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} eq 0x0022) { # "  
         !!!cp (182);  
         $self->{ct}->{pubid} = ''; # DOCTYPE  
         $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} eq 0x0027) { # '  
         !!!cp (183);  
         $self->{ct}->{pubid} = ''; # DOCTYPE  
         $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} eq 0x003E) { # >  
         !!!cp (184);  
         !!!parse-error (type => 'no PUBLIC literal');  
   
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (185);  
         !!!parse-error (type => 'unclosed DOCTYPE');  
   
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } else {  
         !!!cp (186);  
         !!!parse-error (type => 'string after PUBLIC');  
         $self->{ct}->{quirks} = 1;  
   
         $self->{state} = BOGUS_DOCTYPE_STATE;  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE) {  
       if ($self->{nc} == 0x0022) { # "  
         !!!cp (187);  
         $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         !!!cp (188);  
         !!!parse-error (type => 'unclosed PUBLIC literal');  
   
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (189);  
         !!!parse-error (type => 'unclosed PUBLIC literal');  
   
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } else {  
         !!!cp (190);  
         $self->{ct}->{pubid} # DOCTYPE  
             .= chr $self->{nc};  
         $self->{read_until}->($self->{ct}->{pubid}, q[">],  
                               length $self->{ct}->{pubid});  
   
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE) {  
       if ($self->{nc} == 0x0027) { # '  
         !!!cp (191);  
         $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         !!!cp (192);  
         !!!parse-error (type => 'unclosed PUBLIC literal');  
   
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (193);  
         !!!parse-error (type => 'unclosed PUBLIC literal');  
   
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } else {  
         !!!cp (194);  
         $self->{ct}->{pubid} # DOCTYPE  
             .= chr $self->{nc};  
         $self->{read_until}->($self->{ct}->{pubid}, q['>],  
                               length $self->{ct}->{pubid});  
   
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {  
       if ($is_space->{$self->{nc}}) {  
         !!!cp (195);  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x0022) { # "  
         !!!cp (196);  
         $self->{ct}->{sysid} = ''; # DOCTYPE  
         $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x0027) { # '  
         !!!cp (197);  
         $self->{ct}->{sysid} = ''; # DOCTYPE  
         $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         !!!cp (198);  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (199);  
         !!!parse-error (type => 'unclosed DOCTYPE');  
   
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } else {  
         !!!cp (200);  
         !!!parse-error (type => 'string after PUBLIC literal');  
         $self->{ct}->{quirks} = 1;  
   
         $self->{state} = BOGUS_DOCTYPE_STATE;  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {  
       if ($is_space->{$self->{nc}}) {  
         !!!cp (201);  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x0022) { # "  
         !!!cp (202);  
         $self->{ct}->{sysid} = ''; # DOCTYPE  
         $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x0027) { # '  
         !!!cp (203);  
         $self->{ct}->{sysid} = ''; # DOCTYPE  
         $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         !!!cp (204);  
         !!!parse-error (type => 'no SYSTEM literal');  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (205);  
         !!!parse-error (type => 'unclosed DOCTYPE');  
   
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } else {  
         !!!cp (206);  
         !!!parse-error (type => 'string after SYSTEM');  
         $self->{ct}->{quirks} = 1;  
   
         $self->{state} = BOGUS_DOCTYPE_STATE;  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE) {  
       if ($self->{nc} == 0x0022) { # "  
         !!!cp (207);  
         $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         !!!cp (208);  
         !!!parse-error (type => 'unclosed SYSTEM literal');  
   
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (209);  
         !!!parse-error (type => 'unclosed SYSTEM literal');  
   
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } else {  
         !!!cp (210);  
         $self->{ct}->{sysid} # DOCTYPE  
             .= chr $self->{nc};  
         $self->{read_until}->($self->{ct}->{sysid}, q[">],  
                               length $self->{ct}->{sysid});  
   
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE) {  
       if ($self->{nc} == 0x0027) { # '  
         !!!cp (211);  
         $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         !!!cp (212);  
         !!!parse-error (type => 'unclosed SYSTEM literal');  
   
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (213);  
         !!!parse-error (type => 'unclosed SYSTEM literal');  
   
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } else {  
         !!!cp (214);  
         $self->{ct}->{sysid} # DOCTYPE  
             .= chr $self->{nc};  
         $self->{read_until}->($self->{ct}->{sysid}, q['>],  
                               length $self->{ct}->{sysid});  
   
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {  
       if ($is_space->{$self->{nc}}) {  
         !!!cp (215);  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         !!!cp (216);  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (217);  
         !!!parse-error (type => 'unclosed DOCTYPE');  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } else {  
         !!!cp (218);  
         !!!parse-error (type => 'string after SYSTEM literal');  
         #$self->{ct}->{quirks} = 1;  
   
         $self->{state} = BOGUS_DOCTYPE_STATE;  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == BOGUS_DOCTYPE_STATE) {  
       if ($self->{nc} == 0x003E) { # >  
         !!!cp (219);  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (220);  
         !!!parse-error (type => 'unclosed DOCTYPE');  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } else {  
         !!!cp (221);  
         my $s = '';  
         $self->{read_until}->($s, q[>], 0);  
   
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == CDATA_SECTION_STATE) {  
       ## NOTE: "CDATA section state" in the state is jointly implemented  
       ## by three states, |CDATA_SECTION_STATE|, |CDATA_SECTION_MSE1_STATE|,  
       ## and |CDATA_SECTION_MSE2_STATE|.  
         
       if ($self->{nc} == 0x005D) { # ]  
         !!!cp (221.1);  
         $self->{state} = CDATA_SECTION_MSE1_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == -1) {  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
         if (length $self->{ct}->{data}) { # character  
           !!!cp (221.2);  
           !!!emit ($self->{ct}); # character  
         } else {  
           !!!cp (221.3);  
           ## No token to emit. $self->{ct} is discarded.  
         }          
         redo A;  
       } else {  
         !!!cp (221.4);  
         $self->{ct}->{data} .= chr $self->{nc};  
         $self->{read_until}->($self->{ct}->{data},  
                               q<]>,  
                               length $self->{ct}->{data});  
   
         ## Stay in the state.  
         !!!next-input-character;  
         redo A;  
       }  
   
       ## ISSUE: "text tokens" in spec.  
     } elsif ($self->{state} == CDATA_SECTION_MSE1_STATE) {  
       if ($self->{nc} == 0x005D) { # ]  
         !!!cp (221.5);  
         $self->{state} = CDATA_SECTION_MSE2_STATE;  
         !!!next-input-character;  
         redo A;  
       } else {  
         !!!cp (221.6);  
         $self->{ct}->{data} .= ']';  
         $self->{state} = CDATA_SECTION_STATE;  
         ## Reconsume.  
         redo A;  
       }  
     } elsif ($self->{state} == CDATA_SECTION_MSE2_STATE) {  
       if ($self->{nc} == 0x003E) { # >  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
         if (length $self->{ct}->{data}) { # character  
           !!!cp (221.7);  
           !!!emit ($self->{ct}); # character  
         } else {  
           !!!cp (221.8);  
           ## No token to emit. $self->{ct} is discarded.  
         }  
         redo A;  
       } elsif ($self->{nc} == 0x005D) { # ]  
         !!!cp (221.9); # character  
         $self->{ct}->{data} .= ']'; ## Add first "]" of "]]]".  
         ## Stay in the state.  
         !!!next-input-character;  
         redo A;  
       } else {  
         !!!cp (221.11);  
         $self->{ct}->{data} .= ']]'; # character  
         $self->{state} = CDATA_SECTION_STATE;  
         ## Reconsume.  
         redo A;  
       }  
     } elsif ($self->{state} == ENTITY_STATE) {  
       if ($is_space->{$self->{nc}} or  
           {  
             0x003C => 1, 0x0026 => 1, -1 => 1, # <, &  
             $self->{entity_add} => 1,  
           }->{$self->{nc}}) {  
         !!!cp (1001);  
         ## Don't consume  
         ## No error  
         ## Return nothing.  
         #  
       } elsif ($self->{nc} == 0x0023) { # #  
         !!!cp (999);  
         $self->{state} = ENTITY_HASH_STATE;  
         $self->{s_kwd} = '#';  
         !!!next-input-character;  
         redo A;  
       } elsif ((0x0041 <= $self->{nc} and  
                 $self->{nc} <= 0x005A) or # A..Z  
                (0x0061 <= $self->{nc} and  
                 $self->{nc} <= 0x007A)) { # a..z  
         !!!cp (998);  
         require Whatpm::_NamedEntityList;  
         $self->{state} = ENTITY_NAME_STATE;  
         $self->{s_kwd} = chr $self->{nc};  
         $self->{entity__value} = $self->{s_kwd};  
         $self->{entity__match} = 0;  
         !!!next-input-character;  
         redo A;  
       } else {  
         !!!cp (1027);  
         !!!parse-error (type => 'bare ero');  
         ## Return nothing.  
         #  
       }  
   
       ## NOTE: No character is consumed by the "consume a character  
       ## reference" algorithm.  In other word, there is an "&" character  
       ## that does not introduce a character reference, which would be  
       ## appended to the parent element or the attribute value in later  
       ## process of the tokenizer.  
   
       if ($self->{prev_state} == DATA_STATE) {  
         !!!cp (997);  
         $self->{state} = $self->{prev_state};  
         ## Reconsume.  
         !!!emit ({type => CHARACTER_TOKEN, data => '&',  
                   line => $self->{line_prev},  
                   column => $self->{column_prev},  
                  });  
         redo A;  
       } else {  
         !!!cp (996);  
         $self->{ca}->{value} .= '&';  
         $self->{state} = $self->{prev_state};  
         ## Reconsume.  
         redo A;  
       }  
     } elsif ($self->{state} == ENTITY_HASH_STATE) {  
       if ($self->{nc} == 0x0078 or # x  
           $self->{nc} == 0x0058) { # X  
         !!!cp (995);  
         $self->{state} = HEXREF_X_STATE;  
         $self->{s_kwd} .= chr $self->{nc};  
         !!!next-input-character;  
         redo A;  
       } elsif (0x0030 <= $self->{nc} and  
                $self->{nc} <= 0x0039) { # 0..9  
         !!!cp (994);  
         $self->{state} = NCR_NUM_STATE;  
         $self->{s_kwd} = $self->{nc} - 0x0030;  
         !!!next-input-character;  
         redo A;  
       } else {  
         !!!parse-error (type => 'bare nero',  
                         line => $self->{line_prev},  
                         column => $self->{column_prev} - 1);  
   
         ## NOTE: According to the spec algorithm, nothing is returned,  
         ## and then "&#" is appended to the parent element or the attribute  
         ## value in the later processing.  
   
         if ($self->{prev_state} == DATA_STATE) {  
           !!!cp (1019);  
           $self->{state} = $self->{prev_state};  
           ## Reconsume.  
           !!!emit ({type => CHARACTER_TOKEN,  
                     data => '&#',  
                     line => $self->{line_prev},  
                     column => $self->{column_prev} - 1,  
                    });  
           redo A;  
         } else {  
           !!!cp (993);  
           $self->{ca}->{value} .= '&#';  
           $self->{state} = $self->{prev_state};  
           ## Reconsume.  
           redo A;  
         }  
       }  
     } elsif ($self->{state} == NCR_NUM_STATE) {  
       if (0x0030 <= $self->{nc} and  
           $self->{nc} <= 0x0039) { # 0..9  
         !!!cp (1012);  
         $self->{s_kwd} *= 10;  
         $self->{s_kwd} += $self->{nc} - 0x0030;  
           
         ## Stay in the state.  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003B) { # ;  
         !!!cp (1013);  
         !!!next-input-character;  
         #  
       } else {  
         !!!cp (1014);  
         !!!parse-error (type => 'no refc');  
         ## Reconsume.  
         #  
       }  
   
       my $code = $self->{s_kwd};  
       my $l = $self->{line_prev};  
       my $c = $self->{column_prev};  
       if ($charref_map->{$code}) {  
         !!!cp (1015);  
         !!!parse-error (type => 'invalid character reference',  
                         text => (sprintf 'U+%04X', $code),  
                         line => $l, column => $c);  
         $code = $charref_map->{$code};  
       } elsif ($code > 0x10FFFF) {  
         !!!cp (1016);  
         !!!parse-error (type => 'invalid character reference',  
                         text => (sprintf 'U-%08X', $code),  
                         line => $l, column => $c);  
         $code = 0xFFFD;  
       }  
   
       if ($self->{prev_state} == DATA_STATE) {  
         !!!cp (992);  
         $self->{state} = $self->{prev_state};  
         ## Reconsume.  
         !!!emit ({type => CHARACTER_TOKEN, data => chr $code,  
                   line => $l, column => $c,  
                  });  
         redo A;  
       } else {  
         !!!cp (991);  
         $self->{ca}->{value} .= chr $code;  
         $self->{ca}->{has_reference} = 1;  
         $self->{state} = $self->{prev_state};  
         ## Reconsume.  
         redo A;  
       }  
     } elsif ($self->{state} == HEXREF_X_STATE) {  
       if ((0x0030 <= $self->{nc} and $self->{nc} <= 0x0039) or  
           (0x0041 <= $self->{nc} and $self->{nc} <= 0x0046) or  
           (0x0061 <= $self->{nc} and $self->{nc} <= 0x0066)) {  
         # 0..9, A..F, a..f  
         !!!cp (990);  
         $self->{state} = HEXREF_HEX_STATE;  
         $self->{s_kwd} = 0;  
         ## Reconsume.  
         redo A;  
       } else {  
         !!!parse-error (type => 'bare hcro',  
                         line => $self->{line_prev},  
                         column => $self->{column_prev} - 2);  
   
         ## NOTE: According to the spec algorithm, nothing is returned,  
         ## and then "&#" followed by "X" or "x" is appended to the parent  
         ## element or the attribute value in the later processing.  
   
         if ($self->{prev_state} == DATA_STATE) {  
           !!!cp (1005);  
           $self->{state} = $self->{prev_state};  
           ## Reconsume.  
           !!!emit ({type => CHARACTER_TOKEN,  
                     data => '&' . $self->{s_kwd},  
                     line => $self->{line_prev},  
                     column => $self->{column_prev} - length $self->{s_kwd},  
                    });  
           redo A;  
         } else {  
           !!!cp (989);  
           $self->{ca}->{value} .= '&' . $self->{s_kwd};  
           $self->{state} = $self->{prev_state};  
           ## Reconsume.  
           redo A;  
         }  
       }  
     } elsif ($self->{state} == HEXREF_HEX_STATE) {  
       if (0x0030 <= $self->{nc} and $self->{nc} <= 0x0039) {  
         # 0..9  
         !!!cp (1002);  
         $self->{s_kwd} *= 0x10;  
         $self->{s_kwd} += $self->{nc} - 0x0030;  
         ## Stay in the state.  
         !!!next-input-character;  
         redo A;  
       } elsif (0x0061 <= $self->{nc} and  
                $self->{nc} <= 0x0066) { # a..f  
         !!!cp (1003);  
         $self->{s_kwd} *= 0x10;  
         $self->{s_kwd} += $self->{nc} - 0x0060 + 9;  
         ## Stay in the state.  
         !!!next-input-character;  
         redo A;  
       } elsif (0x0041 <= $self->{nc} and  
                $self->{nc} <= 0x0046) { # A..F  
         !!!cp (1004);  
         $self->{s_kwd} *= 0x10;  
         $self->{s_kwd} += $self->{nc} - 0x0040 + 9;  
         ## Stay in the state.  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003B) { # ;  
         !!!cp (1006);  
         !!!next-input-character;  
         #  
       } else {  
         !!!cp (1007);  
         !!!parse-error (type => 'no refc',  
                         line => $self->{line},  
                         column => $self->{column});  
         ## Reconsume.  
         #  
       }  
   
       my $code = $self->{s_kwd};  
       my $l = $self->{line_prev};  
       my $c = $self->{column_prev};  
       if ($charref_map->{$code}) {  
         !!!cp (1008);  
         !!!parse-error (type => 'invalid character reference',  
                         text => (sprintf 'U+%04X', $code),  
                         line => $l, column => $c);  
         $code = $charref_map->{$code};  
       } elsif ($code > 0x10FFFF) {  
         !!!cp (1009);  
         !!!parse-error (type => 'invalid character reference',  
                         text => (sprintf 'U-%08X', $code),  
                         line => $l, column => $c);  
         $code = 0xFFFD;  
       }  
   
       if ($self->{prev_state} == DATA_STATE) {  
         !!!cp (988);  
         $self->{state} = $self->{prev_state};  
         ## Reconsume.  
         !!!emit ({type => CHARACTER_TOKEN, data => chr $code,  
                   line => $l, column => $c,  
                  });  
         redo A;  
       } else {  
         !!!cp (987);  
         $self->{ca}->{value} .= chr $code;  
         $self->{ca}->{has_reference} = 1;  
         $self->{state} = $self->{prev_state};  
         ## Reconsume.  
         redo A;  
       }  
     } elsif ($self->{state} == ENTITY_NAME_STATE) {  
       if (length $self->{s_kwd} < 30 and  
           ## NOTE: Some number greater than the maximum length of entity name  
           ((0x0041 <= $self->{nc} and # a  
             $self->{nc} <= 0x005A) or # x  
            (0x0061 <= $self->{nc} and # a  
             $self->{nc} <= 0x007A) or # z  
            (0x0030 <= $self->{nc} and # 0  
             $self->{nc} <= 0x0039) or # 9  
            $self->{nc} == 0x003B)) { # ;  
         our $EntityChar;  
         $self->{s_kwd} .= chr $self->{nc};  
         if (defined $EntityChar->{$self->{s_kwd}}) {  
           if ($self->{nc} == 0x003B) { # ;  
             !!!cp (1020);  
             $self->{entity__value} = $EntityChar->{$self->{s_kwd}};  
             $self->{entity__match} = 1;  
             !!!next-input-character;  
             #  
           } else {  
             !!!cp (1021);  
             $self->{entity__value} = $EntityChar->{$self->{s_kwd}};  
             $self->{entity__match} = -1;  
             ## Stay in the state.  
             !!!next-input-character;  
             redo A;  
           }  
         } else {  
           !!!cp (1022);  
           $self->{entity__value} .= chr $self->{nc};  
           $self->{entity__match} *= 2;  
           ## Stay in the state.  
           !!!next-input-character;  
           redo A;  
         }  
       }  
   
       my $data;  
       my $has_ref;  
       if ($self->{entity__match} > 0) {  
         !!!cp (1023);  
         $data = $self->{entity__value};  
         $has_ref = 1;  
         #  
       } elsif ($self->{entity__match} < 0) {  
         !!!parse-error (type => 'no refc');  
         if ($self->{prev_state} != DATA_STATE and # in attribute  
             $self->{entity__match} < -1) {  
           !!!cp (1024);  
           $data = '&' . $self->{s_kwd};  
           #  
         } else {  
           !!!cp (1025);  
           $data = $self->{entity__value};  
           $has_ref = 1;  
           #  
         }  
       } else {  
         !!!cp (1026);  
         !!!parse-error (type => 'bare ero',  
                         line => $self->{line_prev},  
                         column => $self->{column_prev} - length $self->{s_kwd});  
         $data = '&' . $self->{s_kwd};  
         #  
       }  
     
       ## NOTE: In these cases, when a character reference is found,  
       ## it is consumed and a character token is returned, or, otherwise,  
       ## nothing is consumed and returned, according to the spec algorithm.  
       ## In this implementation, anything that has been examined by the  
       ## tokenizer is appended to the parent element or the attribute value  
       ## as string, either literal string when no character reference or  
       ## entity-replaced string otherwise, in this stage, since any characters  
       ## that would not be consumed are appended in the data state or in an  
       ## appropriate attribute value state anyway.  
   
       if ($self->{prev_state} == DATA_STATE) {  
         !!!cp (986);  
         $self->{state} = $self->{prev_state};  
         ## Reconsume.  
         !!!emit ({type => CHARACTER_TOKEN,  
                   data => $data,  
                   line => $self->{line_prev},  
                   column => $self->{column_prev} + 1 - length $self->{s_kwd},  
                  });  
         redo A;  
       } else {  
         !!!cp (985);  
         $self->{ca}->{value} .= $data;  
         $self->{ca}->{has_reference} = 1 if $has_ref;  
         $self->{state} = $self->{prev_state};  
         ## Reconsume.  
         redo A;  
       }  
     } else {  
       die "$0: $self->{state}: Unknown state";  
     }  
   } # A    
   
   die "$0: _get_next_token: unexpected case";  
 } # _get_next_token  
   
850  sub _initialize_tree_constructor ($) {  sub _initialize_tree_constructor ($) {
851    my $self = shift;    my $self = shift;
852    ## NOTE: $self->{document} MUST be specified before this method is called    ## NOTE: $self->{document} MUST be specified before this method is called
# Line 3467  sub _construct_tree ($) { Line 875  sub _construct_tree ($) {
875    ## When an interactive UA render the $self->{document} available    ## When an interactive UA render the $self->{document} available
876    ## to the user, or when it begin accepting user input, are    ## to the user, or when it begin accepting user input, are
877    ## not defined.    ## not defined.
   
   ## Append a character: collect it and all subsequent consecutive  
   ## characters and insert one Text node whose data is concatenation  
   ## of all those characters. # MUST  
878        
879    !!!next-token;    !!!next-token;
880    
881    undef $self->{form_element};    undef $self->{form_element};
882    undef $self->{head_element};    undef $self->{head_element};
883      undef $self->{head_element_inserted};
884    $self->{open_elements} = [];    $self->{open_elements} = [];
885    undef $self->{inner_html_node};    undef $self->{inner_html_node};
886      undef $self->{ignore_newline};
887    
888    ## NOTE: The "initial" insertion mode.    ## NOTE: The "initial" insertion mode.
889    $self->_tree_construction_initial; # MUST    $self->_tree_construction_initial; # MUST
# Line 3530  sub _tree_construction_initial ($) { Line 936  sub _tree_construction_initial ($) {
936        $doctype->public_id ($token->{pubid}) if defined $token->{pubid};        $doctype->public_id ($token->{pubid}) if defined $token->{pubid};
937        $doctype->system_id ($token->{sysid}) if defined $token->{sysid};        $doctype->system_id ($token->{sysid}) if defined $token->{sysid};
938        ## NOTE: Other DocumentType attributes are null or empty lists.        ## NOTE: Other DocumentType attributes are null or empty lists.
939        ## ISSUE: internalSubset = null??        ## In Firefox3, |internalSubset| attribute is set to the empty
940          ## string, while |null| is an allowed value for the attribute
941          ## according to DOM3 Core.
942        $self->{document}->append_child ($doctype);        $self->{document}->append_child ($doctype);
943                
944        if ($token->{quirks} or $doctype_name ne 'HTML') {        if ($token->{quirks} or $doctype_name ne 'HTML') {
# Line 3783  sub _tree_construction_root_element ($) Line 1191  sub _tree_construction_root_element ($)
1191      ## NOTE: Reprocess the token.      ## NOTE: Reprocess the token.
1192      !!!ack-later;      !!!ack-later;
1193      return; ## Go to the "before head" insertion mode.      return; ## Go to the "before head" insertion mode.
   
     ## ISSUE: There is an issue in the spec  
1194    } # B    } # B
1195    
1196    die "$0: _tree_construction_root_element: This should never be reached";    die "$0: _tree_construction_root_element: This should never be reached";
# Line 3820  sub _reset_insertion_mode ($) { Line 1226  sub _reset_insertion_mode ($) {
1226          ## SVG elements.  Currently the HTML syntax supports only MathML and          ## SVG elements.  Currently the HTML syntax supports only MathML and
1227          ## SVG elements as foreigners.          ## SVG elements as foreigners.
1228          $new_mode = IN_BODY_IM | IN_FOREIGN_CONTENT_IM;          $new_mode = IN_BODY_IM | IN_FOREIGN_CONTENT_IM;
1229        } elsif ($node->[1] & TABLE_CELL_EL) {        } elsif ($node->[1] == TABLE_CELL_EL) {
1230          if ($last) {          if ($last) {
1231            !!!cp ('t28.2');            !!!cp ('t28.2');
1232            #            #
# Line 3849  sub _reset_insertion_mode ($) { Line 1255  sub _reset_insertion_mode ($) {
1255        $self->{insertion_mode} = $new_mode and return if defined $new_mode;        $self->{insertion_mode} = $new_mode and return if defined $new_mode;
1256                
1257        ## Step 15        ## Step 15
1258        if ($node->[1] & HTML_EL) {        if ($node->[1] == HTML_EL) {
1259          unless (defined $self->{head_element}) {          unless (defined $self->{head_element}) {
1260            !!!cp ('t29');            !!!cp ('t29');
1261            $self->{insertion_mode} = BEFORE_HEAD_IM;            $self->{insertion_mode} = BEFORE_HEAD_IM;
# Line 3981  sub _tree_construction_main ($) { Line 1387  sub _tree_construction_main ($) {
1387    
1388      ## Step 1      ## Step 1
1389      my $start_tag_name = $token->{tag_name};      my $start_tag_name = $token->{tag_name};
1390      my $el;      !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
     !!!create-element ($el, $HTML_NS, $start_tag_name, $token->{attributes}, $token);  
1391    
1392      ## Step 2      ## Step 2
     $insert->($el);  
   
     ## Step 3  
1393      $self->{content_model} = $content_model_flag; # CDATA or RCDATA      $self->{content_model} = $content_model_flag; # CDATA or RCDATA
1394      delete $self->{escape}; # MUST      delete $self->{escape}; # MUST
1395    
1396      ## Step 4      ## Step 3, 4
1397      my $text = '';      $self->{insertion_mode} |= IN_CDATA_RCDATA_IM;
     !!!nack ('t40.1');  
     !!!next-token;  
     while ($token->{type} == CHARACTER_TOKEN) { # or until stop tokenizing  
       !!!cp ('t40');  
       $text .= $token->{data};  
       !!!next-token;  
     }  
   
     ## Step 5  
     if (length $text) {  
       !!!cp ('t41');  
       my $text = $self->{document}->create_text_node ($text);  
       $el->append_child ($text);  
     }  
   
     ## Step 6  
     $self->{content_model} = PCDATA_CONTENT_MODEL;  
1398    
1399      ## Step 7      !!!nack ('t40.1');
     if ($token->{type} == END_TAG_TOKEN and  
         $token->{tag_name} eq $start_tag_name) {  
       !!!cp ('t42');  
       ## Ignore the token  
     } else {  
       ## NOTE: An end-of-file token.  
       if ($content_model_flag == CDATA_CONTENT_MODEL) {  
         !!!cp ('t43');  
         !!!parse-error (type => 'in CDATA:#eof', token => $token);  
       } elsif ($content_model_flag == RCDATA_CONTENT_MODEL) {  
         !!!cp ('t44');  
         !!!parse-error (type => 'in RCDATA:#eof', token => $token);  
       } else {  
         die "$0: $content_model_flag in parse_rcdata";  
       }  
     }  
1400      !!!next-token;      !!!next-token;
1401    }; # $parse_rcdata    }; # $parse_rcdata
1402    
1403    my $script_start_tag = sub () {    my $script_start_tag = sub () {
1404        ## Step 1
1405      my $script_el;      my $script_el;
1406      !!!create-element ($script_el, $HTML_NS, 'script', $token->{attributes}, $token);      !!!create-element ($script_el, $HTML_NS, 'script', $token->{attributes}, $token);
1407    
1408        ## Step 2
1409      ## TODO: mark as "parser-inserted"      ## TODO: mark as "parser-inserted"
1410    
1411        ## Step 3
1412        ## TODO: Mark as "already executed", if ...
1413    
1414        ## Step 4
1415        $insert->($script_el);
1416    
1417        ## ISSUE: $script_el is not put into the stack
1418        push @{$self->{open_elements}}, [$script_el, $el_category->{script}];
1419    
1420        ## Step 5
1421      $self->{content_model} = CDATA_CONTENT_MODEL;      $self->{content_model} = CDATA_CONTENT_MODEL;
1422      delete $self->{escape}; # MUST      delete $self->{escape}; # MUST
       
     my $text = '';  
     !!!nack ('t45.1');  
     !!!next-token;  
     while ($token->{type} == CHARACTER_TOKEN) {  
       !!!cp ('t45');  
       $text .= $token->{data};  
       !!!next-token;  
     } # stop if non-character token or tokenizer stops tokenising  
     if (length $text) {  
       !!!cp ('t46');  
       $script_el->manakai_append_text ($text);  
     }  
                 
     $self->{content_model} = PCDATA_CONTENT_MODEL;  
1423    
1424      if ($token->{type} == END_TAG_TOKEN and      ## Step 6-7
1425          $token->{tag_name} eq 'script') {      $self->{insertion_mode} |= IN_CDATA_RCDATA_IM;
       !!!cp ('t47');  
       ## Ignore the token  
     } else {  
       !!!cp ('t48');  
       !!!parse-error (type => 'in CDATA:#eof', token => $token);  
       ## ISSUE: And ignore?  
       ## TODO: mark as "already executed"  
     }  
       
     if (defined $self->{inner_html_node}) {  
       !!!cp ('t49');  
       ## TODO: mark as "already executed"  
     } else {  
       !!!cp ('t50');  
       ## TODO: $old_insertion_point = current insertion point  
       ## TODO: insertion point = just before the next input character  
1426    
1427        $insert->($script_el);      !!!nack ('t40.2');
         
       ## TODO: insertion point = $old_insertion_point (might be "undefined")  
         
       ## TODO: if there is a script that will execute as soon as the parser resume, then...  
     }  
       
1428      !!!next-token;      !!!next-token;
1429    }; # $script_start_tag    }; # $script_start_tag
1430    
1431    ## NOTE: $open_tables->[-1]->[0] is the "current table" element node.    ## NOTE: $open_tables->[-1]->[0] is the "current table" element node.
1432    ## NOTE: $open_tables->[-1]->[1] is the "tainted" flag.    ## NOTE: $open_tables->[-1]->[1] is the "tainted" flag.
1433      ## NOTE: $open_tables->[-1]->[2] is set false when non-Text node inserted.
1434    my $open_tables = [[$self->{open_elements}->[0]->[0]]];    my $open_tables = [[$self->{open_elements}->[0]->[0]]];
1435    
1436    my $formatting_end_tag = sub {    my $formatting_end_tag = sub {
# Line 4169  sub _tree_construction_main ($) { Line 1515  sub _tree_construction_main ($) {
1515            !!!cp ('t59');            !!!cp ('t59');
1516            $furthest_block = $node;            $furthest_block = $node;
1517            $furthest_block_i_in_open = $_;            $furthest_block_i_in_open = $_;
1518              ## NOTE: The topmost (eldest) node.
1519          } elsif ($node->[0] eq $formatting_element->[0]) {          } elsif ($node->[0] eq $formatting_element->[0]) {
1520            !!!cp ('t60');            !!!cp ('t60');
1521            last OE;            last OE;
# Line 4255  sub _tree_construction_main ($) { Line 1602  sub _tree_construction_main ($) {
1602          my $foster_parent_element;          my $foster_parent_element;
1603          my $next_sibling;          my $next_sibling;
1604          OE: for (reverse 0..$#{$self->{open_elements}}) {          OE: for (reverse 0..$#{$self->{open_elements}}) {
1605            if ($self->{open_elements}->[$_]->[1] & TABLE_EL) {            if ($self->{open_elements}->[$_]->[1] == TABLE_EL) {
1606                               my $parent = $self->{open_elements}->[$_]->[0]->parent_node;                               my $parent = $self->{open_elements}->[$_]->[0]->parent_node;
1607                               if (defined $parent and $parent->node_type == 1) {                               if (defined $parent and $parent->node_type == 1) {
1608                                 !!!cp ('t65.1');                                 !!!cp ('t65.1');
# Line 4315  sub _tree_construction_main ($) { Line 1662  sub _tree_construction_main ($) {
1662            $i = $_;            $i = $_;
1663          }          }
1664        } # OE        } # OE
1665        splice @{$self->{open_elements}}, $i + 1, 1, $clone;        splice @{$self->{open_elements}}, $i + 1, 0, $clone;
1666                
1667        ## Step 14        ## Step 14
1668        redo FET;        redo FET;
# Line 4333  sub _tree_construction_main ($) { Line 1680  sub _tree_construction_main ($) {
1680        my $foster_parent_element;        my $foster_parent_element;
1681        my $next_sibling;        my $next_sibling;
1682        OE: for (reverse 0..$#{$self->{open_elements}}) {        OE: for (reverse 0..$#{$self->{open_elements}}) {
1683          if ($self->{open_elements}->[$_]->[1] & TABLE_EL) {          if ($self->{open_elements}->[$_]->[1] == TABLE_EL) {
1684                               my $parent = $self->{open_elements}->[$_]->[0]->parent_node;                               my $parent = $self->{open_elements}->[$_]->[0]->parent_node;
1685                               if (defined $parent and $parent->node_type == 1) {                               if (defined $parent and $parent->node_type == 1) {
1686                                 !!!cp ('t70');                                 !!!cp ('t70');
# Line 4358  sub _tree_construction_main ($) { Line 1705  sub _tree_construction_main ($) {
1705      }      }
1706    }; # $insert_to_foster    }; # $insert_to_foster
1707    
1708      ## NOTE: Insert a character (MUST): When a character is inserted, if
1709      ## the last node that was inserted by the parser is a Text node and
1710      ## the character has to be inserted after that node, then the
1711      ## character is appended to the Text node.  However, if any other
1712      ## node is inserted by the parser, then a new Text node is created
1713      ## and the character is appended as that Text node.  If I'm not
1714      ## wrong, for a parser with scripting disabled, there are only two
1715      ## cases where this occurs.  One is the case where an element node
1716      ## is inserted to the |head| element.  This is covered by using the
1717      ## |$self->{head_element_inserted}| flag.  Another is the case where
1718      ## an element or comment is inserted into the |table| subtree while
1719      ## foster parenting happens.  This is covered by using the [2] flag
1720      ## of the |$open_tables| structure.  All other cases are handled
1721      ## simply by calling |manakai_append_text| method.
1722    
1723      ## TODO: |<body><script>document.write("a<br>");
1724      ## document.body.removeChild (document.body.lastChild);
1725      ## document.write ("b")</script>|
1726    
1727    B: while (1) {    B: while (1) {
1728      if ($token->{type} == DOCTYPE_TOKEN) {      if ($token->{type} == DOCTYPE_TOKEN) {
1729        !!!cp ('t73');        !!!cp ('t73');
# Line 4405  sub _tree_construction_main ($) { Line 1771  sub _tree_construction_main ($) {
1771        } else {        } else {
1772          !!!cp ('t87');          !!!cp ('t87');
1773          $self->{open_elements}->[-1]->[0]->append_child ($comment);          $self->{open_elements}->[-1]->[0]->append_child ($comment);
1774            $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
1775        }        }
1776        !!!next-token;        !!!next-token;
1777        next B;        next B;
1778        } elsif ($self->{insertion_mode} & IN_CDATA_RCDATA_IM) {
1779          if ($token->{type} == CHARACTER_TOKEN) {
1780            $token->{data} =~ s/^\x0A// if $self->{ignore_newline};
1781            delete $self->{ignore_newline};
1782    
1783            if (length $token->{data}) {
1784              !!!cp ('t43');
1785              $self->{open_elements}->[-1]->[0]->manakai_append_text
1786                  ($token->{data});
1787            } else {
1788              !!!cp ('t43.1');
1789            }
1790            !!!next-token;
1791            next B;
1792          } elsif ($token->{type} == END_TAG_TOKEN) {
1793            delete $self->{ignore_newline};
1794    
1795            if ($token->{tag_name} eq 'script') {
1796              !!!cp ('t50');
1797              
1798              ## Para 1-2
1799              my $script = pop @{$self->{open_elements}};
1800              
1801              ## Para 3
1802              $self->{insertion_mode} &= ~ IN_CDATA_RCDATA_IM;
1803    
1804              ## Para 4
1805              ## TODO: $old_insertion_point = $current_insertion_point;
1806              ## TODO: $current_insertion_point = just before $self->{nc};
1807    
1808              ## Para 5
1809              ## TODO: Run the $script->[0].
1810    
1811              ## Para 6
1812              ## TODO: $current_insertion_point = $old_insertion_point;
1813    
1814              ## Para 7
1815              ## TODO: if ($pending_external_script) {
1816                ## TODO: ...
1817              ## TODO: }
1818    
1819              !!!next-token;
1820              next B;
1821            } else {
1822              !!!cp ('t42');
1823    
1824              pop @{$self->{open_elements}};
1825    
1826              $self->{insertion_mode} &= ~ IN_CDATA_RCDATA_IM;
1827              !!!next-token;
1828              next B;
1829            }
1830          } elsif ($token->{type} == END_OF_FILE_TOKEN) {
1831            delete $self->{ignore_newline};
1832    
1833            !!!cp ('t44');
1834            !!!parse-error (type => 'not closed',
1835                            text => $self->{open_elements}->[-1]->[0]
1836                                ->manakai_local_name,
1837                            token => $token);
1838    
1839            #if ($self->{open_elements}->[-1]->[1] == SCRIPT_EL) {
1840            #  ## TODO: Mark as "already executed"
1841            #}
1842    
1843            pop @{$self->{open_elements}};
1844    
1845            $self->{insertion_mode} &= ~ IN_CDATA_RCDATA_IM;
1846            ## Reprocess.
1847            next B;
1848          } else {
1849            die "$0: $token->{type}: In CDATA/RCDATA: Unknown token type";        
1850          }
1851      } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM) {      } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM) {
1852        if ($token->{type} == CHARACTER_TOKEN) {        if ($token->{type} == CHARACTER_TOKEN) {
1853          !!!cp ('t87.1');          !!!cp ('t87.1');
# Line 4419  sub _tree_construction_main ($) { Line 1859  sub _tree_construction_main ($) {
1859               $self->{open_elements}->[-1]->[1] & FOREIGN_FLOW_CONTENT_EL) or               $self->{open_elements}->[-1]->[1] & FOREIGN_FLOW_CONTENT_EL) or
1860              not ($self->{open_elements}->[-1]->[1] & FOREIGN_EL) or              not ($self->{open_elements}->[-1]->[1] & FOREIGN_EL) or
1861              ($token->{tag_name} eq 'svg' and              ($token->{tag_name} eq 'svg' and
1862               $self->{open_elements}->[-1]->[1] & MML_AXML_EL)) {               $self->{open_elements}->[-1]->[1] == MML_AXML_EL)) {
1863            ## NOTE: "using the rules for secondary insertion mode"then"continue"            ## NOTE: "using the rules for secondary insertion mode"then"continue"
1864            !!!cp ('t87.2');            !!!cp ('t87.2');
1865            #            #
# Line 4520  sub _tree_construction_main ($) { Line 1960  sub _tree_construction_main ($) {
1960          pop @{$self->{open_elements}}          pop @{$self->{open_elements}}
1961              while $self->{open_elements}->[-1]->[1] & FOREIGN_EL;              while $self->{open_elements}->[-1]->[1] & FOREIGN_EL;
1962    
1963            ## NOTE: |<span><svg>| ... two parse errors, |<svg>| ... a parse error.
1964    
1965          $self->{insertion_mode} &= ~ IN_FOREIGN_CONTENT_IM;          $self->{insertion_mode} &= ~ IN_FOREIGN_CONTENT_IM;
1966          ## Reprocess.          ## Reprocess.
1967          next B;          next B;
# Line 4532  sub _tree_construction_main ($) { Line 1974  sub _tree_construction_main ($) {
1974        if ($token->{type} == CHARACTER_TOKEN) {        if ($token->{type} == CHARACTER_TOKEN) {
1975          if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {          if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
1976            unless ($self->{insertion_mode} == BEFORE_HEAD_IM) {            unless ($self->{insertion_mode} == BEFORE_HEAD_IM) {
1977              !!!cp ('t88.2');              if ($self->{head_element_inserted}) {
1978              $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);                !!!cp ('t88.3');
1979              #                $self->{open_elements}->[-1]->[0]->append_child
1980                    ($self->{document}->create_text_node ($1));
1981                  delete $self->{head_element_inserted};
1982                  ## NOTE: |</head> <link> |
1983                  #
1984                } else {
1985                  !!!cp ('t88.2');
1986                  $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
1987                  ## NOTE: |</head> &#x20;|
1988                  #
1989                }
1990            } else {            } else {
1991              !!!cp ('t88.1');              !!!cp ('t88.1');
1992              ## Ignore the token.              ## Ignore the token.
# Line 4630  sub _tree_construction_main ($) { Line 2082  sub _tree_construction_main ($) {
2082            !!!cp ('t97');            !!!cp ('t97');
2083          }          }
2084    
2085              if ($token->{tag_name} eq 'base') {          if ($token->{tag_name} eq 'base') {
2086                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {            if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
2087                  !!!cp ('t98');              !!!cp ('t98');
2088                  ## As if </noscript>              ## As if </noscript>
2089                  pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
2090                  !!!parse-error (type => 'in noscript', text => 'base',              !!!parse-error (type => 'in noscript', text => 'base',
2091                                  token => $token);                              token => $token);
2092                            
2093                  $self->{insertion_mode} = IN_HEAD_IM;              $self->{insertion_mode} = IN_HEAD_IM;
2094                  ## Reprocess in the "in head" insertion mode...              ## Reprocess in the "in head" insertion mode...
2095                } else {            } else {
2096                  !!!cp ('t99');              !!!cp ('t99');
2097                }            }
2098    
2099                ## NOTE: There is a "as if in head" code clone.            ## NOTE: There is a "as if in head" code clone.
2100                if ($self->{insertion_mode} == AFTER_HEAD_IM) {            if ($self->{insertion_mode} == AFTER_HEAD_IM) {
2101                  !!!cp ('t100');              !!!cp ('t100');
2102                  !!!parse-error (type => 'after head',              !!!parse-error (type => 'after head',
2103                                  text => $token->{tag_name}, token => $token);                              text => $token->{tag_name}, token => $token);
2104                  push @{$self->{open_elements}},              push @{$self->{open_elements}},
2105                      [$self->{head_element}, $el_category->{head}];                  [$self->{head_element}, $el_category->{head}];
2106                } else {              $self->{head_element_inserted} = 1;
2107                  !!!cp ('t101');            } else {
2108                }              !!!cp ('t101');
2109                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);            }
2110                pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.            !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
2111                pop @{$self->{open_elements}} # <head>            pop @{$self->{open_elements}};
2112                    if $self->{insertion_mode} == AFTER_HEAD_IM;            pop @{$self->{open_elements}} # <head>
2113                !!!nack ('t101.1');                if $self->{insertion_mode} == AFTER_HEAD_IM;
2114                !!!next-token;            !!!nack ('t101.1');
2115                next B;            !!!next-token;
2116              } elsif ($token->{tag_name} eq 'link') {            next B;
2117                ## NOTE: There is a "as if in head" code clone.          } elsif ($token->{tag_name} eq 'link') {
2118                if ($self->{insertion_mode} == AFTER_HEAD_IM) {            ## NOTE: There is a "as if in head" code clone.
2119                  !!!cp ('t102');            if ($self->{insertion_mode} == AFTER_HEAD_IM) {
2120                  !!!parse-error (type => 'after head',              !!!cp ('t102');
2121                                  text => $token->{tag_name}, token => $token);              !!!parse-error (type => 'after head',
2122                  push @{$self->{open_elements}},                              text => $token->{tag_name}, token => $token);
2123                      [$self->{head_element}, $el_category->{head}];              push @{$self->{open_elements}},
2124                } else {                  [$self->{head_element}, $el_category->{head}];
2125                  !!!cp ('t103');              $self->{head_element_inserted} = 1;
2126                }            } else {
2127                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);              !!!cp ('t103');
2128                pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.            }
2129                pop @{$self->{open_elements}} # <head>            !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
2130                    if $self->{insertion_mode} == AFTER_HEAD_IM;            pop @{$self->{open_elements}};
2131                !!!ack ('t103.1');            pop @{$self->{open_elements}} # <head>
2132                !!!next-token;                if $self->{insertion_mode} == AFTER_HEAD_IM;
2133                next B;            !!!ack ('t103.1');
2134              } elsif ($token->{tag_name} eq 'meta') {            !!!next-token;
2135                ## NOTE: There is a "as if in head" code clone.            next B;
2136                if ($self->{insertion_mode} == AFTER_HEAD_IM) {          } elsif ($token->{tag_name} eq 'command' or
2137                  !!!cp ('t104');                   $token->{tag_name} eq 'eventsource') {
2138                  !!!parse-error (type => 'after head',            if ($self->{insertion_mode} == IN_HEAD_IM) {
2139                                  text => $token->{tag_name}, token => $token);              ## NOTE: If the insertion mode at the time of the emission
2140                  push @{$self->{open_elements}},              ## of the token was "before head", $self->{insertion_mode}
2141                      [$self->{head_element}, $el_category->{head}];              ## is already changed to |IN_HEAD_IM|.
2142                } else {  
2143                  !!!cp ('t105');              ## NOTE: There is a "as if in head" code clone.
2144                }              !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
2145                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);              pop @{$self->{open_elements}};
2146                my $meta_el = pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.              pop @{$self->{open_elements}} # <head>
2147                    if $self->{insertion_mode} == AFTER_HEAD_IM;
2148                !!!ack ('t103.2');
2149                !!!next-token;
2150                next B;
2151              } else {
2152                ## NOTE: "in head noscript" or "after head" insertion mode
2153                ## - in these cases, these tags are treated as same as
2154                ## normal in-body tags.
2155                !!!cp ('t103.3');
2156                #
2157              }
2158            } elsif ($token->{tag_name} eq 'meta') {
2159              ## NOTE: There is a "as if in head" code clone.
2160              if ($self->{insertion_mode} == AFTER_HEAD_IM) {
2161                !!!cp ('t104');
2162                !!!parse-error (type => 'after head',
2163                                text => $token->{tag_name}, token => $token);
2164                push @{$self->{open_elements}},
2165                    [$self->{head_element}, $el_category->{head}];
2166                $self->{head_element_inserted} = 1;
2167              } else {
2168                !!!cp ('t105');
2169              }
2170              !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
2171              my $meta_el = pop @{$self->{open_elements}};
2172    
2173                unless ($self->{confident}) {                unless ($self->{confident}) {
2174                  if ($token->{attributes}->{charset}) {                  if ($token->{attributes}->{charset}) {
# Line 4749  sub _tree_construction_main ($) { Line 2226  sub _tree_construction_main ($) {
2226                !!!ack ('t110.1');                !!!ack ('t110.1');
2227                !!!next-token;                !!!next-token;
2228                next B;                next B;
2229              } elsif ($token->{tag_name} eq 'title') {          } elsif ($token->{tag_name} eq 'title') {
2230                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {            if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
2231                  !!!cp ('t111');              !!!cp ('t111');
2232                  ## As if </noscript>              ## As if </noscript>
2233                  pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
2234                  !!!parse-error (type => 'in noscript', text => 'title',              !!!parse-error (type => 'in noscript', text => 'title',
2235                                  token => $token);                              token => $token);
2236                            
2237                  $self->{insertion_mode} = IN_HEAD_IM;              $self->{insertion_mode} = IN_HEAD_IM;
2238                  ## Reprocess in the "in head" insertion mode...              ## Reprocess in the "in head" insertion mode...
2239                } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {            } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
2240                  !!!cp ('t112');              !!!cp ('t112');
2241                  !!!parse-error (type => 'after head',              !!!parse-error (type => 'after head',
2242                                  text => $token->{tag_name}, token => $token);                              text => $token->{tag_name}, token => $token);
2243                  push @{$self->{open_elements}},              push @{$self->{open_elements}},
2244                      [$self->{head_element}, $el_category->{head}];                  [$self->{head_element}, $el_category->{head}];
2245                } else {              $self->{head_element_inserted} = 1;
2246                  !!!cp ('t113');            } else {
2247                }              !!!cp ('t113');
2248              }
2249    
2250                ## NOTE: There is a "as if in head" code clone.            ## NOTE: There is a "as if in head" code clone.
2251                my $parent = defined $self->{head_element} ? $self->{head_element}            $parse_rcdata->(RCDATA_CONTENT_MODEL);
2252                    : $self->{open_elements}->[-1]->[0];            ## ISSUE: A spec bug [Bug 6038]
2253                $parse_rcdata->(RCDATA_CONTENT_MODEL);            splice @{$self->{open_elements}}, -2, 1, () # <head>
2254                pop @{$self->{open_elements}} # <head>                if ($self->{insertion_mode} & IM_MASK) == AFTER_HEAD_IM;
2255                    if $self->{insertion_mode} == AFTER_HEAD_IM;            next B;
2256                next B;          } elsif ($token->{tag_name} eq 'style' or
2257              } elsif ($token->{tag_name} eq 'style' or                   $token->{tag_name} eq 'noframes') {
2258                       $token->{tag_name} eq 'noframes') {            ## NOTE: Or (scripting is enabled and tag_name eq 'noscript' and
2259                ## NOTE: Or (scripting is enabled and tag_name eq 'noscript' and            ## insertion mode IN_HEAD_IM)
2260                ## insertion mode IN_HEAD_IM)            ## NOTE: There is a "as if in head" code clone.
2261                ## NOTE: There is a "as if in head" code clone.            if ($self->{insertion_mode} == AFTER_HEAD_IM) {
2262                if ($self->{insertion_mode} == AFTER_HEAD_IM) {              !!!cp ('t114');
2263                  !!!cp ('t114');              !!!parse-error (type => 'after head',
2264                  !!!parse-error (type => 'after head',                              text => $token->{tag_name}, token => $token);
2265                                  text => $token->{tag_name}, token => $token);              push @{$self->{open_elements}},
2266                  push @{$self->{open_elements}},                  [$self->{head_element}, $el_category->{head}];
2267                      [$self->{head_element}, $el_category->{head}];              $self->{head_element_inserted} = 1;
2268                } else {            } else {
2269                  !!!cp ('t115');              !!!cp ('t115');
2270                }            }
2271                $parse_rcdata->(CDATA_CONTENT_MODEL);            $parse_rcdata->(CDATA_CONTENT_MODEL);
2272                pop @{$self->{open_elements}} # <head>            ## ISSUE: A spec bug [Bug 6038]
2273                    if $self->{insertion_mode} == AFTER_HEAD_IM;            splice @{$self->{open_elements}}, -2, 1, () # <head>
2274                next B;                if ($self->{insertion_mode} & IM_MASK) == AFTER_HEAD_IM;
2275              } elsif ($token->{tag_name} eq 'noscript') {            next B;
2276            } elsif ($token->{tag_name} eq 'noscript') {
2277                if ($self->{insertion_mode} == IN_HEAD_IM) {                if ($self->{insertion_mode} == IN_HEAD_IM) {
2278                  !!!cp ('t116');                  !!!cp ('t116');
2279                  ## NOTE: and scripting is disalbed                  ## NOTE: and scripting is disalbed
# Line 4815  sub _tree_construction_main ($) { Line 2294  sub _tree_construction_main ($) {
2294                  !!!cp ('t118');                  !!!cp ('t118');
2295                  #                  #
2296                }                }
2297              } elsif ($token->{tag_name} eq 'script') {          } elsif ($token->{tag_name} eq 'script') {
2298                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {            if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
2299                  !!!cp ('t119');              !!!cp ('t119');
2300                  ## As if </noscript>              ## As if </noscript>
2301                  pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
2302                  !!!parse-error (type => 'in noscript', text => 'script',              !!!parse-error (type => 'in noscript', text => 'script',
2303                                  token => $token);                              token => $token);
2304                            
2305                  $self->{insertion_mode} = IN_HEAD_IM;              $self->{insertion_mode} = IN_HEAD_IM;
2306                  ## Reprocess in the "in head" insertion mode...              ## Reprocess in the "in head" insertion mode...
2307                } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {            } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
2308                  !!!cp ('t120');              !!!cp ('t120');
2309                  !!!parse-error (type => 'after head',              !!!parse-error (type => 'after head',
2310                                  text => $token->{tag_name}, token => $token);                              text => $token->{tag_name}, token => $token);
2311                  push @{$self->{open_elements}},              push @{$self->{open_elements}},
2312                      [$self->{head_element}, $el_category->{head}];                  [$self->{head_element}, $el_category->{head}];
2313                } else {              $self->{head_element_inserted} = 1;
2314                  !!!cp ('t121');            } else {
2315                }              !!!cp ('t121');
2316              }
2317    
2318                ## NOTE: There is a "as if in head" code clone.            ## NOTE: There is a "as if in head" code clone.
2319                $script_start_tag->();            $script_start_tag->();
2320                pop @{$self->{open_elements}} # <head>            ## ISSUE: A spec bug  [Bug 6038]
2321                    if $self->{insertion_mode} == AFTER_HEAD_IM;            splice @{$self->{open_elements}}, -2, 1 # <head>
2322                next B;                if ($self->{insertion_mode} & IM_MASK) == AFTER_HEAD_IM;
2323              } elsif ($token->{tag_name} eq 'body' or            next B;
2324                       $token->{tag_name} eq 'frameset') {          } elsif ($token->{tag_name} eq 'body' or
2325                     $token->{tag_name} eq 'frameset') {
2326                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
2327                  !!!cp ('t122');                  !!!cp ('t122');
2328                  ## As if </noscript>                  ## As if </noscript>
# Line 4976  sub _tree_construction_main ($) { Line 2457  sub _tree_construction_main ($) {
2457              } elsif ({              } elsif ({
2458                        body => 1, html => 1,                        body => 1, html => 1,
2459                       }->{$token->{tag_name}}) {                       }->{$token->{tag_name}}) {
2460                if ($self->{insertion_mode} == BEFORE_HEAD_IM or                ## TODO: This branch is entirely redundant.
2461                  if ($self->{insertion_mode} == BEFORE_HEAD_IM or
2462                    $self->{insertion_mode} == IN_HEAD_IM or                    $self->{insertion_mode} == IN_HEAD_IM or
2463                    $self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {                    $self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
2464                  !!!cp ('t140');                  !!!cp ('t140');
# Line 5020  sub _tree_construction_main ($) { Line 2502  sub _tree_construction_main ($) {
2502                  ## Reprocess in the "after head" insertion mode...                  ## Reprocess in the "after head" insertion mode...
2503                } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {                } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
2504                  !!!cp ('t143.3');                  !!!cp ('t143.3');
2505                  ## ISSUE: Two parse errors for <head><noscript></br>                  ## NOTE: Two parse errors for <head><noscript></br>
2506                  !!!parse-error (type => 'unmatched end tag',                  !!!parse-error (type => 'unmatched end tag',
2507                                  text => 'br', token => $token);                                  text => 'br', token => $token);
2508                  ## As if </noscript>                  ## As if </noscript>
# Line 5148  sub _tree_construction_main ($) { Line 2630  sub _tree_construction_main ($) {
2630        } else {        } else {
2631          die "$0: $token->{type}: Unknown token type";          die "$0: $token->{type}: Unknown token type";
2632        }        }
   
           ## ISSUE: An issue in the spec.  
2633      } elsif ($self->{insertion_mode} & BODY_IMS) {      } elsif ($self->{insertion_mode} & BODY_IMS) {
2634            if ($token->{type} == CHARACTER_TOKEN) {            if ($token->{type} == CHARACTER_TOKEN) {
2635              !!!cp ('t150');              !!!cp ('t150');
# Line 5165  sub _tree_construction_main ($) { Line 2645  sub _tree_construction_main ($) {
2645                   caption => 1, col => 1, colgroup => 1, tbody => 1,                   caption => 1, col => 1, colgroup => 1, tbody => 1,
2646                   td => 1, tfoot => 1, th => 1, thead => 1, tr => 1,                   td => 1, tfoot => 1, th => 1, thead => 1, tr => 1,
2647                  }->{$token->{tag_name}}) {                  }->{$token->{tag_name}}) {
2648                if ($self->{insertion_mode} == IN_CELL_IM) {                if (($self->{insertion_mode} & IM_MASK) == IN_CELL_IM) {
2649                  ## have an element in table scope                  ## have an element in table scope
2650                  for (reverse 0..$#{$self->{open_elements}}) {                  for (reverse 0..$#{$self->{open_elements}}) {
2651                    my $node = $self->{open_elements}->[$_];                    my $node = $self->{open_elements}->[$_];
2652                    if ($node->[1] & TABLE_CELL_EL) {                    if ($node->[1] == TABLE_CELL_EL) {
2653                      !!!cp ('t151');                      !!!cp ('t151');
2654    
2655                      ## Close the cell                      ## Close the cell
# Line 5193  sub _tree_construction_main ($) { Line 2673  sub _tree_construction_main ($) {
2673                  !!!nack ('t153.1');                  !!!nack ('t153.1');
2674                  !!!next-token;                  !!!next-token;
2675                  next B;                  next B;
2676                } elsif ($self->{insertion_mode} == IN_CAPTION_IM) {                } elsif (($self->{insertion_mode} & IM_MASK) == IN_CAPTION_IM) {
2677                  !!!parse-error (type => 'not closed', text => 'caption',                  !!!parse-error (type => 'not closed', text => 'caption',
2678                                  token => $token);                                  token => $token);
2679                                    
# Line 5203  sub _tree_construction_main ($) { Line 2683  sub _tree_construction_main ($) {
2683                  INSCOPE: {                  INSCOPE: {
2684                    for (reverse 0..$#{$self->{open_elements}}) {                    for (reverse 0..$#{$self->{open_elements}}) {
2685                      my $node = $self->{open_elements}->[$_];                      my $node = $self->{open_elements}->[$_];
2686                      if ($node->[1] & CAPTION_EL) {                      if ($node->[1] == CAPTION_EL) {
2687                        !!!cp ('t155');                        !!!cp ('t155');
2688                        $i = $_;                        $i = $_;
2689                        last INSCOPE;                        last INSCOPE;
# Line 5229  sub _tree_construction_main ($) { Line 2709  sub _tree_construction_main ($) {
2709                    pop @{$self->{open_elements}};                    pop @{$self->{open_elements}};
2710                  }                  }
2711    
2712                  unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {                  unless ($self->{open_elements}->[-1]->[1] == CAPTION_EL) {
2713                    !!!cp ('t159');                    !!!cp ('t159');
2714                    !!!parse-error (type => 'not closed',                    !!!parse-error (type => 'not closed',
2715                                    text => $self->{open_elements}->[-1]->[0]                                    text => $self->{open_elements}->[-1]->[0]
# Line 5258  sub _tree_construction_main ($) { Line 2738  sub _tree_construction_main ($) {
2738              }              }
2739            } elsif ($token->{type} == END_TAG_TOKEN) {            } elsif ($token->{type} == END_TAG_TOKEN) {
2740              if ($token->{tag_name} eq 'td' or $token->{tag_name} eq 'th') {              if ($token->{tag_name} eq 'td' or $token->{tag_name} eq 'th') {
2741                if ($self->{insertion_mode} == IN_CELL_IM) {                if (($self->{insertion_mode} & IM_MASK) == IN_CELL_IM) {
2742                  ## have an element in table scope                  ## have an element in table scope
2743                  my $i;                  my $i;
2744                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
# Line 5308  sub _tree_construction_main ($) { Line 2788  sub _tree_construction_main ($) {
2788                                    
2789                  !!!next-token;                  !!!next-token;
2790                  next B;                  next B;
2791                } elsif ($self->{insertion_mode} == IN_CAPTION_IM) {                } elsif (($self->{insertion_mode} & IM_MASK) == IN_CAPTION_IM) {
2792                  !!!cp ('t169');                  !!!cp ('t169');
2793                  !!!parse-error (type => 'unmatched end tag',                  !!!parse-error (type => 'unmatched end tag',
2794                                  text => $token->{tag_name}, token => $token);                                  text => $token->{tag_name}, token => $token);
# Line 5320  sub _tree_construction_main ($) { Line 2800  sub _tree_construction_main ($) {
2800                  #                  #
2801                }                }
2802              } elsif ($token->{tag_name} eq 'caption') {              } elsif ($token->{tag_name} eq 'caption') {
2803                if ($self->{insertion_mode} == IN_CAPTION_IM) {                if (($self->{insertion_mode} & IM_MASK) == IN_CAPTION_IM) {
2804                  ## have a table element in table scope                  ## have a table element in table scope
2805                  my $i;                  my $i;
2806                  INSCOPE: {                  INSCOPE: {
2807                    for (reverse 0..$#{$self->{open_elements}}) {                    for (reverse 0..$#{$self->{open_elements}}) {
2808                      my $node = $self->{open_elements}->[$_];                      my $node = $self->{open_elements}->[$_];
2809                      if ($node->[1] & CAPTION_EL) {                      if ($node->[1] == CAPTION_EL) {
2810                        !!!cp ('t171');                        !!!cp ('t171');
2811                        $i = $_;                        $i = $_;
2812                        last INSCOPE;                        last INSCOPE;
# Line 5351  sub _tree_construction_main ($) { Line 2831  sub _tree_construction_main ($) {
2831                    pop @{$self->{open_elements}};                    pop @{$self->{open_elements}};
2832                  }                  }
2833                                    
2834                  unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {                  unless ($self->{open_elements}->[-1]->[1] == CAPTION_EL) {
2835                    !!!cp ('t175');                    !!!cp ('t175');
2836                    !!!parse-error (type => 'not closed',                    !!!parse-error (type => 'not closed',
2837                                    text => $self->{open_elements}->[-1]->[0]                                    text => $self->{open_elements}->[-1]->[0]
# Line 5369  sub _tree_construction_main ($) { Line 2849  sub _tree_construction_main ($) {
2849                                    
2850                  !!!next-token;                  !!!next-token;
2851                  next B;                  next B;
2852                } elsif ($self->{insertion_mode} == IN_CELL_IM) {                } elsif (($self->{insertion_mode} & IM_MASK) == IN_CELL_IM) {
2853                  !!!cp ('t177');                  !!!cp ('t177');
2854                  !!!parse-error (type => 'unmatched end tag',                  !!!parse-error (type => 'unmatched end tag',
2855                                  text => $token->{tag_name}, token => $token);                                  text => $token->{tag_name}, token => $token);
# Line 5384  sub _tree_construction_main ($) { Line 2864  sub _tree_construction_main ($) {
2864                        table => 1, tbody => 1, tfoot => 1,                        table => 1, tbody => 1, tfoot => 1,
2865                        thead => 1, tr => 1,                        thead => 1, tr => 1,
2866                       }->{$token->{tag_name}} and                       }->{$token->{tag_name}} and
2867                       $self->{insertion_mode} == IN_CELL_IM) {                       ($self->{insertion_mode} & IM_MASK) == IN_CELL_IM) {
2868                ## have an element in table scope                ## have an element in table scope
2869                my $i;                my $i;
2870                my $tn;                my $tn;
# Line 5401  sub _tree_construction_main ($) { Line 2881  sub _tree_construction_main ($) {
2881                                line => $token->{line},                                line => $token->{line},
2882                                column => $token->{column}};                                column => $token->{column}};
2883                      next B;                      next B;
2884                    } elsif ($node->[1] & TABLE_CELL_EL) {                    } elsif ($node->[1] == TABLE_CELL_EL) {
2885                      !!!cp ('t180');                      !!!cp ('t180');
2886                      $tn = $node->[0]->manakai_local_name;                      $tn = $node->[0]->manakai_local_name;
2887                      ## NOTE: There is exactly one |td| or |th| element                      ## NOTE: There is exactly one |td| or |th| element
# Line 5421  sub _tree_construction_main ($) { Line 2901  sub _tree_construction_main ($) {
2901                  next B;                  next B;
2902                } # INSCOPE                } # INSCOPE
2903              } elsif ($token->{tag_name} eq 'table' and              } elsif ($token->{tag_name} eq 'table' and
2904                       $self->{insertion_mode} == IN_CAPTION_IM) {                       ($self->{insertion_mode} & IM_MASK) == IN_CAPTION_IM) {
2905                !!!parse-error (type => 'not closed', text => 'caption',                !!!parse-error (type => 'not closed', text => 'caption',
2906                                token => $token);                                token => $token);
2907    
# Line 5430  sub _tree_construction_main ($) { Line 2910  sub _tree_construction_main ($) {
2910                my $i;                my $i;
2911                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
2912                  my $node = $self->{open_elements}->[$_];                  my $node = $self->{open_elements}->[$_];
2913                  if ($node->[1] & CAPTION_EL) {                  if ($node->[1] == CAPTION_EL) {
2914                    !!!cp ('t184');                    !!!cp ('t184');
2915                    $i = $_;                    $i = $_;
2916                    last INSCOPE;                    last INSCOPE;
# Line 5441  sub _tree_construction_main ($) { Line 2921  sub _tree_construction_main ($) {
2921                } # INSCOPE                } # INSCOPE
2922                unless (defined $i) {                unless (defined $i) {
2923                  !!!cp ('t186');                  !!!cp ('t186');
2924            ## TODO: Wrong error type?
2925                  !!!parse-error (type => 'unmatched end tag',                  !!!parse-error (type => 'unmatched end tag',
2926                                  text => 'caption', token => $token);                                  text => 'caption', token => $token);
2927                  ## Ignore the token                  ## Ignore the token
# Line 5454  sub _tree_construction_main ($) { Line 2935  sub _tree_construction_main ($) {
2935                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
2936                }                }
2937    
2938                unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {                unless ($self->{open_elements}->[-1]->[1] == CAPTION_EL) {
2939                  !!!cp ('t188');                  !!!cp ('t188');
2940                  !!!parse-error (type => 'not closed',                  !!!parse-error (type => 'not closed',
2941                                  text => $self->{open_elements}->[-1]->[0]                                  text => $self->{open_elements}->[-1]->[0]
# Line 5486  sub _tree_construction_main ($) { Line 2967  sub _tree_construction_main ($) {
2967                  !!!cp ('t191');                  !!!cp ('t191');
2968                  #                  #
2969                }                }
2970              } elsif ({          } elsif ({
2971                        tbody => 1, tfoot => 1,                    tbody => 1, tfoot => 1,
2972                        thead => 1, tr => 1,                    thead => 1, tr => 1,
2973                       }->{$token->{tag_name}} and                   }->{$token->{tag_name}} and
2974                       $self->{insertion_mode} == IN_CAPTION_IM) {                   ($self->{insertion_mode} & IM_MASK) == IN_CAPTION_IM) {
2975                !!!cp ('t192');            !!!cp ('t192');
2976                !!!parse-error (type => 'unmatched end tag',            !!!parse-error (type => 'unmatched end tag',
2977                                text => $token->{tag_name}, token => $token);                            text => $token->{tag_name}, token => $token);
2978                ## Ignore the token            ## Ignore the token
2979                !!!next-token;            !!!next-token;
2980                next B;            next B;
2981              } else {          } else {
2982                !!!cp ('t193');            !!!cp ('t193');
2983                #            #
2984              }          }
2985        } elsif ($token->{type} == END_OF_FILE_TOKEN) {        } elsif ($token->{type} == END_OF_FILE_TOKEN) {
2986          for my $entry (@{$self->{open_elements}}) {          for my $entry (@{$self->{open_elements}}) {
2987            unless ($entry->[1] & ALL_END_TAG_OPTIONAL_EL) {            unless ($entry->[1] & ALL_END_TAG_OPTIONAL_EL) {
# Line 5535  sub _tree_construction_main ($) { Line 3016  sub _tree_construction_main ($) {
3016    
3017          !!!parse-error (type => 'in table:#text', token => $token);          !!!parse-error (type => 'in table:#text', token => $token);
3018    
3019              ## As if in body, but insert into foster parent element          ## NOTE: As if in body, but insert into the foster parent element.
3020              ## ISSUE: Spec says that "whenever a node would be inserted          $reconstruct_active_formatting_elements->($insert_to_foster);
             ## into the current node" while characters might not be  
             ## result in a new Text node.  
             $reconstruct_active_formatting_elements->($insert_to_foster);  
3021                            
3022              if ($self->{open_elements}->[-1]->[1] & TABLE_ROWS_EL) {          if ($self->{open_elements}->[-1]->[1] & TABLE_ROWS_EL) {
3023                # MUST            # MUST
3024                my $foster_parent_element;            my $foster_parent_element;
3025                my $next_sibling;            my $next_sibling;
3026                my $prev_sibling;            my $prev_sibling;
3027                OE: for (reverse 0..$#{$self->{open_elements}}) {            OE: for (reverse 0..$#{$self->{open_elements}}) {
3028                  if ($self->{open_elements}->[$_]->[1] & TABLE_EL) {              if ($self->{open_elements}->[$_]->[1] == TABLE_EL) {
3029                    my $parent = $self->{open_elements}->[$_]->[0]->parent_node;                my $parent = $self->{open_elements}->[$_]->[0]->parent_node;
3030                    if (defined $parent and $parent->node_type == 1) {                if (defined $parent and $parent->node_type == 1) {
3031                      !!!cp ('t196');                  $foster_parent_element = $parent;
3032                      $foster_parent_element = $parent;                  !!!cp ('t196');
3033                      $next_sibling = $self->{open_elements}->[$_]->[0];                  $next_sibling = $self->{open_elements}->[$_]->[0];
3034                      $prev_sibling = $next_sibling->previous_sibling;                  $prev_sibling = $next_sibling->previous_sibling;
3035                    } else {                  #
                     !!!cp ('t197');  
                     $foster_parent_element = $self->{open_elements}->[$_ - 1]->[0];  
                     $prev_sibling = $foster_parent_element->last_child;  
                   }  
                   last OE;  
                 }  
               } # OE  
               $foster_parent_element = $self->{open_elements}->[0]->[0] and  
               $prev_sibling = $foster_parent_element->last_child  
                 unless defined $foster_parent_element;  
               if (defined $prev_sibling and  
                   $prev_sibling->node_type == 3) {  
                 !!!cp ('t198');  
                 $prev_sibling->manakai_append_text ($token->{data});  
3036                } else {                } else {
3037                  !!!cp ('t199');                  !!!cp ('t197');
3038                  $foster_parent_element->insert_before                  $foster_parent_element = $self->{open_elements}->[$_ - 1]->[0];
3039                    ($self->{document}->create_text_node ($token->{data}),                  $prev_sibling = $foster_parent_element->last_child;
3040                     $next_sibling);                  #
3041                }                }
3042                  last OE;
3043                }
3044              } # OE
3045              $foster_parent_element = $self->{open_elements}->[0]->[0] and
3046              $prev_sibling = $foster_parent_element->last_child
3047                  unless defined $foster_parent_element;
3048              undef $prev_sibling unless $open_tables->[-1]->[2]; # ~node inserted
3049              if (defined $prev_sibling and
3050                  $prev_sibling->node_type == 3) {
3051                !!!cp ('t198');
3052                $prev_sibling->manakai_append_text ($token->{data});
3053              } else {
3054                !!!cp ('t199');
3055                $foster_parent_element->insert_before
3056                    ($self->{document}->create_text_node ($token->{data}),
3057                     $next_sibling);
3058              }
3059            $open_tables->[-1]->[1] = 1; # tainted            $open_tables->[-1]->[1] = 1; # tainted
3060              $open_tables->[-1]->[2] = 1; # ~node inserted
3061          } else {          } else {
3062              ## NOTE: Fragment case or in a foster parent'ed element
3063              ## (e.g. |<table><span>a|).  In fragment case, whether the
3064              ## character is appended to existing node or a new node is
3065              ## created is irrelevant, since the foster parent'ed nodes
3066              ## are discarded and fragment parsing does not invoke any
3067              ## script.
3068            !!!cp ('t200');            !!!cp ('t200');
3069            $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});            $self->{open_elements}->[-1]->[0]->manakai_append_text
3070                  ($token->{data});
3071          }          }
3072                            
3073          !!!next-token;          !!!next-token;
3074          next B;          next B;
3075        } elsif ($token->{type} == START_TAG_TOKEN) {        } elsif ($token->{type} == START_TAG_TOKEN) {
3076          if ({          if ({
3077               tr => ($self->{insertion_mode} != IN_ROW_IM),               tr => (($self->{insertion_mode} & IM_MASK) != IN_ROW_IM),
3078               th => 1, td => 1,               th => 1, td => 1,
3079              }->{$token->{tag_name}}) {              }->{$token->{tag_name}}) {
3080            if ($self->{insertion_mode} == IN_TABLE_IM) {            if (($self->{insertion_mode} & IM_MASK) == IN_TABLE_IM) {
3081              ## Clear back to table context              ## Clear back to table context
3082              while (not ($self->{open_elements}->[-1]->[1]              while (not ($self->{open_elements}->[-1]->[1]
3083                              & TABLE_SCOPING_EL)) {                              & TABLE_SCOPING_EL)) {
# Line 5601  sub _tree_construction_main ($) { Line 3090  sub _tree_construction_main ($) {
3090              ## reprocess in the "in table body" insertion mode...              ## reprocess in the "in table body" insertion mode...
3091            }            }
3092                        
3093            if ($self->{insertion_mode} == IN_TABLE_BODY_IM) {            if (($self->{insertion_mode} & IM_MASK) == IN_TABLE_BODY_IM) {
3094              unless ($token->{tag_name} eq 'tr') {              unless ($token->{tag_name} eq 'tr') {
3095                !!!cp ('t202');                !!!cp ('t202');
3096                !!!parse-error (type => 'missing start tag:tr', token => $token);                !!!parse-error (type => 'missing start tag:tr', token => $token);
# Line 5615  sub _tree_construction_main ($) { Line 3104  sub _tree_construction_main ($) {
3104                pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
3105              }              }
3106                                    
3107                  $self->{insertion_mode} = IN_ROW_IM;              $self->{insertion_mode} = IN_ROW_IM;
3108                  if ($token->{tag_name} eq 'tr') {              if ($token->{tag_name} eq 'tr') {
3109                    !!!cp ('t204');                !!!cp ('t204');
3110                    !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
3111                    !!!nack ('t204');                $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
3112                    !!!next-token;                !!!nack ('t204');
3113                    next B;                !!!next-token;
3114                  } else {                next B;
3115                    !!!cp ('t205');              } else {
3116                    !!!insert-element ('tr',, $token);                !!!cp ('t205');
3117                    ## reprocess in the "in row" insertion mode                !!!insert-element ('tr',, $token);
3118                  }                ## reprocess in the "in row" insertion mode
3119                } else {              }
3120                  !!!cp ('t206');            } else {
3121                }              !!!cp ('t206');
3122              }
3123    
3124                ## Clear back to table row context                ## Clear back to table row context
3125                while (not ($self->{open_elements}->[-1]->[1]                while (not ($self->{open_elements}->[-1]->[1]
# Line 5638  sub _tree_construction_main ($) { Line 3128  sub _tree_construction_main ($) {
3128                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
3129                }                }
3130                                
3131                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);            !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
3132                $self->{insertion_mode} = IN_CELL_IM;            $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
3133              $self->{insertion_mode} = IN_CELL_IM;
3134    
3135                push @$active_formatting_elements, ['#marker', ''];            push @$active_formatting_elements, ['#marker', ''];
3136                                
3137                !!!nack ('t207.1');            !!!nack ('t207.1');
3138              !!!next-token;
3139              next B;
3140            } elsif ({
3141                      caption => 1, col => 1, colgroup => 1,
3142                      tbody => 1, tfoot => 1, thead => 1,
3143                      tr => 1, # $self->{insertion_mode} == IN_ROW_IM
3144                     }->{$token->{tag_name}}) {
3145              if (($self->{insertion_mode} & IM_MASK) == IN_ROW_IM) {
3146                ## As if </tr>
3147                ## have an element in table scope
3148                my $i;
3149                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3150                  my $node = $self->{open_elements}->[$_];
3151                  if ($node->[1] == TABLE_ROW_EL) {
3152                    !!!cp ('t208');
3153                    $i = $_;
3154                    last INSCOPE;
3155                  } elsif ($node->[1] & TABLE_SCOPING_EL) {
3156                    !!!cp ('t209');
3157                    last INSCOPE;
3158                  }
3159                } # INSCOPE
3160                unless (defined $i) {
3161                  !!!cp ('t210');
3162                  ## TODO: This type is wrong.
3163                  !!!parse-error (type => 'unmacthed end tag',
3164                                  text => $token->{tag_name}, token => $token);
3165                  ## Ignore the token
3166                  !!!nack ('t210.1');
3167                !!!next-token;                !!!next-token;
3168                next B;                next B;
3169              } elsif ({              }
                       caption => 1, col => 1, colgroup => 1,  
                       tbody => 1, tfoot => 1, thead => 1,  
                       tr => 1, # $self->{insertion_mode} == IN_ROW_IM  
                      }->{$token->{tag_name}}) {  
               if ($self->{insertion_mode} == IN_ROW_IM) {  
                 ## As if </tr>  
                 ## have an element in table scope  
                 my $i;  
                 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
                   my $node = $self->{open_elements}->[$_];  
                   if ($node->[1] & TABLE_ROW_EL) {  
                     !!!cp ('t208');  
                     $i = $_;  
                     last INSCOPE;  
                   } elsif ($node->[1] & TABLE_SCOPING_EL) {  
                     !!!cp ('t209');  
                     last INSCOPE;  
                   }  
                 } # INSCOPE  
                 unless (defined $i) {  
                   !!!cp ('t210');  
 ## TODO: This type is wrong.  
                   !!!parse-error (type => 'unmacthed end tag',  
                                   text => $token->{tag_name}, token => $token);  
                   ## Ignore the token  
                   !!!nack ('t210.1');  
                   !!!next-token;  
                   next B;  
                 }  
3170                                    
3171                  ## Clear back to table row context                  ## Clear back to table row context
3172                  while (not ($self->{open_elements}->[-1]->[1]                  while (not ($self->{open_elements}->[-1]->[1]
# Line 5698  sub _tree_construction_main ($) { Line 3189  sub _tree_construction_main ($) {
3189                  }                  }
3190                }                }
3191    
3192                if ($self->{insertion_mode} == IN_TABLE_BODY_IM) {                if (($self->{insertion_mode} & IM_MASK) == IN_TABLE_BODY_IM) {
3193                  ## have an element in table scope                  ## have an element in table scope
3194                  my $i;                  my $i;
3195                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3196                    my $node = $self->{open_elements}->[$_];                    my $node = $self->{open_elements}->[$_];
3197                    if ($node->[1] & TABLE_ROW_GROUP_EL) {                    if ($node->[1] == TABLE_ROW_GROUP_EL) {
3198                      !!!cp ('t214');                      !!!cp ('t214');
3199                      $i = $_;                      $i = $_;
3200                      last INSCOPE;                      last INSCOPE;
# Line 5745  sub _tree_construction_main ($) { Line 3236  sub _tree_construction_main ($) {
3236                  !!!cp ('t218');                  !!!cp ('t218');
3237                }                }
3238    
3239                if ($token->{tag_name} eq 'col') {            if ($token->{tag_name} eq 'col') {
3240                  ## Clear back to table context              ## Clear back to table context
3241                  while (not ($self->{open_elements}->[-1]->[1]              while (not ($self->{open_elements}->[-1]->[1]
3242                                  & TABLE_SCOPING_EL)) {                              & TABLE_SCOPING_EL)) {
3243                    !!!cp ('t219');                !!!cp ('t219');
3244                    ## ISSUE: Can this state be reached?                ## ISSUE: Can this state be reached?
3245                    pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
3246                  }              }
3247                                
3248                  !!!insert-element ('colgroup',, $token);              !!!insert-element ('colgroup',, $token);
3249                  $self->{insertion_mode} = IN_COLUMN_GROUP_IM;              $self->{insertion_mode} = IN_COLUMN_GROUP_IM;
3250                  ## reprocess              ## reprocess
3251                  !!!ack-later;              $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
3252                  next B;              !!!ack-later;
3253                } elsif ({              next B;
3254                          caption => 1,            } elsif ({
3255                          colgroup => 1,                      caption => 1,
3256                          tbody => 1, tfoot => 1, thead => 1,                      colgroup => 1,
3257                         }->{$token->{tag_name}}) {                      tbody => 1, tfoot => 1, thead => 1,
3258                  ## Clear back to table context                     }->{$token->{tag_name}}) {
3259                ## Clear back to table context
3260                  while (not ($self->{open_elements}->[-1]->[1]                  while (not ($self->{open_elements}->[-1]->[1]
3261                                  & TABLE_SCOPING_EL)) {                                  & TABLE_SCOPING_EL)) {
3262                    !!!cp ('t220');                    !!!cp ('t220');
# Line 5772  sub _tree_construction_main ($) { Line 3264  sub _tree_construction_main ($) {
3264                    pop @{$self->{open_elements}};                    pop @{$self->{open_elements}};
3265                  }                  }
3266                                    
3267                  push @$active_formatting_elements, ['#marker', '']              push @$active_formatting_elements, ['#marker', '']
3268                      if $token->{tag_name} eq 'caption';                  if $token->{tag_name} eq 'caption';
3269                                    
3270                  !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);              !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
3271                  $self->{insertion_mode} = {              $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
3272                                             caption => IN_CAPTION_IM,              $self->{insertion_mode} = {
3273                                             colgroup => IN_COLUMN_GROUP_IM,                                         caption => IN_CAPTION_IM,
3274                                             tbody => IN_TABLE_BODY_IM,                                         colgroup => IN_COLUMN_GROUP_IM,
3275                                             tfoot => IN_TABLE_BODY_IM,                                         tbody => IN_TABLE_BODY_IM,
3276                                             thead => IN_TABLE_BODY_IM,                                         tfoot => IN_TABLE_BODY_IM,
3277                                            }->{$token->{tag_name}};                                         thead => IN_TABLE_BODY_IM,
3278                  !!!next-token;                                        }->{$token->{tag_name}};
3279                  !!!nack ('t220.1');              !!!next-token;
3280                  next B;              !!!nack ('t220.1');
3281                } else {              next B;
3282                  die "$0: in table: <>: $token->{tag_name}";            } else {
3283                }              die "$0: in table: <>: $token->{tag_name}";
3284              }
3285              } elsif ($token->{tag_name} eq 'table') {              } elsif ($token->{tag_name} eq 'table') {
3286                !!!parse-error (type => 'not closed',                !!!parse-error (type => 'not closed',
3287                                text => $self->{open_elements}->[-1]->[0]                                text => $self->{open_elements}->[-1]->[0]
# Line 5800  sub _tree_construction_main ($) { Line 3293  sub _tree_construction_main ($) {
3293                my $i;                my $i;
3294                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3295                  my $node = $self->{open_elements}->[$_];                  my $node = $self->{open_elements}->[$_];
3296                  if ($node->[1] & TABLE_EL) {                  if ($node->[1] == TABLE_EL) {
3297                    !!!cp ('t221');                    !!!cp ('t221');
3298                    $i = $_;                    $i = $_;
3299                    last INSCOPE;                    last INSCOPE;
# Line 5827  sub _tree_construction_main ($) { Line 3320  sub _tree_construction_main ($) {
3320                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
3321                }                }
3322    
3323                unless ($self->{open_elements}->[-1]->[1] & TABLE_EL) {                unless ($self->{open_elements}->[-1]->[1] == TABLE_EL) {
3324                  !!!cp ('t225');                  !!!cp ('t225');
3325                  ## NOTE: |<table><tr><table>|                  ## NOTE: |<table><tr><table>|
3326                  !!!parse-error (type => 'not closed',                  !!!parse-error (type => 'not closed',
# Line 5851  sub _tree_construction_main ($) { Line 3344  sub _tree_construction_main ($) {
3344              !!!cp ('t227.8');              !!!cp ('t227.8');
3345              ## NOTE: This is a "as if in head" code clone.              ## NOTE: This is a "as if in head" code clone.
3346              $parse_rcdata->(CDATA_CONTENT_MODEL);              $parse_rcdata->(CDATA_CONTENT_MODEL);
3347                $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
3348              next B;              next B;
3349            } else {            } else {
3350              !!!cp ('t227.7');              !!!cp ('t227.7');
# Line 5861  sub _tree_construction_main ($) { Line 3355  sub _tree_construction_main ($) {
3355              !!!cp ('t227.6');              !!!cp ('t227.6');
3356              ## NOTE: This is a "as if in head" code clone.              ## NOTE: This is a "as if in head" code clone.
3357              $script_start_tag->();              $script_start_tag->();
3358                $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
3359              next B;              next B;
3360            } else {            } else {
3361              !!!cp ('t227.5');              !!!cp ('t227.5');
# Line 5876  sub _tree_construction_main ($) { Line 3371  sub _tree_construction_main ($) {
3371                                  text => $token->{tag_name}, token => $token);                                  text => $token->{tag_name}, token => $token);
3372    
3373                  !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);                  !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
3374                    $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
3375    
3376                  ## TODO: form element pointer                  ## TODO: form element pointer
3377    
# Line 5907  sub _tree_construction_main ($) { Line 3403  sub _tree_construction_main ($) {
3403          $insert = $insert_to_foster;          $insert = $insert_to_foster;
3404          #          #
3405        } elsif ($token->{type} == END_TAG_TOKEN) {        } elsif ($token->{type} == END_TAG_TOKEN) {
3406              if ($token->{tag_name} eq 'tr' and          if ($token->{tag_name} eq 'tr' and
3407                  $self->{insertion_mode} == IN_ROW_IM) {              ($self->{insertion_mode} & IM_MASK) == IN_ROW_IM) {
3408                ## have an element in table scope            ## have an element in table scope
3409                my $i;                my $i;
3410                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3411                  my $node = $self->{open_elements}->[$_];                  my $node = $self->{open_elements}->[$_];
3412                  if ($node->[1] & TABLE_ROW_EL) {                  if ($node->[1] == TABLE_ROW_EL) {
3413                    !!!cp ('t228');                    !!!cp ('t228');
3414                    $i = $_;                    $i = $_;
3415                    last INSCOPE;                    last INSCOPE;
# Line 5948  sub _tree_construction_main ($) { Line 3444  sub _tree_construction_main ($) {
3444                !!!nack ('t231.1');                !!!nack ('t231.1');
3445                next B;                next B;
3446              } elsif ($token->{tag_name} eq 'table') {              } elsif ($token->{tag_name} eq 'table') {
3447                if ($self->{insertion_mode} == IN_ROW_IM) {                if (($self->{insertion_mode} & IM_MASK) == IN_ROW_IM) {
3448                  ## As if </tr>                  ## As if </tr>
3449                  ## have an element in table scope                  ## have an element in table scope
3450                  my $i;                  my $i;
3451                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3452                    my $node = $self->{open_elements}->[$_];                    my $node = $self->{open_elements}->[$_];
3453                    if ($node->[1] & TABLE_ROW_EL) {                    if ($node->[1] == TABLE_ROW_EL) {
3454                      !!!cp ('t233');                      !!!cp ('t233');
3455                      $i = $_;                      $i = $_;
3456                      last INSCOPE;                      last INSCOPE;
# Line 5987  sub _tree_construction_main ($) { Line 3483  sub _tree_construction_main ($) {
3483                  ## reprocess in the "in table body" insertion mode...                  ## reprocess in the "in table body" insertion mode...
3484                }                }
3485    
3486                if ($self->{insertion_mode} == IN_TABLE_BODY_IM) {                if (($self->{insertion_mode} & IM_MASK) == IN_TABLE_BODY_IM) {
3487                  ## have an element in table scope                  ## have an element in table scope
3488                  my $i;                  my $i;
3489                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3490                    my $node = $self->{open_elements}->[$_];                    my $node = $self->{open_elements}->[$_];
3491                    if ($node->[1] & TABLE_ROW_GROUP_EL) {                    if ($node->[1] == TABLE_ROW_GROUP_EL) {
3492                      !!!cp ('t237');                      !!!cp ('t237');
3493                      $i = $_;                      $i = $_;
3494                      last INSCOPE;                      last INSCOPE;
# Line 6039  sub _tree_construction_main ($) { Line 3535  sub _tree_construction_main ($) {
3535                my $i;                my $i;
3536                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3537                  my $node = $self->{open_elements}->[$_];                  my $node = $self->{open_elements}->[$_];
3538                  if ($node->[1] & TABLE_EL) {                  if ($node->[1] == TABLE_EL) {
3539                    !!!cp ('t241');                    !!!cp ('t241');
3540                    $i = $_;                    $i = $_;
3541                    last INSCOPE;                    last INSCOPE;
# Line 6069  sub _tree_construction_main ($) { Line 3565  sub _tree_construction_main ($) {
3565                        tbody => 1, tfoot => 1, thead => 1,                        tbody => 1, tfoot => 1, thead => 1,
3566                       }->{$token->{tag_name}} and                       }->{$token->{tag_name}} and
3567                       $self->{insertion_mode} & ROW_IMS) {                       $self->{insertion_mode} & ROW_IMS) {
3568                if ($self->{insertion_mode} == IN_ROW_IM) {                if (($self->{insertion_mode} & IM_MASK) == IN_ROW_IM) {
3569                  ## have an element in table scope                  ## have an element in table scope
3570                  my $i;                  my $i;
3571                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
# Line 6098  sub _tree_construction_main ($) { Line 3594  sub _tree_construction_main ($) {
3594                  my $i;                  my $i;
3595                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3596                    my $node = $self->{open_elements}->[$_];                    my $node = $self->{open_elements}->[$_];
3597                    if ($node->[1] & TABLE_ROW_EL) {                    if ($node->[1] == TABLE_ROW_EL) {
3598                      !!!cp ('t250');                      !!!cp ('t250');
3599                      $i = $_;                      $i = $_;
3600                      last INSCOPE;                      last INSCOPE;
# Line 6188  sub _tree_construction_main ($) { Line 3684  sub _tree_construction_main ($) {
3684            #            #
3685          }          }
3686        } elsif ($token->{type} == END_OF_FILE_TOKEN) {        } elsif ($token->{type} == END_OF_FILE_TOKEN) {
3687          unless ($self->{open_elements}->[-1]->[1] & HTML_EL and          unless ($self->{open_elements}->[-1]->[1] == HTML_EL and
3688                  @{$self->{open_elements}} == 1) { # redundant, maybe                  @{$self->{open_elements}} == 1) { # redundant, maybe
3689            !!!parse-error (type => 'in body:#eof', token => $token);            !!!parse-error (type => 'in body:#eof', token => $token);
3690            !!!cp ('t259.1');            !!!cp ('t259.1');
# Line 6203  sub _tree_construction_main ($) { Line 3699  sub _tree_construction_main ($) {
3699        } else {        } else {
3700          die "$0: $token->{type}: Unknown token type";          die "$0: $token->{type}: Unknown token type";
3701        }        }
3702      } elsif ($self->{insertion_mode} == IN_COLUMN_GROUP_IM) {      } elsif (($self->{insertion_mode} & IM_MASK) == IN_COLUMN_GROUP_IM) {
3703            if ($token->{type} == CHARACTER_TOKEN) {            if ($token->{type} == CHARACTER_TOKEN) {
3704              if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {              if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
3705                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
# Line 6230  sub _tree_construction_main ($) { Line 3726  sub _tree_construction_main ($) {
3726              }              }
3727            } elsif ($token->{type} == END_TAG_TOKEN) {            } elsif ($token->{type} == END_TAG_TOKEN) {
3728              if ($token->{tag_name} eq 'colgroup') {              if ($token->{tag_name} eq 'colgroup') {
3729                if ($self->{open_elements}->[-1]->[1] & HTML_EL) {                if ($self->{open_elements}->[-1]->[1] == HTML_EL) {
3730                  !!!cp ('t264');                  !!!cp ('t264');
3731                  !!!parse-error (type => 'unmatched end tag',                  !!!parse-error (type => 'unmatched end tag',
3732                                  text => 'colgroup', token => $token);                                  text => 'colgroup', token => $token);
# Line 6256  sub _tree_construction_main ($) { Line 3752  sub _tree_construction_main ($) {
3752                #                #
3753              }              }
3754        } elsif ($token->{type} == END_OF_FILE_TOKEN) {        } elsif ($token->{type} == END_OF_FILE_TOKEN) {
3755          if ($self->{open_elements}->[-1]->[1] & HTML_EL and          if ($self->{open_elements}->[-1]->[1] == HTML_EL and
3756              @{$self->{open_elements}} == 1) { # redundant, maybe              @{$self->{open_elements}} == 1) { # redundant, maybe
3757            !!!cp ('t270.2');            !!!cp ('t270.2');
3758            ## Stop parsing.            ## Stop parsing.
# Line 6274  sub _tree_construction_main ($) { Line 3770  sub _tree_construction_main ($) {
3770        }        }
3771    
3772            ## As if </colgroup>            ## As if </colgroup>
3773            if ($self->{open_elements}->[-1]->[1] & HTML_EL) {            if ($self->{open_elements}->[-1]->[1] == HTML_EL) {
3774              !!!cp ('t269');              !!!cp ('t269');
3775  ## TODO: Wrong error type?  ## TODO: Wrong error type?
3776              !!!parse-error (type => 'unmatched end tag',              !!!parse-error (type => 'unmatched end tag',
# Line 6299  sub _tree_construction_main ($) { Line 3795  sub _tree_construction_main ($) {
3795          next B;          next B;
3796        } elsif ($token->{type} == START_TAG_TOKEN) {        } elsif ($token->{type} == START_TAG_TOKEN) {
3797          if ($token->{tag_name} eq 'option') {          if ($token->{tag_name} eq 'option') {
3798            if ($self->{open_elements}->[-1]->[1] & OPTION_EL) {            if ($self->{open_elements}->[-1]->[1] == OPTION_EL) {
3799              !!!cp ('t272');              !!!cp ('t272');
3800              ## As if </option>              ## As if </option>
3801              pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
# Line 6312  sub _tree_construction_main ($) { Line 3808  sub _tree_construction_main ($) {
3808            !!!next-token;            !!!next-token;
3809            next B;            next B;
3810          } elsif ($token->{tag_name} eq 'optgroup') {          } elsif ($token->{tag_name} eq 'optgroup') {
3811            if ($self->{open_elements}->[-1]->[1] & OPTION_EL) {            if ($self->{open_elements}->[-1]->[1] == OPTION_EL) {
3812              !!!cp ('t274');              !!!cp ('t274');
3813              ## As if </option>              ## As if </option>
3814              pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
# Line 6320  sub _tree_construction_main ($) { Line 3816  sub _tree_construction_main ($) {
3816              !!!cp ('t275');              !!!cp ('t275');
3817            }            }
3818    
3819            if ($self->{open_elements}->[-1]->[1] & OPTGROUP_EL) {            if ($self->{open_elements}->[-1]->[1] == OPTGROUP_EL) {
3820              !!!cp ('t276');              !!!cp ('t276');
3821              ## As if </optgroup>              ## As if </optgroup>
3822              pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
# Line 6335  sub _tree_construction_main ($) { Line 3831  sub _tree_construction_main ($) {
3831          } elsif ({          } elsif ({
3832                     select => 1, input => 1, textarea => 1,                     select => 1, input => 1, textarea => 1,
3833                   }->{$token->{tag_name}} or                   }->{$token->{tag_name}} or
3834                   ($self->{insertion_mode} == IN_SELECT_IN_TABLE_IM and                   (($self->{insertion_mode} & IM_MASK)
3835                          == IN_SELECT_IN_TABLE_IM and
3836                    {                    {
3837                     caption => 1, table => 1,                     caption => 1, table => 1,
3838                     tbody => 1, tfoot => 1, thead => 1,                     tbody => 1, tfoot => 1, thead => 1,
# Line 6350  sub _tree_construction_main ($) { Line 3847  sub _tree_construction_main ($) {
3847            my $i;            my $i;
3848            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3849              my $node = $self->{open_elements}->[$_];              my $node = $self->{open_elements}->[$_];
3850              if ($node->[1] & SELECT_EL) {              if ($node->[1] == SELECT_EL) {
3851                !!!cp ('t278');                !!!cp ('t278');
3852                $i = $_;                $i = $_;
3853                last INSCOPE;                last INSCOPE;
# Line 6395  sub _tree_construction_main ($) { Line 3892  sub _tree_construction_main ($) {
3892          }          }
3893        } elsif ($token->{type} == END_TAG_TOKEN) {        } elsif ($token->{type} == END_TAG_TOKEN) {
3894          if ($token->{tag_name} eq 'optgroup') {          if ($token->{tag_name} eq 'optgroup') {
3895            if ($self->{open_elements}->[-1]->[1] & OPTION_EL and            if ($self->{open_elements}->[-1]->[1] == OPTION_EL and
3896                $self->{open_elements}->[-2]->[1] & OPTGROUP_EL) {                $self->{open_elements}->[-2]->[1] == OPTGROUP_EL) {
3897              !!!cp ('t283');              !!!cp ('t283');
3898              ## As if </option>              ## As if </option>
3899              splice @{$self->{open_elements}}, -2;              splice @{$self->{open_elements}}, -2;
3900            } elsif ($self->{open_elements}->[-1]->[1] & OPTGROUP_EL) {            } elsif ($self->{open_elements}->[-1]->[1] == OPTGROUP_EL) {
3901              !!!cp ('t284');              !!!cp ('t284');
3902              pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
3903            } else {            } else {
# Line 6413  sub _tree_construction_main ($) { Line 3910  sub _tree_construction_main ($) {
3910            !!!next-token;            !!!next-token;
3911            next B;            next B;
3912          } elsif ($token->{tag_name} eq 'option') {          } elsif ($token->{tag_name} eq 'option') {
3913            if ($self->{open_elements}->[-1]->[1] & OPTION_EL) {            if ($self->{open_elements}->[-1]->[1] == OPTION_EL) {
3914              !!!cp ('t286');              !!!cp ('t286');
3915              pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
3916            } else {            } else {
# Line 6430  sub _tree_construction_main ($) { Line 3927  sub _tree_construction_main ($) {
3927            my $i;            my $i;
3928            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3929              my $node = $self->{open_elements}->[$_];              my $node = $self->{open_elements}->[$_];
3930              if ($node->[1] & SELECT_EL) {              if ($node->[1] == SELECT_EL) {
3931                !!!cp ('t288');                !!!cp ('t288');
3932                $i = $_;                $i = $_;
3933                last INSCOPE;                last INSCOPE;
# Line 6457  sub _tree_construction_main ($) { Line 3954  sub _tree_construction_main ($) {
3954            !!!nack ('t291.1');            !!!nack ('t291.1');
3955            !!!next-token;            !!!next-token;
3956            next B;            next B;
3957          } elsif ($self->{insertion_mode} == IN_SELECT_IN_TABLE_IM and          } elsif (($self->{insertion_mode} & IM_MASK)
3958                         == IN_SELECT_IN_TABLE_IM and
3959                   {                   {
3960                    caption => 1, table => 1, tbody => 1,                    caption => 1, table => 1, tbody => 1,
3961                    tfoot => 1, thead => 1, tr => 1, td => 1, th => 1,                    tfoot => 1, thead => 1, tr => 1, td => 1, th => 1,
# Line 6492  sub _tree_construction_main ($) { Line 3990  sub _tree_construction_main ($) {
3990            undef $i;            undef $i;
3991            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3992              my $node = $self->{open_elements}->[$_];              my $node = $self->{open_elements}->[$_];
3993              if ($node->[1] & SELECT_EL) {              if ($node->[1] == SELECT_EL) {
3994                !!!cp ('t295');                !!!cp ('t295');
3995                $i = $_;                $i = $_;
3996                last INSCOPE;                last INSCOPE;
# Line 6531  sub _tree_construction_main ($) { Line 4029  sub _tree_construction_main ($) {
4029            next B;            next B;
4030          }          }
4031        } elsif ($token->{type} == END_OF_FILE_TOKEN) {        } elsif ($token->{type} == END_OF_FILE_TOKEN) {
4032          unless ($self->{open_elements}->[-1]->[1] & HTML_EL and          unless ($self->{open_elements}->[-1]->[1] == HTML_EL and
4033                  @{$self->{open_elements}} == 1) { # redundant, maybe                  @{$self->{open_elements}} == 1) { # redundant, maybe
4034            !!!cp ('t299.1');            !!!cp ('t299.1');
4035            !!!parse-error (type => 'in body:#eof', token => $token);            !!!parse-error (type => 'in body:#eof', token => $token);
# Line 6718  sub _tree_construction_main ($) { Line 4216  sub _tree_construction_main ($) {
4216        } elsif ($token->{type} == END_TAG_TOKEN) {        } elsif ($token->{type} == END_TAG_TOKEN) {
4217          if ($token->{tag_name} eq 'frameset' and          if ($token->{tag_name} eq 'frameset' and
4218              $self->{insertion_mode} == IN_FRAMESET_IM) {              $self->{insertion_mode} == IN_FRAMESET_IM) {
4219            if ($self->{open_elements}->[-1]->[1] & HTML_EL and            if ($self->{open_elements}->[-1]->[1] == HTML_EL and
4220                @{$self->{open_elements}} == 1) {                @{$self->{open_elements}} == 1) {
4221              !!!cp ('t325');              !!!cp ('t325');
4222              !!!parse-error (type => 'unmatched end tag',              !!!parse-error (type => 'unmatched end tag',
# Line 6732  sub _tree_construction_main ($) { Line 4230  sub _tree_construction_main ($) {
4230            }            }
4231    
4232            if (not defined $self->{inner_html_node} and            if (not defined $self->{inner_html_node} and
4233                not ($self->{open_elements}->[-1]->[1] & FRAMESET_EL)) {                not ($self->{open_elements}->[-1]->[1] == FRAMESET_EL)) {
4234              !!!cp ('t327');              !!!cp ('t327');
4235              $self->{insertion_mode} = AFTER_FRAMESET_IM;              $self->{insertion_mode} = AFTER_FRAMESET_IM;
4236            } else {            } else {
# Line 6764  sub _tree_construction_main ($) { Line 4262  sub _tree_construction_main ($) {
4262            next B;            next B;
4263          }          }
4264        } elsif ($token->{type} == END_OF_FILE_TOKEN) {        } elsif ($token->{type} == END_OF_FILE_TOKEN) {
4265          unless ($self->{open_elements}->[-1]->[1] & HTML_EL and          unless ($self->{open_elements}->[-1]->[1] == HTML_EL and
4266                  @{$self->{open_elements}} == 1) { # redundant, maybe                  @{$self->{open_elements}} == 1) { # redundant, maybe
4267            !!!cp ('t331.1');            !!!cp ('t331.1');
4268            !!!parse-error (type => 'in body:#eof', token => $token);            !!!parse-error (type => 'in body:#eof', token => $token);
# Line 6777  sub _tree_construction_main ($) { Line 4275  sub _tree_construction_main ($) {
4275        } else {        } else {
4276          die "$0: $token->{type}: Unknown token type";          die "$0: $token->{type}: Unknown token type";
4277        }        }
   
       ## ISSUE: An issue in spec here  
4278      } else {      } else {
4279        die "$0: $self->{insertion_mode}: Unknown insertion mode";        die "$0: $self->{insertion_mode}: Unknown insertion mode";
4280      }      }
# Line 6796  sub _tree_construction_main ($) { Line 4292  sub _tree_construction_main ($) {
4292          $parse_rcdata->(CDATA_CONTENT_MODEL);          $parse_rcdata->(CDATA_CONTENT_MODEL);
4293          next B;          next B;
4294        } elsif ({        } elsif ({
4295                  base => 1, link => 1,                  base => 1, command => 1, eventsource => 1, link => 1,
4296                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
4297          !!!cp ('t334');          !!!cp ('t334');
4298          ## NOTE: This is an "as if in head" code clone, only "-t" differs          ## NOTE: This is an "as if in head" code clone, only "-t" differs
4299          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
4300          pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.          pop @{$self->{open_elements}};
4301          !!!ack ('t334.1');          !!!ack ('t334.1');
4302          !!!next-token;          !!!next-token;
4303          next B;          next B;
4304        } elsif ($token->{tag_name} eq 'meta') {        } elsif ($token->{tag_name} eq 'meta') {
4305          ## NOTE: This is an "as if in head" code clone, only "-t" differs          ## NOTE: This is an "as if in head" code clone, only "-t" differs
4306          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
4307          my $meta_el = pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.          my $meta_el = pop @{$self->{open_elements}};
4308    
4309          unless ($self->{confident}) {          unless ($self->{confident}) {
4310            if ($token->{attributes}->{charset}) {            if ($token->{attributes}->{charset}) {
# Line 6869  sub _tree_construction_main ($) { Line 4365  sub _tree_construction_main ($) {
4365          !!!parse-error (type => 'in body', text => 'body', token => $token);          !!!parse-error (type => 'in body', text => 'body', token => $token);
4366                                
4367          if (@{$self->{open_elements}} == 1 or          if (@{$self->{open_elements}} == 1 or
4368              not ($self->{open_elements}->[1]->[1] & BODY_EL)) {              not ($self->{open_elements}->[1]->[1] == BODY_EL)) {
4369            !!!cp ('t342');            !!!cp ('t342');
4370            ## Ignore the token            ## Ignore the token
4371          } else {          } else {
# Line 6887  sub _tree_construction_main ($) { Line 4383  sub _tree_construction_main ($) {
4383          !!!next-token;          !!!next-token;
4384          next B;          next B;
4385        } elsif ({        } elsif ({
4386                  address => 1, blockquote => 1, center => 1, dir => 1,                  ## NOTE: Start tags for non-phrasing flow content elements
4387                  div => 1, dl => 1, fieldset => 1,  
4388                  h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,                  ## NOTE: The normal one
4389                  menu => 1, ol => 1, p => 1, ul => 1,                  address => 1, article => 1, aside => 1, blockquote => 1,
4390                    center => 1, datagrid => 1, details => 1, dialog => 1,
4391                    dir => 1, div => 1, dl => 1, fieldset => 1, figure => 1,
4392                    footer => 1, h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1,
4393                    h6 => 1, header => 1, menu => 1, nav => 1, ol => 1, p => 1,
4394                    section => 1, ul => 1,
4395                    ## NOTE: As normal, but drops leading newline
4396                  pre => 1, listing => 1,                  pre => 1, listing => 1,
4397                    ## NOTE: As normal, but interacts with the form element pointer
4398                  form => 1,                  form => 1,
4399                    
4400                  table => 1,                  table => 1,
4401                  hr => 1,                  hr => 1,
4402                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
# Line 6907  sub _tree_construction_main ($) { Line 4411  sub _tree_construction_main ($) {
4411    
4412          ## has a p element in scope          ## has a p element in scope
4413          INSCOPE: for (reverse @{$self->{open_elements}}) {          INSCOPE: for (reverse @{$self->{open_elements}}) {
4414            if ($_->[1] & P_EL) {            if ($_->[1] == P_EL) {
4415              !!!cp ('t344');              !!!cp ('t344');
4416              !!!back-token; # <form>              !!!back-token; # <form>
4417              $token = {type => END_TAG_TOKEN, tag_name => 'p',              $token = {type => END_TAG_TOKEN, tag_name => 'p',
# Line 6959  sub _tree_construction_main ($) { Line 4463  sub _tree_construction_main ($) {
4463            !!!next-token;            !!!next-token;
4464          }          }
4465          next B;          next B;
4466        } elsif ({li => 1, dt => 1, dd => 1}->{$token->{tag_name}}) {        } elsif ($token->{tag_name} eq 'li') {
4467          ## has a p element in scope          ## NOTE: As normal, but imply </li> when there's another <li> ...
         INSCOPE: for (reverse @{$self->{open_elements}}) {  
           if ($_->[1] & P_EL) {  
             !!!cp ('t353');  
             !!!back-token; # <x>  
             $token = {type => END_TAG_TOKEN, tag_name => 'p',  
                       line => $token->{line}, column => $token->{column}};  
             next B;  
           } elsif ($_->[1] & SCOPING_EL) {  
             !!!cp ('t354');  
             last INSCOPE;  
           }  
         } # INSCOPE  
4468    
4469          ## NOTE: Special, Scope (<li><foo><li> == <li><foo><li/></foo></li>)          ## NOTE: Special, Scope (<li><foo><li> == <li><foo><li/></foo></li>)
4470            ## Interpreted as <li><foo/></li><li/> (non-conforming)            ## Interpreted as <li><foo/></li><li/> (non-conforming)
# Line 6986  sub _tree_construction_main ($) { Line 4478  sub _tree_construction_main ($) {
4478          ## address, div, p (<li><foo><li> == <li><foo/></li><li/>)          ## address, div, p (<li><foo><li> == <li><foo/></li><li/>)
4479            ## Interpreted as <li><foo><li/></foo></li> (non-conforming)            ## Interpreted as <li><foo><li/></foo></li> (non-conforming)
4480            ## div (Fx, S)            ## div (Fx, S)
4481              
4482          ## Step 1          my $non_optional;
4483          my $i = -1;          my $i = -1;
4484          my $node = $self->{open_elements}->[$i];  
4485          my $li_or_dtdd = {li => {li => 1},          ## 1.
4486                            dt => {dt => 1, dd => 1},          for my $node (reverse @{$self->{open_elements}}) {
4487                            dd => {dt => 1, dd => 1}}->{$token->{tag_name}};            if ($node->[1] == LI_EL) {
4488          LI: {              ## 2. (a) As if </li>
4489            ## Step 2              {
4490            if ($li_or_dtdd->{$node->[0]->manakai_local_name}) {                ## If no </li> - not applied
4491              if ($i != -1) {                #
4492                !!!cp ('t355');  
4493                !!!parse-error (type => 'not closed',                ## Otherwise
4494                                text => $self->{open_elements}->[-1]->[0]  
4495                                    ->manakai_local_name,                ## 1. generate implied end tags, except for </li>
4496                                token => $token);                #
4497              } else {  
4498                !!!cp ('t356');                ## 2. If current node != "li", parse error
4499                  if ($non_optional) {
4500                    !!!parse-error (type => 'not closed',
4501                                    text => $non_optional->[0]->manakai_local_name,
4502                                    token => $token);
4503                    !!!cp ('t355');
4504                  } else {
4505                    !!!cp ('t356');
4506                  }
4507    
4508                  ## 3. Pop
4509                  splice @{$self->{open_elements}}, $i;
4510              }              }
4511              splice @{$self->{open_elements}}, $i;  
4512              last LI;              last; ## 2. (b) goto 5.
4513            } else {            } elsif (
4514                       ## NOTE: not "formatting" and not "phrasing"
4515                       ($node->[1] & SPECIAL_EL or
4516                        $node->[1] & SCOPING_EL) and
4517                       ## NOTE: "li", "dt", and "dd" are in |SPECIAL_EL|.
4518                       (not $node->[1] & ADDRESS_DIV_P_EL)
4519                      ) {
4520                ## 3.
4521              !!!cp ('t357');              !!!cp ('t357');
4522            }              last; ## goto 5.
4523                        } elsif ($node->[1] & END_TAG_OPTIONAL_EL) {
           ## Step 3  
           if (not ($node->[1] & FORMATTING_EL) and  
               #not $phrasing_category->{$node->[1]} and  
               ($node->[1] & SPECIAL_EL or  
                $node->[1] & SCOPING_EL) and  
               not ($node->[1] & ADDRESS_EL) and  
               not ($node->[1] & DIV_EL)) {  
4524              !!!cp ('t358');              !!!cp ('t358');
4525              last LI;              #
4526              } else {
4527                !!!cp ('t359');
4528                $non_optional ||= $node;
4529                #
4530            }            }
4531                        ## 4.
4532            !!!cp ('t359');            ## goto 2.
           ## Step 4  
4533            $i--;            $i--;
4534            $node = $self->{open_elements}->[$i];          }
4535            redo LI;  
4536          } # LI          ## 5. (a) has a |p| element in scope
4537                      INSCOPE: for (reverse @{$self->{open_elements}}) {
4538              if ($_->[1] == P_EL) {
4539                !!!cp ('t353');
4540    
4541                ## NOTE: |<p><li>|, for example.
4542    
4543                !!!back-token; # <x>
4544                $token = {type => END_TAG_TOKEN, tag_name => 'p',
4545                          line => $token->{line}, column => $token->{column}};
4546                next B;
4547              } elsif ($_->[1] & SCOPING_EL) {
4548                !!!cp ('t354');
4549                last INSCOPE;
4550              }
4551            } # INSCOPE
4552    
4553            ## 5. (b) insert
4554          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
4555          !!!nack ('t359.1');          !!!nack ('t359.1');
4556          !!!next-token;          !!!next-token;
4557          next B;          next B;
4558          } elsif ($token->{tag_name} eq 'dt' or
4559                   $token->{tag_name} eq 'dd') {
4560            ## NOTE: As normal, but imply </dt> or </dd> when ...
4561    
4562            my $non_optional;
4563            my $i = -1;
4564    
4565            ## 1.
4566            for my $node (reverse @{$self->{open_elements}}) {
4567              if ($node->[1] == DTDD_EL) {
4568                ## 2. (a) As if </li>
4569                {
4570                  ## If no </li> - not applied
4571                  #
4572    
4573                  ## Otherwise
4574    
4575                  ## 1. generate implied end tags, except for </dt> or </dd>
4576                  #
4577    
4578                  ## 2. If current node != "dt"|"dd", parse error
4579                  if ($non_optional) {
4580                    !!!parse-error (type => 'not closed',
4581                                    text => $non_optional->[0]->manakai_local_name,
4582                                    token => $token);
4583                    !!!cp ('t355.1');
4584                  } else {
4585                    !!!cp ('t356.1');
4586                  }
4587    
4588                  ## 3. Pop
4589                  splice @{$self->{open_elements}}, $i;
4590                }
4591    
4592                last; ## 2. (b) goto 5.
4593              } elsif (
4594                       ## NOTE: not "formatting" and not "phrasing"
4595                       ($node->[1] & SPECIAL_EL or
4596                        $node->[1] & SCOPING_EL) and
4597                       ## NOTE: "li", "dt", and "dd" are in |SPECIAL_EL|.
4598    
4599                       (not $node->[1] & ADDRESS_DIV_P_EL)
4600                      ) {
4601                ## 3.
4602                !!!cp ('t357.1');
4603                last; ## goto 5.
4604              } elsif ($node->[1] & END_TAG_OPTIONAL_EL) {
4605                !!!cp ('t358.1');
4606                #
4607              } else {
4608                !!!cp ('t359.1');
4609                $non_optional ||= $node;
4610                #
4611              }
4612              ## 4.
4613              ## goto 2.
4614              $i--;
4615            }
4616    
4617            ## 5. (a) has a |p| element in scope
4618            INSCOPE: for (reverse @{$self->{open_elements}}) {
4619              if ($_->[1] == P_EL) {
4620                !!!cp ('t353.1');
4621                !!!back-token; # <x>
4622                $token = {type => END_TAG_TOKEN, tag_name => 'p',
4623                          line => $token->{line}, column => $token->{column}};
4624                next B;
4625              } elsif ($_->[1] & SCOPING_EL) {
4626                !!!cp ('t354.1');
4627                last INSCOPE;
4628              }
4629            } # INSCOPE
4630    
4631            ## 5. (b) insert
4632            !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
4633            !!!nack ('t359.2');
4634            !!!next-token;
4635            next B;
4636        } elsif ($token->{tag_name} eq 'plaintext') {        } elsif ($token->{tag_name} eq 'plaintext') {
4637            ## NOTE: As normal, but effectively ends parsing
4638    
4639          ## has a p element in scope          ## has a p element in scope
4640          INSCOPE: for (reverse @{$self->{open_elements}}) {          INSCOPE: for (reverse @{$self->{open_elements}}) {
4641            if ($_->[1] & P_EL) {            if ($_->[1] == P_EL) {
4642              !!!cp ('t367');              !!!cp ('t367');
4643              !!!back-token; # <plaintext>              !!!back-token; # <plaintext>
4644              $token = {type => END_TAG_TOKEN, tag_name => 'p',              $token = {type => END_TAG_TOKEN, tag_name => 'p',
# Line 7058  sub _tree_construction_main ($) { Line 4660  sub _tree_construction_main ($) {
4660        } elsif ($token->{tag_name} eq 'a') {        } elsif ($token->{tag_name} eq 'a') {
4661          AFE: for my $i (reverse 0..$#$active_formatting_elements) {          AFE: for my $i (reverse 0..$#$active_formatting_elements) {
4662            my $node = $active_formatting_elements->[$i];            my $node = $active_formatting_elements->[$i];
4663            if ($node->[1] & A_EL) {            if ($node->[1] == A_EL) {
4664              !!!cp ('t371');              !!!cp ('t371');
4665              !!!parse-error (type => 'in a:a', token => $token);              !!!parse-error (type => 'in a:a', token => $token);
4666                            
# Line 7102  sub _tree_construction_main ($) { Line 4704  sub _tree_construction_main ($) {
4704          ## has a |nobr| element in scope          ## has a |nobr| element in scope
4705          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
4706            my $node = $self->{open_elements}->[$_];            my $node = $self->{open_elements}->[$_];
4707            if ($node->[1] & NOBR_EL) {            if ($node->[1] == NOBR_EL) {
4708              !!!cp ('t376');              !!!cp ('t376');
4709              !!!parse-error (type => 'in nobr:nobr', token => $token);              !!!parse-error (type => 'in nobr:nobr', token => $token);
4710              !!!back-token; # <nobr>              !!!back-token; # <nobr>
# Line 7125  sub _tree_construction_main ($) { Line 4727  sub _tree_construction_main ($) {
4727          ## has a button element in scope          ## has a button element in scope
4728          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
4729            my $node = $self->{open_elements}->[$_];            my $node = $self->{open_elements}->[$_];
4730            if ($node->[1] & BUTTON_EL) {            if ($node->[1] == BUTTON_EL) {
4731              !!!cp ('t378');              !!!cp ('t378');
4732              !!!parse-error (type => 'in button:button', token => $token);              !!!parse-error (type => 'in button:button', token => $token);
4733              !!!back-token; # <button>              !!!back-token; # <button>
# Line 7225  sub _tree_construction_main ($) { Line 4827  sub _tree_construction_main ($) {
4827            next B;            next B;
4828          }          }
4829        } elsif ($token->{tag_name} eq 'textarea') {        } elsif ($token->{tag_name} eq 'textarea') {
4830          my $tag_name = $token->{tag_name};          ## Step 1
4831          my $el;          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
         !!!create-element ($el, $HTML_NS, $token->{tag_name}, $token->{attributes}, $token);  
4832                    
4833            ## Step 2
4834          ## TODO: $self->{form_element} if defined          ## TODO: $self->{form_element} if defined
4835    
4836            ## Step 3
4837            $self->{ignore_newline} = 1;
4838    
4839            ## Step 4
4840            ## ISSUE: This step is wrong. (r2302 enbugged)
4841    
4842            ## Step 5
4843          $self->{content_model} = RCDATA_CONTENT_MODEL;          $self->{content_model} = RCDATA_CONTENT_MODEL;
4844          delete $self->{escape}; # MUST          delete $self->{escape}; # MUST
4845            
4846          $insert->($el);          ## Step 6-7
4847                    $self->{insertion_mode} |= IN_CDATA_RCDATA_IM;
4848          my $text = '';  
4849          !!!nack ('t392.1');          !!!nack ('t392.1');
4850          !!!next-token;          !!!next-token;
4851          if ($token->{type} == CHARACTER_TOKEN) {          next B;
4852            $token->{data} =~ s/^\x0A//;        } elsif ($token->{tag_name} eq 'optgroup' or
4853            unless (length $token->{data}) {                 $token->{tag_name} eq 'option') {
4854              !!!cp ('t392');          ## has an |option| element in scope
4855              !!!next-token;          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
4856            } else {            my $node = $self->{open_elements}->[$_];
4857              !!!cp ('t393');            if ($node->[1] == OPTION_EL) {
4858                !!!cp ('t397.1');
4859                ## NOTE: As if </option>
4860                !!!back-token; # <option> or <optgroup>
4861                $token = {type => END_TAG_TOKEN, tag_name => 'option',
4862                          line => $token->{line}, column => $token->{column}};
4863                next B;
4864              } elsif ($node->[1] & SCOPING_EL) {
4865                !!!cp ('t397.2');
4866                last INSCOPE;
4867            }            }
4868          } else {          } # INSCOPE
4869            !!!cp ('t394');  
4870          }          $reconstruct_active_formatting_elements->($insert_to_current);
4871          while ($token->{type} == CHARACTER_TOKEN) {  
4872            !!!cp ('t395');          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
4873            $text .= $token->{data};  
4874            !!!next-token;          !!!nack ('t397.3');
         }  
         if (length $text) {  
           !!!cp ('t396');  
           $el->manakai_append_text ($text);  
         }  
           
         $self->{content_model} = PCDATA_CONTENT_MODEL;  
           
         if ($token->{type} == END_TAG_TOKEN and  
             $token->{tag_name} eq $tag_name) {  
           !!!cp ('t397');  
           ## Ignore the token  
         } else {  
           !!!cp ('t398');  
           !!!parse-error (type => 'in RCDATA:#eof', token => $token);  
         }  
4875          !!!next-token;          !!!next-token;
4876          next B;          redo B;
4877        } elsif ($token->{tag_name} eq 'rt' or        } elsif ($token->{tag_name} eq 'rt' or
4878                 $token->{tag_name} eq 'rp') {                 $token->{tag_name} eq 'rp') {
4879          ## has a |ruby| element in scope          ## has a |ruby| element in scope
4880          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
4881            my $node = $self->{open_elements}->[$_];            my $node = $self->{open_elements}->[$_];
4882            if ($node->[1] & RUBY_EL) {            if ($node->[1] == RUBY_EL) {
4883              !!!cp ('t398.1');              !!!cp ('t398.1');
4884              ## generate implied end tags              ## generate implied end tags
4885              while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {              while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
4886                !!!cp ('t398.2');                !!!cp ('t398.2');
4887                pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
4888              }              }
4889              unless ($self->{open_elements}->[-1]->[1] & RUBY_EL) {              unless ($self->{open_elements}->[-1]->[1] == RUBY_EL) {
4890                !!!cp ('t398.3');                !!!cp ('t398.3');
4891                !!!parse-error (type => 'not closed',                !!!parse-error (type => 'not closed',
4892                                text => $self->{open_elements}->[-1]->[0]                                text => $self->{open_elements}->[-1]->[0]
4893                                    ->manakai_local_name,                                    ->manakai_local_name,
4894                                token => $token);                                token => $token);
4895                pop @{$self->{open_elements}}                pop @{$self->{open_elements}}
4896                    while not $self->{open_elements}->[-1]->[1] & RUBY_EL;                    while not $self->{open_elements}->[-1]->[1] == RUBY_EL;
4897              }              }
4898              last INSCOPE;              last INSCOPE;
4899            } elsif ($node->[1] & SCOPING_EL) {            } elsif ($node->[1] & SCOPING_EL) {
# Line 7318  sub _tree_construction_main ($) { Line 4921  sub _tree_construction_main ($) {
4921                    
4922          if ($self->{self_closing}) {          if ($self->{self_closing}) {
4923            pop @{$self->{open_elements}};            pop @{$self->{open_elements}};
4924            !!!ack ('t398.1');            !!!ack ('t398.6');
4925          } else {          } else {
4926            !!!cp ('t398.2');            !!!cp ('t398.7');
4927            $self->{insertion_mode} |= IN_FOREIGN_CONTENT_IM;            $self->{insertion_mode} |= IN_FOREIGN_CONTENT_IM;
4928            ## NOTE: |<body><math><mi><svg>| -> "in foreign content" insertion            ## NOTE: |<body><math><mi><svg>| -> "in foreign content" insertion
4929            ## mode, "in body" (not "in foreign content") secondary insertion            ## mode, "in body" (not "in foreign content") secondary insertion
# Line 7331  sub _tree_construction_main ($) { Line 4934  sub _tree_construction_main ($) {
4934          next B;          next B;
4935        } elsif ({        } elsif ({
4936                  caption => 1, col => 1, colgroup => 1, frame => 1,                  caption => 1, col => 1, colgroup => 1, frame => 1,
4937                  frameset => 1, head => 1, option => 1, optgroup => 1,                  frameset => 1, head => 1,
4938                  tbody => 1, td => 1, tfoot => 1, th => 1,                  tbody => 1, td => 1, tfoot => 1, th => 1,
4939                  thead => 1, tr => 1,                  thead => 1, tr => 1,
4940                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
# Line 7342  sub _tree_construction_main ($) { Line 4945  sub _tree_construction_main ($) {
4945          !!!nack ('t401.1'); ## NOTE: |<col/>| or |<frame/>| here is an error.          !!!nack ('t401.1'); ## NOTE: |<col/>| or |<frame/>| here is an error.
4946          !!!next-token;          !!!next-token;
4947          next B;          next B;
4948                  } elsif ($token->{tag_name} eq 'param' or
4949          ## ISSUE: An issue on HTML5 new elements in the spec.                 $token->{tag_name} eq 'source') {
4950            !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
4951            pop @{$self->{open_elements}};
4952    
4953            !!!ack ('t398.5');
4954            !!!next-token;
4955            redo B;
4956        } else {        } else {
4957          if ($token->{tag_name} eq 'image') {          if ($token->{tag_name} eq 'image') {
4958            !!!cp ('t384');            !!!cp ('t384');
# Line 7379  sub _tree_construction_main ($) { Line 4988  sub _tree_construction_main ($) {
4988            !!!ack ('t388.2');            !!!ack ('t388.2');
4989          } elsif ({          } elsif ({
4990                    area => 1, basefont => 1, bgsound => 1, br => 1,                    area => 1, basefont => 1, bgsound => 1, br => 1,
4991                    embed => 1, img => 1, param => 1, spacer => 1, wbr => 1,                    embed => 1, img => 1, spacer => 1, wbr => 1,
                   #image => 1,  
4992                   }->{$token->{tag_name}}) {                   }->{$token->{tag_name}}) {
4993            !!!cp ('t388.1');            !!!cp ('t388.1');
4994            pop @{$self->{open_elements}};            pop @{$self->{open_elements}};
# Line 7390  sub _tree_construction_main ($) { Line 4998  sub _tree_construction_main ($) {
4998                    
4999            if ($self->{insertion_mode} & TABLE_IMS or            if ($self->{insertion_mode} & TABLE_IMS or
5000                $self->{insertion_mode} & BODY_TABLE_IMS or                $self->{insertion_mode} & BODY_TABLE_IMS or
5001                $self->{insertion_mode} == IN_COLUMN_GROUP_IM) {                ($self->{insertion_mode} & IM_MASK) == IN_COLUMN_GROUP_IM) {
5002              !!!cp ('t400.1');              !!!cp ('t400.1');
5003              $self->{insertion_mode} = IN_SELECT_IN_TABLE_IM;              $self->{insertion_mode} = IN_SELECT_IN_TABLE_IM;
5004            } else {            } else {
# Line 7411  sub _tree_construction_main ($) { Line 5019  sub _tree_construction_main ($) {
5019          my $i;          my $i;
5020          INSCOPE: {          INSCOPE: {
5021            for (reverse @{$self->{open_elements}}) {            for (reverse @{$self->{open_elements}}) {
5022              if ($_->[1] & BODY_EL) {              if ($_->[1] == BODY_EL) {
5023                !!!cp ('t405');                !!!cp ('t405');
5024                $i = $_;                $i = $_;
5025                last INSCOPE;                last INSCOPE;
# Line 7421  sub _tree_construction_main ($) { Line 5029  sub _tree_construction_main ($) {
5029              }              }
5030            }            }
5031    
5032            !!!parse-error (type => 'start tag not allowed',            ## NOTE: |<marquee></body>|, |<svg><foreignobject></body>|
5033    
5034              !!!parse-error (type => 'unmatched end tag',
5035                            text => $token->{tag_name}, token => $token);                            text => $token->{tag_name}, token => $token);
5036            ## NOTE: Ignore the token.            ## NOTE: Ignore the token.
5037            !!!next-token;            !!!next-token;
# Line 7447  sub _tree_construction_main ($) { Line 5057  sub _tree_construction_main ($) {
5057          ## TODO: Update this code.  It seems that the code below is not          ## TODO: Update this code.  It seems that the code below is not
5058          ## up-to-date, though it has same effect as speced.          ## up-to-date, though it has same effect as speced.
5059          if (@{$self->{open_elements}} > 1 and          if (@{$self->{open_elements}} > 1 and
5060              $self->{open_elements}->[1]->[1] & BODY_EL) {              $self->{open_elements}->[1]->[1] == BODY_EL) {
5061            ## ISSUE: There is an issue in the spec.            unless ($self->{open_elements}->[-1]->[1] == BODY_EL) {
           unless ($self->{open_elements}->[-1]->[1] & BODY_EL) {  
5062              !!!cp ('t406');              !!!cp ('t406');
5063              !!!parse-error (type => 'not closed',              !!!parse-error (type => 'not closed',
5064                              text => $self->{open_elements}->[1]->[0]                              text => $self->{open_elements}->[1]->[0]
# Line 7470  sub _tree_construction_main ($) { Line 5079  sub _tree_construction_main ($) {
5079            next B;            next B;
5080          }          }
5081        } elsif ({        } elsif ({
5082                  address => 1, blockquote => 1, center => 1, dir => 1,                  ## NOTE: End tags for non-phrasing flow content elements
5083                  div => 1, dl => 1, fieldset => 1, listing => 1,  
5084                  menu => 1, ol => 1, pre => 1, ul => 1,                  ## NOTE: The normal ones
5085                    address => 1, article => 1, aside => 1, blockquote => 1,
5086                    center => 1, datagrid => 1, details => 1, dialog => 1,
5087                    dir => 1, div => 1, dl => 1, fieldset => 1, figure => 1,
5088                    footer => 1, header => 1, listing => 1, menu => 1, nav => 1,
5089                    ol => 1, pre => 1, section => 1, ul => 1,
5090    
5091                    ## NOTE: As normal, but ... optional tags
5092                  dd => 1, dt => 1, li => 1,                  dd => 1, dt => 1, li => 1,
5093    
5094                  applet => 1, button => 1, marquee => 1, object => 1,                  applet => 1, button => 1, marquee => 1, object => 1,
5095                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
5096            ## NOTE: Code for <li> start tags includes "as if </li>" code.
5097            ## Code for <dt> or <dd> start tags includes "as if </dt> or
5098            ## </dd>" code.
5099    
5100          ## has an element in scope          ## has an element in scope
5101          my $i;          my $i;
5102          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
# Line 7502  sub _tree_construction_main ($) { Line 5123  sub _tree_construction_main ($) {
5123                    dd => ($token->{tag_name} ne 'dd'),                    dd => ($token->{tag_name} ne 'dd'),
5124                    dt => ($token->{tag_name} ne 'dt'),                    dt => ($token->{tag_name} ne 'dt'),
5125                    li => ($token->{tag_name} ne 'li'),                    li => ($token->{tag_name} ne 'li'),
5126                      option => 1,
5127                      optgroup => 1,
5128                    p => 1,                    p => 1,
5129                    rt => 1,                    rt => 1,
5130                    rp => 1,                    rp => 1,
# Line 7534  sub _tree_construction_main ($) { Line 5157  sub _tree_construction_main ($) {
5157          !!!next-token;          !!!next-token;
5158          next B;          next B;
5159        } elsif ($token->{tag_name} eq 'form') {        } elsif ($token->{tag_name} eq 'form') {
5160            ## NOTE: As normal, but interacts with the form element pointer
5161    
5162          undef $self->{form_element};          undef $self->{form_element};
5163    
5164          ## has an element in scope          ## has an element in scope
5165          my $i;          my $i;
5166          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5167            my $node = $self->{open_elements}->[$_];            my $node = $self->{open_elements}->[$_];
5168            if ($node->[1] & FORM_EL) {            if ($node->[1] == FORM_EL) {
5169              !!!cp ('t418');              !!!cp ('t418');
5170              $i = $_;              $i = $_;
5171              last INSCOPE;              last INSCOPE;
# Line 7581  sub _tree_construction_main ($) { Line 5206  sub _tree_construction_main ($) {
5206          !!!next-token;          !!!next-token;
5207          next B;          next B;
5208        } elsif ({        } elsif ({
5209                    ## NOTE: As normal, except acts as a closer for any ...
5210                  h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,                  h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,
5211                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
5212          ## has an element in scope          ## has an element in scope
5213          my $i;          my $i;
5214          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5215            my $node = $self->{open_elements}->[$_];            my $node = $self->{open_elements}->[$_];
5216            if ($node->[1] & HEADING_EL) {            if ($node->[1] == HEADING_EL) {
5217              !!!cp ('t423');              !!!cp ('t423');
5218              $i = $_;              $i = $_;
5219              last INSCOPE;              last INSCOPE;
# Line 7626  sub _tree_construction_main ($) { Line 5252  sub _tree_construction_main ($) {
5252          !!!next-token;          !!!next-token;
5253          next B;          next B;
5254        } elsif ($token->{tag_name} eq 'p') {        } elsif ($token->{tag_name} eq 'p') {
5255            ## NOTE: As normal, except </p> implies <p> and ...
5256    
5257          ## has an element in scope          ## has an element in scope
5258            my $non_optional;
5259          my $i;          my $i;
5260          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5261            my $node = $self->{open_elements}->[$_];            my $node = $self->{open_elements}->[$_];
5262            if ($node->[1] & P_EL) {            if ($node->[1] == P_EL) {
5263              !!!cp ('t410.1');              !!!cp ('t410.1');
5264              $i = $_;              $i = $_;
5265              last INSCOPE;              last INSCOPE;
5266            } elsif ($node->[1] & SCOPING_EL) {            } elsif ($node->[1] & SCOPING_EL) {
5267              !!!cp ('t411.1');              !!!cp ('t411.1');
5268              last INSCOPE;              last INSCOPE;
5269              } elsif ($node->[1] & END_TAG_OPTIONAL_EL) {
5270                ## NOTE: |END_TAG_OPTIONAL_EL| includes "p"
5271                !!!cp ('t411.2');
5272                #
5273              } else {
5274                !!!cp ('t411.3');
5275                $non_optional ||= $node;
5276                #
5277            }            }
5278          } # INSCOPE          } # INSCOPE
5279    
5280          if (defined $i) {          if (defined $i) {
5281            if ($self->{open_elements}->[-1]->[0]->manakai_local_name            ## 1. Generate implied end tags
5282                    ne $token->{tag_name}) {            #
5283    
5284              ## 2. If current node != "p", parse error
5285              if ($non_optional) {
5286              !!!cp ('t412.1');              !!!cp ('t412.1');
5287              !!!parse-error (type => 'not closed',              !!!parse-error (type => 'not closed',
5288                              text => $self->{open_elements}->[-1]->[0]                              text => $non_optional->[0]->manakai_local_name,
                                 ->manakai_local_name,  
5289                              token => $token);                              token => $token);
5290            } else {            } else {
5291              !!!cp ('t414.1');              !!!cp ('t414.1');
5292            }            }
5293    
5294              ## 3. Pop
5295            splice @{$self->{open_elements}}, $i;            splice @{$self->{open_elements}}, $i;
5296          } else {          } else {
5297            !!!cp ('t413.1');            !!!cp ('t413.1');
# Line 7692  sub _tree_construction_main ($) { Line 5332  sub _tree_construction_main ($) {
5332          ## Ignore the token.          ## Ignore the token.
5333          !!!next-token;          !!!next-token;
5334          next B;          next B;
       } elsif ({  
                 caption => 1, col => 1, colgroup => 1, frame => 1,  
                 frameset => 1, head => 1, option => 1, optgroup => 1,  
                 tbody => 1, td => 1, tfoot => 1, th => 1,  
                 thead => 1, tr => 1,  
                 area => 1, basefont => 1, bgsound => 1,  
                 embed => 1, hr => 1, iframe => 1, image => 1,  
                 img => 1, input => 1, isindex => 1, noembed => 1,  
                 noframes => 1, param => 1, select => 1, spacer => 1,  
                 table => 1, textarea => 1, wbr => 1,  
                 noscript => 0, ## TODO: if scripting is enabled  
                }->{$token->{tag_name}}) {  
         !!!cp ('t429');  
         !!!parse-error (type => 'unmatched end tag',  
                         text => $token->{tag_name}, token => $token);  
         ## Ignore the token  
         !!!next-token;  
         next B;  
           
         ## ISSUE: Issue on HTML5 new elements in spec  
           
5335        } else {        } else {
5336            if ($token->{tag_name} eq 'sarcasm') {
5337              sleep 0.001; # take a deep breath
5338            }
5339    
5340          ## Step 1          ## Step 1
5341          my $node_i = -1;          my $node_i = -1;
5342          my $node = $self->{open_elements}->[$node_i];          my $node = $self->{open_elements}->[$node_i];
5343    
5344          ## Step 2          ## Step 2
5345          S2: {          S2: {
5346            if ($node->[0]->manakai_local_name eq $token->{tag_name}) {            my $node_tag_name = $node->[0]->manakai_local_name;
5347              $node_tag_name =~ tr/A-Z/a-z/; # for SVG camelCase tag names
5348              if ($node_tag_name eq $token->{tag_name}) {
5349              ## Step 1              ## Step 1
5350              ## generate implied end tags              ## generate implied end tags
5351              while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {              while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
# Line 7733  sub _tree_construction_main ($) { Line 5358  sub _tree_construction_main ($) {
5358              }              }
5359                    
5360              ## Step 2              ## Step 2
5361              if ($self->{open_elements}->[-1]->[0]->manakai_local_name              my $current_tag_name
5362                      ne $token->{tag_name}) {                  = $self->{open_elements}->[-1]->[0]->manakai_local_name;
5363                $current_tag_name =~ tr/A-Z/a-z/;
5364                if ($current_tag_name ne $token->{tag_name}) {
5365                !!!cp ('t431');                !!!cp ('t431');
5366                ## NOTE: <x><y></x>                ## NOTE: <x><y></x>
5367                !!!parse-error (type => 'not closed',                !!!parse-error (type => 'not closed',
# Line 7988  sub set_inner_html ($$$$;$) { Line 5615  sub set_inner_html ($$$$;$) {
5615      }->{$node_ln};      }->{$node_ln};
5616      $p->{content_model} = PCDATA_CONTENT_MODEL      $p->{content_model} = PCDATA_CONTENT_MODEL
5617          unless defined $p->{content_model};          unless defined $p->{content_model};
         ## ISSUE: What is "the name of the element"? local name?  
5618    
5619      $p->{inner_html_node} = [$node, $el_category->{$node_ln}];      $p->{inner_html_node} = [$node, $el_category->{$node_ln}];
5620        ## TODO: Foreign element OK?        ## TODO: Foreign element OK?
# Line 8004  sub set_inner_html ($$$$;$) { Line 5630  sub set_inner_html ($$$$;$) {
5630      push @{$p->{open_elements}}, [$root, $el_category->{html}];      push @{$p->{open_elements}}, [$root, $el_category->{html}];
5631    
5632      undef $p->{head_element};      undef $p->{head_element};
5633        undef $p->{head_element_inserted};
5634    
5635      ## Step 6 # MUST      ## Step 6 # MUST
5636      $p->_reset_insertion_mode;      $p->_reset_insertion_mode;

Legend:
Removed from v.1.193  
changed lines
  Added in v.1.211

[email protected]
ViewVC Help
Powered by ViewVC 1.1.24