/[suikacvs]/markup/html/whatpm/Whatpm/HTML.pm.src
Suika

Diff of /markup/html/whatpm/Whatpm/HTML.pm.src

Parent Directory Parent Directory | Revision Log Revision Log | View Patch Patch

revision 1.194 by wakaba, Sat Oct 4 05:53:45 2008 UTC revision 1.210 by wakaba, Tue Oct 14 13:24:52 2008 UTC
# Line 3  use strict; Line 3  use strict;
3  our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r};  our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r};
4  use Error qw(:try);  use Error qw(:try);
5    
6    use Whatpm::HTML::Tokenizer;
7    
8  ## NOTE: This module don't check all HTML5 parse errors; character  ## NOTE: This module don't check all HTML5 parse errors; character
9  ## encoding related parse errors are expected to be handled by relevant  ## encoding related parse errors are expected to be handled by relevant
10  ## modules.  ## modules.
# Line 21  use Error qw(:try); Line 23  use Error qw(:try);
23    
24  require IO::Handle;  require IO::Handle;
25    
26    ## Namespace URLs
27    
28  my $HTML_NS = q<http://www.w3.org/1999/xhtml>;  my $HTML_NS = q<http://www.w3.org/1999/xhtml>;
29  my $MML_NS = q<http://www.w3.org/1998/Math/MathML>;  my $MML_NS = q<http://www.w3.org/1998/Math/MathML>;
30  my $SVG_NS = q<http://www.w3.org/2000/svg>;  my $SVG_NS = q<http://www.w3.org/2000/svg>;
# Line 28  my $XLINK_NS = q<http://www.w3.org/1999/ Line 32  my $XLINK_NS = q<http://www.w3.org/1999/
32  my $XML_NS = q<http://www.w3.org/XML/1998/namespace>;  my $XML_NS = q<http://www.w3.org/XML/1998/namespace>;
33  my $XMLNS_NS = q<http://www.w3.org/2000/xmlns/>;  my $XMLNS_NS = q<http://www.w3.org/2000/xmlns/>;
34    
35  sub A_EL () { 0b1 }  ## Element categories
36  sub ADDRESS_EL () { 0b10 }  
37  sub BODY_EL () { 0b100 }  ## Bits 12-15
38  sub BUTTON_EL () { 0b1000 }  sub SPECIAL_EL () { 0b1_000000000000000 }
39  sub CAPTION_EL () { 0b10000 }  sub SCOPING_EL () { 0b1_00000000000000 }
40  sub DD_EL () { 0b100000 }  sub FORMATTING_EL () { 0b1_0000000000000 }
41  sub DIV_EL () { 0b1000000 }  sub PHRASING_EL () { 0b1_000000000000 }
42  sub DT_EL () { 0b10000000 }  
43  sub FORM_EL () { 0b100000000 }  ## Bits 10-11
44  sub FORMATTING_EL () { 0b1000000000 }  #sub FOREIGN_EL () { 0b1_00000000000 } # see Whatpm::HTML::Tokenizer
45  sub FRAMESET_EL () { 0b10000000000 }  sub FOREIGN_FLOW_CONTENT_EL () { 0b1_0000000000 }
46  sub HEADING_EL () { 0b100000000000 }  
47  sub HTML_EL () { 0b1000000000000 }  ## Bits 6-9
48  sub LI_EL () { 0b10000000000000 }  sub TABLE_SCOPING_EL () { 0b1_000000000 }
49  sub NOBR_EL () { 0b100000000000000 }  sub TABLE_ROWS_SCOPING_EL () { 0b1_00000000 }
50  sub OPTION_EL () { 0b1000000000000000 }  sub TABLE_ROW_SCOPING_EL () { 0b1_0000000 }
51  sub OPTGROUP_EL () { 0b10000000000000000 }  sub TABLE_ROWS_EL () { 0b1_000000 }
52  sub P_EL () { 0b100000000000000000 }  
53  sub SELECT_EL () { 0b1000000000000000000 }  ## Bit 5
54  sub TABLE_EL () { 0b10000000000000000000 }  sub ADDRESS_DIV_P_EL () { 0b1_00000 }
55  sub TABLE_CELL_EL () { 0b100000000000000000000 }  
56  sub TABLE_ROW_EL () { 0b1000000000000000000000 }  ## NOTE: Used in </body> and EOF algorithms.
57  sub TABLE_ROW_GROUP_EL () { 0b10000000000000000000000 }  ## Bit 4
58  sub MISC_SCOPING_EL () { 0b100000000000000000000000 }  sub ALL_END_TAG_OPTIONAL_EL () { 0b1_0000 }
 sub MISC_SPECIAL_EL () { 0b1000000000000000000000000 }  
 sub FOREIGN_EL () { 0b10000000000000000000000000 }  
 sub FOREIGN_FLOW_CONTENT_EL () { 0b100000000000000000000000000 }  
 sub MML_AXML_EL () { 0b1000000000000000000000000000 }  
 sub RUBY_EL () { 0b10000000000000000000000000000 }  
 sub RUBY_COMPONENT_EL () { 0b100000000000000000000000000000 }  
   
 sub TABLE_ROWS_EL () {  
   TABLE_EL |  
   TABLE_ROW_EL |  
   TABLE_ROW_GROUP_EL  
 }  
59    
60  ## NOTE: Used in "generate implied end tags" algorithm.  ## NOTE: Used in "generate implied end tags" algorithm.
61  ## NOTE: There is a code where a modified version of  ## NOTE: There is a code where a modified version of
62  ## END_TAG_OPTIONAL_EL is used in "generate implied end tags"  ## END_TAG_OPTIONAL_EL is used in "generate implied end tags"
63  ## implementation (search for the algorithm name).  ## implementation (search for the algorithm name).
64  sub END_TAG_OPTIONAL_EL () {  ## Bit 3
65    DD_EL |  sub END_TAG_OPTIONAL_EL () { 0b1_000 }
   DT_EL |  
   LI_EL |  
   OPTION_EL |  
   OPTGROUP_EL |  
   P_EL |  
   RUBY_COMPONENT_EL  
 }  
66    
67  ## NOTE: Used in </body> and EOF algorithms.  ## Bits 0-2
 sub ALL_END_TAG_OPTIONAL_EL () {  
   DD_EL |  
   DT_EL |  
   LI_EL |  
   P_EL |  
   
   BODY_EL |  
   HTML_EL |  
   TABLE_CELL_EL |  
   TABLE_ROW_EL |  
   TABLE_ROW_GROUP_EL  
 }  
68    
69  sub SCOPING_EL () {  sub MISC_SPECIAL_EL () { SPECIAL_EL | 0b000 }
70    BUTTON_EL |  sub FORM_EL () { SPECIAL_EL | 0b001 }
71    CAPTION_EL |  sub FRAMESET_EL () { SPECIAL_EL | 0b010 }
72    HTML_EL |  sub HEADING_EL () { SPECIAL_EL | 0b011 }
73    TABLE_EL |  sub SELECT_EL () { SPECIAL_EL | 0b100 }
74    TABLE_CELL_EL |  sub SCRIPT_EL () { SPECIAL_EL | 0b101 }
75    MISC_SCOPING_EL  
76    sub ADDRESS_DIV_EL () { SPECIAL_EL | ADDRESS_DIV_P_EL | 0b001 }
77    sub BODY_EL () { SPECIAL_EL | ALL_END_TAG_OPTIONAL_EL | 0b001 }
78    
79    sub DTDD_EL () {
80      SPECIAL_EL |
81      END_TAG_OPTIONAL_EL |
82      ALL_END_TAG_OPTIONAL_EL |
83      0b010
84  }  }
85    sub LI_EL () {
86  sub TABLE_SCOPING_EL () {    SPECIAL_EL |
87    HTML_EL |    END_TAG_OPTIONAL_EL |
88    TABLE_EL    ALL_END_TAG_OPTIONAL_EL |
89      0b100
90  }  }
91    sub P_EL () {
92  sub TABLE_ROWS_SCOPING_EL () {    SPECIAL_EL |
93    HTML_EL |    ADDRESS_DIV_P_EL |
94    TABLE_ROW_GROUP_EL    END_TAG_OPTIONAL_EL |
95      ALL_END_TAG_OPTIONAL_EL |
96      0b001
97  }  }
98    
99  sub TABLE_ROW_SCOPING_EL () {  sub TABLE_ROW_EL () {
100    HTML_EL |    SPECIAL_EL |
101    TABLE_ROW_EL    TABLE_ROWS_EL |
102      TABLE_ROW_SCOPING_EL |
103      ALL_END_TAG_OPTIONAL_EL |
104      0b001
105    }
106    sub TABLE_ROW_GROUP_EL () {
107      SPECIAL_EL |
108      TABLE_ROWS_EL |
109      TABLE_ROWS_SCOPING_EL |
110      ALL_END_TAG_OPTIONAL_EL |
111      0b001
112  }  }
113    
114  sub SPECIAL_EL () {  sub MISC_SCOPING_EL () { SCOPING_EL | 0b000 }
115    ADDRESS_EL |  sub BUTTON_EL () { SCOPING_EL | 0b001 }
116    BODY_EL |  sub CAPTION_EL () { SCOPING_EL | 0b010 }
117    DIV_EL |  sub HTML_EL () {
118      SCOPING_EL |
119    DD_EL |    TABLE_SCOPING_EL |
120    DT_EL |    TABLE_ROWS_SCOPING_EL |
121    LI_EL |    TABLE_ROW_SCOPING_EL |
122    P_EL |    ALL_END_TAG_OPTIONAL_EL |
123      0b001
124    FORM_EL |  }
125    FRAMESET_EL |  sub TABLE_EL () {
126    HEADING_EL |    SCOPING_EL |
127    OPTION_EL |    TABLE_ROWS_EL |
128    OPTGROUP_EL |    TABLE_SCOPING_EL |
129    SELECT_EL |    0b001
130    TABLE_ROW_EL |  }
131    TABLE_ROW_GROUP_EL |  sub TABLE_CELL_EL () {
132    MISC_SPECIAL_EL    SCOPING_EL |
133      TABLE_ROW_SCOPING_EL |
134      ALL_END_TAG_OPTIONAL_EL |
135      0b001
136  }  }
137    
138    sub MISC_FORMATTING_EL () { FORMATTING_EL | 0b000 }
139    sub A_EL () { FORMATTING_EL | 0b001 }
140    sub NOBR_EL () { FORMATTING_EL | 0b010 }
141    
142    sub RUBY_EL () { PHRASING_EL | 0b001 }
143    
144    ## ISSUE: ALL_END_TAG_OPTIONAL_EL?
145    sub OPTGROUP_EL () { PHRASING_EL | END_TAG_OPTIONAL_EL | 0b001 }
146    sub OPTION_EL () { PHRASING_EL | END_TAG_OPTIONAL_EL | 0b010 }
147    sub RUBY_COMPONENT_EL () { PHRASING_EL | END_TAG_OPTIONAL_EL | 0b100 }
148    
149    sub MML_AXML_EL () { PHRASING_EL | FOREIGN_EL | 0b001 }
150    
151  my $el_category = {  my $el_category = {
152    a => A_EL | FORMATTING_EL,    a => A_EL,
153    address => ADDRESS_EL,    address => ADDRESS_DIV_EL,
154    applet => MISC_SCOPING_EL,    applet => MISC_SCOPING_EL,
155    area => MISC_SPECIAL_EL,    area => MISC_SPECIAL_EL,
156    article => MISC_SPECIAL_EL,    article => MISC_SPECIAL_EL,
# Line 160  my $el_category = { Line 170  my $el_category = {
170    colgroup => MISC_SPECIAL_EL,    colgroup => MISC_SPECIAL_EL,
171    command => MISC_SPECIAL_EL,    command => MISC_SPECIAL_EL,
172    datagrid => MISC_SPECIAL_EL,    datagrid => MISC_SPECIAL_EL,
173    dd => DD_EL,    dd => DTDD_EL,
174    details => MISC_SPECIAL_EL,    details => MISC_SPECIAL_EL,
175    dialog => MISC_SPECIAL_EL,    dialog => MISC_SPECIAL_EL,
176    dir => MISC_SPECIAL_EL,    dir => MISC_SPECIAL_EL,
177    div => DIV_EL,    div => ADDRESS_DIV_EL,
178    dl => MISC_SPECIAL_EL,    dl => MISC_SPECIAL_EL,
179    dt => DT_EL,    dt => DTDD_EL,
180    em => FORMATTING_EL,    em => FORMATTING_EL,
181    embed => MISC_SPECIAL_EL,    embed => MISC_SPECIAL_EL,
182    eventsource => MISC_SPECIAL_EL,    eventsource => MISC_SPECIAL_EL,
# Line 200  my $el_category = { Line 210  my $el_category = {
210    menu => MISC_SPECIAL_EL,    menu => MISC_SPECIAL_EL,
211    meta => MISC_SPECIAL_EL,    meta => MISC_SPECIAL_EL,
212    nav => MISC_SPECIAL_EL,    nav => MISC_SPECIAL_EL,
213    nobr => NOBR_EL | FORMATTING_EL,    nobr => NOBR_EL,
214    noembed => MISC_SPECIAL_EL,    noembed => MISC_SPECIAL_EL,
215    noframes => MISC_SPECIAL_EL,    noframes => MISC_SPECIAL_EL,
216    noscript => MISC_SPECIAL_EL,    noscript => MISC_SPECIAL_EL,
# Line 242  my $el_category = { Line 252  my $el_category = {
252  my $el_category_f = {  my $el_category_f = {
253    $MML_NS => {    $MML_NS => {
254      'annotation-xml' => MML_AXML_EL,      'annotation-xml' => MML_AXML_EL,
255      mi => FOREIGN_FLOW_CONTENT_EL,      mi => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
256      mo => FOREIGN_FLOW_CONTENT_EL,      mo => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
257      mn => FOREIGN_FLOW_CONTENT_EL,      mn => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
258      ms => FOREIGN_FLOW_CONTENT_EL,      ms => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
259      mtext => FOREIGN_FLOW_CONTENT_EL,      mtext => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
260    },    },
261    $SVG_NS => {    $SVG_NS => {
262      foreignObject => FOREIGN_FLOW_CONTENT_EL,      foreignObject => SCOPING_EL | FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
263      desc => FOREIGN_FLOW_CONTENT_EL,      desc => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
264      title => FOREIGN_FLOW_CONTENT_EL,      title => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
265    },    },
266    ## NOTE: In addition, FOREIGN_EL is set to non-HTML elements.    ## NOTE: In addition, FOREIGN_EL is set to non-HTML elements.
267  };  };
# Line 338  my $foreign_attr_xname = { Line 348  my $foreign_attr_xname = {
348    
349  ## ISSUE: xmlns:xlink="non-xlink-ns" is not an error.  ## ISSUE: xmlns:xlink="non-xlink-ns" is not an error.
350    
 my $charref_map = {  
   0x0D => 0x000A,  
   0x80 => 0x20AC,  
   0x81 => 0xFFFD,  
   0x82 => 0x201A,  
   0x83 => 0x0192,  
   0x84 => 0x201E,  
   0x85 => 0x2026,  
   0x86 => 0x2020,  
   0x87 => 0x2021,  
   0x88 => 0x02C6,  
   0x89 => 0x2030,  
   0x8A => 0x0160,  
   0x8B => 0x2039,  
   0x8C => 0x0152,  
   0x8D => 0xFFFD,  
   0x8E => 0x017D,  
   0x8F => 0xFFFD,  
   0x90 => 0xFFFD,  
   0x91 => 0x2018,  
   0x92 => 0x2019,  
   0x93 => 0x201C,  
   0x94 => 0x201D,  
   0x95 => 0x2022,  
   0x96 => 0x2013,  
   0x97 => 0x2014,  
   0x98 => 0x02DC,  
   0x99 => 0x2122,  
   0x9A => 0x0161,  
   0x9B => 0x203A,  
   0x9C => 0x0153,  
   0x9D => 0xFFFD,  
   0x9E => 0x017E,  
   0x9F => 0x0178,  
 }; # $charref_map  
 $charref_map->{$_} = 0xFFFD  
     for 0x0000..0x0008, 0x000B, 0x000E..0x001F, 0x007F,  
         0xD800..0xDFFF, 0xFDD0..0xFDDF, ## ISSUE: 0xFDEF  
         0xFFFE, 0xFFFF, 0x1FFFE, 0x1FFFF, 0x2FFFE, 0x2FFFF, 0x3FFFE, 0x3FFFF,  
         0x4FFFE, 0x4FFFF, 0x5FFFE, 0x5FFFF, 0x6FFFE, 0x6FFFF, 0x7FFFE,  
         0x7FFFF, 0x8FFFE, 0x8FFFF, 0x9FFFE, 0x9FFFF, 0xAFFFE, 0xAFFFF,  
         0xBFFFE, 0xBFFFF, 0xCFFFE, 0xCFFFF, 0xDFFFE, 0xDFFFF, 0xEFFFE,  
         0xEFFFF, 0xFFFFE, 0xFFFFF, 0x10FFFE, 0x10FFFF;  
   
351  ## TODO: Invoke the reset algorithm when a resettable element is  ## TODO: Invoke the reset algorithm when a resettable element is
352  ## created (cf. HTML5 revision 2259).  ## created (cf. HTML5 revision 2259).
353    
# Line 488  sub parse_byte_stream ($$$$;$$) { Line 454  sub parse_byte_stream ($$$$;$$) {
454      if (defined $charset_name) {      if (defined $charset_name) {
455        $charset = Message::Charset::Info->get_by_html_name ($charset_name);        $charset = Message::Charset::Info->get_by_html_name ($charset_name);
456    
       ## ISSUE: Unsupported encoding is not ignored according to the spec.  
457        require Whatpm::Charset::DecodeHandle;        require Whatpm::Charset::DecodeHandle;
458        $buffer = Whatpm::Charset::DecodeHandle::ByteBuffer->new        $buffer = Whatpm::Charset::DecodeHandle::ByteBuffer->new
459            ($byte_stream);            ($byte_stream);
# Line 835  sub new ($) { Line 800  sub new ($) {
800    return $self;    return $self;
801  } # new  } # new
802    
803  sub CM_ENTITY () { 0b001 } # & markup in data  ## Insertion modes
 sub CM_LIMITED_MARKUP () { 0b010 } # < markup in data (limited)  
 sub CM_FULL_MARKUP () { 0b100 } # < markup in data (any)  
   
 sub PLAINTEXT_CONTENT_MODEL () { 0 }  
 sub CDATA_CONTENT_MODEL () { CM_LIMITED_MARKUP }  
 sub RCDATA_CONTENT_MODEL () { CM_ENTITY | CM_LIMITED_MARKUP }  
 sub PCDATA_CONTENT_MODEL () { CM_ENTITY | CM_FULL_MARKUP }  
   
 sub DATA_STATE () { 0 }  
 #sub ENTITY_DATA_STATE () { 1 }  
 sub TAG_OPEN_STATE () { 2 }  
 sub CLOSE_TAG_OPEN_STATE () { 3 }  
 sub TAG_NAME_STATE () { 4 }  
 sub BEFORE_ATTRIBUTE_NAME_STATE () { 5 }  
 sub ATTRIBUTE_NAME_STATE () { 6 }  
 sub AFTER_ATTRIBUTE_NAME_STATE () { 7 }  
 sub BEFORE_ATTRIBUTE_VALUE_STATE () { 8 }  
 sub ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE () { 9 }  
 sub ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE () { 10 }  
 sub ATTRIBUTE_VALUE_UNQUOTED_STATE () { 11 }  
 #sub ENTITY_IN_ATTRIBUTE_VALUE_STATE () { 12 }  
 sub MARKUP_DECLARATION_OPEN_STATE () { 13 }  
 sub COMMENT_START_STATE () { 14 }  
 sub COMMENT_START_DASH_STATE () { 15 }  
 sub COMMENT_STATE () { 16 }  
 sub COMMENT_END_STATE () { 17 }  
 sub COMMENT_END_DASH_STATE () { 18 }  
 sub BOGUS_COMMENT_STATE () { 19 }  
 sub DOCTYPE_STATE () { 20 }  
 sub BEFORE_DOCTYPE_NAME_STATE () { 21 }  
 sub DOCTYPE_NAME_STATE () { 22 }  
 sub AFTER_DOCTYPE_NAME_STATE () { 23 }  
 sub BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE () { 24 }  
 sub DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE () { 25 }  
 sub DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE () { 26 }  
 sub AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE () { 27 }  
 sub BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE () { 28 }  
 sub DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE () { 29 }  
 sub DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE () { 30 }  
 sub AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE () { 31 }  
 sub BOGUS_DOCTYPE_STATE () { 32 }  
 sub AFTER_ATTRIBUTE_VALUE_QUOTED_STATE () { 33 }  
 sub SELF_CLOSING_START_TAG_STATE () { 34 }  
 sub CDATA_SECTION_STATE () { 35 }  
 sub MD_HYPHEN_STATE () { 36 } # "markup declaration open state" in the spec  
 sub MD_DOCTYPE_STATE () { 37 } # "markup declaration open state" in the spec  
 sub MD_CDATA_STATE () { 38 } # "markup declaration open state" in the spec  
 sub CDATA_RCDATA_CLOSE_TAG_STATE () { 39 } # "close tag open state" in the spec  
 sub CDATA_SECTION_MSE1_STATE () { 40 } # "CDATA section state" in the spec  
 sub CDATA_SECTION_MSE2_STATE () { 41 } # "CDATA section state" in the spec  
 sub PUBLIC_STATE () { 42 } # "after DOCTYPE name state" in the spec  
 sub SYSTEM_STATE () { 43 } # "after DOCTYPE name state" in the spec  
 ## NOTE: "Entity data state", "entity in attribute value state", and  
 ## "consume a character reference" algorithm are jointly implemented  
 ## using the following six states:  
 sub ENTITY_STATE () { 44 }  
 sub ENTITY_HASH_STATE () { 45 }  
 sub NCR_NUM_STATE () { 46 }  
 sub HEXREF_X_STATE () { 47 }  
 sub HEXREF_HEX_STATE () { 48 }  
 sub ENTITY_NAME_STATE () { 49 }  
 sub PCDATA_STATE () { 50 } # "data state" in the spec  
   
 sub DOCTYPE_TOKEN () { 1 }  
 sub COMMENT_TOKEN () { 2 }  
 sub START_TAG_TOKEN () { 3 }  
 sub END_TAG_TOKEN () { 4 }  
 sub END_OF_FILE_TOKEN () { 5 }  
 sub CHARACTER_TOKEN () { 6 }  
804    
805  sub AFTER_HTML_IMS () { 0b100 }  sub AFTER_HTML_IMS () { 0b100 }
806  sub HEAD_IMS ()       { 0b1000 }  sub HEAD_IMS ()       { 0b1000 }
# Line 915  sub ROW_IMS ()        { 0b10000000 } Line 811  sub ROW_IMS ()        { 0b10000000 }
811  sub BODY_AFTER_IMS () { 0b100000000 }  sub BODY_AFTER_IMS () { 0b100000000 }
812  sub FRAME_IMS ()      { 0b1000000000 }  sub FRAME_IMS ()      { 0b1000000000 }
813  sub SELECT_IMS ()     { 0b10000000000 }  sub SELECT_IMS ()     { 0b10000000000 }
814  sub IN_FOREIGN_CONTENT_IM () { 0b100000000000 }  #sub IN_FOREIGN_CONTENT_IM () { 0b100000000000 } # see Whatpm::HTML::Tokenizer
815      ## NOTE: "in foreign content" insertion mode is special; it is combined      ## NOTE: "in foreign content" insertion mode is special; it is combined
816      ## with the secondary insertion mode.  In this parser, they are stored      ## with the secondary insertion mode.  In this parser, they are stored
817      ## together in the bit-or'ed form.      ## together in the bit-or'ed form.
818    sub IN_CDATA_RCDATA_IM () { 0b1000000000000 }
819        ## NOTE: "in CDATA/RCDATA" insertion mode is also special; it is
820        ## combined with the original insertion mode.  In thie parser,
821        ## they are stored together in the bit-or'ed form.
822    
823    sub IM_MASK () { 0b11111111111 }
824    
825  ## NOTE: "initial" and "before html" insertion modes have no constants.  ## NOTE: "initial" and "before html" insertion modes have no constants.
826    
# Line 945  sub IN_SELECT_IM () { SELECT_IMS | 0b01 Line 847  sub IN_SELECT_IM () { SELECT_IMS | 0b01
847  sub IN_SELECT_IN_TABLE_IM () { SELECT_IMS | 0b10 }  sub IN_SELECT_IN_TABLE_IM () { SELECT_IMS | 0b10 }
848  sub IN_COLUMN_GROUP_IM () { 0b10 }  sub IN_COLUMN_GROUP_IM () { 0b10 }
849    
 ## Implementations MUST act as if state machine in the spec  
   
 sub _initialize_tokenizer ($) {  
   my $self = shift;  
   $self->{state} = DATA_STATE; # MUST  
   #$self->{s_kwd}; # state keyword - initialized when used  
   #$self->{entity__value}; # initialized when used  
   #$self->{entity__match}; # initialized when used  
   $self->{content_model} = PCDATA_CONTENT_MODEL; # be  
   undef $self->{ct}; # current token  
   undef $self->{ca}; # current attribute  
   undef $self->{last_stag_name}; # last emitted start tag name  
   #$self->{prev_state}; # initialized when used  
   delete $self->{self_closing};  
   $self->{char_buffer} = '';  
   $self->{char_buffer_pos} = 0;  
   $self->{nc} = -1; # next input character  
   #$self->{next_nc}  
   !!!next-input-character;  
   $self->{token} = [];  
   # $self->{escape}  
 } # _initialize_tokenizer  
   
 ## A token has:  
 ##   ->{type} == DOCTYPE_TOKEN, START_TAG_TOKEN, END_TAG_TOKEN, COMMENT_TOKEN,  
 ##       CHARACTER_TOKEN, or END_OF_FILE_TOKEN  
 ##   ->{name} (DOCTYPE_TOKEN)  
 ##   ->{tag_name} (START_TAG_TOKEN, END_TAG_TOKEN)  
 ##   ->{pubid} (DOCTYPE_TOKEN)  
 ##   ->{sysid} (DOCTYPE_TOKEN)  
 ##   ->{quirks} == 1 or 0 (DOCTYPE_TOKEN): "force-quirks" flag  
 ##   ->{attributes} isa HASH (START_TAG_TOKEN, END_TAG_TOKEN)  
 ##        ->{name}  
 ##        ->{value}  
 ##        ->{has_reference} == 1 or 0  
 ##   ->{data} (COMMENT_TOKEN, CHARACTER_TOKEN)  
 ## NOTE: The "self-closing flag" is hold as |$self->{self_closing}|.  
 ##     |->{self_closing}| is used to save the value of |$self->{self_closing}|  
 ##     while the token is pushed back to the stack.  
   
 ## Emitted token MUST immediately be handled by the tree construction state.  
   
 ## Before each step, UA MAY check to see if either one of the scripts in  
 ## "list of scripts that will execute as soon as possible" or the first  
 ## script in the "list of scripts that will execute asynchronously",  
 ## has completed loading.  If one has, then it MUST be executed  
 ## and removed from the list.  
   
 ## TODO: Polytheistic slash SHOULD NOT be used. (Applied only to atheists.)  
 ## (This requirement was dropped from HTML5 spec, unfortunately.)  
   
 my $is_space = {  
   0x0009 => 1, # CHARACTER TABULATION (HT)  
   0x000A => 1, # LINE FEED (LF)  
   #0x000B => 0, # LINE TABULATION (VT)  
   0x000C => 1, # FORM FEED (FF)  
   #0x000D => 1, # CARRIAGE RETURN (CR)  
   0x0020 => 1, # SPACE (SP)  
 };  
   
 sub _get_next_token ($) {  
   my $self = shift;  
   
   if ($self->{self_closing}) {  
     !!!parse-error (type => 'nestc', token => $self->{ct});  
     ## NOTE: The |self_closing| flag is only set by start tag token.  
     ## In addition, when a start tag token is emitted, it is always set to  
     ## |ct|.  
     delete $self->{self_closing};  
   }  
   
   if (@{$self->{token}}) {  
     $self->{self_closing} = $self->{token}->[0]->{self_closing};  
     return shift @{$self->{token}};  
   }  
   
   A: {  
     if ($self->{state} == PCDATA_STATE) {  
       ## NOTE: Same as |DATA_STATE|, but only for |PCDATA| content model.  
   
       if ($self->{nc} == 0x0026) { # &  
         !!!cp (0.1);  
         ## NOTE: In the spec, the tokenizer is switched to the  
         ## "entity data state".  In this implementation, the tokenizer  
         ## is switched to the |ENTITY_STATE|, which is an implementation  
         ## of the "consume a character reference" algorithm.  
         $self->{entity_add} = -1;  
         $self->{prev_state} = DATA_STATE;  
         $self->{state} = ENTITY_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003C) { # <  
         !!!cp (0.2);  
         $self->{state} = TAG_OPEN_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (0.3);  
         !!!emit ({type => END_OF_FILE_TOKEN,  
                   line => $self->{line}, column => $self->{column}});  
         last A; ## TODO: ok?  
       } else {  
         !!!cp (0.4);  
         #  
       }  
   
       # Anything else  
       my $token = {type => CHARACTER_TOKEN,  
                    data => chr $self->{nc},  
                    line => $self->{line}, column => $self->{column},  
                   };  
       $self->{read_until}->($token->{data}, q[<&], length $token->{data});  
   
       ## Stay in the state.  
       !!!next-input-character;  
       !!!emit ($token);  
       redo A;  
     } elsif ($self->{state} == DATA_STATE) {  
       $self->{s_kwd} = '' unless defined $self->{s_kwd};  
       if ($self->{nc} == 0x0026) { # &  
         $self->{s_kwd} = '';  
         if ($self->{content_model} & CM_ENTITY and # PCDATA | RCDATA  
             not $self->{escape}) {  
           !!!cp (1);  
           ## NOTE: In the spec, the tokenizer is switched to the  
           ## "entity data state".  In this implementation, the tokenizer  
           ## is switched to the |ENTITY_STATE|, which is an implementation  
           ## of the "consume a character reference" algorithm.  
           $self->{entity_add} = -1;  
           $self->{prev_state} = DATA_STATE;  
           $self->{state} = ENTITY_STATE;  
           !!!next-input-character;  
           redo A;  
         } else {  
           !!!cp (2);  
           #  
         }  
       } elsif ($self->{nc} == 0x002D) { # -  
         if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA  
           $self->{s_kwd} .= '-';  
             
           if ($self->{s_kwd} eq '<!--') {  
             !!!cp (3);  
             $self->{escape} = 1; # unless $self->{escape};  
             $self->{s_kwd} = '--';  
             #  
           } elsif ($self->{s_kwd} eq '---') {  
             !!!cp (4);  
             $self->{s_kwd} = '--';  
             #  
           } else {  
             !!!cp (5);  
             #  
           }  
         }  
           
         #  
       } elsif ($self->{nc} == 0x0021) { # !  
         if (length $self->{s_kwd}) {  
           !!!cp (5.1);  
           $self->{s_kwd} .= '!';  
           #  
         } else {  
           !!!cp (5.2);  
           #$self->{s_kwd} = '';  
           #  
         }  
         #  
       } elsif ($self->{nc} == 0x003C) { # <  
         if ($self->{content_model} & CM_FULL_MARKUP or # PCDATA  
             (($self->{content_model} & CM_LIMITED_MARKUP) and # CDATA | RCDATA  
              not $self->{escape})) {  
           !!!cp (6);  
           $self->{state} = TAG_OPEN_STATE;  
           !!!next-input-character;  
           redo A;  
         } else {  
           !!!cp (7);  
           $self->{s_kwd} = '';  
           #  
         }  
       } elsif ($self->{nc} == 0x003E) { # >  
         if ($self->{escape} and  
             ($self->{content_model} & CM_LIMITED_MARKUP)) { # RCDATA | CDATA  
           if ($self->{s_kwd} eq '--') {  
             !!!cp (8);  
             delete $self->{escape};  
           } else {  
             !!!cp (9);  
           }  
         } else {  
           !!!cp (10);  
         }  
           
         $self->{s_kwd} = '';  
         #  
       } elsif ($self->{nc} == -1) {  
         !!!cp (11);  
         $self->{s_kwd} = '';  
         !!!emit ({type => END_OF_FILE_TOKEN,  
                   line => $self->{line}, column => $self->{column}});  
         last A; ## TODO: ok?  
       } else {  
         !!!cp (12);  
         $self->{s_kwd} = '';  
         #  
       }  
   
       # Anything else  
       my $token = {type => CHARACTER_TOKEN,  
                    data => chr $self->{nc},  
                    line => $self->{line}, column => $self->{column},  
                   };  
       if ($self->{read_until}->($token->{data}, q[-!<>&],  
                                 length $token->{data})) {  
         $self->{s_kwd} = '';  
       }  
   
       ## Stay in the data state.  
       if ($self->{content_model} == PCDATA_CONTENT_MODEL) {  
         !!!cp (13);  
         $self->{state} = PCDATA_STATE;  
       } else {  
         !!!cp (14);  
         ## Stay in the state.  
       }  
       !!!next-input-character;  
       !!!emit ($token);  
       redo A;  
     } elsif ($self->{state} == TAG_OPEN_STATE) {  
       if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA  
         if ($self->{nc} == 0x002F) { # /  
           !!!cp (15);  
           !!!next-input-character;  
           $self->{state} = CLOSE_TAG_OPEN_STATE;  
           redo A;  
         } elsif ($self->{nc} == 0x0021) { # !  
           !!!cp (15.1);  
           $self->{s_kwd} = '<' unless $self->{escape};  
           #  
         } else {  
           !!!cp (16);  
           #  
         }  
   
         ## reconsume  
         $self->{state} = DATA_STATE;  
         !!!emit ({type => CHARACTER_TOKEN, data => '<',  
                   line => $self->{line_prev},  
                   column => $self->{column_prev},  
                  });  
         redo A;  
       } elsif ($self->{content_model} & CM_FULL_MARKUP) { # PCDATA  
         if ($self->{nc} == 0x0021) { # !  
           !!!cp (17);  
           $self->{state} = MARKUP_DECLARATION_OPEN_STATE;  
           !!!next-input-character;  
           redo A;  
         } elsif ($self->{nc} == 0x002F) { # /  
           !!!cp (18);  
           $self->{state} = CLOSE_TAG_OPEN_STATE;  
           !!!next-input-character;  
           redo A;  
         } elsif (0x0041 <= $self->{nc} and  
                  $self->{nc} <= 0x005A) { # A..Z  
           !!!cp (19);  
           $self->{ct}  
             = {type => START_TAG_TOKEN,  
                tag_name => chr ($self->{nc} + 0x0020),  
                line => $self->{line_prev},  
                column => $self->{column_prev}};  
           $self->{state} = TAG_NAME_STATE;  
           !!!next-input-character;  
           redo A;  
         } elsif (0x0061 <= $self->{nc} and  
                  $self->{nc} <= 0x007A) { # a..z  
           !!!cp (20);  
           $self->{ct} = {type => START_TAG_TOKEN,  
                                     tag_name => chr ($self->{nc}),  
                                     line => $self->{line_prev},  
                                     column => $self->{column_prev}};  
           $self->{state} = TAG_NAME_STATE;  
           !!!next-input-character;  
           redo A;  
         } elsif ($self->{nc} == 0x003E) { # >  
           !!!cp (21);  
           !!!parse-error (type => 'empty start tag',  
                           line => $self->{line_prev},  
                           column => $self->{column_prev});  
           $self->{state} = DATA_STATE;  
           !!!next-input-character;  
   
           !!!emit ({type => CHARACTER_TOKEN, data => '<>',  
                     line => $self->{line_prev},  
                     column => $self->{column_prev},  
                    });  
   
           redo A;  
         } elsif ($self->{nc} == 0x003F) { # ?  
           !!!cp (22);  
           !!!parse-error (type => 'pio',  
                           line => $self->{line_prev},  
                           column => $self->{column_prev});  
           $self->{state} = BOGUS_COMMENT_STATE;  
           $self->{ct} = {type => COMMENT_TOKEN, data => '',  
                                     line => $self->{line_prev},  
                                     column => $self->{column_prev},  
                                    };  
           ## $self->{nc} is intentionally left as is  
           redo A;  
         } else {  
           !!!cp (23);  
           !!!parse-error (type => 'bare stago',  
                           line => $self->{line_prev},  
                           column => $self->{column_prev});  
           $self->{state} = DATA_STATE;  
           ## reconsume  
   
           !!!emit ({type => CHARACTER_TOKEN, data => '<',  
                     line => $self->{line_prev},  
                     column => $self->{column_prev},  
                    });  
   
           redo A;  
         }  
       } else {  
         die "$0: $self->{content_model} in tag open";  
       }  
     } elsif ($self->{state} == CLOSE_TAG_OPEN_STATE) {  
       ## NOTE: The "close tag open state" in the spec is implemented as  
       ## |CLOSE_TAG_OPEN_STATE| and |CDATA_RCDATA_CLOSE_TAG_STATE|.  
   
       my ($l, $c) = ($self->{line_prev}, $self->{column_prev} - 1); # "<"of"</"  
       if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA  
         if (defined $self->{last_stag_name}) {  
           $self->{state} = CDATA_RCDATA_CLOSE_TAG_STATE;  
           $self->{s_kwd} = '';  
           ## Reconsume.  
           redo A;  
         } else {  
           ## No start tag token has ever been emitted  
           ## NOTE: See <http://krijnhoetmer.nl/irc-logs/whatwg/20070626#l-564>.  
           !!!cp (28);  
           $self->{state} = DATA_STATE;  
           ## Reconsume.  
           !!!emit ({type => CHARACTER_TOKEN, data => '</',  
                     line => $l, column => $c,  
                    });  
           redo A;  
         }  
       }  
   
       if (0x0041 <= $self->{nc} and  
           $self->{nc} <= 0x005A) { # A..Z  
         !!!cp (29);  
         $self->{ct}  
             = {type => END_TAG_TOKEN,  
                tag_name => chr ($self->{nc} + 0x0020),  
                line => $l, column => $c};  
         $self->{state} = TAG_NAME_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif (0x0061 <= $self->{nc} and  
                $self->{nc} <= 0x007A) { # a..z  
         !!!cp (30);  
         $self->{ct} = {type => END_TAG_TOKEN,  
                                   tag_name => chr ($self->{nc}),  
                                   line => $l, column => $c};  
         $self->{state} = TAG_NAME_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         !!!cp (31);  
         !!!parse-error (type => 'empty end tag',  
                         line => $self->{line_prev}, ## "<" in "</>"  
                         column => $self->{column_prev} - 1);  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (32);  
         !!!parse-error (type => 'bare etago');  
         $self->{state} = DATA_STATE;  
         # reconsume  
   
         !!!emit ({type => CHARACTER_TOKEN, data => '</',  
                   line => $l, column => $c,  
                  });  
   
         redo A;  
       } else {  
         !!!cp (33);  
         !!!parse-error (type => 'bogus end tag');  
         $self->{state} = BOGUS_COMMENT_STATE;  
         $self->{ct} = {type => COMMENT_TOKEN, data => '',  
                                   line => $self->{line_prev}, # "<" of "</"  
                                   column => $self->{column_prev} - 1,  
                                  };  
         ## NOTE: $self->{nc} is intentionally left as is.  
         ## Although the "anything else" case of the spec not explicitly  
         ## states that the next input character is to be reconsumed,  
         ## it will be included to the |data| of the comment token  
         ## generated from the bogus end tag, as defined in the  
         ## "bogus comment state" entry.  
         redo A;  
       }  
     } elsif ($self->{state} == CDATA_RCDATA_CLOSE_TAG_STATE) {  
       my $ch = substr $self->{last_stag_name}, length $self->{s_kwd}, 1;  
       if (length $ch) {  
         my $CH = $ch;  
         $ch =~ tr/a-z/A-Z/;  
         my $nch = chr $self->{nc};  
         if ($nch eq $ch or $nch eq $CH) {  
           !!!cp (24);  
           ## Stay in the state.  
           $self->{s_kwd} .= $nch;  
           !!!next-input-character;  
           redo A;  
         } else {  
           !!!cp (25);  
           $self->{state} = DATA_STATE;  
           ## Reconsume.  
           !!!emit ({type => CHARACTER_TOKEN,  
                     data => '</' . $self->{s_kwd},  
                     line => $self->{line_prev},  
                     column => $self->{column_prev} - 1 - length $self->{s_kwd},  
                    });  
           redo A;  
         }  
       } else { # after "<{tag-name}"  
         unless ($is_space->{$self->{nc}} or  
                 {  
                  0x003E => 1, # >  
                  0x002F => 1, # /  
                  -1 => 1, # EOF  
                 }->{$self->{nc}}) {  
           !!!cp (26);  
           ## Reconsume.  
           $self->{state} = DATA_STATE;  
           !!!emit ({type => CHARACTER_TOKEN,  
                     data => '</' . $self->{s_kwd},  
                     line => $self->{line_prev},  
                     column => $self->{column_prev} - 1 - length $self->{s_kwd},  
                    });  
           redo A;  
         } else {  
           !!!cp (27);  
           $self->{ct}  
               = {type => END_TAG_TOKEN,  
                  tag_name => $self->{last_stag_name},  
                  line => $self->{line_prev},  
                  column => $self->{column_prev} - 1 - length $self->{s_kwd}};  
           $self->{state} = TAG_NAME_STATE;  
           ## Reconsume.  
           redo A;  
         }  
       }  
     } elsif ($self->{state} == TAG_NAME_STATE) {  
       if ($is_space->{$self->{nc}}) {  
         !!!cp (34);  
         $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (35);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           #if ($self->{ct}->{attributes}) {  
           #  ## NOTE: This should never be reached.  
           #  !!! cp (36);  
           #  !!! parse-error (type => 'end tag attribute');  
           #} else {  
             !!!cp (37);  
           #}  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } elsif (0x0041 <= $self->{nc} and  
                $self->{nc} <= 0x005A) { # A..Z  
         !!!cp (38);  
         $self->{ct}->{tag_name} .= chr ($self->{nc} + 0x0020);  
           # start tag or end tag  
         ## Stay in this state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!parse-error (type => 'unclosed tag');  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (39);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           #if ($self->{ct}->{attributes}) {  
           #  ## NOTE: This state should never be reached.  
           #  !!! cp (40);  
           #  !!! parse-error (type => 'end tag attribute');  
           #} else {  
             !!!cp (41);  
           #}  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         # reconsume  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } elsif ($self->{nc} == 0x002F) { # /  
         !!!cp (42);  
         $self->{state} = SELF_CLOSING_START_TAG_STATE;  
         !!!next-input-character;  
         redo A;  
       } else {  
         !!!cp (44);  
         $self->{ct}->{tag_name} .= chr $self->{nc};  
           # start tag or end tag  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == BEFORE_ATTRIBUTE_NAME_STATE) {  
       if ($is_space->{$self->{nc}}) {  
         !!!cp (45);  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (46);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           if ($self->{ct}->{attributes}) {  
             !!!cp (47);  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             !!!cp (48);  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } elsif (0x0041 <= $self->{nc} and  
                $self->{nc} <= 0x005A) { # A..Z  
         !!!cp (49);  
         $self->{ca}  
             = {name => chr ($self->{nc} + 0x0020),  
                value => '',  
                line => $self->{line}, column => $self->{column}};  
         $self->{state} = ATTRIBUTE_NAME_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x002F) { # /  
         !!!cp (50);  
         $self->{state} = SELF_CLOSING_START_TAG_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!parse-error (type => 'unclosed tag');  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (52);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           if ($self->{ct}->{attributes}) {  
             !!!cp (53);  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             !!!cp (54);  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         # reconsume  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } else {  
         if ({  
              0x0022 => 1, # "  
              0x0027 => 1, # '  
              0x003D => 1, # =  
             }->{$self->{nc}}) {  
           !!!cp (55);  
           !!!parse-error (type => 'bad attribute name');  
         } else {  
           !!!cp (56);  
         }  
         $self->{ca}  
             = {name => chr ($self->{nc}),  
                value => '',  
                line => $self->{line}, column => $self->{column}};  
         $self->{state} = ATTRIBUTE_NAME_STATE;  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == ATTRIBUTE_NAME_STATE) {  
       my $before_leave = sub {  
         if (exists $self->{ct}->{attributes} # start tag or end tag  
             ->{$self->{ca}->{name}}) { # MUST  
           !!!cp (57);  
           !!!parse-error (type => 'duplicate attribute', text => $self->{ca}->{name}, line => $self->{ca}->{line}, column => $self->{ca}->{column});  
           ## Discard $self->{ca} # MUST  
         } else {  
           !!!cp (58);  
           $self->{ct}->{attributes}->{$self->{ca}->{name}}  
             = $self->{ca};  
         }  
       }; # $before_leave  
   
       if ($is_space->{$self->{nc}}) {  
         !!!cp (59);  
         $before_leave->();  
         $self->{state} = AFTER_ATTRIBUTE_NAME_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003D) { # =  
         !!!cp (60);  
         $before_leave->();  
         $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         $before_leave->();  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (61);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           !!!cp (62);  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           if ($self->{ct}->{attributes}) {  
             !!!parse-error (type => 'end tag attribute');  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } elsif (0x0041 <= $self->{nc} and  
                $self->{nc} <= 0x005A) { # A..Z  
         !!!cp (63);  
         $self->{ca}->{name} .= chr ($self->{nc} + 0x0020);  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x002F) { # /  
         !!!cp (64);  
         $before_leave->();  
         $self->{state} = SELF_CLOSING_START_TAG_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!parse-error (type => 'unclosed tag');  
         $before_leave->();  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (66);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           if ($self->{ct}->{attributes}) {  
             !!!cp (67);  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             ## NOTE: This state should never be reached.  
             !!!cp (68);  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         # reconsume  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } else {  
         if ($self->{nc} == 0x0022 or # "  
             $self->{nc} == 0x0027) { # '  
           !!!cp (69);  
           !!!parse-error (type => 'bad attribute name');  
         } else {  
           !!!cp (70);  
         }  
         $self->{ca}->{name} .= chr ($self->{nc});  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == AFTER_ATTRIBUTE_NAME_STATE) {  
       if ($is_space->{$self->{nc}}) {  
         !!!cp (71);  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003D) { # =  
         !!!cp (72);  
         $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (73);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           if ($self->{ct}->{attributes}) {  
             !!!cp (74);  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             ## NOTE: This state should never be reached.  
             !!!cp (75);  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } elsif (0x0041 <= $self->{nc} and  
                $self->{nc} <= 0x005A) { # A..Z  
         !!!cp (76);  
         $self->{ca}  
             = {name => chr ($self->{nc} + 0x0020),  
                value => '',  
                line => $self->{line}, column => $self->{column}};  
         $self->{state} = ATTRIBUTE_NAME_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x002F) { # /  
         !!!cp (77);  
         $self->{state} = SELF_CLOSING_START_TAG_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!parse-error (type => 'unclosed tag');  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (79);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           if ($self->{ct}->{attributes}) {  
             !!!cp (80);  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             ## NOTE: This state should never be reached.  
             !!!cp (81);  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         # reconsume  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } else {  
         if ($self->{nc} == 0x0022 or # "  
             $self->{nc} == 0x0027) { # '  
           !!!cp (78);  
           !!!parse-error (type => 'bad attribute name');  
         } else {  
           !!!cp (82);  
         }  
         $self->{ca}  
             = {name => chr ($self->{nc}),  
                value => '',  
                line => $self->{line}, column => $self->{column}};  
         $self->{state} = ATTRIBUTE_NAME_STATE;  
         !!!next-input-character;  
         redo A;          
       }  
     } elsif ($self->{state} == BEFORE_ATTRIBUTE_VALUE_STATE) {  
       if ($is_space->{$self->{nc}}) {  
         !!!cp (83);  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x0022) { # "  
         !!!cp (84);  
         $self->{state} = ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x0026) { # &  
         !!!cp (85);  
         $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;  
         ## reconsume  
         redo A;  
       } elsif ($self->{nc} == 0x0027) { # '  
         !!!cp (86);  
         $self->{state} = ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         !!!parse-error (type => 'empty unquoted attribute value');  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (87);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           if ($self->{ct}->{attributes}) {  
             !!!cp (88);  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             ## NOTE: This state should never be reached.  
             !!!cp (89);  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!parse-error (type => 'unclosed tag');  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (90);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           if ($self->{ct}->{attributes}) {  
             !!!cp (91);  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             ## NOTE: This state should never be reached.  
             !!!cp (92);  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } else {  
         if ($self->{nc} == 0x003D) { # =  
           !!!cp (93);  
           !!!parse-error (type => 'bad attribute value');  
         } else {  
           !!!cp (94);  
         }  
         $self->{ca}->{value} .= chr ($self->{nc});  
         $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE) {  
       if ($self->{nc} == 0x0022) { # "  
         !!!cp (95);  
         $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x0026) { # &  
         !!!cp (96);  
         ## NOTE: In the spec, the tokenizer is switched to the  
         ## "entity in attribute value state".  In this implementation, the  
         ## tokenizer is switched to the |ENTITY_STATE|, which is an  
         ## implementation of the "consume a character reference" algorithm.  
         $self->{prev_state} = $self->{state};  
         $self->{entity_add} = 0x0022; # "  
         $self->{state} = ENTITY_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!parse-error (type => 'unclosed attribute value');  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (97);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           if ($self->{ct}->{attributes}) {  
             !!!cp (98);  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             ## NOTE: This state should never be reached.  
             !!!cp (99);  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } else {  
         !!!cp (100);  
         $self->{ca}->{value} .= chr ($self->{nc});  
         $self->{read_until}->($self->{ca}->{value},  
                               q["&],  
                               length $self->{ca}->{value});  
   
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE) {  
       if ($self->{nc} == 0x0027) { # '  
         !!!cp (101);  
         $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x0026) { # &  
         !!!cp (102);  
         ## NOTE: In the spec, the tokenizer is switched to the  
         ## "entity in attribute value state".  In this implementation, the  
         ## tokenizer is switched to the |ENTITY_STATE|, which is an  
         ## implementation of the "consume a character reference" algorithm.  
         $self->{entity_add} = 0x0027; # '  
         $self->{prev_state} = $self->{state};  
         $self->{state} = ENTITY_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!parse-error (type => 'unclosed attribute value');  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (103);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           if ($self->{ct}->{attributes}) {  
             !!!cp (104);  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             ## NOTE: This state should never be reached.  
             !!!cp (105);  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } else {  
         !!!cp (106);  
         $self->{ca}->{value} .= chr ($self->{nc});  
         $self->{read_until}->($self->{ca}->{value},  
                               q['&],  
                               length $self->{ca}->{value});  
   
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == ATTRIBUTE_VALUE_UNQUOTED_STATE) {  
       if ($is_space->{$self->{nc}}) {  
         !!!cp (107);  
         $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x0026) { # &  
         !!!cp (108);  
         ## NOTE: In the spec, the tokenizer is switched to the  
         ## "entity in attribute value state".  In this implementation, the  
         ## tokenizer is switched to the |ENTITY_STATE|, which is an  
         ## implementation of the "consume a character reference" algorithm.  
         $self->{entity_add} = -1;  
         $self->{prev_state} = $self->{state};  
         $self->{state} = ENTITY_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (109);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           if ($self->{ct}->{attributes}) {  
             !!!cp (110);  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             ## NOTE: This state should never be reached.  
             !!!cp (111);  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!parse-error (type => 'unclosed tag');  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (112);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           if ($self->{ct}->{attributes}) {  
             !!!cp (113);  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             ## NOTE: This state should never be reached.  
             !!!cp (114);  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } else {  
         if ({  
              0x0022 => 1, # "  
              0x0027 => 1, # '  
              0x003D => 1, # =  
             }->{$self->{nc}}) {  
           !!!cp (115);  
           !!!parse-error (type => 'bad attribute value');  
         } else {  
           !!!cp (116);  
         }  
         $self->{ca}->{value} .= chr ($self->{nc});  
         $self->{read_until}->($self->{ca}->{value},  
                               q["'=& >],  
                               length $self->{ca}->{value});  
   
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == AFTER_ATTRIBUTE_VALUE_QUOTED_STATE) {  
       if ($is_space->{$self->{nc}}) {  
         !!!cp (118);  
         $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (119);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           if ($self->{ct}->{attributes}) {  
             !!!cp (120);  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             ## NOTE: This state should never be reached.  
             !!!cp (121);  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } elsif ($self->{nc} == 0x002F) { # /  
         !!!cp (122);  
         $self->{state} = SELF_CLOSING_START_TAG_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!parse-error (type => 'unclosed tag');  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (122.3);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           if ($self->{ct}->{attributes}) {  
             !!!cp (122.1);  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             ## NOTE: This state should never be reached.  
             !!!cp (122.2);  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         ## Reconsume.  
         !!!emit ($self->{ct}); # start tag or end tag  
         redo A;  
       } else {  
         !!!cp ('124.1');  
         !!!parse-error (type => 'no space between attributes');  
         $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;  
         ## reconsume  
         redo A;  
       }  
     } elsif ($self->{state} == SELF_CLOSING_START_TAG_STATE) {  
       if ($self->{nc} == 0x003E) { # >  
         if ($self->{ct}->{type} == END_TAG_TOKEN) {  
           !!!cp ('124.2');  
           !!!parse-error (type => 'nestc', token => $self->{ct});  
           ## TODO: Different type than slash in start tag  
           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST  
           if ($self->{ct}->{attributes}) {  
             !!!cp ('124.4');  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             !!!cp ('124.5');  
           }  
           ## TODO: Test |<title></title/>|  
         } else {  
           !!!cp ('124.3');  
           $self->{self_closing} = 1;  
         }  
   
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # start tag or end tag  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!parse-error (type => 'unclosed tag');  
         if ($self->{ct}->{type} == START_TAG_TOKEN) {  
           !!!cp (124.7);  
           $self->{last_stag_name} = $self->{ct}->{tag_name};  
         } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {  
           if ($self->{ct}->{attributes}) {  
             !!!cp (124.5);  
             !!!parse-error (type => 'end tag attribute');  
           } else {  
             ## NOTE: This state should never be reached.  
             !!!cp (124.6);  
           }  
         } else {  
           die "$0: $self->{ct}->{type}: Unknown token type";  
         }  
         $self->{state} = DATA_STATE;  
         ## Reconsume.  
         !!!emit ($self->{ct}); # start tag or end tag  
         redo A;  
       } else {  
         !!!cp ('124.4');  
         !!!parse-error (type => 'nestc');  
         ## TODO: This error type is wrong.  
         $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;  
         ## Reconsume.  
         redo A;  
       }  
     } elsif ($self->{state} == BOGUS_COMMENT_STATE) {  
       ## (only happen if PCDATA state)  
   
       ## NOTE: Unlike spec's "bogus comment state", this implementation  
       ## consumes characters one-by-one basis.  
         
       if ($self->{nc} == 0x003E) { # >  
         !!!cp (124);  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # comment  
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (125);  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         !!!emit ($self->{ct}); # comment  
         redo A;  
       } else {  
         !!!cp (126);  
         $self->{ct}->{data} .= chr ($self->{nc}); # comment  
         $self->{read_until}->($self->{ct}->{data},  
                               q[>],  
                               length $self->{ct}->{data});  
   
         ## Stay in the state.  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == MARKUP_DECLARATION_OPEN_STATE) {  
       ## (only happen if PCDATA state)  
         
       if ($self->{nc} == 0x002D) { # -  
         !!!cp (133);  
         $self->{state} = MD_HYPHEN_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x0044 or # D  
                $self->{nc} == 0x0064) { # d  
         ## ASCII case-insensitive.  
         !!!cp (130);  
         $self->{state} = MD_DOCTYPE_STATE;  
         $self->{s_kwd} = chr $self->{nc};  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM and  
                $self->{open_elements}->[-1]->[1] & FOREIGN_EL and  
                $self->{nc} == 0x005B) { # [  
         !!!cp (135.4);                  
         $self->{state} = MD_CDATA_STATE;  
         $self->{s_kwd} = '[';  
         !!!next-input-character;  
         redo A;  
       } else {  
         !!!cp (136);  
       }  
   
       !!!parse-error (type => 'bogus comment',  
                       line => $self->{line_prev},  
                       column => $self->{column_prev} - 1);  
       ## Reconsume.  
       $self->{state} = BOGUS_COMMENT_STATE;  
       $self->{ct} = {type => COMMENT_TOKEN, data => '',  
                                 line => $self->{line_prev},  
                                 column => $self->{column_prev} - 1,  
                                };  
       redo A;  
     } elsif ($self->{state} == MD_HYPHEN_STATE) {  
       if ($self->{nc} == 0x002D) { # -  
         !!!cp (127);  
         $self->{ct} = {type => COMMENT_TOKEN, data => '',  
                                   line => $self->{line_prev},  
                                   column => $self->{column_prev} - 2,  
                                  };  
         $self->{state} = COMMENT_START_STATE;  
         !!!next-input-character;  
         redo A;  
       } else {  
         !!!cp (128);  
         !!!parse-error (type => 'bogus comment',  
                         line => $self->{line_prev},  
                         column => $self->{column_prev} - 2);  
         $self->{state} = BOGUS_COMMENT_STATE;  
         ## Reconsume.  
         $self->{ct} = {type => COMMENT_TOKEN,  
                                   data => '-',  
                                   line => $self->{line_prev},  
                                   column => $self->{column_prev} - 2,  
                                  };  
         redo A;  
       }  
     } elsif ($self->{state} == MD_DOCTYPE_STATE) {  
       ## ASCII case-insensitive.  
       if ($self->{nc} == [  
             undef,  
             0x004F, # O  
             0x0043, # C  
             0x0054, # T  
             0x0059, # Y  
             0x0050, # P  
           ]->[length $self->{s_kwd}] or  
           $self->{nc} == [  
             undef,  
             0x006F, # o  
             0x0063, # c  
             0x0074, # t  
             0x0079, # y  
             0x0070, # p  
           ]->[length $self->{s_kwd}]) {  
         !!!cp (131);  
         ## Stay in the state.  
         $self->{s_kwd} .= chr $self->{nc};  
         !!!next-input-character;  
         redo A;  
       } elsif ((length $self->{s_kwd}) == 6 and  
                ($self->{nc} == 0x0045 or # E  
                 $self->{nc} == 0x0065)) { # e  
         !!!cp (129);  
         $self->{state} = DOCTYPE_STATE;  
         $self->{ct} = {type => DOCTYPE_TOKEN,  
                                   quirks => 1,  
                                   line => $self->{line_prev},  
                                   column => $self->{column_prev} - 7,  
                                  };  
         !!!next-input-character;  
         redo A;  
       } else {  
         !!!cp (132);          
         !!!parse-error (type => 'bogus comment',  
                         line => $self->{line_prev},  
                         column => $self->{column_prev} - 1 - length $self->{s_kwd});  
         $self->{state} = BOGUS_COMMENT_STATE;  
         ## Reconsume.  
         $self->{ct} = {type => COMMENT_TOKEN,  
                                   data => $self->{s_kwd},  
                                   line => $self->{line_prev},  
                                   column => $self->{column_prev} - 1 - length $self->{s_kwd},  
                                  };  
         redo A;  
       }  
     } elsif ($self->{state} == MD_CDATA_STATE) {  
       if ($self->{nc} == {  
             '[' => 0x0043, # C  
             '[C' => 0x0044, # D  
             '[CD' => 0x0041, # A  
             '[CDA' => 0x0054, # T  
             '[CDAT' => 0x0041, # A  
           }->{$self->{s_kwd}}) {  
         !!!cp (135.1);  
         ## Stay in the state.  
         $self->{s_kwd} .= chr $self->{nc};  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{s_kwd} eq '[CDATA' and  
                $self->{nc} == 0x005B) { # [  
         !!!cp (135.2);  
         $self->{ct} = {type => CHARACTER_TOKEN,  
                                   data => '',  
                                   line => $self->{line_prev},  
                                   column => $self->{column_prev} - 7};  
         $self->{state} = CDATA_SECTION_STATE;  
         !!!next-input-character;  
         redo A;  
       } else {  
         !!!cp (135.3);  
         !!!parse-error (type => 'bogus comment',  
                         line => $self->{line_prev},  
                         column => $self->{column_prev} - 1 - length $self->{s_kwd});  
         $self->{state} = BOGUS_COMMENT_STATE;  
         ## Reconsume.  
         $self->{ct} = {type => COMMENT_TOKEN,  
                                   data => $self->{s_kwd},  
                                   line => $self->{line_prev},  
                                   column => $self->{column_prev} - 1 - length $self->{s_kwd},  
                                  };  
         redo A;  
       }  
     } elsif ($self->{state} == COMMENT_START_STATE) {  
       if ($self->{nc} == 0x002D) { # -  
         !!!cp (137);  
         $self->{state} = COMMENT_START_DASH_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         !!!cp (138);  
         !!!parse-error (type => 'bogus comment');  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # comment  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (139);  
         !!!parse-error (type => 'unclosed comment');  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         !!!emit ($self->{ct}); # comment  
   
         redo A;  
       } else {  
         !!!cp (140);  
         $self->{ct}->{data} # comment  
             .= chr ($self->{nc});  
         $self->{state} = COMMENT_STATE;  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == COMMENT_START_DASH_STATE) {  
       if ($self->{nc} == 0x002D) { # -  
         !!!cp (141);  
         $self->{state} = COMMENT_END_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         !!!cp (142);  
         !!!parse-error (type => 'bogus comment');  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # comment  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (143);  
         !!!parse-error (type => 'unclosed comment');  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         !!!emit ($self->{ct}); # comment  
   
         redo A;  
       } else {  
         !!!cp (144);  
         $self->{ct}->{data} # comment  
             .= '-' . chr ($self->{nc});  
         $self->{state} = COMMENT_STATE;  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == COMMENT_STATE) {  
       if ($self->{nc} == 0x002D) { # -  
         !!!cp (145);  
         $self->{state} = COMMENT_END_DASH_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (146);  
         !!!parse-error (type => 'unclosed comment');  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         !!!emit ($self->{ct}); # comment  
   
         redo A;  
       } else {  
         !!!cp (147);  
         $self->{ct}->{data} .= chr ($self->{nc}); # comment  
         $self->{read_until}->($self->{ct}->{data},  
                               q[-],  
                               length $self->{ct}->{data});  
   
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == COMMENT_END_DASH_STATE) {  
       if ($self->{nc} == 0x002D) { # -  
         !!!cp (148);  
         $self->{state} = COMMENT_END_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (149);  
         !!!parse-error (type => 'unclosed comment');  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         !!!emit ($self->{ct}); # comment  
   
         redo A;  
       } else {  
         !!!cp (150);  
         $self->{ct}->{data} .= '-' . chr ($self->{nc}); # comment  
         $self->{state} = COMMENT_STATE;  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == COMMENT_END_STATE) {  
       if ($self->{nc} == 0x003E) { # >  
         !!!cp (151);  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # comment  
   
         redo A;  
       } elsif ($self->{nc} == 0x002D) { # -  
         !!!cp (152);  
         !!!parse-error (type => 'dash in comment',  
                         line => $self->{line_prev},  
                         column => $self->{column_prev});  
         $self->{ct}->{data} .= '-'; # comment  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (153);  
         !!!parse-error (type => 'unclosed comment');  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         !!!emit ($self->{ct}); # comment  
   
         redo A;  
       } else {  
         !!!cp (154);  
         !!!parse-error (type => 'dash in comment',  
                         line => $self->{line_prev},  
                         column => $self->{column_prev});  
         $self->{ct}->{data} .= '--' . chr ($self->{nc}); # comment  
         $self->{state} = COMMENT_STATE;  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == DOCTYPE_STATE) {  
       if ($is_space->{$self->{nc}}) {  
         !!!cp (155);  
         $self->{state} = BEFORE_DOCTYPE_NAME_STATE;  
         !!!next-input-character;  
         redo A;  
       } else {  
         !!!cp (156);  
         !!!parse-error (type => 'no space before DOCTYPE name');  
         $self->{state} = BEFORE_DOCTYPE_NAME_STATE;  
         ## reconsume  
         redo A;  
       }  
     } elsif ($self->{state} == BEFORE_DOCTYPE_NAME_STATE) {  
       if ($is_space->{$self->{nc}}) {  
         !!!cp (157);  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         !!!cp (158);  
         !!!parse-error (type => 'no DOCTYPE name');  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # DOCTYPE (quirks)  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (159);  
         !!!parse-error (type => 'no DOCTYPE name');  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         !!!emit ($self->{ct}); # DOCTYPE (quirks)  
   
         redo A;  
       } else {  
         !!!cp (160);  
         $self->{ct}->{name} = chr $self->{nc};  
         delete $self->{ct}->{quirks};  
 ## ISSUE: "Set the token's name name to the" in the spec  
         $self->{state} = DOCTYPE_NAME_STATE;  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == DOCTYPE_NAME_STATE) {  
 ## ISSUE: Redundant "First," in the spec.  
       if ($is_space->{$self->{nc}}) {  
         !!!cp (161);  
         $self->{state} = AFTER_DOCTYPE_NAME_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         !!!cp (162);  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (163);  
         !!!parse-error (type => 'unclosed DOCTYPE');  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } else {  
         !!!cp (164);  
         $self->{ct}->{name}  
           .= chr ($self->{nc}); # DOCTYPE  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == AFTER_DOCTYPE_NAME_STATE) {  
       if ($is_space->{$self->{nc}}) {  
         !!!cp (165);  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         !!!cp (166);  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (167);  
         !!!parse-error (type => 'unclosed DOCTYPE');  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } elsif ($self->{nc} == 0x0050 or # P  
                $self->{nc} == 0x0070) { # p  
         $self->{state} = PUBLIC_STATE;  
         $self->{s_kwd} = chr $self->{nc};  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x0053 or # S  
                $self->{nc} == 0x0073) { # s  
         $self->{state} = SYSTEM_STATE;  
         $self->{s_kwd} = chr $self->{nc};  
         !!!next-input-character;  
         redo A;  
       } else {  
         !!!cp (180);  
         !!!parse-error (type => 'string after DOCTYPE name');  
         $self->{ct}->{quirks} = 1;  
   
         $self->{state} = BOGUS_DOCTYPE_STATE;  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == PUBLIC_STATE) {  
       ## ASCII case-insensitive  
       if ($self->{nc} == [  
             undef,  
             0x0055, # U  
             0x0042, # B  
             0x004C, # L  
             0x0049, # I  
           ]->[length $self->{s_kwd}] or  
           $self->{nc} == [  
             undef,  
             0x0075, # u  
             0x0062, # b  
             0x006C, # l  
             0x0069, # i  
           ]->[length $self->{s_kwd}]) {  
         !!!cp (175);  
         ## Stay in the state.  
         $self->{s_kwd} .= chr $self->{nc};  
         !!!next-input-character;  
         redo A;  
       } elsif ((length $self->{s_kwd}) == 5 and  
                ($self->{nc} == 0x0043 or # C  
                 $self->{nc} == 0x0063)) { # c  
         !!!cp (168);  
         $self->{state} = BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE;  
         !!!next-input-character;  
         redo A;  
       } else {  
         !!!cp (169);  
         !!!parse-error (type => 'string after DOCTYPE name',  
                         line => $self->{line_prev},  
                         column => $self->{column_prev} + 1 - length $self->{s_kwd});  
         $self->{ct}->{quirks} = 1;  
   
         $self->{state} = BOGUS_DOCTYPE_STATE;  
         ## Reconsume.  
         redo A;  
       }  
     } elsif ($self->{state} == SYSTEM_STATE) {  
       ## ASCII case-insensitive  
       if ($self->{nc} == [  
             undef,  
             0x0059, # Y  
             0x0053, # S  
             0x0054, # T  
             0x0045, # E  
           ]->[length $self->{s_kwd}] or  
           $self->{nc} == [  
             undef,  
             0x0079, # y  
             0x0073, # s  
             0x0074, # t  
             0x0065, # e  
           ]->[length $self->{s_kwd}]) {  
         !!!cp (170);  
         ## Stay in the state.  
         $self->{s_kwd} .= chr $self->{nc};  
         !!!next-input-character;  
         redo A;  
       } elsif ((length $self->{s_kwd}) == 5 and  
                ($self->{nc} == 0x004D or # M  
                 $self->{nc} == 0x006D)) { # m  
         !!!cp (171);  
         $self->{state} = BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE;  
         !!!next-input-character;  
         redo A;  
       } else {  
         !!!cp (172);  
         !!!parse-error (type => 'string after DOCTYPE name',  
                         line => $self->{line_prev},  
                         column => $self->{column_prev} + 1 - length $self->{s_kwd});  
         $self->{ct}->{quirks} = 1;  
   
         $self->{state} = BOGUS_DOCTYPE_STATE;  
         ## Reconsume.  
         redo A;  
       }  
     } elsif ($self->{state} == BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {  
       if ($is_space->{$self->{nc}}) {  
         !!!cp (181);  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} eq 0x0022) { # "  
         !!!cp (182);  
         $self->{ct}->{pubid} = ''; # DOCTYPE  
         $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} eq 0x0027) { # '  
         !!!cp (183);  
         $self->{ct}->{pubid} = ''; # DOCTYPE  
         $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} eq 0x003E) { # >  
         !!!cp (184);  
         !!!parse-error (type => 'no PUBLIC literal');  
   
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (185);  
         !!!parse-error (type => 'unclosed DOCTYPE');  
   
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } else {  
         !!!cp (186);  
         !!!parse-error (type => 'string after PUBLIC');  
         $self->{ct}->{quirks} = 1;  
   
         $self->{state} = BOGUS_DOCTYPE_STATE;  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE) {  
       if ($self->{nc} == 0x0022) { # "  
         !!!cp (187);  
         $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         !!!cp (188);  
         !!!parse-error (type => 'unclosed PUBLIC literal');  
   
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (189);  
         !!!parse-error (type => 'unclosed PUBLIC literal');  
   
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } else {  
         !!!cp (190);  
         $self->{ct}->{pubid} # DOCTYPE  
             .= chr $self->{nc};  
         $self->{read_until}->($self->{ct}->{pubid}, q[">],  
                               length $self->{ct}->{pubid});  
   
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE) {  
       if ($self->{nc} == 0x0027) { # '  
         !!!cp (191);  
         $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         !!!cp (192);  
         !!!parse-error (type => 'unclosed PUBLIC literal');  
   
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (193);  
         !!!parse-error (type => 'unclosed PUBLIC literal');  
   
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } else {  
         !!!cp (194);  
         $self->{ct}->{pubid} # DOCTYPE  
             .= chr $self->{nc};  
         $self->{read_until}->($self->{ct}->{pubid}, q['>],  
                               length $self->{ct}->{pubid});  
   
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {  
       if ($is_space->{$self->{nc}}) {  
         !!!cp (195);  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x0022) { # "  
         !!!cp (196);  
         $self->{ct}->{sysid} = ''; # DOCTYPE  
         $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x0027) { # '  
         !!!cp (197);  
         $self->{ct}->{sysid} = ''; # DOCTYPE  
         $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         !!!cp (198);  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (199);  
         !!!parse-error (type => 'unclosed DOCTYPE');  
   
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } else {  
         !!!cp (200);  
         !!!parse-error (type => 'string after PUBLIC literal');  
         $self->{ct}->{quirks} = 1;  
   
         $self->{state} = BOGUS_DOCTYPE_STATE;  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {  
       if ($is_space->{$self->{nc}}) {  
         !!!cp (201);  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x0022) { # "  
         !!!cp (202);  
         $self->{ct}->{sysid} = ''; # DOCTYPE  
         $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x0027) { # '  
         !!!cp (203);  
         $self->{ct}->{sysid} = ''; # DOCTYPE  
         $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         !!!cp (204);  
         !!!parse-error (type => 'no SYSTEM literal');  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (205);  
         !!!parse-error (type => 'unclosed DOCTYPE');  
   
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } else {  
         !!!cp (206);  
         !!!parse-error (type => 'string after SYSTEM');  
         $self->{ct}->{quirks} = 1;  
   
         $self->{state} = BOGUS_DOCTYPE_STATE;  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE) {  
       if ($self->{nc} == 0x0022) { # "  
         !!!cp (207);  
         $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         !!!cp (208);  
         !!!parse-error (type => 'unclosed SYSTEM literal');  
   
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (209);  
         !!!parse-error (type => 'unclosed SYSTEM literal');  
   
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } else {  
         !!!cp (210);  
         $self->{ct}->{sysid} # DOCTYPE  
             .= chr $self->{nc};  
         $self->{read_until}->($self->{ct}->{sysid}, q[">],  
                               length $self->{ct}->{sysid});  
   
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE) {  
       if ($self->{nc} == 0x0027) { # '  
         !!!cp (211);  
         $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         !!!cp (212);  
         !!!parse-error (type => 'unclosed SYSTEM literal');  
   
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (213);  
         !!!parse-error (type => 'unclosed SYSTEM literal');  
   
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } else {  
         !!!cp (214);  
         $self->{ct}->{sysid} # DOCTYPE  
             .= chr $self->{nc};  
         $self->{read_until}->($self->{ct}->{sysid}, q['>],  
                               length $self->{ct}->{sysid});  
   
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {  
       if ($is_space->{$self->{nc}}) {  
         !!!cp (215);  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003E) { # >  
         !!!cp (216);  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (217);  
         !!!parse-error (type => 'unclosed DOCTYPE');  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         $self->{ct}->{quirks} = 1;  
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } else {  
         !!!cp (218);  
         !!!parse-error (type => 'string after SYSTEM literal');  
         #$self->{ct}->{quirks} = 1;  
   
         $self->{state} = BOGUS_DOCTYPE_STATE;  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == BOGUS_DOCTYPE_STATE) {  
       if ($self->{nc} == 0x003E) { # >  
         !!!cp (219);  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
   
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } elsif ($self->{nc} == -1) {  
         !!!cp (220);  
         !!!parse-error (type => 'unclosed DOCTYPE');  
         $self->{state} = DATA_STATE;  
         ## reconsume  
   
         !!!emit ($self->{ct}); # DOCTYPE  
   
         redo A;  
       } else {  
         !!!cp (221);  
         my $s = '';  
         $self->{read_until}->($s, q[>], 0);  
   
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} == CDATA_SECTION_STATE) {  
       ## NOTE: "CDATA section state" in the state is jointly implemented  
       ## by three states, |CDATA_SECTION_STATE|, |CDATA_SECTION_MSE1_STATE|,  
       ## and |CDATA_SECTION_MSE2_STATE|.  
         
       if ($self->{nc} == 0x005D) { # ]  
         !!!cp (221.1);  
         $self->{state} = CDATA_SECTION_MSE1_STATE;  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == -1) {  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
         if (length $self->{ct}->{data}) { # character  
           !!!cp (221.2);  
           !!!emit ($self->{ct}); # character  
         } else {  
           !!!cp (221.3);  
           ## No token to emit. $self->{ct} is discarded.  
         }          
         redo A;  
       } else {  
         !!!cp (221.4);  
         $self->{ct}->{data} .= chr $self->{nc};  
         $self->{read_until}->($self->{ct}->{data},  
                               q<]>,  
                               length $self->{ct}->{data});  
   
         ## Stay in the state.  
         !!!next-input-character;  
         redo A;  
       }  
   
       ## ISSUE: "text tokens" in spec.  
     } elsif ($self->{state} == CDATA_SECTION_MSE1_STATE) {  
       if ($self->{nc} == 0x005D) { # ]  
         !!!cp (221.5);  
         $self->{state} = CDATA_SECTION_MSE2_STATE;  
         !!!next-input-character;  
         redo A;  
       } else {  
         !!!cp (221.6);  
         $self->{ct}->{data} .= ']';  
         $self->{state} = CDATA_SECTION_STATE;  
         ## Reconsume.  
         redo A;  
       }  
     } elsif ($self->{state} == CDATA_SECTION_MSE2_STATE) {  
       if ($self->{nc} == 0x003E) { # >  
         $self->{state} = DATA_STATE;  
         !!!next-input-character;  
         if (length $self->{ct}->{data}) { # character  
           !!!cp (221.7);  
           !!!emit ($self->{ct}); # character  
         } else {  
           !!!cp (221.8);  
           ## No token to emit. $self->{ct} is discarded.  
         }  
         redo A;  
       } elsif ($self->{nc} == 0x005D) { # ]  
         !!!cp (221.9); # character  
         $self->{ct}->{data} .= ']'; ## Add first "]" of "]]]".  
         ## Stay in the state.  
         !!!next-input-character;  
         redo A;  
       } else {  
         !!!cp (221.11);  
         $self->{ct}->{data} .= ']]'; # character  
         $self->{state} = CDATA_SECTION_STATE;  
         ## Reconsume.  
         redo A;  
       }  
     } elsif ($self->{state} == ENTITY_STATE) {  
       if ($is_space->{$self->{nc}} or  
           {  
             0x003C => 1, 0x0026 => 1, -1 => 1, # <, &  
             $self->{entity_add} => 1,  
           }->{$self->{nc}}) {  
         !!!cp (1001);  
         ## Don't consume  
         ## No error  
         ## Return nothing.  
         #  
       } elsif ($self->{nc} == 0x0023) { # #  
         !!!cp (999);  
         $self->{state} = ENTITY_HASH_STATE;  
         $self->{s_kwd} = '#';  
         !!!next-input-character;  
         redo A;  
       } elsif ((0x0041 <= $self->{nc} and  
                 $self->{nc} <= 0x005A) or # A..Z  
                (0x0061 <= $self->{nc} and  
                 $self->{nc} <= 0x007A)) { # a..z  
         !!!cp (998);  
         require Whatpm::_NamedEntityList;  
         $self->{state} = ENTITY_NAME_STATE;  
         $self->{s_kwd} = chr $self->{nc};  
         $self->{entity__value} = $self->{s_kwd};  
         $self->{entity__match} = 0;  
         !!!next-input-character;  
         redo A;  
       } else {  
         !!!cp (1027);  
         !!!parse-error (type => 'bare ero');  
         ## Return nothing.  
         #  
       }  
   
       ## NOTE: No character is consumed by the "consume a character  
       ## reference" algorithm.  In other word, there is an "&" character  
       ## that does not introduce a character reference, which would be  
       ## appended to the parent element or the attribute value in later  
       ## process of the tokenizer.  
   
       if ($self->{prev_state} == DATA_STATE) {  
         !!!cp (997);  
         $self->{state} = $self->{prev_state};  
         ## Reconsume.  
         !!!emit ({type => CHARACTER_TOKEN, data => '&',  
                   line => $self->{line_prev},  
                   column => $self->{column_prev},  
                  });  
         redo A;  
       } else {  
         !!!cp (996);  
         $self->{ca}->{value} .= '&';  
         $self->{state} = $self->{prev_state};  
         ## Reconsume.  
         redo A;  
       }  
     } elsif ($self->{state} == ENTITY_HASH_STATE) {  
       if ($self->{nc} == 0x0078 or # x  
           $self->{nc} == 0x0058) { # X  
         !!!cp (995);  
         $self->{state} = HEXREF_X_STATE;  
         $self->{s_kwd} .= chr $self->{nc};  
         !!!next-input-character;  
         redo A;  
       } elsif (0x0030 <= $self->{nc} and  
                $self->{nc} <= 0x0039) { # 0..9  
         !!!cp (994);  
         $self->{state} = NCR_NUM_STATE;  
         $self->{s_kwd} = $self->{nc} - 0x0030;  
         !!!next-input-character;  
         redo A;  
       } else {  
         !!!parse-error (type => 'bare nero',  
                         line => $self->{line_prev},  
                         column => $self->{column_prev} - 1);  
   
         ## NOTE: According to the spec algorithm, nothing is returned,  
         ## and then "&#" is appended to the parent element or the attribute  
         ## value in the later processing.  
   
         if ($self->{prev_state} == DATA_STATE) {  
           !!!cp (1019);  
           $self->{state} = $self->{prev_state};  
           ## Reconsume.  
           !!!emit ({type => CHARACTER_TOKEN,  
                     data => '&#',  
                     line => $self->{line_prev},  
                     column => $self->{column_prev} - 1,  
                    });  
           redo A;  
         } else {  
           !!!cp (993);  
           $self->{ca}->{value} .= '&#';  
           $self->{state} = $self->{prev_state};  
           ## Reconsume.  
           redo A;  
         }  
       }  
     } elsif ($self->{state} == NCR_NUM_STATE) {  
       if (0x0030 <= $self->{nc} and  
           $self->{nc} <= 0x0039) { # 0..9  
         !!!cp (1012);  
         $self->{s_kwd} *= 10;  
         $self->{s_kwd} += $self->{nc} - 0x0030;  
           
         ## Stay in the state.  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003B) { # ;  
         !!!cp (1013);  
         !!!next-input-character;  
         #  
       } else {  
         !!!cp (1014);  
         !!!parse-error (type => 'no refc');  
         ## Reconsume.  
         #  
       }  
   
       my $code = $self->{s_kwd};  
       my $l = $self->{line_prev};  
       my $c = $self->{column_prev};  
       if ($charref_map->{$code}) {  
         !!!cp (1015);  
         !!!parse-error (type => 'invalid character reference',  
                         text => (sprintf 'U+%04X', $code),  
                         line => $l, column => $c);  
         $code = $charref_map->{$code};  
       } elsif ($code > 0x10FFFF) {  
         !!!cp (1016);  
         !!!parse-error (type => 'invalid character reference',  
                         text => (sprintf 'U-%08X', $code),  
                         line => $l, column => $c);  
         $code = 0xFFFD;  
       }  
   
       if ($self->{prev_state} == DATA_STATE) {  
         !!!cp (992);  
         $self->{state} = $self->{prev_state};  
         ## Reconsume.  
         !!!emit ({type => CHARACTER_TOKEN, data => chr $code,  
                   line => $l, column => $c,  
                  });  
         redo A;  
       } else {  
         !!!cp (991);  
         $self->{ca}->{value} .= chr $code;  
         $self->{ca}->{has_reference} = 1;  
         $self->{state} = $self->{prev_state};  
         ## Reconsume.  
         redo A;  
       }  
     } elsif ($self->{state} == HEXREF_X_STATE) {  
       if ((0x0030 <= $self->{nc} and $self->{nc} <= 0x0039) or  
           (0x0041 <= $self->{nc} and $self->{nc} <= 0x0046) or  
           (0x0061 <= $self->{nc} and $self->{nc} <= 0x0066)) {  
         # 0..9, A..F, a..f  
         !!!cp (990);  
         $self->{state} = HEXREF_HEX_STATE;  
         $self->{s_kwd} = 0;  
         ## Reconsume.  
         redo A;  
       } else {  
         !!!parse-error (type => 'bare hcro',  
                         line => $self->{line_prev},  
                         column => $self->{column_prev} - 2);  
   
         ## NOTE: According to the spec algorithm, nothing is returned,  
         ## and then "&#" followed by "X" or "x" is appended to the parent  
         ## element or the attribute value in the later processing.  
   
         if ($self->{prev_state} == DATA_STATE) {  
           !!!cp (1005);  
           $self->{state} = $self->{prev_state};  
           ## Reconsume.  
           !!!emit ({type => CHARACTER_TOKEN,  
                     data => '&' . $self->{s_kwd},  
                     line => $self->{line_prev},  
                     column => $self->{column_prev} - length $self->{s_kwd},  
                    });  
           redo A;  
         } else {  
           !!!cp (989);  
           $self->{ca}->{value} .= '&' . $self->{s_kwd};  
           $self->{state} = $self->{prev_state};  
           ## Reconsume.  
           redo A;  
         }  
       }  
     } elsif ($self->{state} == HEXREF_HEX_STATE) {  
       if (0x0030 <= $self->{nc} and $self->{nc} <= 0x0039) {  
         # 0..9  
         !!!cp (1002);  
         $self->{s_kwd} *= 0x10;  
         $self->{s_kwd} += $self->{nc} - 0x0030;  
         ## Stay in the state.  
         !!!next-input-character;  
         redo A;  
       } elsif (0x0061 <= $self->{nc} and  
                $self->{nc} <= 0x0066) { # a..f  
         !!!cp (1003);  
         $self->{s_kwd} *= 0x10;  
         $self->{s_kwd} += $self->{nc} - 0x0060 + 9;  
         ## Stay in the state.  
         !!!next-input-character;  
         redo A;  
       } elsif (0x0041 <= $self->{nc} and  
                $self->{nc} <= 0x0046) { # A..F  
         !!!cp (1004);  
         $self->{s_kwd} *= 0x10;  
         $self->{s_kwd} += $self->{nc} - 0x0040 + 9;  
         ## Stay in the state.  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{nc} == 0x003B) { # ;  
         !!!cp (1006);  
         !!!next-input-character;  
         #  
       } else {  
         !!!cp (1007);  
         !!!parse-error (type => 'no refc',  
                         line => $self->{line},  
                         column => $self->{column});  
         ## Reconsume.  
         #  
       }  
   
       my $code = $self->{s_kwd};  
       my $l = $self->{line_prev};  
       my $c = $self->{column_prev};  
       if ($charref_map->{$code}) {  
         !!!cp (1008);  
         !!!parse-error (type => 'invalid character reference',  
                         text => (sprintf 'U+%04X', $code),  
                         line => $l, column => $c);  
         $code = $charref_map->{$code};  
       } elsif ($code > 0x10FFFF) {  
         !!!cp (1009);  
         !!!parse-error (type => 'invalid character reference',  
                         text => (sprintf 'U-%08X', $code),  
                         line => $l, column => $c);  
         $code = 0xFFFD;  
       }  
   
       if ($self->{prev_state} == DATA_STATE) {  
         !!!cp (988);  
         $self->{state} = $self->{prev_state};  
         ## Reconsume.  
         !!!emit ({type => CHARACTER_TOKEN, data => chr $code,  
                   line => $l, column => $c,  
                  });  
         redo A;  
       } else {  
         !!!cp (987);  
         $self->{ca}->{value} .= chr $code;  
         $self->{ca}->{has_reference} = 1;  
         $self->{state} = $self->{prev_state};  
         ## Reconsume.  
         redo A;  
       }  
     } elsif ($self->{state} == ENTITY_NAME_STATE) {  
       if (length $self->{s_kwd} < 30 and  
           ## NOTE: Some number greater than the maximum length of entity name  
           ((0x0041 <= $self->{nc} and # a  
             $self->{nc} <= 0x005A) or # x  
            (0x0061 <= $self->{nc} and # a  
             $self->{nc} <= 0x007A) or # z  
            (0x0030 <= $self->{nc} and # 0  
             $self->{nc} <= 0x0039) or # 9  
            $self->{nc} == 0x003B)) { # ;  
         our $EntityChar;  
         $self->{s_kwd} .= chr $self->{nc};  
         if (defined $EntityChar->{$self->{s_kwd}}) {  
           if ($self->{nc} == 0x003B) { # ;  
             !!!cp (1020);  
             $self->{entity__value} = $EntityChar->{$self->{s_kwd}};  
             $self->{entity__match} = 1;  
             !!!next-input-character;  
             #  
           } else {  
             !!!cp (1021);  
             $self->{entity__value} = $EntityChar->{$self->{s_kwd}};  
             $self->{entity__match} = -1;  
             ## Stay in the state.  
             !!!next-input-character;  
             redo A;  
           }  
         } else {  
           !!!cp (1022);  
           $self->{entity__value} .= chr $self->{nc};  
           $self->{entity__match} *= 2;  
           ## Stay in the state.  
           !!!next-input-character;  
           redo A;  
         }  
       }  
   
       my $data;  
       my $has_ref;  
       if ($self->{entity__match} > 0) {  
         !!!cp (1023);  
         $data = $self->{entity__value};  
         $has_ref = 1;  
         #  
       } elsif ($self->{entity__match} < 0) {  
         !!!parse-error (type => 'no refc');  
         if ($self->{prev_state} != DATA_STATE and # in attribute  
             $self->{entity__match} < -1) {  
           !!!cp (1024);  
           $data = '&' . $self->{s_kwd};  
           #  
         } else {  
           !!!cp (1025);  
           $data = $self->{entity__value};  
           $has_ref = 1;  
           #  
         }  
       } else {  
         !!!cp (1026);  
         !!!parse-error (type => 'bare ero',  
                         line => $self->{line_prev},  
                         column => $self->{column_prev} - length $self->{s_kwd});  
         $data = '&' . $self->{s_kwd};  
         #  
       }  
     
       ## NOTE: In these cases, when a character reference is found,  
       ## it is consumed and a character token is returned, or, otherwise,  
       ## nothing is consumed and returned, according to the spec algorithm.  
       ## In this implementation, anything that has been examined by the  
       ## tokenizer is appended to the parent element or the attribute value  
       ## as string, either literal string when no character reference or  
       ## entity-replaced string otherwise, in this stage, since any characters  
       ## that would not be consumed are appended in the data state or in an  
       ## appropriate attribute value state anyway.  
   
       if ($self->{prev_state} == DATA_STATE) {  
         !!!cp (986);  
         $self->{state} = $self->{prev_state};  
         ## Reconsume.  
         !!!emit ({type => CHARACTER_TOKEN,  
                   data => $data,  
                   line => $self->{line_prev},  
                   column => $self->{column_prev} + 1 - length $self->{s_kwd},  
                  });  
         redo A;  
       } else {  
         !!!cp (985);  
         $self->{ca}->{value} .= $data;  
         $self->{ca}->{has_reference} = 1 if $has_ref;  
         $self->{state} = $self->{prev_state};  
         ## Reconsume.  
         redo A;  
       }  
     } else {  
       die "$0: $self->{state}: Unknown state";  
     }  
   } # A    
   
   die "$0: _get_next_token: unexpected case";  
 } # _get_next_token  
   
850  sub _initialize_tree_constructor ($) {  sub _initialize_tree_constructor ($) {
851    my $self = shift;    my $self = shift;
852    ## NOTE: $self->{document} MUST be specified before this method is called    ## NOTE: $self->{document} MUST be specified before this method is called
# Line 3469  sub _construct_tree ($) { Line 875  sub _construct_tree ($) {
875    ## When an interactive UA render the $self->{document} available    ## When an interactive UA render the $self->{document} available
876    ## to the user, or when it begin accepting user input, are    ## to the user, or when it begin accepting user input, are
877    ## not defined.    ## not defined.
   
   ## Append a character: collect it and all subsequent consecutive  
   ## characters and insert one Text node whose data is concatenation  
   ## of all those characters. # MUST  
878        
879    !!!next-token;    !!!next-token;
880    
881    undef $self->{form_element};    undef $self->{form_element};
882    undef $self->{head_element};    undef $self->{head_element};
883      undef $self->{head_element_inserted};
884    $self->{open_elements} = [];    $self->{open_elements} = [];
885    undef $self->{inner_html_node};    undef $self->{inner_html_node};
886      undef $self->{ignore_newline};
887    
888    ## NOTE: The "initial" insertion mode.    ## NOTE: The "initial" insertion mode.
889    $self->_tree_construction_initial; # MUST    $self->_tree_construction_initial; # MUST
# Line 3785  sub _tree_construction_root_element ($) Line 1189  sub _tree_construction_root_element ($)
1189      ## NOTE: Reprocess the token.      ## NOTE: Reprocess the token.
1190      !!!ack-later;      !!!ack-later;
1191      return; ## Go to the "before head" insertion mode.      return; ## Go to the "before head" insertion mode.
   
     ## ISSUE: There is an issue in the spec  
1192    } # B    } # B
1193    
1194    die "$0: _tree_construction_root_element: This should never be reached";    die "$0: _tree_construction_root_element: This should never be reached";
# Line 3822  sub _reset_insertion_mode ($) { Line 1224  sub _reset_insertion_mode ($) {
1224          ## SVG elements.  Currently the HTML syntax supports only MathML and          ## SVG elements.  Currently the HTML syntax supports only MathML and
1225          ## SVG elements as foreigners.          ## SVG elements as foreigners.
1226          $new_mode = IN_BODY_IM | IN_FOREIGN_CONTENT_IM;          $new_mode = IN_BODY_IM | IN_FOREIGN_CONTENT_IM;
1227        } elsif ($node->[1] & TABLE_CELL_EL) {        } elsif ($node->[1] == TABLE_CELL_EL) {
1228          if ($last) {          if ($last) {
1229            !!!cp ('t28.2');            !!!cp ('t28.2');
1230            #            #
# Line 3851  sub _reset_insertion_mode ($) { Line 1253  sub _reset_insertion_mode ($) {
1253        $self->{insertion_mode} = $new_mode and return if defined $new_mode;        $self->{insertion_mode} = $new_mode and return if defined $new_mode;
1254                
1255        ## Step 15        ## Step 15
1256        if ($node->[1] & HTML_EL) {        if ($node->[1] == HTML_EL) {
1257          unless (defined $self->{head_element}) {          unless (defined $self->{head_element}) {
1258            !!!cp ('t29');            !!!cp ('t29');
1259            $self->{insertion_mode} = BEFORE_HEAD_IM;            $self->{insertion_mode} = BEFORE_HEAD_IM;
# Line 3983  sub _tree_construction_main ($) { Line 1385  sub _tree_construction_main ($) {
1385    
1386      ## Step 1      ## Step 1
1387      my $start_tag_name = $token->{tag_name};      my $start_tag_name = $token->{tag_name};
1388      my $el;      !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
     !!!create-element ($el, $HTML_NS, $start_tag_name, $token->{attributes}, $token);  
1389    
1390      ## Step 2      ## Step 2
     $insert->($el);  
   
     ## Step 3  
1391      $self->{content_model} = $content_model_flag; # CDATA or RCDATA      $self->{content_model} = $content_model_flag; # CDATA or RCDATA
1392      delete $self->{escape}; # MUST      delete $self->{escape}; # MUST
1393    
1394      ## Step 4      ## Step 3, 4
1395      my $text = '';      $self->{insertion_mode} |= IN_CDATA_RCDATA_IM;
     !!!nack ('t40.1');  
     !!!next-token;  
     while ($token->{type} == CHARACTER_TOKEN) { # or until stop tokenizing  
       !!!cp ('t40');  
       $text .= $token->{data};  
       !!!next-token;  
     }  
   
     ## Step 5  
     if (length $text) {  
       !!!cp ('t41');  
       my $text = $self->{document}->create_text_node ($text);  
       $el->append_child ($text);  
     }  
   
     ## Step 6  
     $self->{content_model} = PCDATA_CONTENT_MODEL;  
1396    
1397      ## Step 7      !!!nack ('t40.1');
     if ($token->{type} == END_TAG_TOKEN and  
         $token->{tag_name} eq $start_tag_name) {  
       !!!cp ('t42');  
       ## Ignore the token  
     } else {  
       ## NOTE: An end-of-file token.  
       if ($content_model_flag == CDATA_CONTENT_MODEL) {  
         !!!cp ('t43');  
         !!!parse-error (type => 'in CDATA:#eof', token => $token);  
       } elsif ($content_model_flag == RCDATA_CONTENT_MODEL) {  
         !!!cp ('t44');  
         !!!parse-error (type => 'in RCDATA:#eof', token => $token);  
       } else {  
         die "$0: $content_model_flag in parse_rcdata";  
       }  
     }  
1398      !!!next-token;      !!!next-token;
1399    }; # $parse_rcdata    }; # $parse_rcdata
1400    
1401    my $script_start_tag = sub () {    my $script_start_tag = sub () {
1402        ## Step 1
1403      my $script_el;      my $script_el;
1404      !!!create-element ($script_el, $HTML_NS, 'script', $token->{attributes}, $token);      !!!create-element ($script_el, $HTML_NS, 'script', $token->{attributes}, $token);
1405    
1406        ## Step 2
1407      ## TODO: mark as "parser-inserted"      ## TODO: mark as "parser-inserted"
1408    
1409        ## Step 3
1410        ## TODO: Mark as "already executed", if ...
1411    
1412        ## Step 4
1413        $insert->($script_el);
1414    
1415        ## ISSUE: $script_el is not put into the stack
1416        push @{$self->{open_elements}}, [$script_el, $el_category->{script}];
1417    
1418        ## Step 5
1419      $self->{content_model} = CDATA_CONTENT_MODEL;      $self->{content_model} = CDATA_CONTENT_MODEL;
1420      delete $self->{escape}; # MUST      delete $self->{escape}; # MUST
       
     my $text = '';  
     !!!nack ('t45.1');  
     !!!next-token;  
     while ($token->{type} == CHARACTER_TOKEN) {  
       !!!cp ('t45');  
       $text .= $token->{data};  
       !!!next-token;  
     } # stop if non-character token or tokenizer stops tokenising  
     if (length $text) {  
       !!!cp ('t46');  
       $script_el->manakai_append_text ($text);  
     }  
                 
     $self->{content_model} = PCDATA_CONTENT_MODEL;  
1421    
1422      if ($token->{type} == END_TAG_TOKEN and      ## Step 6-7
1423          $token->{tag_name} eq 'script') {      $self->{insertion_mode} |= IN_CDATA_RCDATA_IM;
       !!!cp ('t47');  
       ## Ignore the token  
     } else {  
       !!!cp ('t48');  
       !!!parse-error (type => 'in CDATA:#eof', token => $token);  
       ## ISSUE: And ignore?  
       ## TODO: mark as "already executed"  
     }  
       
     if (defined $self->{inner_html_node}) {  
       !!!cp ('t49');  
       ## TODO: mark as "already executed"  
     } else {  
       !!!cp ('t50');  
       ## TODO: $old_insertion_point = current insertion point  
       ## TODO: insertion point = just before the next input character  
1424    
1425        $insert->($script_el);      !!!nack ('t40.2');
         
       ## TODO: insertion point = $old_insertion_point (might be "undefined")  
         
       ## TODO: if there is a script that will execute as soon as the parser resume, then...  
     }  
       
1426      !!!next-token;      !!!next-token;
1427    }; # $script_start_tag    }; # $script_start_tag
1428    
1429    ## NOTE: $open_tables->[-1]->[0] is the "current table" element node.    ## NOTE: $open_tables->[-1]->[0] is the "current table" element node.
1430    ## NOTE: $open_tables->[-1]->[1] is the "tainted" flag.    ## NOTE: $open_tables->[-1]->[1] is the "tainted" flag.
1431      ## NOTE: $open_tables->[-1]->[2] is set false when non-Text node inserted.
1432    my $open_tables = [[$self->{open_elements}->[0]->[0]]];    my $open_tables = [[$self->{open_elements}->[0]->[0]]];
1433    
1434    my $formatting_end_tag = sub {    my $formatting_end_tag = sub {
# Line 4171  sub _tree_construction_main ($) { Line 1513  sub _tree_construction_main ($) {
1513            !!!cp ('t59');            !!!cp ('t59');
1514            $furthest_block = $node;            $furthest_block = $node;
1515            $furthest_block_i_in_open = $_;            $furthest_block_i_in_open = $_;
1516              ## NOTE: The topmost (eldest) node.
1517          } elsif ($node->[0] eq $formatting_element->[0]) {          } elsif ($node->[0] eq $formatting_element->[0]) {
1518            !!!cp ('t60');            !!!cp ('t60');
1519            last OE;            last OE;
# Line 4257  sub _tree_construction_main ($) { Line 1600  sub _tree_construction_main ($) {
1600          my $foster_parent_element;          my $foster_parent_element;
1601          my $next_sibling;          my $next_sibling;
1602          OE: for (reverse 0..$#{$self->{open_elements}}) {          OE: for (reverse 0..$#{$self->{open_elements}}) {
1603            if ($self->{open_elements}->[$_]->[1] & TABLE_EL) {            if ($self->{open_elements}->[$_]->[1] == TABLE_EL) {
1604                               my $parent = $self->{open_elements}->[$_]->[0]->parent_node;                               my $parent = $self->{open_elements}->[$_]->[0]->parent_node;
1605                               if (defined $parent and $parent->node_type == 1) {                               if (defined $parent and $parent->node_type == 1) {
1606                                 !!!cp ('t65.1');                                 !!!cp ('t65.1');
# Line 4317  sub _tree_construction_main ($) { Line 1660  sub _tree_construction_main ($) {
1660            $i = $_;            $i = $_;
1661          }          }
1662        } # OE        } # OE
1663        splice @{$self->{open_elements}}, $i + 1, 1, $clone;        splice @{$self->{open_elements}}, $i + 1, 0, $clone;
1664                
1665        ## Step 14        ## Step 14
1666        redo FET;        redo FET;
# Line 4335  sub _tree_construction_main ($) { Line 1678  sub _tree_construction_main ($) {
1678        my $foster_parent_element;        my $foster_parent_element;
1679        my $next_sibling;        my $next_sibling;
1680        OE: for (reverse 0..$#{$self->{open_elements}}) {        OE: for (reverse 0..$#{$self->{open_elements}}) {
1681          if ($self->{open_elements}->[$_]->[1] & TABLE_EL) {          if ($self->{open_elements}->[$_]->[1] == TABLE_EL) {
1682                               my $parent = $self->{open_elements}->[$_]->[0]->parent_node;                               my $parent = $self->{open_elements}->[$_]->[0]->parent_node;
1683                               if (defined $parent and $parent->node_type == 1) {                               if (defined $parent and $parent->node_type == 1) {
1684                                 !!!cp ('t70');                                 !!!cp ('t70');
# Line 4360  sub _tree_construction_main ($) { Line 1703  sub _tree_construction_main ($) {
1703      }      }
1704    }; # $insert_to_foster    }; # $insert_to_foster
1705    
1706      ## NOTE: Insert a character (MUST): When a character is inserted, if
1707      ## the last node that was inserted by the parser is a Text node and
1708      ## the character has to be inserted after that node, then the
1709      ## character is appended to the Text node.  However, if any other
1710      ## node is inserted by the parser, then a new Text node is created
1711      ## and the character is appended as that Text node.  If I'm not
1712      ## wrong, for a parser with scripting disabled, there are only two
1713      ## cases where this occurs.  One is the case where an element node
1714      ## is inserted to the |head| element.  This is covered by using the
1715      ## |$self->{head_element_inserted}| flag.  Another is the case where
1716      ## an element or comment is inserted into the |table| subtree while
1717      ## foster parenting happens.  This is covered by using the [2] flag
1718      ## of the |$open_tables| structure.  All other cases are handled
1719      ## simply by calling |manakai_append_text| method.
1720    
1721      ## TODO: |<body><script>document.write("a<br>");
1722      ## document.body.removeChild (document.body.lastChild);
1723      ## document.write ("b")</script>|
1724    
1725    B: while (1) {    B: while (1) {
1726      if ($token->{type} == DOCTYPE_TOKEN) {      if ($token->{type} == DOCTYPE_TOKEN) {
1727        !!!cp ('t73');        !!!cp ('t73');
# Line 4407  sub _tree_construction_main ($) { Line 1769  sub _tree_construction_main ($) {
1769        } else {        } else {
1770          !!!cp ('t87');          !!!cp ('t87');
1771          $self->{open_elements}->[-1]->[0]->append_child ($comment);          $self->{open_elements}->[-1]->[0]->append_child ($comment);
1772            $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
1773        }        }
1774        !!!next-token;        !!!next-token;
1775        next B;        next B;
1776        } elsif ($self->{insertion_mode} & IN_CDATA_RCDATA_IM) {
1777          if ($token->{type} == CHARACTER_TOKEN) {
1778            $token->{data} =~ s/^\x0A// if $self->{ignore_newline};
1779            delete $self->{ignore_newline};
1780    
1781            if (length $token->{data}) {
1782              !!!cp ('t43');
1783              $self->{open_elements}->[-1]->[0]->manakai_append_text
1784                  ($token->{data});
1785            } else {
1786              !!!cp ('t43.1');
1787            }
1788            !!!next-token;
1789            next B;
1790          } elsif ($token->{type} == END_TAG_TOKEN) {
1791            delete $self->{ignore_newline};
1792    
1793            if ($token->{tag_name} eq 'script') {
1794              !!!cp ('t50');
1795              
1796              ## Para 1-2
1797              my $script = pop @{$self->{open_elements}};
1798              
1799              ## Para 3
1800              $self->{insertion_mode} &= ~ IN_CDATA_RCDATA_IM;
1801    
1802              ## Para 4
1803              ## TODO: $old_insertion_point = $current_insertion_point;
1804              ## TODO: $current_insertion_point = just before $self->{nc};
1805    
1806              ## Para 5
1807              ## TODO: Run the $script->[0].
1808    
1809              ## Para 6
1810              ## TODO: $current_insertion_point = $old_insertion_point;
1811    
1812              ## Para 7
1813              ## TODO: if ($pending_external_script) {
1814                ## TODO: ...
1815              ## TODO: }
1816    
1817              !!!next-token;
1818              next B;
1819            } else {
1820              !!!cp ('t42');
1821    
1822              pop @{$self->{open_elements}};
1823    
1824              $self->{insertion_mode} &= ~ IN_CDATA_RCDATA_IM;
1825              !!!next-token;
1826              next B;
1827            }
1828          } elsif ($token->{type} == END_OF_FILE_TOKEN) {
1829            delete $self->{ignore_newline};
1830    
1831            !!!cp ('t44');
1832            !!!parse-error (type => 'not closed',
1833                            text => $self->{open_elements}->[-1]->[0]
1834                                ->manakai_local_name,
1835                            token => $token);
1836    
1837            #if ($self->{open_elements}->[-1]->[1] == SCRIPT_EL) {
1838            #  ## TODO: Mark as "already executed"
1839            #}
1840    
1841            pop @{$self->{open_elements}};
1842    
1843            $self->{insertion_mode} &= ~ IN_CDATA_RCDATA_IM;
1844            ## Reprocess.
1845            next B;
1846          } else {
1847            die "$0: $token->{type}: In CDATA/RCDATA: Unknown token type";        
1848          }
1849      } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM) {      } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM) {
1850        if ($token->{type} == CHARACTER_TOKEN) {        if ($token->{type} == CHARACTER_TOKEN) {
1851          !!!cp ('t87.1');          !!!cp ('t87.1');
# Line 4421  sub _tree_construction_main ($) { Line 1857  sub _tree_construction_main ($) {
1857               $self->{open_elements}->[-1]->[1] & FOREIGN_FLOW_CONTENT_EL) or               $self->{open_elements}->[-1]->[1] & FOREIGN_FLOW_CONTENT_EL) or
1858              not ($self->{open_elements}->[-1]->[1] & FOREIGN_EL) or              not ($self->{open_elements}->[-1]->[1] & FOREIGN_EL) or
1859              ($token->{tag_name} eq 'svg' and              ($token->{tag_name} eq 'svg' and
1860               $self->{open_elements}->[-1]->[1] & MML_AXML_EL)) {               $self->{open_elements}->[-1]->[1] == MML_AXML_EL)) {
1861            ## NOTE: "using the rules for secondary insertion mode"then"continue"            ## NOTE: "using the rules for secondary insertion mode"then"continue"
1862            !!!cp ('t87.2');            !!!cp ('t87.2');
1863            #            #
# Line 4522  sub _tree_construction_main ($) { Line 1958  sub _tree_construction_main ($) {
1958          pop @{$self->{open_elements}}          pop @{$self->{open_elements}}
1959              while $self->{open_elements}->[-1]->[1] & FOREIGN_EL;              while $self->{open_elements}->[-1]->[1] & FOREIGN_EL;
1960    
1961            ## NOTE: |<span><svg>| ... two parse errors, |<svg>| ... a parse error.
1962    
1963          $self->{insertion_mode} &= ~ IN_FOREIGN_CONTENT_IM;          $self->{insertion_mode} &= ~ IN_FOREIGN_CONTENT_IM;
1964          ## Reprocess.          ## Reprocess.
1965          next B;          next B;
# Line 4534  sub _tree_construction_main ($) { Line 1972  sub _tree_construction_main ($) {
1972        if ($token->{type} == CHARACTER_TOKEN) {        if ($token->{type} == CHARACTER_TOKEN) {
1973          if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {          if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
1974            unless ($self->{insertion_mode} == BEFORE_HEAD_IM) {            unless ($self->{insertion_mode} == BEFORE_HEAD_IM) {
1975              !!!cp ('t88.2');              if ($self->{head_element_inserted}) {
1976              $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);                !!!cp ('t88.3');
1977              #                $self->{open_elements}->[-1]->[0]->append_child
1978                    ($self->{document}->create_text_node ($1));
1979                  delete $self->{head_element_inserted};
1980                  ## NOTE: |</head> <link> |
1981                  #
1982                } else {
1983                  !!!cp ('t88.2');
1984                  $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
1985                  ## NOTE: |</head> &#x20;|
1986                  #
1987                }
1988            } else {            } else {
1989              !!!cp ('t88.1');              !!!cp ('t88.1');
1990              ## Ignore the token.              ## Ignore the token.
# Line 4632  sub _tree_construction_main ($) { Line 2080  sub _tree_construction_main ($) {
2080            !!!cp ('t97');            !!!cp ('t97');
2081          }          }
2082    
2083              if ($token->{tag_name} eq 'base') {          if ($token->{tag_name} eq 'base') {
2084                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {            if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
2085                  !!!cp ('t98');              !!!cp ('t98');
2086                  ## As if </noscript>              ## As if </noscript>
2087                  pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
2088                  !!!parse-error (type => 'in noscript', text => 'base',              !!!parse-error (type => 'in noscript', text => 'base',
2089                                  token => $token);                              token => $token);
2090                            
2091                  $self->{insertion_mode} = IN_HEAD_IM;              $self->{insertion_mode} = IN_HEAD_IM;
2092                  ## Reprocess in the "in head" insertion mode...              ## Reprocess in the "in head" insertion mode...
2093                } else {            } else {
2094                  !!!cp ('t99');              !!!cp ('t99');
2095                }            }
2096    
2097                ## NOTE: There is a "as if in head" code clone.            ## NOTE: There is a "as if in head" code clone.
2098                if ($self->{insertion_mode} == AFTER_HEAD_IM) {            if ($self->{insertion_mode} == AFTER_HEAD_IM) {
2099                  !!!cp ('t100');              !!!cp ('t100');
2100                  !!!parse-error (type => 'after head',              !!!parse-error (type => 'after head',
2101                                  text => $token->{tag_name}, token => $token);                              text => $token->{tag_name}, token => $token);
2102                  push @{$self->{open_elements}},              push @{$self->{open_elements}},
2103                      [$self->{head_element}, $el_category->{head}];                  [$self->{head_element}, $el_category->{head}];
2104                } else {              $self->{head_element_inserted} = 1;
2105                  !!!cp ('t101');            } else {
2106                }              !!!cp ('t101');
2107                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);            }
2108                pop @{$self->{open_elements}};            !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
2109                pop @{$self->{open_elements}} # <head>            pop @{$self->{open_elements}};
2110                    if $self->{insertion_mode} == AFTER_HEAD_IM;            pop @{$self->{open_elements}} # <head>
2111                !!!nack ('t101.1');                if $self->{insertion_mode} == AFTER_HEAD_IM;
2112                !!!next-token;            !!!nack ('t101.1');
2113                next B;            !!!next-token;
2114              next B;
2115          } elsif ($token->{tag_name} eq 'link') {          } elsif ($token->{tag_name} eq 'link') {
2116            ## NOTE: There is a "as if in head" code clone.            ## NOTE: There is a "as if in head" code clone.
2117            if ($self->{insertion_mode} == AFTER_HEAD_IM) {            if ($self->{insertion_mode} == AFTER_HEAD_IM) {
# Line 4671  sub _tree_construction_main ($) { Line 2120  sub _tree_construction_main ($) {
2120                              text => $token->{tag_name}, token => $token);                              text => $token->{tag_name}, token => $token);
2121              push @{$self->{open_elements}},              push @{$self->{open_elements}},
2122                  [$self->{head_element}, $el_category->{head}];                  [$self->{head_element}, $el_category->{head}];
2123                $self->{head_element_inserted} = 1;
2124            } else {            } else {
2125              !!!cp ('t103');              !!!cp ('t103');
2126            }            }
# Line 4703  sub _tree_construction_main ($) { Line 2153  sub _tree_construction_main ($) {
2153              !!!cp ('t103.3');              !!!cp ('t103.3');
2154              #              #
2155            }            }
2156              } elsif ($token->{tag_name} eq 'meta') {          } elsif ($token->{tag_name} eq 'meta') {
2157                ## NOTE: There is a "as if in head" code clone.            ## NOTE: There is a "as if in head" code clone.
2158                if ($self->{insertion_mode} == AFTER_HEAD_IM) {            if ($self->{insertion_mode} == AFTER_HEAD_IM) {
2159                  !!!cp ('t104');              !!!cp ('t104');
2160                  !!!parse-error (type => 'after head',              !!!parse-error (type => 'after head',
2161                                  text => $token->{tag_name}, token => $token);                              text => $token->{tag_name}, token => $token);
2162                  push @{$self->{open_elements}},              push @{$self->{open_elements}},
2163                      [$self->{head_element}, $el_category->{head}];                  [$self->{head_element}, $el_category->{head}];
2164                } else {              $self->{head_element_inserted} = 1;
2165                  !!!cp ('t105');            } else {
2166                }              !!!cp ('t105');
2167                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);            }
2168                my $meta_el = pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.            !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
2169              my $meta_el = pop @{$self->{open_elements}};
2170    
2171                unless ($self->{confident}) {                unless ($self->{confident}) {
2172                  if ($token->{attributes}->{charset}) {                  if ($token->{attributes}->{charset}) {
# Line 4773  sub _tree_construction_main ($) { Line 2224  sub _tree_construction_main ($) {
2224                !!!ack ('t110.1');                !!!ack ('t110.1');
2225                !!!next-token;                !!!next-token;
2226                next B;                next B;
2227              } elsif ($token->{tag_name} eq 'title') {          } elsif ($token->{tag_name} eq 'title') {
2228                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {            if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
2229                  !!!cp ('t111');              !!!cp ('t111');
2230                  ## As if </noscript>              ## As if </noscript>
2231                  pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
2232                  !!!parse-error (type => 'in noscript', text => 'title',              !!!parse-error (type => 'in noscript', text => 'title',
2233                                  token => $token);                              token => $token);
2234                            
2235                  $self->{insertion_mode} = IN_HEAD_IM;              $self->{insertion_mode} = IN_HEAD_IM;
2236                  ## Reprocess in the "in head" insertion mode...              ## Reprocess in the "in head" insertion mode...
2237                } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {            } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
2238                  !!!cp ('t112');              !!!cp ('t112');
2239                  !!!parse-error (type => 'after head',              !!!parse-error (type => 'after head',
2240                                  text => $token->{tag_name}, token => $token);                              text => $token->{tag_name}, token => $token);
2241                  push @{$self->{open_elements}},              push @{$self->{open_elements}},
2242                      [$self->{head_element}, $el_category->{head}];                  [$self->{head_element}, $el_category->{head}];
2243                } else {              $self->{head_element_inserted} = 1;
2244                  !!!cp ('t113');            } else {
2245                }              !!!cp ('t113');
2246              }
2247    
2248                ## NOTE: There is a "as if in head" code clone.            ## NOTE: There is a "as if in head" code clone.
2249                my $parent = defined $self->{head_element} ? $self->{head_element}            $parse_rcdata->(RCDATA_CONTENT_MODEL);
2250                    : $self->{open_elements}->[-1]->[0];            ## ISSUE: A spec bug [Bug 6038]
2251                $parse_rcdata->(RCDATA_CONTENT_MODEL);            splice @{$self->{open_elements}}, -2, 1, () # <head>
2252                pop @{$self->{open_elements}} # <head>                if ($self->{insertion_mode} & IM_MASK) == AFTER_HEAD_IM;
2253                    if $self->{insertion_mode} == AFTER_HEAD_IM;            next B;
2254                next B;          } elsif ($token->{tag_name} eq 'style' or
2255              } elsif ($token->{tag_name} eq 'style' or                   $token->{tag_name} eq 'noframes') {
2256                       $token->{tag_name} eq 'noframes') {            ## NOTE: Or (scripting is enabled and tag_name eq 'noscript' and
2257                ## NOTE: Or (scripting is enabled and tag_name eq 'noscript' and            ## insertion mode IN_HEAD_IM)
2258                ## insertion mode IN_HEAD_IM)            ## NOTE: There is a "as if in head" code clone.
2259                ## NOTE: There is a "as if in head" code clone.            if ($self->{insertion_mode} == AFTER_HEAD_IM) {
2260                if ($self->{insertion_mode} == AFTER_HEAD_IM) {              !!!cp ('t114');
2261                  !!!cp ('t114');              !!!parse-error (type => 'after head',
2262                  !!!parse-error (type => 'after head',                              text => $token->{tag_name}, token => $token);
2263                                  text => $token->{tag_name}, token => $token);              push @{$self->{open_elements}},
2264                  push @{$self->{open_elements}},                  [$self->{head_element}, $el_category->{head}];
2265                      [$self->{head_element}, $el_category->{head}];              $self->{head_element_inserted} = 1;
2266                } else {            } else {
2267                  !!!cp ('t115');              !!!cp ('t115');
2268                }            }
2269                $parse_rcdata->(CDATA_CONTENT_MODEL);            $parse_rcdata->(CDATA_CONTENT_MODEL);
2270                pop @{$self->{open_elements}} # <head>            ## ISSUE: A spec bug [Bug 6038]
2271                    if $self->{insertion_mode} == AFTER_HEAD_IM;            splice @{$self->{open_elements}}, -2, 1, () # <head>
2272                next B;                if ($self->{insertion_mode} & IM_MASK) == AFTER_HEAD_IM;
2273              } elsif ($token->{tag_name} eq 'noscript') {            next B;
2274            } elsif ($token->{tag_name} eq 'noscript') {
2275                if ($self->{insertion_mode} == IN_HEAD_IM) {                if ($self->{insertion_mode} == IN_HEAD_IM) {
2276                  !!!cp ('t116');                  !!!cp ('t116');
2277                  ## NOTE: and scripting is disalbed                  ## NOTE: and scripting is disalbed
# Line 4839  sub _tree_construction_main ($) { Line 2292  sub _tree_construction_main ($) {
2292                  !!!cp ('t118');                  !!!cp ('t118');
2293                  #                  #
2294                }                }
2295              } elsif ($token->{tag_name} eq 'script') {          } elsif ($token->{tag_name} eq 'script') {
2296                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {            if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
2297                  !!!cp ('t119');              !!!cp ('t119');
2298                  ## As if </noscript>              ## As if </noscript>
2299                  pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
2300                  !!!parse-error (type => 'in noscript', text => 'script',              !!!parse-error (type => 'in noscript', text => 'script',
2301                                  token => $token);                              token => $token);
2302                            
2303                  $self->{insertion_mode} = IN_HEAD_IM;              $self->{insertion_mode} = IN_HEAD_IM;
2304                  ## Reprocess in the "in head" insertion mode...              ## Reprocess in the "in head" insertion mode...
2305                } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {            } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
2306                  !!!cp ('t120');              !!!cp ('t120');
2307                  !!!parse-error (type => 'after head',              !!!parse-error (type => 'after head',
2308                                  text => $token->{tag_name}, token => $token);                              text => $token->{tag_name}, token => $token);
2309                  push @{$self->{open_elements}},              push @{$self->{open_elements}},
2310                      [$self->{head_element}, $el_category->{head}];                  [$self->{head_element}, $el_category->{head}];
2311                } else {              $self->{head_element_inserted} = 1;
2312                  !!!cp ('t121');            } else {
2313                }              !!!cp ('t121');
2314              }
2315    
2316                ## NOTE: There is a "as if in head" code clone.            ## NOTE: There is a "as if in head" code clone.
2317                $script_start_tag->();            $script_start_tag->();
2318                pop @{$self->{open_elements}} # <head>            ## ISSUE: A spec bug  [Bug 6038]
2319                    if $self->{insertion_mode} == AFTER_HEAD_IM;            splice @{$self->{open_elements}}, -2, 1 # <head>
2320                next B;                if ($self->{insertion_mode} & IM_MASK) == AFTER_HEAD_IM;
2321              } elsif ($token->{tag_name} eq 'body' or            next B;
2322                       $token->{tag_name} eq 'frameset') {          } elsif ($token->{tag_name} eq 'body' or
2323                     $token->{tag_name} eq 'frameset') {
2324                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
2325                  !!!cp ('t122');                  !!!cp ('t122');
2326                  ## As if </noscript>                  ## As if </noscript>
# Line 5000  sub _tree_construction_main ($) { Line 2455  sub _tree_construction_main ($) {
2455              } elsif ({              } elsif ({
2456                        body => 1, html => 1,                        body => 1, html => 1,
2457                       }->{$token->{tag_name}}) {                       }->{$token->{tag_name}}) {
2458                if ($self->{insertion_mode} == BEFORE_HEAD_IM or                ## TODO: This branch is entirely redundant.
2459                  if ($self->{insertion_mode} == BEFORE_HEAD_IM or
2460                    $self->{insertion_mode} == IN_HEAD_IM or                    $self->{insertion_mode} == IN_HEAD_IM or
2461                    $self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {                    $self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
2462                  !!!cp ('t140');                  !!!cp ('t140');
# Line 5172  sub _tree_construction_main ($) { Line 2628  sub _tree_construction_main ($) {
2628        } else {        } else {
2629          die "$0: $token->{type}: Unknown token type";          die "$0: $token->{type}: Unknown token type";
2630        }        }
   
           ## ISSUE: An issue in the spec.  
2631      } elsif ($self->{insertion_mode} & BODY_IMS) {      } elsif ($self->{insertion_mode} & BODY_IMS) {
2632            if ($token->{type} == CHARACTER_TOKEN) {            if ($token->{type} == CHARACTER_TOKEN) {
2633              !!!cp ('t150');              !!!cp ('t150');
# Line 5189  sub _tree_construction_main ($) { Line 2643  sub _tree_construction_main ($) {
2643                   caption => 1, col => 1, colgroup => 1, tbody => 1,                   caption => 1, col => 1, colgroup => 1, tbody => 1,
2644                   td => 1, tfoot => 1, th => 1, thead => 1, tr => 1,                   td => 1, tfoot => 1, th => 1, thead => 1, tr => 1,
2645                  }->{$token->{tag_name}}) {                  }->{$token->{tag_name}}) {
2646                if ($self->{insertion_mode} == IN_CELL_IM) {                if (($self->{insertion_mode} & IM_MASK) == IN_CELL_IM) {
2647                  ## have an element in table scope                  ## have an element in table scope
2648                  for (reverse 0..$#{$self->{open_elements}}) {                  for (reverse 0..$#{$self->{open_elements}}) {
2649                    my $node = $self->{open_elements}->[$_];                    my $node = $self->{open_elements}->[$_];
2650                    if ($node->[1] & TABLE_CELL_EL) {                    if ($node->[1] == TABLE_CELL_EL) {
2651                      !!!cp ('t151');                      !!!cp ('t151');
2652    
2653                      ## Close the cell                      ## Close the cell
# Line 5217  sub _tree_construction_main ($) { Line 2671  sub _tree_construction_main ($) {
2671                  !!!nack ('t153.1');                  !!!nack ('t153.1');
2672                  !!!next-token;                  !!!next-token;
2673                  next B;                  next B;
2674                } elsif ($self->{insertion_mode} == IN_CAPTION_IM) {                } elsif (($self->{insertion_mode} & IM_MASK) == IN_CAPTION_IM) {
2675                  !!!parse-error (type => 'not closed', text => 'caption',                  !!!parse-error (type => 'not closed', text => 'caption',
2676                                  token => $token);                                  token => $token);
2677                                    
# Line 5227  sub _tree_construction_main ($) { Line 2681  sub _tree_construction_main ($) {
2681                  INSCOPE: {                  INSCOPE: {
2682                    for (reverse 0..$#{$self->{open_elements}}) {                    for (reverse 0..$#{$self->{open_elements}}) {
2683                      my $node = $self->{open_elements}->[$_];                      my $node = $self->{open_elements}->[$_];
2684                      if ($node->[1] & CAPTION_EL) {                      if ($node->[1] == CAPTION_EL) {
2685                        !!!cp ('t155');                        !!!cp ('t155');
2686                        $i = $_;                        $i = $_;
2687                        last INSCOPE;                        last INSCOPE;
# Line 5253  sub _tree_construction_main ($) { Line 2707  sub _tree_construction_main ($) {
2707                    pop @{$self->{open_elements}};                    pop @{$self->{open_elements}};
2708                  }                  }
2709    
2710                  unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {                  unless ($self->{open_elements}->[-1]->[1] == CAPTION_EL) {
2711                    !!!cp ('t159');                    !!!cp ('t159');
2712                    !!!parse-error (type => 'not closed',                    !!!parse-error (type => 'not closed',
2713                                    text => $self->{open_elements}->[-1]->[0]                                    text => $self->{open_elements}->[-1]->[0]
# Line 5282  sub _tree_construction_main ($) { Line 2736  sub _tree_construction_main ($) {
2736              }              }
2737            } elsif ($token->{type} == END_TAG_TOKEN) {            } elsif ($token->{type} == END_TAG_TOKEN) {
2738              if ($token->{tag_name} eq 'td' or $token->{tag_name} eq 'th') {              if ($token->{tag_name} eq 'td' or $token->{tag_name} eq 'th') {
2739                if ($self->{insertion_mode} == IN_CELL_IM) {                if (($self->{insertion_mode} & IM_MASK) == IN_CELL_IM) {
2740                  ## have an element in table scope                  ## have an element in table scope
2741                  my $i;                  my $i;
2742                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
# Line 5332  sub _tree_construction_main ($) { Line 2786  sub _tree_construction_main ($) {
2786                                    
2787                  !!!next-token;                  !!!next-token;
2788                  next B;                  next B;
2789                } elsif ($self->{insertion_mode} == IN_CAPTION_IM) {                } elsif (($self->{insertion_mode} & IM_MASK) == IN_CAPTION_IM) {
2790                  !!!cp ('t169');                  !!!cp ('t169');
2791                  !!!parse-error (type => 'unmatched end tag',                  !!!parse-error (type => 'unmatched end tag',
2792                                  text => $token->{tag_name}, token => $token);                                  text => $token->{tag_name}, token => $token);
# Line 5344  sub _tree_construction_main ($) { Line 2798  sub _tree_construction_main ($) {
2798                  #                  #
2799                }                }
2800              } elsif ($token->{tag_name} eq 'caption') {              } elsif ($token->{tag_name} eq 'caption') {
2801                if ($self->{insertion_mode} == IN_CAPTION_IM) {                if (($self->{insertion_mode} & IM_MASK) == IN_CAPTION_IM) {
2802                  ## have a table element in table scope                  ## have a table element in table scope
2803                  my $i;                  my $i;
2804                  INSCOPE: {                  INSCOPE: {
2805                    for (reverse 0..$#{$self->{open_elements}}) {                    for (reverse 0..$#{$self->{open_elements}}) {
2806                      my $node = $self->{open_elements}->[$_];                      my $node = $self->{open_elements}->[$_];
2807                      if ($node->[1] & CAPTION_EL) {                      if ($node->[1] == CAPTION_EL) {
2808                        !!!cp ('t171');                        !!!cp ('t171');
2809                        $i = $_;                        $i = $_;
2810                        last INSCOPE;                        last INSCOPE;
# Line 5375  sub _tree_construction_main ($) { Line 2829  sub _tree_construction_main ($) {
2829                    pop @{$self->{open_elements}};                    pop @{$self->{open_elements}};
2830                  }                  }
2831                                    
2832                  unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {                  unless ($self->{open_elements}->[-1]->[1] == CAPTION_EL) {
2833                    !!!cp ('t175');                    !!!cp ('t175');
2834                    !!!parse-error (type => 'not closed',                    !!!parse-error (type => 'not closed',
2835                                    text => $self->{open_elements}->[-1]->[0]                                    text => $self->{open_elements}->[-1]->[0]
# Line 5393  sub _tree_construction_main ($) { Line 2847  sub _tree_construction_main ($) {
2847                                    
2848                  !!!next-token;                  !!!next-token;
2849                  next B;                  next B;
2850                } elsif ($self->{insertion_mode} == IN_CELL_IM) {                } elsif (($self->{insertion_mode} & IM_MASK) == IN_CELL_IM) {
2851                  !!!cp ('t177');                  !!!cp ('t177');
2852                  !!!parse-error (type => 'unmatched end tag',                  !!!parse-error (type => 'unmatched end tag',
2853                                  text => $token->{tag_name}, token => $token);                                  text => $token->{tag_name}, token => $token);
# Line 5408  sub _tree_construction_main ($) { Line 2862  sub _tree_construction_main ($) {
2862                        table => 1, tbody => 1, tfoot => 1,                        table => 1, tbody => 1, tfoot => 1,
2863                        thead => 1, tr => 1,                        thead => 1, tr => 1,
2864                       }->{$token->{tag_name}} and                       }->{$token->{tag_name}} and
2865                       $self->{insertion_mode} == IN_CELL_IM) {                       ($self->{insertion_mode} & IM_MASK) == IN_CELL_IM) {
2866                ## have an element in table scope                ## have an element in table scope
2867                my $i;                my $i;
2868                my $tn;                my $tn;
# Line 5425  sub _tree_construction_main ($) { Line 2879  sub _tree_construction_main ($) {
2879                                line => $token->{line},                                line => $token->{line},
2880                                column => $token->{column}};                                column => $token->{column}};
2881                      next B;                      next B;
2882                    } elsif ($node->[1] & TABLE_CELL_EL) {                    } elsif ($node->[1] == TABLE_CELL_EL) {
2883                      !!!cp ('t180');                      !!!cp ('t180');
2884                      $tn = $node->[0]->manakai_local_name;                      $tn = $node->[0]->manakai_local_name;
2885                      ## NOTE: There is exactly one |td| or |th| element                      ## NOTE: There is exactly one |td| or |th| element
# Line 5445  sub _tree_construction_main ($) { Line 2899  sub _tree_construction_main ($) {
2899                  next B;                  next B;
2900                } # INSCOPE                } # INSCOPE
2901              } elsif ($token->{tag_name} eq 'table' and              } elsif ($token->{tag_name} eq 'table' and
2902                       $self->{insertion_mode} == IN_CAPTION_IM) {                       ($self->{insertion_mode} & IM_MASK) == IN_CAPTION_IM) {
2903                !!!parse-error (type => 'not closed', text => 'caption',                !!!parse-error (type => 'not closed', text => 'caption',
2904                                token => $token);                                token => $token);
2905    
# Line 5454  sub _tree_construction_main ($) { Line 2908  sub _tree_construction_main ($) {
2908                my $i;                my $i;
2909                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
2910                  my $node = $self->{open_elements}->[$_];                  my $node = $self->{open_elements}->[$_];
2911                  if ($node->[1] & CAPTION_EL) {                  if ($node->[1] == CAPTION_EL) {
2912                    !!!cp ('t184');                    !!!cp ('t184');
2913                    $i = $_;                    $i = $_;
2914                    last INSCOPE;                    last INSCOPE;
# Line 5465  sub _tree_construction_main ($) { Line 2919  sub _tree_construction_main ($) {
2919                } # INSCOPE                } # INSCOPE
2920                unless (defined $i) {                unless (defined $i) {
2921                  !!!cp ('t186');                  !!!cp ('t186');
2922            ## TODO: Wrong error type?
2923                  !!!parse-error (type => 'unmatched end tag',                  !!!parse-error (type => 'unmatched end tag',
2924                                  text => 'caption', token => $token);                                  text => 'caption', token => $token);
2925                  ## Ignore the token                  ## Ignore the token
# Line 5478  sub _tree_construction_main ($) { Line 2933  sub _tree_construction_main ($) {
2933                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
2934                }                }
2935    
2936                unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {                unless ($self->{open_elements}->[-1]->[1] == CAPTION_EL) {
2937                  !!!cp ('t188');                  !!!cp ('t188');
2938                  !!!parse-error (type => 'not closed',                  !!!parse-error (type => 'not closed',
2939                                  text => $self->{open_elements}->[-1]->[0]                                  text => $self->{open_elements}->[-1]->[0]
# Line 5510  sub _tree_construction_main ($) { Line 2965  sub _tree_construction_main ($) {
2965                  !!!cp ('t191');                  !!!cp ('t191');
2966                  #                  #
2967                }                }
2968              } elsif ({          } elsif ({
2969                        tbody => 1, tfoot => 1,                    tbody => 1, tfoot => 1,
2970                        thead => 1, tr => 1,                    thead => 1, tr => 1,
2971                       }->{$token->{tag_name}} and                   }->{$token->{tag_name}} and
2972                       $self->{insertion_mode} == IN_CAPTION_IM) {                   ($self->{insertion_mode} & IM_MASK) == IN_CAPTION_IM) {
2973                !!!cp ('t192');            !!!cp ('t192');
2974                !!!parse-error (type => 'unmatched end tag',            !!!parse-error (type => 'unmatched end tag',
2975                                text => $token->{tag_name}, token => $token);                            text => $token->{tag_name}, token => $token);
2976                ## Ignore the token            ## Ignore the token
2977                !!!next-token;            !!!next-token;
2978                next B;            next B;
2979              } else {          } else {
2980                !!!cp ('t193');            !!!cp ('t193');
2981                #            #
2982              }          }
2983        } elsif ($token->{type} == END_OF_FILE_TOKEN) {        } elsif ($token->{type} == END_OF_FILE_TOKEN) {
2984          for my $entry (@{$self->{open_elements}}) {          for my $entry (@{$self->{open_elements}}) {
2985            unless ($entry->[1] & ALL_END_TAG_OPTIONAL_EL) {            unless ($entry->[1] & ALL_END_TAG_OPTIONAL_EL) {
# Line 5559  sub _tree_construction_main ($) { Line 3014  sub _tree_construction_main ($) {
3014    
3015          !!!parse-error (type => 'in table:#text', token => $token);          !!!parse-error (type => 'in table:#text', token => $token);
3016    
3017              ## As if in body, but insert into foster parent element          ## NOTE: As if in body, but insert into the foster parent element.
3018              ## ISSUE: Spec says that "whenever a node would be inserted          $reconstruct_active_formatting_elements->($insert_to_foster);
             ## into the current node" while characters might not be  
             ## result in a new Text node.  
             $reconstruct_active_formatting_elements->($insert_to_foster);  
3019                            
3020              if ($self->{open_elements}->[-1]->[1] & TABLE_ROWS_EL) {          if ($self->{open_elements}->[-1]->[1] & TABLE_ROWS_EL) {
3021                # MUST            # MUST
3022                my $foster_parent_element;            my $foster_parent_element;
3023                my $next_sibling;            my $next_sibling;
3024                my $prev_sibling;            my $prev_sibling;
3025                OE: for (reverse 0..$#{$self->{open_elements}}) {            OE: for (reverse 0..$#{$self->{open_elements}}) {
3026                  if ($self->{open_elements}->[$_]->[1] & TABLE_EL) {              if ($self->{open_elements}->[$_]->[1] == TABLE_EL) {
3027                    my $parent = $self->{open_elements}->[$_]->[0]->parent_node;                my $parent = $self->{open_elements}->[$_]->[0]->parent_node;
3028                    if (defined $parent and $parent->node_type == 1) {                if (defined $parent and $parent->node_type == 1) {
3029                      !!!cp ('t196');                  $foster_parent_element = $parent;
3030                      $foster_parent_element = $parent;                  !!!cp ('t196');
3031                      $next_sibling = $self->{open_elements}->[$_]->[0];                  $next_sibling = $self->{open_elements}->[$_]->[0];
3032                      $prev_sibling = $next_sibling->previous_sibling;                  $prev_sibling = $next_sibling->previous_sibling;
3033                    } else {                  #
                     !!!cp ('t197');  
                     $foster_parent_element = $self->{open_elements}->[$_ - 1]->[0];  
                     $prev_sibling = $foster_parent_element->last_child;  
                   }  
                   last OE;  
                 }  
               } # OE  
               $foster_parent_element = $self->{open_elements}->[0]->[0] and  
               $prev_sibling = $foster_parent_element->last_child  
                 unless defined $foster_parent_element;  
               if (defined $prev_sibling and  
                   $prev_sibling->node_type == 3) {  
                 !!!cp ('t198');  
                 $prev_sibling->manakai_append_text ($token->{data});  
3034                } else {                } else {
3035                  !!!cp ('t199');                  !!!cp ('t197');
3036                  $foster_parent_element->insert_before                  $foster_parent_element = $self->{open_elements}->[$_ - 1]->[0];
3037                    ($self->{document}->create_text_node ($token->{data}),                  $prev_sibling = $foster_parent_element->last_child;
3038                     $next_sibling);                  #
3039                }                }
3040                  last OE;
3041                }
3042              } # OE
3043              $foster_parent_element = $self->{open_elements}->[0]->[0] and
3044              $prev_sibling = $foster_parent_element->last_child
3045                  unless defined $foster_parent_element;
3046              undef $prev_sibling unless $open_tables->[-1]->[2]; # ~node inserted
3047              if (defined $prev_sibling and
3048                  $prev_sibling->node_type == 3) {
3049                !!!cp ('t198');
3050                $prev_sibling->manakai_append_text ($token->{data});
3051              } else {
3052                !!!cp ('t199');
3053                $foster_parent_element->insert_before
3054                    ($self->{document}->create_text_node ($token->{data}),
3055                     $next_sibling);
3056              }
3057            $open_tables->[-1]->[1] = 1; # tainted            $open_tables->[-1]->[1] = 1; # tainted
3058              $open_tables->[-1]->[2] = 1; # ~node inserted
3059          } else {          } else {
3060              ## NOTE: Fragment case or in a foster parent'ed element
3061              ## (e.g. |<table><span>a|).  In fragment case, whether the
3062              ## character is appended to existing node or a new node is
3063              ## created is irrelevant, since the foster parent'ed nodes
3064              ## are discarded and fragment parsing does not invoke any
3065              ## script.
3066            !!!cp ('t200');            !!!cp ('t200');
3067            $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});            $self->{open_elements}->[-1]->[0]->manakai_append_text
3068                  ($token->{data});
3069          }          }
3070                            
3071          !!!next-token;          !!!next-token;
3072          next B;          next B;
3073        } elsif ($token->{type} == START_TAG_TOKEN) {        } elsif ($token->{type} == START_TAG_TOKEN) {
3074          if ({          if ({
3075               tr => ($self->{insertion_mode} != IN_ROW_IM),               tr => (($self->{insertion_mode} & IM_MASK) != IN_ROW_IM),
3076               th => 1, td => 1,               th => 1, td => 1,
3077              }->{$token->{tag_name}}) {              }->{$token->{tag_name}}) {
3078            if ($self->{insertion_mode} == IN_TABLE_IM) {            if (($self->{insertion_mode} & IM_MASK) == IN_TABLE_IM) {
3079              ## Clear back to table context              ## Clear back to table context
3080              while (not ($self->{open_elements}->[-1]->[1]              while (not ($self->{open_elements}->[-1]->[1]
3081                              & TABLE_SCOPING_EL)) {                              & TABLE_SCOPING_EL)) {
# Line 5625  sub _tree_construction_main ($) { Line 3088  sub _tree_construction_main ($) {
3088              ## reprocess in the "in table body" insertion mode...              ## reprocess in the "in table body" insertion mode...
3089            }            }
3090                        
3091            if ($self->{insertion_mode} == IN_TABLE_BODY_IM) {            if (($self->{insertion_mode} & IM_MASK) == IN_TABLE_BODY_IM) {
3092              unless ($token->{tag_name} eq 'tr') {              unless ($token->{tag_name} eq 'tr') {
3093                !!!cp ('t202');                !!!cp ('t202');
3094                !!!parse-error (type => 'missing start tag:tr', token => $token);                !!!parse-error (type => 'missing start tag:tr', token => $token);
# Line 5639  sub _tree_construction_main ($) { Line 3102  sub _tree_construction_main ($) {
3102                pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
3103              }              }
3104                                    
3105                  $self->{insertion_mode} = IN_ROW_IM;              $self->{insertion_mode} = IN_ROW_IM;
3106                  if ($token->{tag_name} eq 'tr') {              if ($token->{tag_name} eq 'tr') {
3107                    !!!cp ('t204');                !!!cp ('t204');
3108                    !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
3109                    !!!nack ('t204');                $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
3110                    !!!next-token;                !!!nack ('t204');
3111                    next B;                !!!next-token;
3112                  } else {                next B;
3113                    !!!cp ('t205');              } else {
3114                    !!!insert-element ('tr',, $token);                !!!cp ('t205');
3115                    ## reprocess in the "in row" insertion mode                !!!insert-element ('tr',, $token);
3116                  }                ## reprocess in the "in row" insertion mode
3117                } else {              }
3118                  !!!cp ('t206');            } else {
3119                }              !!!cp ('t206');
3120              }
3121    
3122                ## Clear back to table row context                ## Clear back to table row context
3123                while (not ($self->{open_elements}->[-1]->[1]                while (not ($self->{open_elements}->[-1]->[1]
# Line 5662  sub _tree_construction_main ($) { Line 3126  sub _tree_construction_main ($) {
3126                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
3127                }                }
3128                                
3129                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);            !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
3130                $self->{insertion_mode} = IN_CELL_IM;            $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
3131              $self->{insertion_mode} = IN_CELL_IM;
3132    
3133                push @$active_formatting_elements, ['#marker', ''];            push @$active_formatting_elements, ['#marker', ''];
3134                                
3135                !!!nack ('t207.1');            !!!nack ('t207.1');
3136              !!!next-token;
3137              next B;
3138            } elsif ({
3139                      caption => 1, col => 1, colgroup => 1,
3140                      tbody => 1, tfoot => 1, thead => 1,
3141                      tr => 1, # $self->{insertion_mode} == IN_ROW_IM
3142                     }->{$token->{tag_name}}) {
3143              if (($self->{insertion_mode} & IM_MASK) == IN_ROW_IM) {
3144                ## As if </tr>
3145                ## have an element in table scope
3146                my $i;
3147                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3148                  my $node = $self->{open_elements}->[$_];
3149                  if ($node->[1] == TABLE_ROW_EL) {
3150                    !!!cp ('t208');
3151                    $i = $_;
3152                    last INSCOPE;
3153                  } elsif ($node->[1] & TABLE_SCOPING_EL) {
3154                    !!!cp ('t209');
3155                    last INSCOPE;
3156                  }
3157                } # INSCOPE
3158                unless (defined $i) {
3159                  !!!cp ('t210');
3160                  ## TODO: This type is wrong.
3161                  !!!parse-error (type => 'unmacthed end tag',
3162                                  text => $token->{tag_name}, token => $token);
3163                  ## Ignore the token
3164                  !!!nack ('t210.1');
3165                !!!next-token;                !!!next-token;
3166                next B;                next B;
3167              } elsif ({              }
                       caption => 1, col => 1, colgroup => 1,  
                       tbody => 1, tfoot => 1, thead => 1,  
                       tr => 1, # $self->{insertion_mode} == IN_ROW_IM  
                      }->{$token->{tag_name}}) {  
               if ($self->{insertion_mode} == IN_ROW_IM) {  
                 ## As if </tr>  
                 ## have an element in table scope  
                 my $i;  
                 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
                   my $node = $self->{open_elements}->[$_];  
                   if ($node->[1] & TABLE_ROW_EL) {  
                     !!!cp ('t208');  
                     $i = $_;  
                     last INSCOPE;  
                   } elsif ($node->[1] & TABLE_SCOPING_EL) {  
                     !!!cp ('t209');  
                     last INSCOPE;  
                   }  
                 } # INSCOPE  
                 unless (defined $i) {  
                   !!!cp ('t210');  
 ## TODO: This type is wrong.  
                   !!!parse-error (type => 'unmacthed end tag',  
                                   text => $token->{tag_name}, token => $token);  
                   ## Ignore the token  
                   !!!nack ('t210.1');  
                   !!!next-token;  
                   next B;  
                 }  
3168                                    
3169                  ## Clear back to table row context                  ## Clear back to table row context
3170                  while (not ($self->{open_elements}->[-1]->[1]                  while (not ($self->{open_elements}->[-1]->[1]
# Line 5722  sub _tree_construction_main ($) { Line 3187  sub _tree_construction_main ($) {
3187                  }                  }
3188                }                }
3189    
3190                if ($self->{insertion_mode} == IN_TABLE_BODY_IM) {                if (($self->{insertion_mode} & IM_MASK) == IN_TABLE_BODY_IM) {
3191                  ## have an element in table scope                  ## have an element in table scope
3192                  my $i;                  my $i;
3193                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3194                    my $node = $self->{open_elements}->[$_];                    my $node = $self->{open_elements}->[$_];
3195                    if ($node->[1] & TABLE_ROW_GROUP_EL) {                    if ($node->[1] == TABLE_ROW_GROUP_EL) {
3196                      !!!cp ('t214');                      !!!cp ('t214');
3197                      $i = $_;                      $i = $_;
3198                      last INSCOPE;                      last INSCOPE;
# Line 5769  sub _tree_construction_main ($) { Line 3234  sub _tree_construction_main ($) {
3234                  !!!cp ('t218');                  !!!cp ('t218');
3235                }                }
3236    
3237                if ($token->{tag_name} eq 'col') {            if ($token->{tag_name} eq 'col') {
3238                  ## Clear back to table context              ## Clear back to table context
3239                  while (not ($self->{open_elements}->[-1]->[1]              while (not ($self->{open_elements}->[-1]->[1]
3240                                  & TABLE_SCOPING_EL)) {                              & TABLE_SCOPING_EL)) {
3241                    !!!cp ('t219');                !!!cp ('t219');
3242                    ## ISSUE: Can this state be reached?                ## ISSUE: Can this state be reached?
3243                    pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
3244                  }              }
3245                                
3246                  !!!insert-element ('colgroup',, $token);              !!!insert-element ('colgroup',, $token);
3247                  $self->{insertion_mode} = IN_COLUMN_GROUP_IM;              $self->{insertion_mode} = IN_COLUMN_GROUP_IM;
3248                  ## reprocess              ## reprocess
3249                  !!!ack-later;              $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
3250                  next B;              !!!ack-later;
3251                } elsif ({              next B;
3252                          caption => 1,            } elsif ({
3253                          colgroup => 1,                      caption => 1,
3254                          tbody => 1, tfoot => 1, thead => 1,                      colgroup => 1,
3255                         }->{$token->{tag_name}}) {                      tbody => 1, tfoot => 1, thead => 1,
3256                  ## Clear back to table context                     }->{$token->{tag_name}}) {
3257                ## Clear back to table context
3258                  while (not ($self->{open_elements}->[-1]->[1]                  while (not ($self->{open_elements}->[-1]->[1]
3259                                  & TABLE_SCOPING_EL)) {                                  & TABLE_SCOPING_EL)) {
3260                    !!!cp ('t220');                    !!!cp ('t220');
# Line 5796  sub _tree_construction_main ($) { Line 3262  sub _tree_construction_main ($) {
3262                    pop @{$self->{open_elements}};                    pop @{$self->{open_elements}};
3263                  }                  }
3264                                    
3265                  push @$active_formatting_elements, ['#marker', '']              push @$active_formatting_elements, ['#marker', '']
3266                      if $token->{tag_name} eq 'caption';                  if $token->{tag_name} eq 'caption';
3267                                    
3268                  !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);              !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
3269                  $self->{insertion_mode} = {              $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
3270                                             caption => IN_CAPTION_IM,              $self->{insertion_mode} = {
3271                                             colgroup => IN_COLUMN_GROUP_IM,                                         caption => IN_CAPTION_IM,
3272                                             tbody => IN_TABLE_BODY_IM,                                         colgroup => IN_COLUMN_GROUP_IM,
3273                                             tfoot => IN_TABLE_BODY_IM,                                         tbody => IN_TABLE_BODY_IM,
3274                                             thead => IN_TABLE_BODY_IM,                                         tfoot => IN_TABLE_BODY_IM,
3275                                            }->{$token->{tag_name}};                                         thead => IN_TABLE_BODY_IM,
3276                  !!!next-token;                                        }->{$token->{tag_name}};
3277                  !!!nack ('t220.1');              !!!next-token;
3278                  next B;              !!!nack ('t220.1');
3279                } else {              next B;
3280                  die "$0: in table: <>: $token->{tag_name}";            } else {
3281                }              die "$0: in table: <>: $token->{tag_name}";
3282              }
3283              } elsif ($token->{tag_name} eq 'table') {              } elsif ($token->{tag_name} eq 'table') {
3284                !!!parse-error (type => 'not closed',                !!!parse-error (type => 'not closed',
3285                                text => $self->{open_elements}->[-1]->[0]                                text => $self->{open_elements}->[-1]->[0]
# Line 5824  sub _tree_construction_main ($) { Line 3291  sub _tree_construction_main ($) {
3291                my $i;                my $i;
3292                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3293                  my $node = $self->{open_elements}->[$_];                  my $node = $self->{open_elements}->[$_];
3294                  if ($node->[1] & TABLE_EL) {                  if ($node->[1] == TABLE_EL) {
3295                    !!!cp ('t221');                    !!!cp ('t221');
3296                    $i = $_;                    $i = $_;
3297                    last INSCOPE;                    last INSCOPE;
# Line 5851  sub _tree_construction_main ($) { Line 3318  sub _tree_construction_main ($) {
3318                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
3319                }                }
3320    
3321                unless ($self->{open_elements}->[-1]->[1] & TABLE_EL) {                unless ($self->{open_elements}->[-1]->[1] == TABLE_EL) {
3322                  !!!cp ('t225');                  !!!cp ('t225');
3323                  ## NOTE: |<table><tr><table>|                  ## NOTE: |<table><tr><table>|
3324                  !!!parse-error (type => 'not closed',                  !!!parse-error (type => 'not closed',
# Line 5875  sub _tree_construction_main ($) { Line 3342  sub _tree_construction_main ($) {
3342              !!!cp ('t227.8');              !!!cp ('t227.8');
3343              ## NOTE: This is a "as if in head" code clone.              ## NOTE: This is a "as if in head" code clone.
3344              $parse_rcdata->(CDATA_CONTENT_MODEL);              $parse_rcdata->(CDATA_CONTENT_MODEL);
3345                $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
3346              next B;              next B;
3347            } else {            } else {
3348              !!!cp ('t227.7');              !!!cp ('t227.7');
# Line 5885  sub _tree_construction_main ($) { Line 3353  sub _tree_construction_main ($) {
3353              !!!cp ('t227.6');              !!!cp ('t227.6');
3354              ## NOTE: This is a "as if in head" code clone.              ## NOTE: This is a "as if in head" code clone.
3355              $script_start_tag->();              $script_start_tag->();
3356                $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
3357              next B;              next B;
3358            } else {            } else {
3359              !!!cp ('t227.5');              !!!cp ('t227.5');
# Line 5900  sub _tree_construction_main ($) { Line 3369  sub _tree_construction_main ($) {
3369                                  text => $token->{tag_name}, token => $token);                                  text => $token->{tag_name}, token => $token);
3370    
3371                  !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);                  !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
3372                    $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
3373    
3374                  ## TODO: form element pointer                  ## TODO: form element pointer
3375    
# Line 5931  sub _tree_construction_main ($) { Line 3401  sub _tree_construction_main ($) {
3401          $insert = $insert_to_foster;          $insert = $insert_to_foster;
3402          #          #
3403        } elsif ($token->{type} == END_TAG_TOKEN) {        } elsif ($token->{type} == END_TAG_TOKEN) {
3404              if ($token->{tag_name} eq 'tr' and          if ($token->{tag_name} eq 'tr' and
3405                  $self->{insertion_mode} == IN_ROW_IM) {              ($self->{insertion_mode} & IM_MASK) == IN_ROW_IM) {
3406                ## have an element in table scope            ## have an element in table scope
3407                my $i;                my $i;
3408                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3409                  my $node = $self->{open_elements}->[$_];                  my $node = $self->{open_elements}->[$_];
3410                  if ($node->[1] & TABLE_ROW_EL) {                  if ($node->[1] == TABLE_ROW_EL) {
3411                    !!!cp ('t228');                    !!!cp ('t228');
3412                    $i = $_;                    $i = $_;
3413                    last INSCOPE;                    last INSCOPE;
# Line 5972  sub _tree_construction_main ($) { Line 3442  sub _tree_construction_main ($) {
3442                !!!nack ('t231.1');                !!!nack ('t231.1');
3443                next B;                next B;
3444              } elsif ($token->{tag_name} eq 'table') {              } elsif ($token->{tag_name} eq 'table') {
3445                if ($self->{insertion_mode} == IN_ROW_IM) {                if (($self->{insertion_mode} & IM_MASK) == IN_ROW_IM) {
3446                  ## As if </tr>                  ## As if </tr>
3447                  ## have an element in table scope                  ## have an element in table scope
3448                  my $i;                  my $i;
3449                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3450                    my $node = $self->{open_elements}->[$_];                    my $node = $self->{open_elements}->[$_];
3451                    if ($node->[1] & TABLE_ROW_EL) {                    if ($node->[1] == TABLE_ROW_EL) {
3452                      !!!cp ('t233');                      !!!cp ('t233');
3453                      $i = $_;                      $i = $_;
3454                      last INSCOPE;                      last INSCOPE;
# Line 6011  sub _tree_construction_main ($) { Line 3481  sub _tree_construction_main ($) {
3481                  ## reprocess in the "in table body" insertion mode...                  ## reprocess in the "in table body" insertion mode...
3482                }                }
3483    
3484                if ($self->{insertion_mode} == IN_TABLE_BODY_IM) {                if (($self->{insertion_mode} & IM_MASK) == IN_TABLE_BODY_IM) {
3485                  ## have an element in table scope                  ## have an element in table scope
3486                  my $i;                  my $i;
3487                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3488                    my $node = $self->{open_elements}->[$_];                    my $node = $self->{open_elements}->[$_];
3489                    if ($node->[1] & TABLE_ROW_GROUP_EL) {                    if ($node->[1] == TABLE_ROW_GROUP_EL) {
3490                      !!!cp ('t237');                      !!!cp ('t237');
3491                      $i = $_;                      $i = $_;
3492                      last INSCOPE;                      last INSCOPE;
# Line 6063  sub _tree_construction_main ($) { Line 3533  sub _tree_construction_main ($) {
3533                my $i;                my $i;
3534                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3535                  my $node = $self->{open_elements}->[$_];                  my $node = $self->{open_elements}->[$_];
3536                  if ($node->[1] & TABLE_EL) {                  if ($node->[1] == TABLE_EL) {
3537                    !!!cp ('t241');                    !!!cp ('t241');
3538                    $i = $_;                    $i = $_;
3539                    last INSCOPE;                    last INSCOPE;
# Line 6093  sub _tree_construction_main ($) { Line 3563  sub _tree_construction_main ($) {
3563                        tbody => 1, tfoot => 1, thead => 1,                        tbody => 1, tfoot => 1, thead => 1,
3564                       }->{$token->{tag_name}} and                       }->{$token->{tag_name}} and
3565                       $self->{insertion_mode} & ROW_IMS) {                       $self->{insertion_mode} & ROW_IMS) {
3566                if ($self->{insertion_mode} == IN_ROW_IM) {                if (($self->{insertion_mode} & IM_MASK) == IN_ROW_IM) {
3567                  ## have an element in table scope                  ## have an element in table scope
3568                  my $i;                  my $i;
3569                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
# Line 6122  sub _tree_construction_main ($) { Line 3592  sub _tree_construction_main ($) {
3592                  my $i;                  my $i;
3593                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3594                    my $node = $self->{open_elements}->[$_];                    my $node = $self->{open_elements}->[$_];
3595                    if ($node->[1] & TABLE_ROW_EL) {                    if ($node->[1] == TABLE_ROW_EL) {
3596                      !!!cp ('t250');                      !!!cp ('t250');
3597                      $i = $_;                      $i = $_;
3598                      last INSCOPE;                      last INSCOPE;
# Line 6212  sub _tree_construction_main ($) { Line 3682  sub _tree_construction_main ($) {
3682            #            #
3683          }          }
3684        } elsif ($token->{type} == END_OF_FILE_TOKEN) {        } elsif ($token->{type} == END_OF_FILE_TOKEN) {
3685          unless ($self->{open_elements}->[-1]->[1] & HTML_EL and          unless ($self->{open_elements}->[-1]->[1] == HTML_EL and
3686                  @{$self->{open_elements}} == 1) { # redundant, maybe                  @{$self->{open_elements}} == 1) { # redundant, maybe
3687            !!!parse-error (type => 'in body:#eof', token => $token);            !!!parse-error (type => 'in body:#eof', token => $token);
3688            !!!cp ('t259.1');            !!!cp ('t259.1');
# Line 6227  sub _tree_construction_main ($) { Line 3697  sub _tree_construction_main ($) {
3697        } else {        } else {
3698          die "$0: $token->{type}: Unknown token type";          die "$0: $token->{type}: Unknown token type";
3699        }        }
3700      } elsif ($self->{insertion_mode} == IN_COLUMN_GROUP_IM) {      } elsif (($self->{insertion_mode} & IM_MASK) == IN_COLUMN_GROUP_IM) {
3701            if ($token->{type} == CHARACTER_TOKEN) {            if ($token->{type} == CHARACTER_TOKEN) {
3702              if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {              if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
3703                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
# Line 6254  sub _tree_construction_main ($) { Line 3724  sub _tree_construction_main ($) {
3724              }              }
3725            } elsif ($token->{type} == END_TAG_TOKEN) {            } elsif ($token->{type} == END_TAG_TOKEN) {
3726              if ($token->{tag_name} eq 'colgroup') {              if ($token->{tag_name} eq 'colgroup') {
3727                if ($self->{open_elements}->[-1]->[1] & HTML_EL) {                if ($self->{open_elements}->[-1]->[1] == HTML_EL) {
3728                  !!!cp ('t264');                  !!!cp ('t264');
3729                  !!!parse-error (type => 'unmatched end tag',                  !!!parse-error (type => 'unmatched end tag',
3730                                  text => 'colgroup', token => $token);                                  text => 'colgroup', token => $token);
# Line 6280  sub _tree_construction_main ($) { Line 3750  sub _tree_construction_main ($) {
3750                #                #
3751              }              }
3752        } elsif ($token->{type} == END_OF_FILE_TOKEN) {        } elsif ($token->{type} == END_OF_FILE_TOKEN) {
3753          if ($self->{open_elements}->[-1]->[1] & HTML_EL and          if ($self->{open_elements}->[-1]->[1] == HTML_EL and
3754              @{$self->{open_elements}} == 1) { # redundant, maybe              @{$self->{open_elements}} == 1) { # redundant, maybe
3755            !!!cp ('t270.2');            !!!cp ('t270.2');
3756            ## Stop parsing.            ## Stop parsing.
# Line 6298  sub _tree_construction_main ($) { Line 3768  sub _tree_construction_main ($) {
3768        }        }
3769    
3770            ## As if </colgroup>            ## As if </colgroup>
3771            if ($self->{open_elements}->[-1]->[1] & HTML_EL) {            if ($self->{open_elements}->[-1]->[1] == HTML_EL) {
3772              !!!cp ('t269');              !!!cp ('t269');
3773  ## TODO: Wrong error type?  ## TODO: Wrong error type?
3774              !!!parse-error (type => 'unmatched end tag',              !!!parse-error (type => 'unmatched end tag',
# Line 6323  sub _tree_construction_main ($) { Line 3793  sub _tree_construction_main ($) {
3793          next B;          next B;
3794        } elsif ($token->{type} == START_TAG_TOKEN) {        } elsif ($token->{type} == START_TAG_TOKEN) {
3795          if ($token->{tag_name} eq 'option') {          if ($token->{tag_name} eq 'option') {
3796            if ($self->{open_elements}->[-1]->[1] & OPTION_EL) {            if ($self->{open_elements}->[-1]->[1] == OPTION_EL) {
3797              !!!cp ('t272');              !!!cp ('t272');
3798              ## As if </option>              ## As if </option>
3799              pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
# Line 6336  sub _tree_construction_main ($) { Line 3806  sub _tree_construction_main ($) {
3806            !!!next-token;            !!!next-token;
3807            next B;            next B;
3808          } elsif ($token->{tag_name} eq 'optgroup') {          } elsif ($token->{tag_name} eq 'optgroup') {
3809            if ($self->{open_elements}->[-1]->[1] & OPTION_EL) {            if ($self->{open_elements}->[-1]->[1] == OPTION_EL) {
3810              !!!cp ('t274');              !!!cp ('t274');
3811              ## As if </option>              ## As if </option>
3812              pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
# Line 6344  sub _tree_construction_main ($) { Line 3814  sub _tree_construction_main ($) {
3814              !!!cp ('t275');              !!!cp ('t275');
3815            }            }
3816    
3817            if ($self->{open_elements}->[-1]->[1] & OPTGROUP_EL) {            if ($self->{open_elements}->[-1]->[1] == OPTGROUP_EL) {
3818              !!!cp ('t276');              !!!cp ('t276');
3819              ## As if </optgroup>              ## As if </optgroup>
3820              pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
# Line 6359  sub _tree_construction_main ($) { Line 3829  sub _tree_construction_main ($) {
3829          } elsif ({          } elsif ({
3830                     select => 1, input => 1, textarea => 1,                     select => 1, input => 1, textarea => 1,
3831                   }->{$token->{tag_name}} or                   }->{$token->{tag_name}} or
3832                   ($self->{insertion_mode} == IN_SELECT_IN_TABLE_IM and                   (($self->{insertion_mode} & IM_MASK)
3833                          == IN_SELECT_IN_TABLE_IM and
3834                    {                    {
3835                     caption => 1, table => 1,                     caption => 1, table => 1,
3836                     tbody => 1, tfoot => 1, thead => 1,                     tbody => 1, tfoot => 1, thead => 1,
# Line 6374  sub _tree_construction_main ($) { Line 3845  sub _tree_construction_main ($) {
3845            my $i;            my $i;
3846            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3847              my $node = $self->{open_elements}->[$_];              my $node = $self->{open_elements}->[$_];
3848              if ($node->[1] & SELECT_EL) {              if ($node->[1] == SELECT_EL) {
3849                !!!cp ('t278');                !!!cp ('t278');
3850                $i = $_;                $i = $_;
3851                last INSCOPE;                last INSCOPE;
# Line 6419  sub _tree_construction_main ($) { Line 3890  sub _tree_construction_main ($) {
3890          }          }
3891        } elsif ($token->{type} == END_TAG_TOKEN) {        } elsif ($token->{type} == END_TAG_TOKEN) {
3892          if ($token->{tag_name} eq 'optgroup') {          if ($token->{tag_name} eq 'optgroup') {
3893            if ($self->{open_elements}->[-1]->[1] & OPTION_EL and            if ($self->{open_elements}->[-1]->[1] == OPTION_EL and
3894                $self->{open_elements}->[-2]->[1] & OPTGROUP_EL) {                $self->{open_elements}->[-2]->[1] == OPTGROUP_EL) {
3895              !!!cp ('t283');              !!!cp ('t283');
3896              ## As if </option>              ## As if </option>
3897              splice @{$self->{open_elements}}, -2;              splice @{$self->{open_elements}}, -2;
3898            } elsif ($self->{open_elements}->[-1]->[1] & OPTGROUP_EL) {            } elsif ($self->{open_elements}->[-1]->[1] == OPTGROUP_EL) {
3899              !!!cp ('t284');              !!!cp ('t284');
3900              pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
3901            } else {            } else {
# Line 6437  sub _tree_construction_main ($) { Line 3908  sub _tree_construction_main ($) {
3908            !!!next-token;            !!!next-token;
3909            next B;            next B;
3910          } elsif ($token->{tag_name} eq 'option') {          } elsif ($token->{tag_name} eq 'option') {
3911            if ($self->{open_elements}->[-1]->[1] & OPTION_EL) {            if ($self->{open_elements}->[-1]->[1] == OPTION_EL) {
3912              !!!cp ('t286');              !!!cp ('t286');
3913              pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
3914            } else {            } else {
# Line 6454  sub _tree_construction_main ($) { Line 3925  sub _tree_construction_main ($) {
3925            my $i;            my $i;
3926            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3927              my $node = $self->{open_elements}->[$_];              my $node = $self->{open_elements}->[$_];
3928              if ($node->[1] & SELECT_EL) {              if ($node->[1] == SELECT_EL) {
3929                !!!cp ('t288');                !!!cp ('t288');
3930                $i = $_;                $i = $_;
3931                last INSCOPE;                last INSCOPE;
# Line 6481  sub _tree_construction_main ($) { Line 3952  sub _tree_construction_main ($) {
3952            !!!nack ('t291.1');            !!!nack ('t291.1');
3953            !!!next-token;            !!!next-token;
3954            next B;            next B;
3955          } elsif ($self->{insertion_mode} == IN_SELECT_IN_TABLE_IM and          } elsif (($self->{insertion_mode} & IM_MASK)
3956                         == IN_SELECT_IN_TABLE_IM and
3957                   {                   {
3958                    caption => 1, table => 1, tbody => 1,                    caption => 1, table => 1, tbody => 1,
3959                    tfoot => 1, thead => 1, tr => 1, td => 1, th => 1,                    tfoot => 1, thead => 1, tr => 1, td => 1, th => 1,
# Line 6516  sub _tree_construction_main ($) { Line 3988  sub _tree_construction_main ($) {
3988            undef $i;            undef $i;
3989            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3990              my $node = $self->{open_elements}->[$_];              my $node = $self->{open_elements}->[$_];
3991              if ($node->[1] & SELECT_EL) {              if ($node->[1] == SELECT_EL) {
3992                !!!cp ('t295');                !!!cp ('t295');
3993                $i = $_;                $i = $_;
3994                last INSCOPE;                last INSCOPE;
# Line 6555  sub _tree_construction_main ($) { Line 4027  sub _tree_construction_main ($) {
4027            next B;            next B;
4028          }          }
4029        } elsif ($token->{type} == END_OF_FILE_TOKEN) {        } elsif ($token->{type} == END_OF_FILE_TOKEN) {
4030          unless ($self->{open_elements}->[-1]->[1] & HTML_EL and          unless ($self->{open_elements}->[-1]->[1] == HTML_EL and
4031                  @{$self->{open_elements}} == 1) { # redundant, maybe                  @{$self->{open_elements}} == 1) { # redundant, maybe
4032            !!!cp ('t299.1');            !!!cp ('t299.1');
4033            !!!parse-error (type => 'in body:#eof', token => $token);            !!!parse-error (type => 'in body:#eof', token => $token);
# Line 6742  sub _tree_construction_main ($) { Line 4214  sub _tree_construction_main ($) {
4214        } elsif ($token->{type} == END_TAG_TOKEN) {        } elsif ($token->{type} == END_TAG_TOKEN) {
4215          if ($token->{tag_name} eq 'frameset' and          if ($token->{tag_name} eq 'frameset' and
4216              $self->{insertion_mode} == IN_FRAMESET_IM) {              $self->{insertion_mode} == IN_FRAMESET_IM) {
4217            if ($self->{open_elements}->[-1]->[1] & HTML_EL and            if ($self->{open_elements}->[-1]->[1] == HTML_EL and
4218                @{$self->{open_elements}} == 1) {                @{$self->{open_elements}} == 1) {
4219              !!!cp ('t325');              !!!cp ('t325');
4220              !!!parse-error (type => 'unmatched end tag',              !!!parse-error (type => 'unmatched end tag',
# Line 6756  sub _tree_construction_main ($) { Line 4228  sub _tree_construction_main ($) {
4228            }            }
4229    
4230            if (not defined $self->{inner_html_node} and            if (not defined $self->{inner_html_node} and
4231                not ($self->{open_elements}->[-1]->[1] & FRAMESET_EL)) {                not ($self->{open_elements}->[-1]->[1] == FRAMESET_EL)) {
4232              !!!cp ('t327');              !!!cp ('t327');
4233              $self->{insertion_mode} = AFTER_FRAMESET_IM;              $self->{insertion_mode} = AFTER_FRAMESET_IM;
4234            } else {            } else {
# Line 6788  sub _tree_construction_main ($) { Line 4260  sub _tree_construction_main ($) {
4260            next B;            next B;
4261          }          }
4262        } elsif ($token->{type} == END_OF_FILE_TOKEN) {        } elsif ($token->{type} == END_OF_FILE_TOKEN) {
4263          unless ($self->{open_elements}->[-1]->[1] & HTML_EL and          unless ($self->{open_elements}->[-1]->[1] == HTML_EL and
4264                  @{$self->{open_elements}} == 1) { # redundant, maybe                  @{$self->{open_elements}} == 1) { # redundant, maybe
4265            !!!cp ('t331.1');            !!!cp ('t331.1');
4266            !!!parse-error (type => 'in body:#eof', token => $token);            !!!parse-error (type => 'in body:#eof', token => $token);
# Line 6801  sub _tree_construction_main ($) { Line 4273  sub _tree_construction_main ($) {
4273        } else {        } else {
4274          die "$0: $token->{type}: Unknown token type";          die "$0: $token->{type}: Unknown token type";
4275        }        }
   
       ## ISSUE: An issue in spec here  
4276      } else {      } else {
4277        die "$0: $self->{insertion_mode}: Unknown insertion mode";        die "$0: $self->{insertion_mode}: Unknown insertion mode";
4278      }      }
# Line 6893  sub _tree_construction_main ($) { Line 4363  sub _tree_construction_main ($) {
4363          !!!parse-error (type => 'in body', text => 'body', token => $token);          !!!parse-error (type => 'in body', text => 'body', token => $token);
4364                                
4365          if (@{$self->{open_elements}} == 1 or          if (@{$self->{open_elements}} == 1 or
4366              not ($self->{open_elements}->[1]->[1] & BODY_EL)) {              not ($self->{open_elements}->[1]->[1] == BODY_EL)) {
4367            !!!cp ('t342');            !!!cp ('t342');
4368            ## Ignore the token            ## Ignore the token
4369          } else {          } else {
# Line 6911  sub _tree_construction_main ($) { Line 4381  sub _tree_construction_main ($) {
4381          !!!next-token;          !!!next-token;
4382          next B;          next B;
4383        } elsif ({        } elsif ({
4384                  address => 1, blockquote => 1, center => 1, dir => 1,                  ## NOTE: Start tags for non-phrasing flow content elements
4385                  div => 1, dl => 1, fieldset => 1,  
4386                  h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,                  ## NOTE: The normal one
4387                  menu => 1, ol => 1, p => 1, ul => 1,                  address => 1, article => 1, aside => 1, blockquote => 1,
4388                    center => 1, datagrid => 1, details => 1, dialog => 1,
4389                    dir => 1, div => 1, dl => 1, fieldset => 1, figure => 1,
4390                    footer => 1, h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1,
4391                    h6 => 1, header => 1, menu => 1, nav => 1, ol => 1, p => 1,
4392                    section => 1, ul => 1,
4393                    ## NOTE: As normal, but drops leading newline
4394                  pre => 1, listing => 1,                  pre => 1, listing => 1,
4395                    ## NOTE: As normal, but interacts with the form element pointer
4396                  form => 1,                  form => 1,
4397                    
4398                  table => 1,                  table => 1,
4399                  hr => 1,                  hr => 1,
4400                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
# Line 6931  sub _tree_construction_main ($) { Line 4409  sub _tree_construction_main ($) {
4409    
4410          ## has a p element in scope          ## has a p element in scope
4411          INSCOPE: for (reverse @{$self->{open_elements}}) {          INSCOPE: for (reverse @{$self->{open_elements}}) {
4412            if ($_->[1] & P_EL) {            if ($_->[1] == P_EL) {
4413              !!!cp ('t344');              !!!cp ('t344');
4414              !!!back-token; # <form>              !!!back-token; # <form>
4415              $token = {type => END_TAG_TOKEN, tag_name => 'p',              $token = {type => END_TAG_TOKEN, tag_name => 'p',
# Line 6983  sub _tree_construction_main ($) { Line 4461  sub _tree_construction_main ($) {
4461            !!!next-token;            !!!next-token;
4462          }          }
4463          next B;          next B;
4464        } elsif ({li => 1, dt => 1, dd => 1}->{$token->{tag_name}}) {        } elsif ($token->{tag_name} eq 'li') {
4465          ## has a p element in scope          ## NOTE: As normal, but imply </li> when there's another <li> ...
         INSCOPE: for (reverse @{$self->{open_elements}}) {  
           if ($_->[1] & P_EL) {  
             !!!cp ('t353');  
             !!!back-token; # <x>  
             $token = {type => END_TAG_TOKEN, tag_name => 'p',  
                       line => $token->{line}, column => $token->{column}};  
             next B;  
           } elsif ($_->[1] & SCOPING_EL) {  
             !!!cp ('t354');  
             last INSCOPE;  
           }  
         } # INSCOPE  
4466    
4467          ## NOTE: Special, Scope (<li><foo><li> == <li><foo><li/></foo></li>)          ## NOTE: Special, Scope (<li><foo><li> == <li><foo><li/></foo></li>)
4468            ## Interpreted as <li><foo/></li><li/> (non-conforming)            ## Interpreted as <li><foo/></li><li/> (non-conforming)
# Line 7010  sub _tree_construction_main ($) { Line 4476  sub _tree_construction_main ($) {
4476          ## address, div, p (<li><foo><li> == <li><foo/></li><li/>)          ## address, div, p (<li><foo><li> == <li><foo/></li><li/>)
4477            ## Interpreted as <li><foo><li/></foo></li> (non-conforming)            ## Interpreted as <li><foo><li/></foo></li> (non-conforming)
4478            ## div (Fx, S)            ## div (Fx, S)
4479              
4480          ## Step 1          my $non_optional;
4481          my $i = -1;          my $i = -1;
4482          my $node = $self->{open_elements}->[$i];  
4483          my $li_or_dtdd = {li => {li => 1},          ## 1.
4484                            dt => {dt => 1, dd => 1},          for my $node (reverse @{$self->{open_elements}}) {
4485                            dd => {dt => 1, dd => 1}}->{$token->{tag_name}};            if ($node->[1] == LI_EL) {
4486          LI: {              ## 2. (a) As if </li>
4487            ## Step 2              {
4488            if ($li_or_dtdd->{$node->[0]->manakai_local_name}) {                ## If no </li> - not applied
4489              if ($i != -1) {                #
4490                !!!cp ('t355');  
4491                !!!parse-error (type => 'not closed',                ## Otherwise
4492                                text => $self->{open_elements}->[-1]->[0]  
4493                                    ->manakai_local_name,                ## 1. generate implied end tags, except for </li>
4494                                token => $token);                #
4495              } else {  
4496                !!!cp ('t356');                ## 2. If current node != "li", parse error
4497                  if ($non_optional) {
4498                    !!!parse-error (type => 'not closed',
4499                                    text => $non_optional->[0]->manakai_local_name,
4500                                    token => $token);
4501                    !!!cp ('t355');
4502                  } else {
4503                    !!!cp ('t356');
4504                  }
4505    
4506                  ## 3. Pop
4507                  splice @{$self->{open_elements}}, $i;
4508              }              }
4509              splice @{$self->{open_elements}}, $i;  
4510              last LI;              last; ## 2. (b) goto 5.
4511            } else {            } elsif (
4512                       ## NOTE: not "formatting" and not "phrasing"
4513                       ($node->[1] & SPECIAL_EL or
4514                        $node->[1] & SCOPING_EL) and
4515                       ## NOTE: "li", "dt", and "dd" are in |SPECIAL_EL|.
4516                       (not $node->[1] & ADDRESS_DIV_P_EL)
4517                      ) {
4518                ## 3.
4519              !!!cp ('t357');              !!!cp ('t357');
4520            }              last; ## goto 5.
4521                        } elsif ($node->[1] & END_TAG_OPTIONAL_EL) {
           ## Step 3  
           if (not ($node->[1] & FORMATTING_EL) and  
               #not $phrasing_category->{$node->[1]} and  
               ($node->[1] & SPECIAL_EL or  
                $node->[1] & SCOPING_EL) and  
               not ($node->[1] & ADDRESS_EL) and  
               not ($node->[1] & DIV_EL)) {  
4522              !!!cp ('t358');              !!!cp ('t358');
4523              last LI;              #
4524              } else {
4525                !!!cp ('t359');
4526                $non_optional ||= $node;
4527                #
4528            }            }
4529                        ## 4.
4530            !!!cp ('t359');            ## goto 2.
           ## Step 4  
4531            $i--;            $i--;
4532            $node = $self->{open_elements}->[$i];          }
4533            redo LI;  
4534          } # LI          ## 5. (a) has a |p| element in scope
4535                      INSCOPE: for (reverse @{$self->{open_elements}}) {
4536              if ($_->[1] == P_EL) {
4537                !!!cp ('t353');
4538    
4539                ## NOTE: |<p><li>|, for example.
4540    
4541                !!!back-token; # <x>
4542                $token = {type => END_TAG_TOKEN, tag_name => 'p',
4543                          line => $token->{line}, column => $token->{column}};
4544                next B;
4545              } elsif ($_->[1] & SCOPING_EL) {
4546                !!!cp ('t354');
4547                last INSCOPE;
4548              }
4549            } # INSCOPE
4550    
4551            ## 5. (b) insert
4552          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
4553          !!!nack ('t359.1');          !!!nack ('t359.1');
4554          !!!next-token;          !!!next-token;
4555          next B;          next B;
4556          } elsif ($token->{tag_name} eq 'dt' or
4557                   $token->{tag_name} eq 'dd') {
4558            ## NOTE: As normal, but imply </dt> or </dd> when ...
4559    
4560            my $non_optional;
4561            my $i = -1;
4562    
4563            ## 1.
4564            for my $node (reverse @{$self->{open_elements}}) {
4565              if ($node->[1] == DTDD_EL) {
4566                ## 2. (a) As if </li>
4567                {
4568                  ## If no </li> - not applied
4569                  #
4570    
4571                  ## Otherwise
4572    
4573                  ## 1. generate implied end tags, except for </dt> or </dd>
4574                  #
4575    
4576                  ## 2. If current node != "dt"|"dd", parse error
4577                  if ($non_optional) {
4578                    !!!parse-error (type => 'not closed',
4579                                    text => $non_optional->[0]->manakai_local_name,
4580                                    token => $token);
4581                    !!!cp ('t355.1');
4582                  } else {
4583                    !!!cp ('t356.1');
4584                  }
4585    
4586                  ## 3. Pop
4587                  splice @{$self->{open_elements}}, $i;
4588                }
4589    
4590                last; ## 2. (b) goto 5.
4591              } elsif (
4592                       ## NOTE: not "formatting" and not "phrasing"
4593                       ($node->[1] & SPECIAL_EL or
4594                        $node->[1] & SCOPING_EL) and
4595                       ## NOTE: "li", "dt", and "dd" are in |SPECIAL_EL|.
4596    
4597                       (not $node->[1] & ADDRESS_DIV_P_EL)
4598                      ) {
4599                ## 3.
4600                !!!cp ('t357.1');
4601                last; ## goto 5.
4602              } elsif ($node->[1] & END_TAG_OPTIONAL_EL) {
4603                !!!cp ('t358.1');
4604                #
4605              } else {
4606                !!!cp ('t359.1');
4607                $non_optional ||= $node;
4608                #
4609              }
4610              ## 4.
4611              ## goto 2.
4612              $i--;
4613            }
4614    
4615            ## 5. (a) has a |p| element in scope
4616            INSCOPE: for (reverse @{$self->{open_elements}}) {
4617              if ($_->[1] == P_EL) {
4618                !!!cp ('t353.1');
4619                !!!back-token; # <x>
4620                $token = {type => END_TAG_TOKEN, tag_name => 'p',
4621                          line => $token->{line}, column => $token->{column}};
4622                next B;
4623              } elsif ($_->[1] & SCOPING_EL) {
4624                !!!cp ('t354.1');
4625                last INSCOPE;
4626              }
4627            } # INSCOPE
4628    
4629            ## 5. (b) insert
4630            !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
4631            !!!nack ('t359.2');
4632            !!!next-token;
4633            next B;
4634        } elsif ($token->{tag_name} eq 'plaintext') {        } elsif ($token->{tag_name} eq 'plaintext') {
4635            ## NOTE: As normal, but effectively ends parsing
4636    
4637          ## has a p element in scope          ## has a p element in scope
4638          INSCOPE: for (reverse @{$self->{open_elements}}) {          INSCOPE: for (reverse @{$self->{open_elements}}) {
4639            if ($_->[1] & P_EL) {            if ($_->[1] == P_EL) {
4640              !!!cp ('t367');              !!!cp ('t367');
4641              !!!back-token; # <plaintext>              !!!back-token; # <plaintext>
4642              $token = {type => END_TAG_TOKEN, tag_name => 'p',              $token = {type => END_TAG_TOKEN, tag_name => 'p',
# Line 7082  sub _tree_construction_main ($) { Line 4658  sub _tree_construction_main ($) {
4658        } elsif ($token->{tag_name} eq 'a') {        } elsif ($token->{tag_name} eq 'a') {
4659          AFE: for my $i (reverse 0..$#$active_formatting_elements) {          AFE: for my $i (reverse 0..$#$active_formatting_elements) {
4660            my $node = $active_formatting_elements->[$i];            my $node = $active_formatting_elements->[$i];
4661            if ($node->[1] & A_EL) {            if ($node->[1] == A_EL) {
4662              !!!cp ('t371');              !!!cp ('t371');
4663              !!!parse-error (type => 'in a:a', token => $token);              !!!parse-error (type => 'in a:a', token => $token);
4664                            
# Line 7126  sub _tree_construction_main ($) { Line 4702  sub _tree_construction_main ($) {
4702          ## has a |nobr| element in scope          ## has a |nobr| element in scope
4703          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
4704            my $node = $self->{open_elements}->[$_];            my $node = $self->{open_elements}->[$_];
4705            if ($node->[1] & NOBR_EL) {            if ($node->[1] == NOBR_EL) {
4706              !!!cp ('t376');              !!!cp ('t376');
4707              !!!parse-error (type => 'in nobr:nobr', token => $token);              !!!parse-error (type => 'in nobr:nobr', token => $token);
4708              !!!back-token; # <nobr>              !!!back-token; # <nobr>
# Line 7149  sub _tree_construction_main ($) { Line 4725  sub _tree_construction_main ($) {
4725          ## has a button element in scope          ## has a button element in scope
4726          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
4727            my $node = $self->{open_elements}->[$_];            my $node = $self->{open_elements}->[$_];
4728            if ($node->[1] & BUTTON_EL) {            if ($node->[1] == BUTTON_EL) {
4729              !!!cp ('t378');              !!!cp ('t378');
4730              !!!parse-error (type => 'in button:button', token => $token);              !!!parse-error (type => 'in button:button', token => $token);
4731              !!!back-token; # <button>              !!!back-token; # <button>
# Line 7249  sub _tree_construction_main ($) { Line 4825  sub _tree_construction_main ($) {
4825            next B;            next B;
4826          }          }
4827        } elsif ($token->{tag_name} eq 'textarea') {        } elsif ($token->{tag_name} eq 'textarea') {
4828          my $tag_name = $token->{tag_name};          ## Step 1
4829          my $el;          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
         !!!create-element ($el, $HTML_NS, $token->{tag_name}, $token->{attributes}, $token);  
4830                    
4831            ## Step 2
4832          ## TODO: $self->{form_element} if defined          ## TODO: $self->{form_element} if defined
4833    
4834            ## Step 3
4835            $self->{ignore_newline} = 1;
4836    
4837            ## Step 4
4838            ## ISSUE: This step is wrong. (r2302 enbugged)
4839    
4840            ## Step 5
4841          $self->{content_model} = RCDATA_CONTENT_MODEL;          $self->{content_model} = RCDATA_CONTENT_MODEL;
4842          delete $self->{escape}; # MUST          delete $self->{escape}; # MUST
4843            
4844          $insert->($el);          ## Step 6-7
4845                    $self->{insertion_mode} |= IN_CDATA_RCDATA_IM;
4846          my $text = '';  
4847          !!!nack ('t392.1');          !!!nack ('t392.1');
4848          !!!next-token;          !!!next-token;
4849          if ($token->{type} == CHARACTER_TOKEN) {          next B;
4850            $token->{data} =~ s/^\x0A//;        } elsif ($token->{tag_name} eq 'optgroup' or
4851            unless (length $token->{data}) {                 $token->{tag_name} eq 'option') {
4852              !!!cp ('t392');          ## has an |option| element in scope
4853              !!!next-token;          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
4854            } else {            my $node = $self->{open_elements}->[$_];
4855              !!!cp ('t393');            if ($node->[1] == OPTION_EL) {
4856                !!!cp ('t397.1');
4857                ## NOTE: As if </option>
4858                !!!back-token; # <option> or <optgroup>
4859                $token = {type => END_TAG_TOKEN, tag_name => 'option',
4860                          line => $token->{line}, column => $token->{column}};
4861                next B;
4862              } elsif ($node->[1] & SCOPING_EL) {
4863                !!!cp ('t397.2');
4864                last INSCOPE;
4865            }            }
4866          } else {          } # INSCOPE
4867            !!!cp ('t394');  
4868          }          $reconstruct_active_formatting_elements->($insert_to_current);
4869          while ($token->{type} == CHARACTER_TOKEN) {  
4870            !!!cp ('t395');          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
4871            $text .= $token->{data};  
4872            !!!next-token;          !!!nack ('t397.3');
         }  
         if (length $text) {  
           !!!cp ('t396');  
           $el->manakai_append_text ($text);  
         }  
           
         $self->{content_model} = PCDATA_CONTENT_MODEL;  
           
         if ($token->{type} == END_TAG_TOKEN and  
             $token->{tag_name} eq $tag_name) {  
           !!!cp ('t397');  
           ## Ignore the token  
         } else {  
           !!!cp ('t398');  
           !!!parse-error (type => 'in RCDATA:#eof', token => $token);  
         }  
4873          !!!next-token;          !!!next-token;
4874          next B;          redo B;
4875        } elsif ($token->{tag_name} eq 'rt' or        } elsif ($token->{tag_name} eq 'rt' or
4876                 $token->{tag_name} eq 'rp') {                 $token->{tag_name} eq 'rp') {
4877          ## has a |ruby| element in scope          ## has a |ruby| element in scope
4878          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
4879            my $node = $self->{open_elements}->[$_];            my $node = $self->{open_elements}->[$_];
4880            if ($node->[1] & RUBY_EL) {            if ($node->[1] == RUBY_EL) {
4881              !!!cp ('t398.1');              !!!cp ('t398.1');
4882              ## generate implied end tags              ## generate implied end tags
4883              while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {              while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
4884                !!!cp ('t398.2');                !!!cp ('t398.2');
4885                pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
4886              }              }
4887              unless ($self->{open_elements}->[-1]->[1] & RUBY_EL) {              unless ($self->{open_elements}->[-1]->[1] == RUBY_EL) {
4888                !!!cp ('t398.3');                !!!cp ('t398.3');
4889                !!!parse-error (type => 'not closed',                !!!parse-error (type => 'not closed',
4890                                text => $self->{open_elements}->[-1]->[0]                                text => $self->{open_elements}->[-1]->[0]
4891                                    ->manakai_local_name,                                    ->manakai_local_name,
4892                                token => $token);                                token => $token);
4893                pop @{$self->{open_elements}}                pop @{$self->{open_elements}}
4894                    while not $self->{open_elements}->[-1]->[1] & RUBY_EL;                    while not $self->{open_elements}->[-1]->[1] == RUBY_EL;
4895              }              }
4896              last INSCOPE;              last INSCOPE;
4897            } elsif ($node->[1] & SCOPING_EL) {            } elsif ($node->[1] & SCOPING_EL) {
# Line 7342  sub _tree_construction_main ($) { Line 4919  sub _tree_construction_main ($) {
4919                    
4920          if ($self->{self_closing}) {          if ($self->{self_closing}) {
4921            pop @{$self->{open_elements}};            pop @{$self->{open_elements}};
4922            !!!ack ('t398.1');            !!!ack ('t398.6');
4923          } else {          } else {
4924            !!!cp ('t398.2');            !!!cp ('t398.7');
4925            $self->{insertion_mode} |= IN_FOREIGN_CONTENT_IM;            $self->{insertion_mode} |= IN_FOREIGN_CONTENT_IM;
4926            ## NOTE: |<body><math><mi><svg>| -> "in foreign content" insertion            ## NOTE: |<body><math><mi><svg>| -> "in foreign content" insertion
4927            ## mode, "in body" (not "in foreign content") secondary insertion            ## mode, "in body" (not "in foreign content") secondary insertion
# Line 7355  sub _tree_construction_main ($) { Line 4932  sub _tree_construction_main ($) {
4932          next B;          next B;
4933        } elsif ({        } elsif ({
4934                  caption => 1, col => 1, colgroup => 1, frame => 1,                  caption => 1, col => 1, colgroup => 1, frame => 1,
4935                  frameset => 1, head => 1, option => 1, optgroup => 1,                  frameset => 1, head => 1,
4936                  tbody => 1, td => 1, tfoot => 1, th => 1,                  tbody => 1, td => 1, tfoot => 1, th => 1,
4937                  thead => 1, tr => 1,                  thead => 1, tr => 1,
4938                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
# Line 7366  sub _tree_construction_main ($) { Line 4943  sub _tree_construction_main ($) {
4943          !!!nack ('t401.1'); ## NOTE: |<col/>| or |<frame/>| here is an error.          !!!nack ('t401.1'); ## NOTE: |<col/>| or |<frame/>| here is an error.
4944          !!!next-token;          !!!next-token;
4945          next B;          next B;
4946                  } elsif ($token->{tag_name} eq 'param' or
4947          ## ISSUE: An issue on HTML5 new elements in the spec.                 $token->{tag_name} eq 'source') {
4948            !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
4949            pop @{$self->{open_elements}};
4950    
4951            !!!ack ('t398.5');
4952            !!!next-token;
4953            redo B;
4954        } else {        } else {
4955          if ($token->{tag_name} eq 'image') {          if ($token->{tag_name} eq 'image') {
4956            !!!cp ('t384');            !!!cp ('t384');
# Line 7403  sub _tree_construction_main ($) { Line 4986  sub _tree_construction_main ($) {
4986            !!!ack ('t388.2');            !!!ack ('t388.2');
4987          } elsif ({          } elsif ({
4988                    area => 1, basefont => 1, bgsound => 1, br => 1,                    area => 1, basefont => 1, bgsound => 1, br => 1,
4989                    embed => 1, img => 1, param => 1, spacer => 1, wbr => 1,                    embed => 1, img => 1, spacer => 1, wbr => 1,
                   #image => 1,  
4990                   }->{$token->{tag_name}}) {                   }->{$token->{tag_name}}) {
4991            !!!cp ('t388.1');            !!!cp ('t388.1');
4992            pop @{$self->{open_elements}};            pop @{$self->{open_elements}};
# Line 7414  sub _tree_construction_main ($) { Line 4996  sub _tree_construction_main ($) {
4996                    
4997            if ($self->{insertion_mode} & TABLE_IMS or            if ($self->{insertion_mode} & TABLE_IMS or
4998                $self->{insertion_mode} & BODY_TABLE_IMS or                $self->{insertion_mode} & BODY_TABLE_IMS or
4999                $self->{insertion_mode} == IN_COLUMN_GROUP_IM) {                ($self->{insertion_mode} & IM_MASK) == IN_COLUMN_GROUP_IM) {
5000              !!!cp ('t400.1');              !!!cp ('t400.1');
5001              $self->{insertion_mode} = IN_SELECT_IN_TABLE_IM;              $self->{insertion_mode} = IN_SELECT_IN_TABLE_IM;
5002            } else {            } else {
# Line 7435  sub _tree_construction_main ($) { Line 5017  sub _tree_construction_main ($) {
5017          my $i;          my $i;
5018          INSCOPE: {          INSCOPE: {
5019            for (reverse @{$self->{open_elements}}) {            for (reverse @{$self->{open_elements}}) {
5020              if ($_->[1] & BODY_EL) {              if ($_->[1] == BODY_EL) {
5021                !!!cp ('t405');                !!!cp ('t405');
5022                $i = $_;                $i = $_;
5023                last INSCOPE;                last INSCOPE;
# Line 7445  sub _tree_construction_main ($) { Line 5027  sub _tree_construction_main ($) {
5027              }              }
5028            }            }
5029    
5030            !!!parse-error (type => 'start tag not allowed',            ## NOTE: |<marquee></body>|, |<svg><foreignobject></body>|
5031    
5032              !!!parse-error (type => 'unmatched end tag',
5033                            text => $token->{tag_name}, token => $token);                            text => $token->{tag_name}, token => $token);
5034            ## NOTE: Ignore the token.            ## NOTE: Ignore the token.
5035            !!!next-token;            !!!next-token;
# Line 7471  sub _tree_construction_main ($) { Line 5055  sub _tree_construction_main ($) {
5055          ## TODO: Update this code.  It seems that the code below is not          ## TODO: Update this code.  It seems that the code below is not
5056          ## up-to-date, though it has same effect as speced.          ## up-to-date, though it has same effect as speced.
5057          if (@{$self->{open_elements}} > 1 and          if (@{$self->{open_elements}} > 1 and
5058              $self->{open_elements}->[1]->[1] & BODY_EL) {              $self->{open_elements}->[1]->[1] == BODY_EL) {
5059            ## ISSUE: There is an issue in the spec.            unless ($self->{open_elements}->[-1]->[1] == BODY_EL) {
           unless ($self->{open_elements}->[-1]->[1] & BODY_EL) {  
5060              !!!cp ('t406');              !!!cp ('t406');
5061              !!!parse-error (type => 'not closed',              !!!parse-error (type => 'not closed',
5062                              text => $self->{open_elements}->[1]->[0]                              text => $self->{open_elements}->[1]->[0]
# Line 7494  sub _tree_construction_main ($) { Line 5077  sub _tree_construction_main ($) {
5077            next B;            next B;
5078          }          }
5079        } elsif ({        } elsif ({
5080                  address => 1, blockquote => 1, center => 1, dir => 1,                  ## NOTE: End tags for non-phrasing flow content elements
5081                  div => 1, dl => 1, fieldset => 1, listing => 1,  
5082                  menu => 1, ol => 1, pre => 1, ul => 1,                  ## NOTE: The normal ones
5083                    address => 1, article => 1, aside => 1, blockquote => 1,
5084                    center => 1, datagrid => 1, details => 1, dialog => 1,
5085                    dir => 1, div => 1, dl => 1, fieldset => 1, figure => 1,
5086                    footer => 1, header => 1, listing => 1, menu => 1, nav => 1,
5087                    ol => 1, pre => 1, section => 1, ul => 1,
5088    
5089                    ## NOTE: As normal, but ... optional tags
5090                  dd => 1, dt => 1, li => 1,                  dd => 1, dt => 1, li => 1,
5091    
5092                  applet => 1, button => 1, marquee => 1, object => 1,                  applet => 1, button => 1, marquee => 1, object => 1,
5093                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
5094            ## NOTE: Code for <li> start tags includes "as if </li>" code.
5095            ## Code for <dt> or <dd> start tags includes "as if </dt> or
5096            ## </dd>" code.
5097    
5098          ## has an element in scope          ## has an element in scope
5099          my $i;          my $i;
5100          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
# Line 7560  sub _tree_construction_main ($) { Line 5155  sub _tree_construction_main ($) {
5155          !!!next-token;          !!!next-token;
5156          next B;          next B;
5157        } elsif ($token->{tag_name} eq 'form') {        } elsif ($token->{tag_name} eq 'form') {
5158            ## NOTE: As normal, but interacts with the form element pointer
5159    
5160          undef $self->{form_element};          undef $self->{form_element};
5161    
5162          ## has an element in scope          ## has an element in scope
5163          my $i;          my $i;
5164          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5165            my $node = $self->{open_elements}->[$_];            my $node = $self->{open_elements}->[$_];
5166            if ($node->[1] & FORM_EL) {            if ($node->[1] == FORM_EL) {
5167              !!!cp ('t418');              !!!cp ('t418');
5168              $i = $_;              $i = $_;
5169              last INSCOPE;              last INSCOPE;
# Line 7607  sub _tree_construction_main ($) { Line 5204  sub _tree_construction_main ($) {
5204          !!!next-token;          !!!next-token;
5205          next B;          next B;
5206        } elsif ({        } elsif ({
5207                    ## NOTE: As normal, except acts as a closer for any ...
5208                  h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,                  h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,
5209                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
5210          ## has an element in scope          ## has an element in scope
5211          my $i;          my $i;
5212          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5213            my $node = $self->{open_elements}->[$_];            my $node = $self->{open_elements}->[$_];
5214            if ($node->[1] & HEADING_EL) {            if ($node->[1] == HEADING_EL) {
5215              !!!cp ('t423');              !!!cp ('t423');
5216              $i = $_;              $i = $_;
5217              last INSCOPE;              last INSCOPE;
# Line 7652  sub _tree_construction_main ($) { Line 5250  sub _tree_construction_main ($) {
5250          !!!next-token;          !!!next-token;
5251          next B;          next B;
5252        } elsif ($token->{tag_name} eq 'p') {        } elsif ($token->{tag_name} eq 'p') {
5253            ## NOTE: As normal, except </p> implies <p> and ...
5254    
5255          ## has an element in scope          ## has an element in scope
5256            my $non_optional;
5257          my $i;          my $i;
5258          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5259            my $node = $self->{open_elements}->[$_];            my $node = $self->{open_elements}->[$_];
5260            if ($node->[1] & P_EL) {            if ($node->[1] == P_EL) {
5261              !!!cp ('t410.1');              !!!cp ('t410.1');
5262              $i = $_;              $i = $_;
5263              last INSCOPE;              last INSCOPE;
5264            } elsif ($node->[1] & SCOPING_EL) {            } elsif ($node->[1] & SCOPING_EL) {
5265              !!!cp ('t411.1');              !!!cp ('t411.1');
5266              last INSCOPE;              last INSCOPE;
5267              } elsif ($node->[1] & END_TAG_OPTIONAL_EL) {
5268                ## NOTE: |END_TAG_OPTIONAL_EL| includes "p"
5269                !!!cp ('t411.2');
5270                #
5271              } else {
5272                !!!cp ('t411.3');
5273                $non_optional ||= $node;
5274                #
5275            }            }
5276          } # INSCOPE          } # INSCOPE
5277    
5278          if (defined $i) {          if (defined $i) {
5279            if ($self->{open_elements}->[-1]->[0]->manakai_local_name            ## 1. Generate implied end tags
5280                    ne $token->{tag_name}) {            #
5281    
5282              ## 2. If current node != "p", parse error
5283              if ($non_optional) {
5284              !!!cp ('t412.1');              !!!cp ('t412.1');
5285              !!!parse-error (type => 'not closed',              !!!parse-error (type => 'not closed',
5286                              text => $self->{open_elements}->[-1]->[0]                              text => $non_optional->[0]->manakai_local_name,
                                 ->manakai_local_name,  
5287                              token => $token);                              token => $token);
5288            } else {            } else {
5289              !!!cp ('t414.1');              !!!cp ('t414.1');
5290            }            }
5291    
5292              ## 3. Pop
5293            splice @{$self->{open_elements}}, $i;            splice @{$self->{open_elements}}, $i;
5294          } else {          } else {
5295            !!!cp ('t413.1');            !!!cp ('t413.1');
# Line 7718  sub _tree_construction_main ($) { Line 5330  sub _tree_construction_main ($) {
5330          ## Ignore the token.          ## Ignore the token.
5331          !!!next-token;          !!!next-token;
5332          next B;          next B;
       } elsif ({  
                 caption => 1, col => 1, colgroup => 1, frame => 1,  
                 frameset => 1, head => 1, option => 1, optgroup => 1,  
                 tbody => 1, td => 1, tfoot => 1, th => 1,  
                 thead => 1, tr => 1,  
                 area => 1, basefont => 1, bgsound => 1,  
                 embed => 1, hr => 1, iframe => 1, image => 1,  
                 img => 1, input => 1, isindex => 1, noembed => 1,  
                 noframes => 1, param => 1, select => 1, spacer => 1,  
                 table => 1, textarea => 1, wbr => 1,  
                 noscript => 0, ## TODO: if scripting is enabled  
                }->{$token->{tag_name}}) {  
         !!!cp ('t429');  
         !!!parse-error (type => 'unmatched end tag',  
                         text => $token->{tag_name}, token => $token);  
         ## Ignore the token  
         !!!next-token;  
         next B;  
           
         ## ISSUE: Issue on HTML5 new elements in spec  
           
5333        } else {        } else {
5334            if ($token->{tag_name} eq 'sarcasm') {
5335              sleep 0.001; # take a deep breath
5336            }
5337    
5338          ## Step 1          ## Step 1
5339          my $node_i = -1;          my $node_i = -1;
5340          my $node = $self->{open_elements}->[$node_i];          my $node = $self->{open_elements}->[$node_i];
5341    
5342          ## Step 2          ## Step 2
5343          S2: {          S2: {
5344            if ($node->[0]->manakai_local_name eq $token->{tag_name}) {            my $node_tag_name = $node->[0]->manakai_local_name;
5345              $node_tag_name =~ tr/A-Z/a-z/; # for SVG camelCase tag names
5346              if ($node_tag_name eq $token->{tag_name}) {
5347              ## Step 1              ## Step 1
5348              ## generate implied end tags              ## generate implied end tags
5349              while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {              while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
# Line 7759  sub _tree_construction_main ($) { Line 5356  sub _tree_construction_main ($) {
5356              }              }
5357                    
5358              ## Step 2              ## Step 2
5359              if ($self->{open_elements}->[-1]->[0]->manakai_local_name              my $current_tag_name
5360                      ne $token->{tag_name}) {                  = $self->{open_elements}->[-1]->[0]->manakai_local_name;
5361                $current_tag_name =~ tr/A-Z/a-z/;
5362                if ($current_tag_name ne $token->{tag_name}) {
5363                !!!cp ('t431');                !!!cp ('t431');
5364                ## NOTE: <x><y></x>                ## NOTE: <x><y></x>
5365                !!!parse-error (type => 'not closed',                !!!parse-error (type => 'not closed',
# Line 8030  sub set_inner_html ($$$$;$) { Line 5629  sub set_inner_html ($$$$;$) {
5629      push @{$p->{open_elements}}, [$root, $el_category->{html}];      push @{$p->{open_elements}}, [$root, $el_category->{html}];
5630    
5631      undef $p->{head_element};      undef $p->{head_element};
5632        undef $p->{head_element_inserted};
5633    
5634      ## Step 6 # MUST      ## Step 6 # MUST
5635      $p->_reset_insertion_mode;      $p->_reset_insertion_mode;

Legend:
Removed from v.1.194  
changed lines
  Added in v.1.210

[email protected]
ViewVC Help
Powered by ViewVC 1.1.24