/[suikacvs]/markup/html/whatpm/Whatpm/HTML.pm.src
Suika

Diff of /markup/html/whatpm/Whatpm/HTML.pm.src

Parent Directory Parent Directory | Revision Log Revision Log | View Patch Patch

revision 1.153 by wakaba, Fri Aug 15 08:32:41 2008 UTC revision 1.206 by wakaba, Mon Oct 13 08:22:30 2008 UTC
# Line 3  use strict; Line 3  use strict;
3  our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r};  our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r};
4  use Error qw(:try);  use Error qw(:try);
5    
6    ## NOTE: This module don't check all HTML5 parse errors; character
7    ## encoding related parse errors are expected to be handled by relevant
8    ## modules.
9    ## Parse errors for control characters that are not allowed in HTML5
10    ## documents, for surrogate code points, and for noncharacter code
11    ## points, as well as U+FFFD substitions for characters whose code points
12    ## is higher than U+10FFFF may be detected by combining the parser with
13    ## the checker implemented by Whatpm::Charset::UnicodeChecker (for its
14    ## usage example, see |t/HTML-tree.t| in the Whatpm package or the
15    ## WebHACC::Language::HTML module in the WebHACC package).
16    
17  ## ISSUE:  ## ISSUE:
18  ## var doc = implementation.createDocument (null, null, null);  ## var doc = implementation.createDocument (null, null, null);
19  ## doc.write ('');  ## doc.write ('');
# Line 17  my $XLINK_NS = q<http://www.w3.org/1999/ Line 28  my $XLINK_NS = q<http://www.w3.org/1999/
28  my $XML_NS = q<http://www.w3.org/XML/1998/namespace>;  my $XML_NS = q<http://www.w3.org/XML/1998/namespace>;
29  my $XMLNS_NS = q<http://www.w3.org/2000/xmlns/>;  my $XMLNS_NS = q<http://www.w3.org/2000/xmlns/>;
30    
31  sub A_EL () { 0b1 }  ## Bits 12-15
32  sub ADDRESS_EL () { 0b10 }  sub SPECIAL_EL () { 0b1_000000000000000 }
33  sub BODY_EL () { 0b100 }  sub SCOPING_EL () { 0b1_00000000000000 }
34  sub BUTTON_EL () { 0b1000 }  sub FORMATTING_EL () { 0b1_0000000000000 }
35  sub CAPTION_EL () { 0b10000 }  sub PHRASING_EL () { 0b1_000000000000 }
36  sub DD_EL () { 0b100000 }  
37  sub DIV_EL () { 0b1000000 }  ## Bits 10-11
38  sub DT_EL () { 0b10000000 }  sub FOREIGN_EL () { 0b1_00000000000 }
39  sub FORM_EL () { 0b100000000 }  sub FOREIGN_FLOW_CONTENT_EL () { 0b1_0000000000 }
40  sub FORMATTING_EL () { 0b1000000000 }  
41  sub FRAMESET_EL () { 0b10000000000 }  ## Bits 6-9
42  sub HEADING_EL () { 0b100000000000 }  sub TABLE_SCOPING_EL () { 0b1_000000000 }
43  sub HTML_EL () { 0b1000000000000 }  sub TABLE_ROWS_SCOPING_EL () { 0b1_00000000 }
44  sub LI_EL () { 0b10000000000000 }  sub TABLE_ROW_SCOPING_EL () { 0b1_0000000 }
45  sub NOBR_EL () { 0b100000000000000 }  sub TABLE_ROWS_EL () { 0b1_000000 }
 sub OPTION_EL () { 0b1000000000000000 }  
 sub OPTGROUP_EL () { 0b10000000000000000 }  
 sub P_EL () { 0b100000000000000000 }  
 sub SELECT_EL () { 0b1000000000000000000 }  
 sub TABLE_EL () { 0b10000000000000000000 }  
 sub TABLE_CELL_EL () { 0b100000000000000000000 }  
 sub TABLE_ROW_EL () { 0b1000000000000000000000 }  
 sub TABLE_ROW_GROUP_EL () { 0b10000000000000000000000 }  
 sub MISC_SCOPING_EL () { 0b100000000000000000000000 }  
 sub MISC_SPECIAL_EL () { 0b1000000000000000000000000 }  
 sub FOREIGN_EL () { 0b10000000000000000000000000 }  
 sub FOREIGN_FLOW_CONTENT_EL () { 0b100000000000000000000000000 }  
 sub MML_AXML_EL () { 0b1000000000000000000000000000 }  
 sub RUBY_EL () { 0b10000000000000000000000000000 }  
 sub RUBY_COMPONENT_EL () { 0b100000000000000000000000000000 }  
   
 sub TABLE_ROWS_EL () {  
   TABLE_EL |  
   TABLE_ROW_EL |  
   TABLE_ROW_GROUP_EL  
 }  
46    
47  ## NOTE: Used in "generate implied end tags" algorithm.  ## Bit 5
48  ## NOTE: There is a code where a modified version of END_TAG_OPTIONAL_EL  sub ADDRESS_DIV_P_EL () { 0b1_00000 }
 ## is used in "generate implied end tags" implementation (search for the  
 ## function mae).  
 sub END_TAG_OPTIONAL_EL () {  
   DD_EL |  
   DT_EL |  
   LI_EL |  
   P_EL |  
   RUBY_COMPONENT_EL  
 }  
49    
50  ## NOTE: Used in </body> and EOF algorithms.  ## NOTE: Used in </body> and EOF algorithms.
51  sub ALL_END_TAG_OPTIONAL_EL () {  ## Bit 4
52    DD_EL |  sub ALL_END_TAG_OPTIONAL_EL () { 0b1_0000 }
   DT_EL |  
   LI_EL |  
   P_EL |  
   
   BODY_EL |  
   HTML_EL |  
   TABLE_CELL_EL |  
   TABLE_ROW_EL |  
   TABLE_ROW_GROUP_EL  
 }  
53    
54  sub SCOPING_EL () {  ## NOTE: Used in "generate implied end tags" algorithm.
55    BUTTON_EL |  ## NOTE: There is a code where a modified version of
56    CAPTION_EL |  ## END_TAG_OPTIONAL_EL is used in "generate implied end tags"
57    HTML_EL |  ## implementation (search for the algorithm name).
58    TABLE_EL |  ## Bit 3
59    TABLE_CELL_EL |  sub END_TAG_OPTIONAL_EL () { 0b1_000 }
60    MISC_SCOPING_EL  
61    ## Bits 0-2
62    
63    sub MISC_SPECIAL_EL () { SPECIAL_EL | 0b000 }
64    sub FORM_EL () { SPECIAL_EL | 0b001 }
65    sub FRAMESET_EL () { SPECIAL_EL | 0b010 }
66    sub HEADING_EL () { SPECIAL_EL | 0b011 }
67    sub SELECT_EL () { SPECIAL_EL | 0b100 }
68    sub SCRIPT_EL () { SPECIAL_EL | 0b101 }
69    
70    sub ADDRESS_DIV_EL () { SPECIAL_EL | ADDRESS_DIV_P_EL | 0b001 }
71    sub BODY_EL () { SPECIAL_EL | ALL_END_TAG_OPTIONAL_EL | 0b001 }
72    
73    sub DD_EL () {
74      SPECIAL_EL |
75      END_TAG_OPTIONAL_EL |
76      ALL_END_TAG_OPTIONAL_EL |
77      0b001
78  }  }
79    sub DT_EL () {
80  sub TABLE_SCOPING_EL () {    SPECIAL_EL |
81    HTML_EL |    END_TAG_OPTIONAL_EL |
82    TABLE_EL    ALL_END_TAG_OPTIONAL_EL |
83      0b010
84  }  }
85    sub LI_EL () {
86  sub TABLE_ROWS_SCOPING_EL () {    SPECIAL_EL |
87    HTML_EL |    END_TAG_OPTIONAL_EL |
88    TABLE_ROW_GROUP_EL    ALL_END_TAG_OPTIONAL_EL |
89      0b100
90    }
91    sub P_EL () {
92      SPECIAL_EL |
93      ADDRESS_DIV_P_EL |
94      END_TAG_OPTIONAL_EL |
95      ALL_END_TAG_OPTIONAL_EL |
96      0b001
97  }  }
98    
99  sub TABLE_ROW_SCOPING_EL () {  sub TABLE_ROW_EL () {
100    HTML_EL |    SPECIAL_EL |
101    TABLE_ROW_EL    TABLE_ROWS_EL |
102      TABLE_ROW_SCOPING_EL |
103      ALL_END_TAG_OPTIONAL_EL |
104      0b001
105    }
106    sub TABLE_ROW_GROUP_EL () {
107      SPECIAL_EL |
108      TABLE_ROWS_EL |
109      TABLE_ROWS_SCOPING_EL |
110      ALL_END_TAG_OPTIONAL_EL |
111      0b001
112  }  }
113    
114  sub SPECIAL_EL () {  sub MISC_SCOPING_EL () { SCOPING_EL | 0b000 }
115    ADDRESS_EL |  sub BUTTON_EL () { SCOPING_EL | 0b001 }
116    BODY_EL |  sub CAPTION_EL () { SCOPING_EL | 0b010 }
117    DIV_EL |  sub HTML_EL () {
118      SCOPING_EL |
119    DD_EL |    TABLE_SCOPING_EL |
120    DT_EL |    TABLE_ROWS_SCOPING_EL |
121    LI_EL |    TABLE_ROW_SCOPING_EL |
122    P_EL |    ALL_END_TAG_OPTIONAL_EL |
123      0b001
124    FORM_EL |  }
125    FRAMESET_EL |  sub TABLE_EL () {
126    HEADING_EL |    SCOPING_EL |
127    OPTION_EL |    TABLE_ROWS_EL |
128    OPTGROUP_EL |    TABLE_SCOPING_EL |
129    SELECT_EL |    0b001
130    TABLE_ROW_EL |  }
131    TABLE_ROW_GROUP_EL |  sub TABLE_CELL_EL () {
132    MISC_SPECIAL_EL    SCOPING_EL |
133      TABLE_ROW_SCOPING_EL |
134      ALL_END_TAG_OPTIONAL_EL |
135      0b001
136  }  }
137    
138    sub MISC_FORMATTING_EL () { FORMATTING_EL | 0b000 }
139    sub A_EL () { FORMATTING_EL | 0b001 }
140    sub NOBR_EL () { FORMATTING_EL | 0b010 }
141    
142    sub RUBY_EL () { PHRASING_EL | 0b001 }
143    
144    ## ISSUE: ALL_END_TAG_OPTIONAL_EL?
145    sub OPTGROUP_EL () { PHRASING_EL | END_TAG_OPTIONAL_EL | 0b001 }
146    sub OPTION_EL () { PHRASING_EL | END_TAG_OPTIONAL_EL | 0b010 }
147    sub RUBY_COMPONENT_EL () { PHRASING_EL | END_TAG_OPTIONAL_EL | 0b100 }
148    
149    sub MML_AXML_EL () { PHRASING_EL | FOREIGN_EL | 0b001 }
150    
151  my $el_category = {  my $el_category = {
152    a => A_EL | FORMATTING_EL,    a => A_EL,
153    address => ADDRESS_EL,    address => ADDRESS_DIV_EL,
154    applet => MISC_SCOPING_EL,    applet => MISC_SCOPING_EL,
155    area => MISC_SPECIAL_EL,    area => MISC_SPECIAL_EL,
156      article => MISC_SPECIAL_EL,
157      aside => MISC_SPECIAL_EL,
158    b => FORMATTING_EL,    b => FORMATTING_EL,
159    base => MISC_SPECIAL_EL,    base => MISC_SPECIAL_EL,
160    basefont => MISC_SPECIAL_EL,    basefont => MISC_SPECIAL_EL,
# Line 143  my $el_category = { Line 168  my $el_category = {
168    center => MISC_SPECIAL_EL,    center => MISC_SPECIAL_EL,
169    col => MISC_SPECIAL_EL,    col => MISC_SPECIAL_EL,
170    colgroup => MISC_SPECIAL_EL,    colgroup => MISC_SPECIAL_EL,
171      command => MISC_SPECIAL_EL,
172      datagrid => MISC_SPECIAL_EL,
173    dd => DD_EL,    dd => DD_EL,
174      details => MISC_SPECIAL_EL,
175      dialog => MISC_SPECIAL_EL,
176    dir => MISC_SPECIAL_EL,    dir => MISC_SPECIAL_EL,
177    div => DIV_EL,    div => ADDRESS_DIV_EL,
178    dl => MISC_SPECIAL_EL,    dl => MISC_SPECIAL_EL,
179    dt => DT_EL,    dt => DT_EL,
180    em => FORMATTING_EL,    em => FORMATTING_EL,
181    embed => MISC_SPECIAL_EL,    embed => MISC_SPECIAL_EL,
182      eventsource => MISC_SPECIAL_EL,
183    fieldset => MISC_SPECIAL_EL,    fieldset => MISC_SPECIAL_EL,
184      figure => MISC_SPECIAL_EL,
185    font => FORMATTING_EL,    font => FORMATTING_EL,
186      footer => MISC_SPECIAL_EL,
187    form => FORM_EL,    form => FORM_EL,
188    frame => MISC_SPECIAL_EL,    frame => MISC_SPECIAL_EL,
189    frameset => FRAMESET_EL,    frameset => FRAMESET_EL,
# Line 162  my $el_category = { Line 194  my $el_category = {
194    h5 => HEADING_EL,    h5 => HEADING_EL,
195    h6 => HEADING_EL,    h6 => HEADING_EL,
196    head => MISC_SPECIAL_EL,    head => MISC_SPECIAL_EL,
197      header => MISC_SPECIAL_EL,
198    hr => MISC_SPECIAL_EL,    hr => MISC_SPECIAL_EL,
199    html => HTML_EL,    html => HTML_EL,
200    i => FORMATTING_EL,    i => FORMATTING_EL,
201    iframe => MISC_SPECIAL_EL,    iframe => MISC_SPECIAL_EL,
202    img => MISC_SPECIAL_EL,    img => MISC_SPECIAL_EL,
203      #image => MISC_SPECIAL_EL, ## NOTE: Commented out in the spec.
204    input => MISC_SPECIAL_EL,    input => MISC_SPECIAL_EL,
205    isindex => MISC_SPECIAL_EL,    isindex => MISC_SPECIAL_EL,
206    li => LI_EL,    li => LI_EL,
# Line 175  my $el_category = { Line 209  my $el_category = {
209    marquee => MISC_SCOPING_EL,    marquee => MISC_SCOPING_EL,
210    menu => MISC_SPECIAL_EL,    menu => MISC_SPECIAL_EL,
211    meta => MISC_SPECIAL_EL,    meta => MISC_SPECIAL_EL,
212    nobr => NOBR_EL | FORMATTING_EL,    nav => MISC_SPECIAL_EL,
213      nobr => NOBR_EL,
214    noembed => MISC_SPECIAL_EL,    noembed => MISC_SPECIAL_EL,
215    noframes => MISC_SPECIAL_EL,    noframes => MISC_SPECIAL_EL,
216    noscript => MISC_SPECIAL_EL,    noscript => MISC_SPECIAL_EL,
# Line 193  my $el_category = { Line 228  my $el_category = {
228    s => FORMATTING_EL,    s => FORMATTING_EL,
229    script => MISC_SPECIAL_EL,    script => MISC_SPECIAL_EL,
230    select => SELECT_EL,    select => SELECT_EL,
231      section => MISC_SPECIAL_EL,
232    small => FORMATTING_EL,    small => FORMATTING_EL,
233    spacer => MISC_SPECIAL_EL,    spacer => MISC_SPECIAL_EL,
234    strike => FORMATTING_EL,    strike => FORMATTING_EL,
# Line 216  my $el_category = { Line 252  my $el_category = {
252  my $el_category_f = {  my $el_category_f = {
253    $MML_NS => {    $MML_NS => {
254      'annotation-xml' => MML_AXML_EL,      'annotation-xml' => MML_AXML_EL,
255      mi => FOREIGN_FLOW_CONTENT_EL,      mi => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
256      mo => FOREIGN_FLOW_CONTENT_EL,      mo => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
257      mn => FOREIGN_FLOW_CONTENT_EL,      mn => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
258      ms => FOREIGN_FLOW_CONTENT_EL,      ms => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
259      mtext => FOREIGN_FLOW_CONTENT_EL,      mtext => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
260    },    },
261    $SVG_NS => {    $SVG_NS => {
262      foreignObject => FOREIGN_FLOW_CONTENT_EL,      foreignObject => SCOPING_EL | FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
263      desc => FOREIGN_FLOW_CONTENT_EL,      desc => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
264      title => FOREIGN_FLOW_CONTENT_EL,      title => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
265    },    },
266    ## NOTE: In addition, FOREIGN_EL is set to non-HTML elements.    ## NOTE: In addition, FOREIGN_EL is set to non-HTML elements.
267  };  };
# Line 312  my $foreign_attr_xname = { Line 348  my $foreign_attr_xname = {
348    
349  ## ISSUE: xmlns:xlink="non-xlink-ns" is not an error.  ## ISSUE: xmlns:xlink="non-xlink-ns" is not an error.
350    
351  my $c1_entity_char = {  my $charref_map = {
352      0x0D => 0x000A,
353    0x80 => 0x20AC,    0x80 => 0x20AC,
354    0x81 => 0xFFFD,    0x81 => 0xFFFD,
355    0x82 => 0x201A,    0x82 => 0x201A,
# Line 345  my $c1_entity_char = { Line 382  my $c1_entity_char = {
382    0x9D => 0xFFFD,    0x9D => 0xFFFD,
383    0x9E => 0x017E,    0x9E => 0x017E,
384    0x9F => 0x0178,    0x9F => 0x0178,
385  }; # $c1_entity_char  }; # $charref_map
386    $charref_map->{$_} = 0xFFFD
387        for 0x0000..0x0008, 0x000B, 0x000E..0x001F, 0x007F,
388            0xD800..0xDFFF, 0xFDD0..0xFDDF, ## ISSUE: 0xFDEF
389            0xFFFE, 0xFFFF, 0x1FFFE, 0x1FFFF, 0x2FFFE, 0x2FFFF, 0x3FFFE, 0x3FFFF,
390            0x4FFFE, 0x4FFFF, 0x5FFFE, 0x5FFFF, 0x6FFFE, 0x6FFFF, 0x7FFFE,
391            0x7FFFF, 0x8FFFE, 0x8FFFF, 0x9FFFE, 0x9FFFF, 0xAFFFE, 0xAFFFF,
392            0xBFFFE, 0xBFFFF, 0xCFFFE, 0xCFFFF, 0xDFFFE, 0xDFFFF, 0xEFFFE,
393            0xEFFFF, 0xFFFFE, 0xFFFFF, 0x10FFFE, 0x10FFFF;
394    
395    ## TODO: Invoke the reset algorithm when a resettable element is
396    ## created (cf. HTML5 revision 2259).
397    
398  sub parse_byte_string ($$$$;$) {  sub parse_byte_string ($$$$;$) {
399    my $self = shift;    my $self = shift;
# Line 354  sub parse_byte_string ($$$$;$) { Line 402  sub parse_byte_string ($$$$;$) {
402    return $self->parse_byte_stream ($charset_name, $input, @_[1..$#_]);    return $self->parse_byte_stream ($charset_name, $input, @_[1..$#_]);
403  } # parse_byte_string  } # parse_byte_string
404    
405  sub parse_byte_stream ($$$$;$) {  sub parse_byte_stream ($$$$;$$) {
406      # my ($self, $charset_name, $byte_stream, $doc, $onerror, $get_wrapper) = @_;
407    my $self = ref $_[0] ? shift : shift->new;    my $self = ref $_[0] ? shift : shift->new;
408    my $charset_name = shift;    my $charset_name = shift;
409    my $byte_stream = $_[0];    my $byte_stream = $_[0];
# Line 365  sub parse_byte_stream ($$$$;$) { Line 414  sub parse_byte_stream ($$$$;$) {
414    };    };
415    $self->{parse_error} = $onerror; # updated later by parse_char_string    $self->{parse_error} = $onerror; # updated later by parse_char_string
416    
417      my $get_wrapper = $_[3] || sub ($) {
418        return $_[0]; # $_[0] = byte stream handle, returned = arg to char handle
419      };
420    
421    ## HTML5 encoding sniffing algorithm    ## HTML5 encoding sniffing algorithm
422    require Message::Charset::Info;    require Message::Charset::Info;
423    my $charset;    my $charset;
# Line 372  sub parse_byte_stream ($$$$;$) { Line 425  sub parse_byte_stream ($$$$;$) {
425    my ($char_stream, $e_status);    my ($char_stream, $e_status);
426    
427    SNIFFING: {    SNIFFING: {
428        ## NOTE: By setting |allow_fallback| option true when the
429        ## |get_decode_handle| method is invoked, we ignore what the HTML5
430        ## spec requires, i.e. unsupported encoding should be ignored.
431          ## TODO: We should not do this unless the parser is invoked
432          ## in the conformance checking mode, in which this behavior
433          ## would be useful.
434    
435      ## Step 1      ## Step 1
436      if (defined $charset_name) {      if (defined $charset_name) {
437        $charset = Message::Charset::Info->get_by_iana_name ($charset_name);        $charset = Message::Charset::Info->get_by_html_name ($charset_name);
438              ## TODO: Is this ok?  Transfer protocol's parameter should be
439              ## interpreted in its semantics?
440    
       ## ISSUE: Unsupported encoding is not ignored according to the spec.  
441        ($char_stream, $e_status) = $charset->get_decode_handle        ($char_stream, $e_status) = $charset->get_decode_handle
442            ($byte_stream, allow_error_reporting => 1,            ($byte_stream, allow_error_reporting => 1,
443             allow_fallback => 1);             allow_fallback => 1);
# Line 385  sub parse_byte_stream ($$$$;$) { Line 445  sub parse_byte_stream ($$$$;$) {
445          $self->{confident} = 1;          $self->{confident} = 1;
446          last SNIFFING;          last SNIFFING;
447        } else {        } else {
448          ## TODO: unsupported error          !!!parse-error (type => 'charset:not supported',
449                            layer => 'encode',
450                            line => 1, column => 1,
451                            value => $charset_name,
452                            level => $self->{level}->{uncertain});
453        }        }
454      }      }
455    
# Line 399  sub parse_byte_stream ($$$$;$) { Line 463  sub parse_byte_stream ($$$$;$) {
463    
464      ## Step 3      ## Step 3
465      if ($byte_buffer =~ /^\xFE\xFF/) {      if ($byte_buffer =~ /^\xFE\xFF/) {
466        $charset = Message::Charset::Info->get_by_iana_name ('utf-16be');        $charset = Message::Charset::Info->get_by_html_name ('utf-16be');
467        ($char_stream, $e_status) = $charset->get_decode_handle        ($char_stream, $e_status) = $charset->get_decode_handle
468            ($byte_stream, allow_error_reporting => 1,            ($byte_stream, allow_error_reporting => 1,
469             allow_fallback => 1, byte_buffer => \$byte_buffer);             allow_fallback => 1, byte_buffer => \$byte_buffer);
470        $self->{confident} = 1;        $self->{confident} = 1;
471        last SNIFFING;        last SNIFFING;
472      } elsif ($byte_buffer =~ /^\xFF\xFE/) {      } elsif ($byte_buffer =~ /^\xFF\xFE/) {
473        $charset = Message::Charset::Info->get_by_iana_name ('utf-16le');        $charset = Message::Charset::Info->get_by_html_name ('utf-16le');
474        ($char_stream, $e_status) = $charset->get_decode_handle        ($char_stream, $e_status) = $charset->get_decode_handle
475            ($byte_stream, allow_error_reporting => 1,            ($byte_stream, allow_error_reporting => 1,
476             allow_fallback => 1, byte_buffer => \$byte_buffer);             allow_fallback => 1, byte_buffer => \$byte_buffer);
477        $self->{confident} = 1;        $self->{confident} = 1;
478        last SNIFFING;        last SNIFFING;
479      } elsif ($byte_buffer =~ /^\xEF\xBB\xBF/) {      } elsif ($byte_buffer =~ /^\xEF\xBB\xBF/) {
480        $charset = Message::Charset::Info->get_by_iana_name ('utf-8');        $charset = Message::Charset::Info->get_by_html_name ('utf-8');
481        ($char_stream, $e_status) = $charset->get_decode_handle        ($char_stream, $e_status) = $charset->get_decode_handle
482            ($byte_stream, allow_error_reporting => 1,            ($byte_stream, allow_error_reporting => 1,
483             allow_fallback => 1, byte_buffer => \$byte_buffer);             allow_fallback => 1, byte_buffer => \$byte_buffer);
# Line 432  sub parse_byte_stream ($$$$;$) { Line 496  sub parse_byte_stream ($$$$;$) {
496      $charset_name = Whatpm::Charset::UniversalCharDet->detect_byte_string      $charset_name = Whatpm::Charset::UniversalCharDet->detect_byte_string
497          ($byte_buffer);          ($byte_buffer);
498      if (defined $charset_name) {      if (defined $charset_name) {
499        $charset = Message::Charset::Info->get_by_iana_name ($charset_name);        $charset = Message::Charset::Info->get_by_html_name ($charset_name);
500    
       ## ISSUE: Unsupported encoding is not ignored according to the spec.  
501        require Whatpm::Charset::DecodeHandle;        require Whatpm::Charset::DecodeHandle;
502        $buffer = Whatpm::Charset::DecodeHandle::ByteBuffer->new        $buffer = Whatpm::Charset::DecodeHandle::ByteBuffer->new
503            ($byte_stream);            ($byte_stream);
# Line 455  sub parse_byte_stream ($$$$;$) { Line 518  sub parse_byte_stream ($$$$;$) {
518    
519      ## Step 7: default      ## Step 7: default
520      ## TODO: Make this configurable.      ## TODO: Make this configurable.
521      $charset = Message::Charset::Info->get_by_iana_name ('windows-1252');      $charset = Message::Charset::Info->get_by_html_name ('windows-1252');
522          ## NOTE: We choose |windows-1252| here, since |utf-8| should be          ## NOTE: We choose |windows-1252| here, since |utf-8| should be
523          ## detectable in the step 6.          ## detectable in the step 6.
524      require Whatpm::Charset::DecodeHandle;      require Whatpm::Charset::DecodeHandle;
# Line 475  sub parse_byte_stream ($$$$;$) { Line 538  sub parse_byte_stream ($$$$;$) {
538      $self->{confident} = 0;      $self->{confident} = 0;
539    } # SNIFFING    } # SNIFFING
540    
   $self->{input_encoding} = $charset->get_iana_name;  
541    if ($e_status & Message::Charset::Info::FALLBACK_ENCODING_IMPL ()) {    if ($e_status & Message::Charset::Info::FALLBACK_ENCODING_IMPL ()) {
542        $self->{input_encoding} = $charset->get_iana_name; ## TODO: Should we set actual charset decoder's encoding name?
543      !!!parse-error (type => 'chardecode:fallback',      !!!parse-error (type => 'chardecode:fallback',
544                      text => $self->{input_encoding},                      #text => $self->{input_encoding},
545                      level => $self->{level}->{uncertain},                      level => $self->{level}->{uncertain},
546                      line => 1, column => 1,                      line => 1, column => 1,
547                      layer => 'encode');                      layer => 'encode');
548    } elsif (not ($e_status &    } elsif (not ($e_status &
549                  Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL())) {                  Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL ())) {
550        $self->{input_encoding} = $charset->get_iana_name;
551      !!!parse-error (type => 'chardecode:no error',      !!!parse-error (type => 'chardecode:no error',
552                      text => $self->{input_encoding},                      text => $self->{input_encoding},
553                      level => $self->{level}->{uncertain},                      level => $self->{level}->{uncertain},
554                      line => 1, column => 1,                      line => 1, column => 1,
555                      layer => 'encode');                      layer => 'encode');
556      } else {
557        $self->{input_encoding} = $charset->get_iana_name;
558    }    }
559    
560    $self->{change_encoding} = sub {    $self->{change_encoding} = sub {
# Line 496  sub parse_byte_stream ($$$$;$) { Line 562  sub parse_byte_stream ($$$$;$) {
562      $charset_name = shift;      $charset_name = shift;
563      my $token = shift;      my $token = shift;
564    
565      $charset = Message::Charset::Info->get_by_iana_name ($charset_name);      $charset = Message::Charset::Info->get_by_html_name ($charset_name);
566      ($char_stream, $e_status) = $charset->get_decode_handle      ($char_stream, $e_status) = $charset->get_decode_handle
567          ($byte_stream, allow_error_reporting => 1, allow_fallback => 1,          ($byte_stream, allow_error_reporting => 1, allow_fallback => 1,
568           byte_buffer => \ $buffer->{buffer});           byte_buffer => \ $buffer->{buffer});
# Line 507  sub parse_byte_stream ($$$$;$) { Line 573  sub parse_byte_stream ($$$$;$) {
573        ## Step 1            ## Step 1    
574        if ($charset->{category} &        if ($charset->{category} &
575            Message::Charset::Info::CHARSET_CATEGORY_UTF16 ()) {            Message::Charset::Info::CHARSET_CATEGORY_UTF16 ()) {
576          $charset = Message::Charset::Info->get_by_iana_name ('utf-8');          $charset = Message::Charset::Info->get_by_html_name ('utf-8');
577          ($char_stream, $e_status) = $charset->get_decode_handle          ($char_stream, $e_status) = $charset->get_decode_handle
578              ($byte_stream,              ($byte_stream,
579               byte_buffer => \ $buffer->{buffer});               byte_buffer => \ $buffer->{buffer});
# Line 545  sub parse_byte_stream ($$$$;$) { Line 611  sub parse_byte_stream ($$$$;$) {
611    my $char_onerror = sub {    my $char_onerror = sub {
612      my (undef, $type, %opt) = @_;      my (undef, $type, %opt) = @_;
613      !!!parse-error (layer => 'encode',      !!!parse-error (layer => 'encode',
614                      %opt, type => $type,                      line => $self->{line}, column => $self->{column} + 1,
615                      line => $self->{line}, column => $self->{column} + 1);                      %opt, type => $type);
616      if ($opt{octets}) {      if ($opt{octets}) {
617        ${$opt{octets}} = "\x{FFFD}"; # relacement character        ${$opt{octets}} = "\x{FFFD}"; # relacement character
618      }      }
619    };    };
   $char_stream->onerror ($char_onerror);  
620    
621    my @args = @_; shift @args; # $s    my $wrapped_char_stream = $get_wrapper->($char_stream);
622      $wrapped_char_stream->onerror ($char_onerror);
623    
624      my @args = ($_[1], $_[2]); # $doc, $onerror - $get_wrapper = undef;
625    my $return;    my $return;
626    try {    try {
627      $return = $self->parse_char_stream ($char_stream, @args);        $return = $self->parse_char_stream ($wrapped_char_stream, @args);  
628    } catch Whatpm::HTML::RestartParser with {    } catch Whatpm::HTML::RestartParser with {
629      ## NOTE: Invoked after {change_encoding}.      ## NOTE: Invoked after {change_encoding}.
630    
     $self->{input_encoding} = $charset->get_iana_name;  
631      if ($e_status & Message::Charset::Info::FALLBACK_ENCODING_IMPL ()) {      if ($e_status & Message::Charset::Info::FALLBACK_ENCODING_IMPL ()) {
632          $self->{input_encoding} = $charset->get_iana_name; ## TODO: Should we set actual charset decoder's encoding name?
633        !!!parse-error (type => 'chardecode:fallback',        !!!parse-error (type => 'chardecode:fallback',
                       text => $self->{input_encoding},  
634                        level => $self->{level}->{uncertain},                        level => $self->{level}->{uncertain},
635                          #text => $self->{input_encoding},
636                        line => 1, column => 1,                        line => 1, column => 1,
637                        layer => 'encode');                        layer => 'encode');
638      } elsif (not ($e_status &      } elsif (not ($e_status &
639                    Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL())) {                    Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL ())) {
640          $self->{input_encoding} = $charset->get_iana_name;
641        !!!parse-error (type => 'chardecode:no error',        !!!parse-error (type => 'chardecode:no error',
642                        text => $self->{input_encoding},                        text => $self->{input_encoding},
643                        level => $self->{level}->{uncertain},                        level => $self->{level}->{uncertain},
644                        line => 1, column => 1,                        line => 1, column => 1,
645                        layer => 'encode');                        layer => 'encode');
646        } else {
647          $self->{input_encoding} = $charset->get_iana_name;
648      }      }
649      $self->{confident} = 1;      $self->{confident} = 1;
650      $char_stream->onerror ($char_onerror);  
651      $return = $self->parse_char_stream ($char_stream, @args);      $wrapped_char_stream = $get_wrapper->($char_stream);
652        $wrapped_char_stream->onerror ($char_onerror);
653    
654        $return = $self->parse_char_stream ($wrapped_char_stream, @args);
655    };    };
656    return $return;    return $return;
657  } # parse_byte_stream  } # parse_byte_stream
# Line 591  sub parse_byte_stream ($$$$;$) { Line 665  sub parse_byte_stream ($$$$;$) {
665  ## such as |parse_byte_string| in this module, must ensure that it does  ## such as |parse_byte_string| in this module, must ensure that it does
666  ## strip the BOM and never strip any ZWNBSP.  ## strip the BOM and never strip any ZWNBSP.
667    
668  sub parse_char_string ($$$;$) {  sub parse_char_string ($$$;$$) {
669      #my ($self, $s, $doc, $onerror, $get_wrapper) = @_;
670    my $self = shift;    my $self = shift;
   require utf8;  
671    my $s = ref $_[0] ? $_[0] : \($_[0]);    my $s = ref $_[0] ? $_[0] : \($_[0]);
672    open my $input, '<' . (utf8::is_utf8 ($$s) ? ':utf8' : ''), $s;    require Whatpm::Charset::DecodeHandle;
673      my $input = Whatpm::Charset::DecodeHandle::CharString->new ($s);
674    return $self->parse_char_stream ($input, @_[1..$#_]);    return $self->parse_char_stream ($input, @_[1..$#_]);
675  } # parse_char_string  } # parse_char_string
676  *parse_string = \&parse_char_string;  *parse_string = \&parse_char_string; ## NOTE: Alias for backward compatibility.
677    
678  sub parse_char_stream ($$$;$) {  sub parse_char_stream ($$$;$$) {
679    my $self = ref $_[0] ? shift : shift->new;    my $self = ref $_[0] ? shift : shift->new;
680    my $input = $_[0];    my $input = $_[0];
681    $self->{document} = $_[1];    $self->{document} = $_[1];
# Line 611  sub parse_char_stream ($$$;$) { Line 686  sub parse_char_stream ($$$;$) {
686    $self->{confident} = 1 unless exists $self->{confident};    $self->{confident} = 1 unless exists $self->{confident};
687    $self->{document}->input_encoding ($self->{input_encoding})    $self->{document}->input_encoding ($self->{input_encoding})
688        if defined $self->{input_encoding};        if defined $self->{input_encoding};
689    ## TODO: |{input_encoding}| is needless?
690    
   my $i = 0;  
691    $self->{line_prev} = $self->{line} = 1;    $self->{line_prev} = $self->{line} = 1;
692    $self->{column_prev} = $self->{column} = 0;    $self->{column_prev} = -1;
693    $self->{set_next_char} = sub {    $self->{column} = 0;
694      $self->{set_nc} = sub {
695      my $self = shift;      my $self = shift;
696    
697      pop @{$self->{prev_char}};      my $char = '';
698      unshift @{$self->{prev_char}}, $self->{next_char};      if (defined $self->{next_nc}) {
699          $char = $self->{next_nc};
700      my $char;        delete $self->{next_nc};
701      if (defined $self->{next_next_char}) {        $self->{nc} = ord $char;
       $char = $self->{next_next_char};  
       delete $self->{next_next_char};  
702      } else {      } else {
703        $char = $input->getc;        $self->{char_buffer} = '';
704          $self->{char_buffer_pos} = 0;
705    
706          my $count = $input->manakai_read_until
707             ($self->{char_buffer}, qr/[^\x00\x0A\x0D]/, $self->{char_buffer_pos});
708          if ($count) {
709            $self->{line_prev} = $self->{line};
710            $self->{column_prev} = $self->{column};
711            $self->{column}++;
712            $self->{nc}
713                = ord substr ($self->{char_buffer}, $self->{char_buffer_pos}++, 1);
714            return;
715          }
716    
717          if ($input->read ($char, 1)) {
718            $self->{nc} = ord $char;
719          } else {
720            $self->{nc} = -1;
721            return;
722          }
723      }      }
     $self->{next_char} = -1 and return unless defined $char;  
     $self->{next_char} = ord $char;  
724    
725      ($self->{line_prev}, $self->{column_prev})      ($self->{line_prev}, $self->{column_prev})
726          = ($self->{line}, $self->{column});          = ($self->{line}, $self->{column});
727      $self->{column}++;      $self->{column}++;
728            
729      if ($self->{next_char} == 0x000A) { # LF      if ($self->{nc} == 0x000A) { # LF
730        !!!cp ('j1');        !!!cp ('j1');
731        $self->{line}++;        $self->{line}++;
732        $self->{column} = 0;        $self->{column} = 0;
733      } elsif ($self->{next_char} == 0x000D) { # CR      } elsif ($self->{nc} == 0x000D) { # CR
734        !!!cp ('j2');        !!!cp ('j2');
735        my $next = $input->getc;  ## TODO: support for abort/streaming
736        if (defined $next and $next ne "\x0A") {        my $next = '';
737          $self->{next_next_char} = $next;        if ($input->read ($next, 1) and $next ne "\x0A") {
738            $self->{next_nc} = $next;
739        }        }
740        $self->{next_char} = 0x000A; # LF # MUST        $self->{nc} = 0x000A; # LF # MUST
741        $self->{line}++;        $self->{line}++;
742        $self->{column} = 0;        $self->{column} = 0;
743      } elsif ($self->{next_char} > 0x10FFFF) {      } elsif ($self->{nc} == 0x0000) { # NULL
       !!!cp ('j3');  
       $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST  
     } elsif ($self->{next_char} == 0x0000) { # NULL  
744        !!!cp ('j4');        !!!cp ('j4');
745        !!!parse-error (type => 'NULL');        !!!parse-error (type => 'NULL');
746        $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST        $self->{nc} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
747      } elsif ($self->{next_char} <= 0x0008 or      }
748               (0x000E <= $self->{next_char} and $self->{next_char} <= 0x001F) or    };
749               (0x007F <= $self->{next_char} and $self->{next_char} <= 0x009F) or  
750               (0xD800 <= $self->{next_char} and $self->{next_char} <= 0xDFFF) or    $self->{read_until} = sub {
751               (0xFDD0 <= $self->{next_char} and $self->{next_char} <= 0xFDDF) or      #my ($scalar, $specials_range, $offset) = @_;
752               {      return 0 if defined $self->{next_nc};
753                0xFFFE => 1, 0xFFFF => 1, 0x1FFFE => 1, 0x1FFFF => 1,  
754                0x2FFFE => 1, 0x2FFFF => 1, 0x3FFFE => 1, 0x3FFFF => 1,      my $pattern = qr/[^$_[1]\x00\x0A\x0D]/;
755                0x4FFFE => 1, 0x4FFFF => 1, 0x5FFFE => 1, 0x5FFFF => 1,      my $offset = $_[2] || 0;
756                0x6FFFE => 1, 0x6FFFF => 1, 0x7FFFE => 1, 0x7FFFF => 1,  
757                0x8FFFE => 1, 0x8FFFF => 1, 0x9FFFE => 1, 0x9FFFF => 1,      if ($self->{char_buffer_pos} < length $self->{char_buffer}) {
758                0xAFFFE => 1, 0xAFFFF => 1, 0xBFFFE => 1, 0xBFFFF => 1,        pos ($self->{char_buffer}) = $self->{char_buffer_pos};
759                0xCFFFE => 1, 0xCFFFF => 1, 0xDFFFE => 1, 0xDFFFF => 1,        if ($self->{char_buffer} =~ /\G(?>$pattern)+/) {
760                0xEFFFE => 1, 0xEFFFF => 1, 0xFFFFE => 1, 0xFFFFF => 1,          substr ($_[0], $offset)
761                0x10FFFE => 1, 0x10FFFF => 1,              = substr ($self->{char_buffer}, $-[0], $+[0] - $-[0]);
762               }->{$self->{next_char}}) {          my $count = $+[0] - $-[0];
763        !!!cp ('j5');          if ($count) {
764        if ($self->{next_char} < 0x10000) {            $self->{column} += $count;
765          !!!parse-error (type => 'control char',            $self->{char_buffer_pos} += $count;
766                          text => (sprintf 'U+%04X', $self->{next_char}));            $self->{line_prev} = $self->{line};
767              $self->{column_prev} = $self->{column} - 1;
768              $self->{nc} = -1;
769            }
770            return $count;
771        } else {        } else {
772          !!!parse-error (type => 'control char',          return 0;
                         text => (sprintf 'U-%08X', $self->{next_char}));  
773        }        }
774        } else {
775          my $count = $input->manakai_read_until ($_[0], $pattern, $_[2]);
776          if ($count) {
777            $self->{column} += $count;
778            $self->{line_prev} = $self->{line};
779            $self->{column_prev} = $self->{column} - 1;
780            $self->{nc} = -1;
781          }
782          return $count;
783      }      }
784    };    }; # $self->{read_until}
   $self->{prev_char} = [-1, -1, -1];  
   $self->{next_char} = -1;  
785    
786    my $onerror = $_[2] || sub {    my $onerror = $_[2] || sub {
787      my (%opt) = @_;      my (%opt) = @_;
# Line 694  sub parse_char_stream ($$$;$) { Line 793  sub parse_char_stream ($$$;$) {
793      $onerror->(line => $self->{line}, column => $self->{column}, @_);      $onerror->(line => $self->{line}, column => $self->{column}, @_);
794    };    };
795    
796      my $char_onerror = sub {
797        my (undef, $type, %opt) = @_;
798        !!!parse-error (layer => 'encode',
799                        line => $self->{line}, column => $self->{column} + 1,
800                        %opt, type => $type);
801      }; # $char_onerror
802    
803      if ($_[3]) {
804        $input = $_[3]->($input);
805        $input->onerror ($char_onerror);
806      } else {
807        $input->onerror ($char_onerror) unless defined $input->onerror;
808      }
809    
810    $self->_initialize_tokenizer;    $self->_initialize_tokenizer;
811    $self->_initialize_tree_constructor;    $self->_initialize_tree_constructor;
812    $self->_construct_tree;    $self->_construct_tree;
# Line 708  sub new ($) { Line 821  sub new ($) {
821    my $class = shift;    my $class = shift;
822    my $self = bless {    my $self = bless {
823      level => {must => 'm',      level => {must => 'm',
824                  should => 's',
825                warn => 'w',                warn => 'w',
826                info => 'i',                info => 'i',
827                uncertain => 'u'},                uncertain => 'u'},
828    }, $class;    }, $class;
829    $self->{set_next_char} = sub {    $self->{set_nc} = sub {
830      $self->{next_char} = -1;      $self->{nc} = -1;
831    };    };
832    $self->{parse_error} = sub {    $self->{parse_error} = sub {
833      #      #
# Line 740  sub RCDATA_CONTENT_MODEL () { CM_ENTITY Line 854  sub RCDATA_CONTENT_MODEL () { CM_ENTITY
854  sub PCDATA_CONTENT_MODEL () { CM_ENTITY | CM_FULL_MARKUP }  sub PCDATA_CONTENT_MODEL () { CM_ENTITY | CM_FULL_MARKUP }
855    
856  sub DATA_STATE () { 0 }  sub DATA_STATE () { 0 }
857  sub ENTITY_DATA_STATE () { 1 }  #sub ENTITY_DATA_STATE () { 1 }
858  sub TAG_OPEN_STATE () { 2 }  sub TAG_OPEN_STATE () { 2 }
859  sub CLOSE_TAG_OPEN_STATE () { 3 }  sub CLOSE_TAG_OPEN_STATE () { 3 }
860  sub TAG_NAME_STATE () { 4 }  sub TAG_NAME_STATE () { 4 }
# Line 751  sub BEFORE_ATTRIBUTE_VALUE_STATE () { 8 Line 865  sub BEFORE_ATTRIBUTE_VALUE_STATE () { 8
865  sub ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE () { 9 }  sub ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE () { 9 }
866  sub ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE () { 10 }  sub ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE () { 10 }
867  sub ATTRIBUTE_VALUE_UNQUOTED_STATE () { 11 }  sub ATTRIBUTE_VALUE_UNQUOTED_STATE () { 11 }
868  sub ENTITY_IN_ATTRIBUTE_VALUE_STATE () { 12 }  #sub ENTITY_IN_ATTRIBUTE_VALUE_STATE () { 12 }
869  sub MARKUP_DECLARATION_OPEN_STATE () { 13 }  sub MARKUP_DECLARATION_OPEN_STATE () { 13 }
870  sub COMMENT_START_STATE () { 14 }  sub COMMENT_START_STATE () { 14 }
871  sub COMMENT_START_DASH_STATE () { 15 }  sub COMMENT_START_DASH_STATE () { 15 }
# Line 774  sub AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STAT Line 888  sub AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STAT
888  sub BOGUS_DOCTYPE_STATE () { 32 }  sub BOGUS_DOCTYPE_STATE () { 32 }
889  sub AFTER_ATTRIBUTE_VALUE_QUOTED_STATE () { 33 }  sub AFTER_ATTRIBUTE_VALUE_QUOTED_STATE () { 33 }
890  sub SELF_CLOSING_START_TAG_STATE () { 34 }  sub SELF_CLOSING_START_TAG_STATE () { 34 }
891  sub CDATA_BLOCK_STATE () { 35 }  sub CDATA_SECTION_STATE () { 35 }
892    sub MD_HYPHEN_STATE () { 36 } # "markup declaration open state" in the spec
893    sub MD_DOCTYPE_STATE () { 37 } # "markup declaration open state" in the spec
894    sub MD_CDATA_STATE () { 38 } # "markup declaration open state" in the spec
895    sub CDATA_RCDATA_CLOSE_TAG_STATE () { 39 } # "close tag open state" in the spec
896    sub CDATA_SECTION_MSE1_STATE () { 40 } # "CDATA section state" in the spec
897    sub CDATA_SECTION_MSE2_STATE () { 41 } # "CDATA section state" in the spec
898    sub PUBLIC_STATE () { 42 } # "after DOCTYPE name state" in the spec
899    sub SYSTEM_STATE () { 43 } # "after DOCTYPE name state" in the spec
900    ## NOTE: "Entity data state", "entity in attribute value state", and
901    ## "consume a character reference" algorithm are jointly implemented
902    ## using the following six states:
903    sub ENTITY_STATE () { 44 }
904    sub ENTITY_HASH_STATE () { 45 }
905    sub NCR_NUM_STATE () { 46 }
906    sub HEXREF_X_STATE () { 47 }
907    sub HEXREF_HEX_STATE () { 48 }
908    sub ENTITY_NAME_STATE () { 49 }
909    sub PCDATA_STATE () { 50 } # "data state" in the spec
910    
911  sub DOCTYPE_TOKEN () { 1 }  sub DOCTYPE_TOKEN () { 1 }
912  sub COMMENT_TOKEN () { 2 }  sub COMMENT_TOKEN () { 2 }
# Line 796  sub IN_FOREIGN_CONTENT_IM () { 0b1000000 Line 928  sub IN_FOREIGN_CONTENT_IM () { 0b1000000
928      ## NOTE: "in foreign content" insertion mode is special; it is combined      ## NOTE: "in foreign content" insertion mode is special; it is combined
929      ## with the secondary insertion mode.  In this parser, they are stored      ## with the secondary insertion mode.  In this parser, they are stored
930      ## together in the bit-or'ed form.      ## together in the bit-or'ed form.
931    sub IN_CDATA_RCDATA_IM () { 0b1000000000000 }
932        ## NOTE: "in CDATA/RCDATA" insertion mode is also special; it is
933        ## combined with the original insertion mode.  In thie parser,
934        ## they are stored together in the bit-or'ed form.
935    
936  ## NOTE: "initial" and "before html" insertion modes have no constants.  ## NOTE: "initial" and "before html" insertion modes have no constants.
937    
# Line 827  sub IN_COLUMN_GROUP_IM () { 0b10 } Line 963  sub IN_COLUMN_GROUP_IM () { 0b10 }
963  sub _initialize_tokenizer ($) {  sub _initialize_tokenizer ($) {
964    my $self = shift;    my $self = shift;
965    $self->{state} = DATA_STATE; # MUST    $self->{state} = DATA_STATE; # MUST
966      #$self->{s_kwd}; # state keyword - initialized when used
967      #$self->{entity__value}; # initialized when used
968      #$self->{entity__match}; # initialized when used
969    $self->{content_model} = PCDATA_CONTENT_MODEL; # be    $self->{content_model} = PCDATA_CONTENT_MODEL; # be
970    undef $self->{current_token}; # start tag, end tag, comment, or DOCTYPE    undef $self->{ct}; # current token
971    undef $self->{current_attribute};    undef $self->{ca}; # current attribute
972    undef $self->{last_emitted_start_tag_name};    undef $self->{last_stag_name}; # last emitted start tag name
973    undef $self->{last_attribute_value_state};    #$self->{prev_state}; # initialized when used
974    delete $self->{self_closing};    delete $self->{self_closing};
975    $self->{char} = [];    $self->{char_buffer} = '';
976    # $self->{next_char}    $self->{char_buffer_pos} = 0;
977      $self->{nc} = -1; # next input character
978      #$self->{next_nc}
979    !!!next-input-character;    !!!next-input-character;
980    $self->{token} = [];    $self->{token} = [];
981    # $self->{escape}    # $self->{escape}
# Line 845  sub _initialize_tokenizer ($) { Line 986  sub _initialize_tokenizer ($) {
986  ##       CHARACTER_TOKEN, or END_OF_FILE_TOKEN  ##       CHARACTER_TOKEN, or END_OF_FILE_TOKEN
987  ##   ->{name} (DOCTYPE_TOKEN)  ##   ->{name} (DOCTYPE_TOKEN)
988  ##   ->{tag_name} (START_TAG_TOKEN, END_TAG_TOKEN)  ##   ->{tag_name} (START_TAG_TOKEN, END_TAG_TOKEN)
989  ##   ->{public_identifier} (DOCTYPE_TOKEN)  ##   ->{pubid} (DOCTYPE_TOKEN)
990  ##   ->{system_identifier} (DOCTYPE_TOKEN)  ##   ->{sysid} (DOCTYPE_TOKEN)
991  ##   ->{quirks} == 1 or 0 (DOCTYPE_TOKEN): "force-quirks" flag  ##   ->{quirks} == 1 or 0 (DOCTYPE_TOKEN): "force-quirks" flag
992  ##   ->{attributes} isa HASH (START_TAG_TOKEN, END_TAG_TOKEN)  ##   ->{attributes} isa HASH (START_TAG_TOKEN, END_TAG_TOKEN)
993  ##        ->{name}  ##        ->{name}
# Line 865  sub _initialize_tokenizer ($) { Line 1006  sub _initialize_tokenizer ($) {
1006  ## has completed loading.  If one has, then it MUST be executed  ## has completed loading.  If one has, then it MUST be executed
1007  ## and removed from the list.  ## and removed from the list.
1008    
1009  ## NOTE: HTML5 "Writing HTML documents" section, applied to  ## TODO: Polytheistic slash SHOULD NOT be used. (Applied only to atheists.)
1010  ## documents and not to user agents and conformance checkers,  ## (This requirement was dropped from HTML5 spec, unfortunately.)
1011  ## contains some requirements that are not detected by the  
1012  ## parsing algorithm:  my $is_space = {
1013  ## - Some requirements on character encoding declarations. ## TODO    0x0009 => 1, # CHARACTER TABULATION (HT)
1014  ## - "Elements MUST NOT contain content that their content model disallows."    0x000A => 1, # LINE FEED (LF)
1015  ##   ... Some are parse error, some are not (will be reported by c.c.).    #0x000B => 0, # LINE TABULATION (VT)
1016  ## - Polytheistic slash SHOULD NOT be used. (Applied only to atheists.) ## TODO    0x000C => 1, # FORM FEED (FF)
1017  ## - Text (in elements, attributes, and comments) SHOULD NOT contain    #0x000D => 1, # CARRIAGE RETURN (CR)
1018  ##   control characters other than space characters. ## TODO: (what is control character? C0, C1 and DEL?  Unicode control character?)    0x0020 => 1, # SPACE (SP)
1019    };
 ## TODO: HTML5 poses authors two SHOULD-level requirements that cannot  
 ## be detected by the HTML5 parsing algorithm:  
 ## - Text,  
1020    
1021  sub _get_next_token ($) {  sub _get_next_token ($) {
1022    my $self = shift;    my $self = shift;
1023    
1024    if ($self->{self_closing}) {    if ($self->{self_closing}) {
1025      !!!parse-error (type => 'nestc', token => $self->{current_token});      !!!parse-error (type => 'nestc', token => $self->{ct});
1026      ## NOTE: The |self_closing| flag is only set by start tag token.      ## NOTE: The |self_closing| flag is only set by start tag token.
1027      ## In addition, when a start tag token is emitted, it is always set to      ## In addition, when a start tag token is emitted, it is always set to
1028      ## |current_token|.      ## |ct|.
1029      delete $self->{self_closing};      delete $self->{self_closing};
1030    }    }
1031    
# Line 897  sub _get_next_token ($) { Line 1035  sub _get_next_token ($) {
1035    }    }
1036    
1037    A: {    A: {
1038      if ($self->{state} == DATA_STATE) {      if ($self->{state} == PCDATA_STATE) {
1039        if ($self->{next_char} == 0x0026) { # &        ## NOTE: Same as |DATA_STATE|, but only for |PCDATA| content model.
1040    
1041          if ($self->{nc} == 0x0026) { # &
1042            !!!cp (0.1);
1043            ## NOTE: In the spec, the tokenizer is switched to the
1044            ## "entity data state".  In this implementation, the tokenizer
1045            ## is switched to the |ENTITY_STATE|, which is an implementation
1046            ## of the "consume a character reference" algorithm.
1047            $self->{entity_add} = -1;
1048            $self->{prev_state} = DATA_STATE;
1049            $self->{state} = ENTITY_STATE;
1050            !!!next-input-character;
1051            redo A;
1052          } elsif ($self->{nc} == 0x003C) { # <
1053            !!!cp (0.2);
1054            $self->{state} = TAG_OPEN_STATE;
1055            !!!next-input-character;
1056            redo A;
1057          } elsif ($self->{nc} == -1) {
1058            !!!cp (0.3);
1059            !!!emit ({type => END_OF_FILE_TOKEN,
1060                      line => $self->{line}, column => $self->{column}});
1061            last A; ## TODO: ok?
1062          } else {
1063            !!!cp (0.4);
1064            #
1065          }
1066    
1067          # Anything else
1068          my $token = {type => CHARACTER_TOKEN,
1069                       data => chr $self->{nc},
1070                       line => $self->{line}, column => $self->{column},
1071                      };
1072          $self->{read_until}->($token->{data}, q[<&], length $token->{data});
1073    
1074          ## Stay in the state.
1075          !!!next-input-character;
1076          !!!emit ($token);
1077          redo A;
1078        } elsif ($self->{state} == DATA_STATE) {
1079          $self->{s_kwd} = '' unless defined $self->{s_kwd};
1080          if ($self->{nc} == 0x0026) { # &
1081            $self->{s_kwd} = '';
1082          if ($self->{content_model} & CM_ENTITY and # PCDATA | RCDATA          if ($self->{content_model} & CM_ENTITY and # PCDATA | RCDATA
1083              not $self->{escape}) {              not $self->{escape}) {
1084            !!!cp (1);            !!!cp (1);
1085            $self->{state} = ENTITY_DATA_STATE;            ## NOTE: In the spec, the tokenizer is switched to the
1086              ## "entity data state".  In this implementation, the tokenizer
1087              ## is switched to the |ENTITY_STATE|, which is an implementation
1088              ## of the "consume a character reference" algorithm.
1089              $self->{entity_add} = -1;
1090              $self->{prev_state} = DATA_STATE;
1091              $self->{state} = ENTITY_STATE;
1092            !!!next-input-character;            !!!next-input-character;
1093            redo A;            redo A;
1094          } else {          } else {
1095            !!!cp (2);            !!!cp (2);
1096            #            #
1097          }          }
1098        } elsif ($self->{next_char} == 0x002D) { # -        } elsif ($self->{nc} == 0x002D) { # -
1099          if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA          if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
1100            unless ($self->{escape}) {            $self->{s_kwd} .= '-';
1101              if ($self->{prev_char}->[0] == 0x002D and # -            
1102                  $self->{prev_char}->[1] == 0x0021 and # !            if ($self->{s_kwd} eq '<!--') {
1103                  $self->{prev_char}->[2] == 0x003C) { # <              !!!cp (3);
1104                !!!cp (3);              $self->{escape} = 1; # unless $self->{escape};
1105                $self->{escape} = 1;              $self->{s_kwd} = '--';
1106              } else {              #
1107                !!!cp (4);            } elsif ($self->{s_kwd} eq '---') {
1108              }              !!!cp (4);
1109                $self->{s_kwd} = '--';
1110                #
1111            } else {            } else {
1112              !!!cp (5);              !!!cp (5);
1113                #
1114            }            }
1115          }          }
1116                    
1117          #          #
1118        } elsif ($self->{next_char} == 0x003C) { # <        } elsif ($self->{nc} == 0x0021) { # !
1119            if (length $self->{s_kwd}) {
1120              !!!cp (5.1);
1121              $self->{s_kwd} .= '!';
1122              #
1123            } else {
1124              !!!cp (5.2);
1125              #$self->{s_kwd} = '';
1126              #
1127            }
1128            #
1129          } elsif ($self->{nc} == 0x003C) { # <
1130          if ($self->{content_model} & CM_FULL_MARKUP or # PCDATA          if ($self->{content_model} & CM_FULL_MARKUP or # PCDATA
1131              (($self->{content_model} & CM_LIMITED_MARKUP) and # CDATA | RCDATA              (($self->{content_model} & CM_LIMITED_MARKUP) and # CDATA | RCDATA
1132               not $self->{escape})) {               not $self->{escape})) {
# Line 936  sub _get_next_token ($) { Line 1136  sub _get_next_token ($) {
1136            redo A;            redo A;
1137          } else {          } else {
1138            !!!cp (7);            !!!cp (7);
1139              $self->{s_kwd} = '';
1140            #            #
1141          }          }
1142        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1143          if ($self->{escape} and          if ($self->{escape} and
1144              ($self->{content_model} & CM_LIMITED_MARKUP)) { # RCDATA | CDATA              ($self->{content_model} & CM_LIMITED_MARKUP)) { # RCDATA | CDATA
1145            if ($self->{prev_char}->[0] == 0x002D and # -            if ($self->{s_kwd} eq '--') {
               $self->{prev_char}->[1] == 0x002D) { # -  
1146              !!!cp (8);              !!!cp (8);
1147              delete $self->{escape};              delete $self->{escape};
1148            } else {            } else {
# Line 952  sub _get_next_token ($) { Line 1152  sub _get_next_token ($) {
1152            !!!cp (10);            !!!cp (10);
1153          }          }
1154                    
1155            $self->{s_kwd} = '';
1156          #          #
1157        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1158          !!!cp (11);          !!!cp (11);
1159            $self->{s_kwd} = '';
1160          !!!emit ({type => END_OF_FILE_TOKEN,          !!!emit ({type => END_OF_FILE_TOKEN,
1161                    line => $self->{line}, column => $self->{column}});                    line => $self->{line}, column => $self->{column}});
1162          last A; ## TODO: ok?          last A; ## TODO: ok?
1163        } else {        } else {
1164          !!!cp (12);          !!!cp (12);
1165            $self->{s_kwd} = '';
1166            #
1167        }        }
1168    
1169        # Anything else        # Anything else
1170        my $token = {type => CHARACTER_TOKEN,        my $token = {type => CHARACTER_TOKEN,
1171                     data => chr $self->{next_char},                     data => chr $self->{nc},
1172                     line => $self->{line}, column => $self->{column},                     line => $self->{line}, column => $self->{column},
1173                    };                    };
1174        ## Stay in the data state        if ($self->{read_until}->($token->{data}, q[-!<>&],
1175        !!!next-input-character;                                  length $token->{data})) {
1176            $self->{s_kwd} = '';
1177        !!!emit ($token);        }
   
       redo A;  
     } elsif ($self->{state} == ENTITY_DATA_STATE) {  
       ## (cannot happen in CDATA state)  
   
       my ($l, $c) = ($self->{line_prev}, $self->{column_prev});  
         
       my $token = $self->_tokenize_attempt_to_consume_an_entity (0, -1);  
   
       $self->{state} = DATA_STATE;  
       # next-input-character is already done  
1178    
1179        unless (defined $token) {        ## Stay in the data state.
1180          if ($self->{content_model} == PCDATA_CONTENT_MODEL) {
1181          !!!cp (13);          !!!cp (13);
1182          !!!emit ({type => CHARACTER_TOKEN, data => '&',          $self->{state} = PCDATA_STATE;
                   line => $l, column => $c,  
                  });  
1183        } else {        } else {
1184          !!!cp (14);          !!!cp (14);
1185          !!!emit ($token);          ## Stay in the state.
1186        }        }
1187          !!!next-input-character;
1188          !!!emit ($token);
1189        redo A;        redo A;
1190      } elsif ($self->{state} == TAG_OPEN_STATE) {      } elsif ($self->{state} == TAG_OPEN_STATE) {
1191        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
1192          if ($self->{next_char} == 0x002F) { # /          if ($self->{nc} == 0x002F) { # /
1193            !!!cp (15);            !!!cp (15);
1194            !!!next-input-character;            !!!next-input-character;
1195            $self->{state} = CLOSE_TAG_OPEN_STATE;            $self->{state} = CLOSE_TAG_OPEN_STATE;
1196            redo A;            redo A;
1197            } elsif ($self->{nc} == 0x0021) { # !
1198              !!!cp (15.1);
1199              $self->{s_kwd} = '<' unless $self->{escape};
1200              #
1201          } else {          } else {
1202            !!!cp (16);            !!!cp (16);
1203            ## reconsume            #
           $self->{state} = DATA_STATE;  
   
           !!!emit ({type => CHARACTER_TOKEN, data => '<',  
                     line => $self->{line_prev},  
                     column => $self->{column_prev},  
                    });  
   
           redo A;  
1204          }          }
1205    
1206            ## reconsume
1207            $self->{state} = DATA_STATE;
1208            !!!emit ({type => CHARACTER_TOKEN, data => '<',
1209                      line => $self->{line_prev},
1210                      column => $self->{column_prev},
1211                     });
1212            redo A;
1213        } elsif ($self->{content_model} & CM_FULL_MARKUP) { # PCDATA        } elsif ($self->{content_model} & CM_FULL_MARKUP) { # PCDATA
1214          if ($self->{next_char} == 0x0021) { # !          if ($self->{nc} == 0x0021) { # !
1215            !!!cp (17);            !!!cp (17);
1216            $self->{state} = MARKUP_DECLARATION_OPEN_STATE;            $self->{state} = MARKUP_DECLARATION_OPEN_STATE;
1217            !!!next-input-character;            !!!next-input-character;
1218            redo A;            redo A;
1219          } elsif ($self->{next_char} == 0x002F) { # /          } elsif ($self->{nc} == 0x002F) { # /
1220            !!!cp (18);            !!!cp (18);
1221            $self->{state} = CLOSE_TAG_OPEN_STATE;            $self->{state} = CLOSE_TAG_OPEN_STATE;
1222            !!!next-input-character;            !!!next-input-character;
1223            redo A;            redo A;
1224          } elsif (0x0041 <= $self->{next_char} and          } elsif (0x0041 <= $self->{nc} and
1225                   $self->{next_char} <= 0x005A) { # A..Z                   $self->{nc} <= 0x005A) { # A..Z
1226            !!!cp (19);            !!!cp (19);
1227            $self->{current_token}            $self->{ct}
1228              = {type => START_TAG_TOKEN,              = {type => START_TAG_TOKEN,
1229                 tag_name => chr ($self->{next_char} + 0x0020),                 tag_name => chr ($self->{nc} + 0x0020),
1230                 line => $self->{line_prev},                 line => $self->{line_prev},
1231                 column => $self->{column_prev}};                 column => $self->{column_prev}};
1232            $self->{state} = TAG_NAME_STATE;            $self->{state} = TAG_NAME_STATE;
1233            !!!next-input-character;            !!!next-input-character;
1234            redo A;            redo A;
1235          } elsif (0x0061 <= $self->{next_char} and          } elsif (0x0061 <= $self->{nc} and
1236                   $self->{next_char} <= 0x007A) { # a..z                   $self->{nc} <= 0x007A) { # a..z
1237            !!!cp (20);            !!!cp (20);
1238            $self->{current_token} = {type => START_TAG_TOKEN,            $self->{ct} = {type => START_TAG_TOKEN,
1239                                      tag_name => chr ($self->{next_char}),                                      tag_name => chr ($self->{nc}),
1240                                      line => $self->{line_prev},                                      line => $self->{line_prev},
1241                                      column => $self->{column_prev}};                                      column => $self->{column_prev}};
1242            $self->{state} = TAG_NAME_STATE;            $self->{state} = TAG_NAME_STATE;
1243            !!!next-input-character;            !!!next-input-character;
1244            redo A;            redo A;
1245          } elsif ($self->{next_char} == 0x003E) { # >          } elsif ($self->{nc} == 0x003E) { # >
1246            !!!cp (21);            !!!cp (21);
1247            !!!parse-error (type => 'empty start tag',            !!!parse-error (type => 'empty start tag',
1248                            line => $self->{line_prev},                            line => $self->{line_prev},
# Line 1058  sub _get_next_token ($) { Line 1256  sub _get_next_token ($) {
1256                     });                     });
1257    
1258            redo A;            redo A;
1259          } elsif ($self->{next_char} == 0x003F) { # ?          } elsif ($self->{nc} == 0x003F) { # ?
1260            !!!cp (22);            !!!cp (22);
1261            !!!parse-error (type => 'pio',            !!!parse-error (type => 'pio',
1262                            line => $self->{line_prev},                            line => $self->{line_prev},
1263                            column => $self->{column_prev});                            column => $self->{column_prev});
1264            $self->{state} = BOGUS_COMMENT_STATE;            $self->{state} = BOGUS_COMMENT_STATE;
1265            $self->{current_token} = {type => COMMENT_TOKEN, data => '',            $self->{ct} = {type => COMMENT_TOKEN, data => '',
1266                                      line => $self->{line_prev},                                      line => $self->{line_prev},
1267                                      column => $self->{column_prev},                                      column => $self->{column_prev},
1268                                     };                                     };
1269            ## $self->{next_char} is intentionally left as is            ## $self->{nc} is intentionally left as is
1270            redo A;            redo A;
1271          } else {          } else {
1272            !!!cp (23);            !!!cp (23);
# Line 1089  sub _get_next_token ($) { Line 1287  sub _get_next_token ($) {
1287          die "$0: $self->{content_model} in tag open";          die "$0: $self->{content_model} in tag open";
1288        }        }
1289      } elsif ($self->{state} == CLOSE_TAG_OPEN_STATE) {      } elsif ($self->{state} == CLOSE_TAG_OPEN_STATE) {
1290          ## NOTE: The "close tag open state" in the spec is implemented as
1291          ## |CLOSE_TAG_OPEN_STATE| and |CDATA_RCDATA_CLOSE_TAG_STATE|.
1292    
1293        my ($l, $c) = ($self->{line_prev}, $self->{column_prev} - 1); # "<"of"</"        my ($l, $c) = ($self->{line_prev}, $self->{column_prev} - 1); # "<"of"</"
1294        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
1295          if (defined $self->{last_emitted_start_tag_name}) {          if (defined $self->{last_stag_name}) {
1296              $self->{state} = CDATA_RCDATA_CLOSE_TAG_STATE;
1297            ## NOTE: <http://krijnhoetmer.nl/irc-logs/whatwg/20070626#l-564>            $self->{s_kwd} = '';
1298            my @next_char;            ## Reconsume.
1299            TAGNAME: for (my $i = 0; $i < length $self->{last_emitted_start_tag_name}; $i++) {            redo A;
             push @next_char, $self->{next_char};  
             my $c = ord substr ($self->{last_emitted_start_tag_name}, $i, 1);  
             my $C = 0x0061 <= $c && $c <= 0x007A ? $c - 0x0020 : $c;  
             if ($self->{next_char} == $c or $self->{next_char} == $C) {  
               !!!cp (24);  
               !!!next-input-character;  
               next TAGNAME;  
             } else {  
               !!!cp (25);  
               $self->{next_char} = shift @next_char; # reconsume  
               !!!back-next-input-character (@next_char);  
               $self->{state} = DATA_STATE;  
   
               !!!emit ({type => CHARACTER_TOKEN, data => '</',  
                         line => $l, column => $c,  
                        });  
     
               redo A;  
             }  
           }  
           push @next_char, $self->{next_char};  
         
           unless ($self->{next_char} == 0x0009 or # HT  
                   $self->{next_char} == 0x000A or # LF  
                   $self->{next_char} == 0x000B or # VT  
                   $self->{next_char} == 0x000C or # FF  
                   $self->{next_char} == 0x0020 or # SP  
                   $self->{next_char} == 0x003E or # >  
                   $self->{next_char} == 0x002F or # /  
                   $self->{next_char} == -1) {  
             !!!cp (26);  
             $self->{next_char} = shift @next_char; # reconsume  
             !!!back-next-input-character (@next_char);  
             $self->{state} = DATA_STATE;  
             !!!emit ({type => CHARACTER_TOKEN, data => '</',  
                       line => $l, column => $c,  
                      });  
             redo A;  
           } else {  
             !!!cp (27);  
             $self->{next_char} = shift @next_char;  
             !!!back-next-input-character (@next_char);  
             # and consume...  
           }  
1300          } else {          } else {
1301            ## No start tag token has ever been emitted            ## No start tag token has ever been emitted
1302              ## NOTE: See <http://krijnhoetmer.nl/irc-logs/whatwg/20070626#l-564>.
1303            !!!cp (28);            !!!cp (28);
           # next-input-character is already done  
1304            $self->{state} = DATA_STATE;            $self->{state} = DATA_STATE;
1305              ## Reconsume.
1306            !!!emit ({type => CHARACTER_TOKEN, data => '</',            !!!emit ({type => CHARACTER_TOKEN, data => '</',
1307                      line => $l, column => $c,                      line => $l, column => $c,
1308                     });                     });
1309            redo A;            redo A;
1310          }          }
1311        }        }
1312          
1313        if (0x0041 <= $self->{next_char} and        if (0x0041 <= $self->{nc} and
1314            $self->{next_char} <= 0x005A) { # A..Z            $self->{nc} <= 0x005A) { # A..Z
1315          !!!cp (29);          !!!cp (29);
1316          $self->{current_token}          $self->{ct}
1317              = {type => END_TAG_TOKEN,              = {type => END_TAG_TOKEN,
1318                 tag_name => chr ($self->{next_char} + 0x0020),                 tag_name => chr ($self->{nc} + 0x0020),
1319                 line => $l, column => $c};                 line => $l, column => $c};
1320          $self->{state} = TAG_NAME_STATE;          $self->{state} = TAG_NAME_STATE;
1321          !!!next-input-character;          !!!next-input-character;
1322          redo A;          redo A;
1323        } elsif (0x0061 <= $self->{next_char} and        } elsif (0x0061 <= $self->{nc} and
1324                 $self->{next_char} <= 0x007A) { # a..z                 $self->{nc} <= 0x007A) { # a..z
1325          !!!cp (30);          !!!cp (30);
1326          $self->{current_token} = {type => END_TAG_TOKEN,          $self->{ct} = {type => END_TAG_TOKEN,
1327                                    tag_name => chr ($self->{next_char}),                                    tag_name => chr ($self->{nc}),
1328                                    line => $l, column => $c};                                    line => $l, column => $c};
1329          $self->{state} = TAG_NAME_STATE;          $self->{state} = TAG_NAME_STATE;
1330          !!!next-input-character;          !!!next-input-character;
1331          redo A;          redo A;
1332        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1333          !!!cp (31);          !!!cp (31);
1334          !!!parse-error (type => 'empty end tag',          !!!parse-error (type => 'empty end tag',
1335                          line => $self->{line_prev}, ## "<" in "</>"                          line => $self->{line_prev}, ## "<" in "</>"
# Line 1179  sub _get_next_token ($) { Line 1337  sub _get_next_token ($) {
1337          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1338          !!!next-input-character;          !!!next-input-character;
1339          redo A;          redo A;
1340        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1341          !!!cp (32);          !!!cp (32);
1342          !!!parse-error (type => 'bare etago');          !!!parse-error (type => 'bare etago');
1343          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
# Line 1194  sub _get_next_token ($) { Line 1352  sub _get_next_token ($) {
1352          !!!cp (33);          !!!cp (33);
1353          !!!parse-error (type => 'bogus end tag');          !!!parse-error (type => 'bogus end tag');
1354          $self->{state} = BOGUS_COMMENT_STATE;          $self->{state} = BOGUS_COMMENT_STATE;
1355          $self->{current_token} = {type => COMMENT_TOKEN, data => '',          $self->{ct} = {type => COMMENT_TOKEN, data => '',
1356                                    line => $self->{line_prev}, # "<" of "</"                                    line => $self->{line_prev}, # "<" of "</"
1357                                    column => $self->{column_prev} - 1,                                    column => $self->{column_prev} - 1,
1358                                   };                                   };
1359          ## $self->{next_char} is intentionally left as is          ## NOTE: $self->{nc} is intentionally left as is.
1360          redo A;          ## Although the "anything else" case of the spec not explicitly
1361            ## states that the next input character is to be reconsumed,
1362            ## it will be included to the |data| of the comment token
1363            ## generated from the bogus end tag, as defined in the
1364            ## "bogus comment state" entry.
1365            redo A;
1366          }
1367        } elsif ($self->{state} == CDATA_RCDATA_CLOSE_TAG_STATE) {
1368          my $ch = substr $self->{last_stag_name}, length $self->{s_kwd}, 1;
1369          if (length $ch) {
1370            my $CH = $ch;
1371            $ch =~ tr/a-z/A-Z/;
1372            my $nch = chr $self->{nc};
1373            if ($nch eq $ch or $nch eq $CH) {
1374              !!!cp (24);
1375              ## Stay in the state.
1376              $self->{s_kwd} .= $nch;
1377              !!!next-input-character;
1378              redo A;
1379            } else {
1380              !!!cp (25);
1381              $self->{state} = DATA_STATE;
1382              ## Reconsume.
1383              !!!emit ({type => CHARACTER_TOKEN,
1384                        data => '</' . $self->{s_kwd},
1385                        line => $self->{line_prev},
1386                        column => $self->{column_prev} - 1 - length $self->{s_kwd},
1387                       });
1388              redo A;
1389            }
1390          } else { # after "<{tag-name}"
1391            unless ($is_space->{$self->{nc}} or
1392                    {
1393                     0x003E => 1, # >
1394                     0x002F => 1, # /
1395                     -1 => 1, # EOF
1396                    }->{$self->{nc}}) {
1397              !!!cp (26);
1398              ## Reconsume.
1399              $self->{state} = DATA_STATE;
1400              !!!emit ({type => CHARACTER_TOKEN,
1401                        data => '</' . $self->{s_kwd},
1402                        line => $self->{line_prev},
1403                        column => $self->{column_prev} - 1 - length $self->{s_kwd},
1404                       });
1405              redo A;
1406            } else {
1407              !!!cp (27);
1408              $self->{ct}
1409                  = {type => END_TAG_TOKEN,
1410                     tag_name => $self->{last_stag_name},
1411                     line => $self->{line_prev},
1412                     column => $self->{column_prev} - 1 - length $self->{s_kwd}};
1413              $self->{state} = TAG_NAME_STATE;
1414              ## Reconsume.
1415              redo A;
1416            }
1417        }        }
1418      } elsif ($self->{state} == TAG_NAME_STATE) {      } elsif ($self->{state} == TAG_NAME_STATE) {
1419        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
1420          !!!cp (34);          !!!cp (34);
1421          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1422          !!!next-input-character;          !!!next-input-character;
1423          redo A;          redo A;
1424        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1425          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1426            !!!cp (35);            !!!cp (35);
1427            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1428          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1429            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1430            #if ($self->{current_token}->{attributes}) {            #if ($self->{ct}->{attributes}) {
1431            #  ## NOTE: This should never be reached.            #  ## NOTE: This should never be reached.
1432            #  !!! cp (36);            #  !!! cp (36);
1433            #  !!! parse-error (type => 'end tag attribute');            #  !!! parse-error (type => 'end tag attribute');
# Line 1225  sub _get_next_token ($) { Line 1435  sub _get_next_token ($) {
1435              !!!cp (37);              !!!cp (37);
1436            #}            #}
1437          } else {          } else {
1438            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1439          }          }
1440          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1441          !!!next-input-character;          !!!next-input-character;
1442    
1443          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1444    
1445          redo A;          redo A;
1446        } elsif (0x0041 <= $self->{next_char} and        } elsif (0x0041 <= $self->{nc} and
1447                 $self->{next_char} <= 0x005A) { # A..Z                 $self->{nc} <= 0x005A) { # A..Z
1448          !!!cp (38);          !!!cp (38);
1449          $self->{current_token}->{tag_name} .= chr ($self->{next_char} + 0x0020);          $self->{ct}->{tag_name} .= chr ($self->{nc} + 0x0020);
1450            # start tag or end tag            # start tag or end tag
1451          ## Stay in this state          ## Stay in this state
1452          !!!next-input-character;          !!!next-input-character;
1453          redo A;          redo A;
1454        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1455          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1456          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1457            !!!cp (39);            !!!cp (39);
1458            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1459          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1460            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1461            #if ($self->{current_token}->{attributes}) {            #if ($self->{ct}->{attributes}) {
1462            #  ## NOTE: This state should never be reached.            #  ## NOTE: This state should never be reached.
1463            #  !!! cp (40);            #  !!! cp (40);
1464            #  !!! parse-error (type => 'end tag attribute');            #  !!! parse-error (type => 'end tag attribute');
# Line 1256  sub _get_next_token ($) { Line 1466  sub _get_next_token ($) {
1466              !!!cp (41);              !!!cp (41);
1467            #}            #}
1468          } else {          } else {
1469            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1470          }          }
1471          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1472          # reconsume          # reconsume
1473    
1474          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1475    
1476          redo A;          redo A;
1477        } elsif ($self->{next_char} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
1478          !!!cp (42);          !!!cp (42);
1479          $self->{state} = SELF_CLOSING_START_TAG_STATE;          $self->{state} = SELF_CLOSING_START_TAG_STATE;
1480          !!!next-input-character;          !!!next-input-character;
1481          redo A;          redo A;
1482        } else {        } else {
1483          !!!cp (44);          !!!cp (44);
1484          $self->{current_token}->{tag_name} .= chr $self->{next_char};          $self->{ct}->{tag_name} .= chr $self->{nc};
1485            # start tag or end tag            # start tag or end tag
1486          ## Stay in the state          ## Stay in the state
1487          !!!next-input-character;          !!!next-input-character;
1488          redo A;          redo A;
1489        }        }
1490      } elsif ($self->{state} == BEFORE_ATTRIBUTE_NAME_STATE) {      } elsif ($self->{state} == BEFORE_ATTRIBUTE_NAME_STATE) {
1491        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
1492          !!!cp (45);          !!!cp (45);
1493          ## Stay in the state          ## Stay in the state
1494          !!!next-input-character;          !!!next-input-character;
1495          redo A;          redo A;
1496        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1497          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1498            !!!cp (46);            !!!cp (46);
1499            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1500          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1501            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1502            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1503              !!!cp (47);              !!!cp (47);
1504              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1505            } else {            } else {
1506              !!!cp (48);              !!!cp (48);
1507            }            }
1508          } else {          } else {
1509            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1510          }          }
1511          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1512          !!!next-input-character;          !!!next-input-character;
1513    
1514          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1515    
1516          redo A;          redo A;
1517        } elsif (0x0041 <= $self->{next_char} and        } elsif (0x0041 <= $self->{nc} and
1518                 $self->{next_char} <= 0x005A) { # A..Z                 $self->{nc} <= 0x005A) { # A..Z
1519          !!!cp (49);          !!!cp (49);
1520          $self->{current_attribute}          $self->{ca}
1521              = {name => chr ($self->{next_char} + 0x0020),              = {name => chr ($self->{nc} + 0x0020),
1522                 value => '',                 value => '',
1523                 line => $self->{line}, column => $self->{column}};                 line => $self->{line}, column => $self->{column}};
1524          $self->{state} = ATTRIBUTE_NAME_STATE;          $self->{state} = ATTRIBUTE_NAME_STATE;
1525          !!!next-input-character;          !!!next-input-character;
1526          redo A;          redo A;
1527        } elsif ($self->{next_char} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
1528          !!!cp (50);          !!!cp (50);
1529          $self->{state} = SELF_CLOSING_START_TAG_STATE;          $self->{state} = SELF_CLOSING_START_TAG_STATE;
1530          !!!next-input-character;          !!!next-input-character;
1531          redo A;          redo A;
1532        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1533          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1534          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1535            !!!cp (52);            !!!cp (52);
1536            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1537          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1538            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1539            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1540              !!!cp (53);              !!!cp (53);
1541              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1542            } else {            } else {
1543              !!!cp (54);              !!!cp (54);
1544            }            }
1545          } else {          } else {
1546            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1547          }          }
1548          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1549          # reconsume          # reconsume
1550    
1551          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1552    
1553          redo A;          redo A;
1554        } else {        } else {
# Line 1350  sub _get_next_token ($) { Line 1556  sub _get_next_token ($) {
1556               0x0022 => 1, # "               0x0022 => 1, # "
1557               0x0027 => 1, # '               0x0027 => 1, # '
1558               0x003D => 1, # =               0x003D => 1, # =
1559              }->{$self->{next_char}}) {              }->{$self->{nc}}) {
1560            !!!cp (55);            !!!cp (55);
1561            !!!parse-error (type => 'bad attribute name');            !!!parse-error (type => 'bad attribute name');
1562          } else {          } else {
1563            !!!cp (56);            !!!cp (56);
1564          }          }
1565          $self->{current_attribute}          $self->{ca}
1566              = {name => chr ($self->{next_char}),              = {name => chr ($self->{nc}),
1567                 value => '',                 value => '',
1568                 line => $self->{line}, column => $self->{column}};                 line => $self->{line}, column => $self->{column}};
1569          $self->{state} = ATTRIBUTE_NAME_STATE;          $self->{state} = ATTRIBUTE_NAME_STATE;
# Line 1366  sub _get_next_token ($) { Line 1572  sub _get_next_token ($) {
1572        }        }
1573      } elsif ($self->{state} == ATTRIBUTE_NAME_STATE) {      } elsif ($self->{state} == ATTRIBUTE_NAME_STATE) {
1574        my $before_leave = sub {        my $before_leave = sub {
1575          if (exists $self->{current_token}->{attributes} # start tag or end tag          if (exists $self->{ct}->{attributes} # start tag or end tag
1576              ->{$self->{current_attribute}->{name}}) { # MUST              ->{$self->{ca}->{name}}) { # MUST
1577            !!!cp (57);            !!!cp (57);
1578            !!!parse-error (type => 'duplicate attribute', text => $self->{current_attribute}->{name}, line => $self->{current_attribute}->{line}, column => $self->{current_attribute}->{column});            !!!parse-error (type => 'duplicate attribute', text => $self->{ca}->{name}, line => $self->{ca}->{line}, column => $self->{ca}->{column});
1579            ## Discard $self->{current_attribute} # MUST            ## Discard $self->{ca} # MUST
1580          } else {          } else {
1581            !!!cp (58);            !!!cp (58);
1582            $self->{current_token}->{attributes}->{$self->{current_attribute}->{name}}            $self->{ct}->{attributes}->{$self->{ca}->{name}}
1583              = $self->{current_attribute};              = $self->{ca};
1584          }          }
1585        }; # $before_leave        }; # $before_leave
1586    
1587        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
1588          !!!cp (59);          !!!cp (59);
1589          $before_leave->();          $before_leave->();
1590          $self->{state} = AFTER_ATTRIBUTE_NAME_STATE;          $self->{state} = AFTER_ATTRIBUTE_NAME_STATE;
1591          !!!next-input-character;          !!!next-input-character;
1592          redo A;          redo A;
1593        } elsif ($self->{next_char} == 0x003D) { # =        } elsif ($self->{nc} == 0x003D) { # =
1594          !!!cp (60);          !!!cp (60);
1595          $before_leave->();          $before_leave->();
1596          $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;          $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;
1597          !!!next-input-character;          !!!next-input-character;
1598          redo A;          redo A;
1599        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1600          $before_leave->();          $before_leave->();
1601          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1602            !!!cp (61);            !!!cp (61);
1603            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1604          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1605            !!!cp (62);            !!!cp (62);
1606            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1607            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1608              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1609            }            }
1610          } else {          } else {
1611            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1612          }          }
1613          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1614          !!!next-input-character;          !!!next-input-character;
1615    
1616          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1617    
1618          redo A;          redo A;
1619        } elsif (0x0041 <= $self->{next_char} and        } elsif (0x0041 <= $self->{nc} and
1620                 $self->{next_char} <= 0x005A) { # A..Z                 $self->{nc} <= 0x005A) { # A..Z
1621          !!!cp (63);          !!!cp (63);
1622          $self->{current_attribute}->{name} .= chr ($self->{next_char} + 0x0020);          $self->{ca}->{name} .= chr ($self->{nc} + 0x0020);
1623          ## Stay in the state          ## Stay in the state
1624          !!!next-input-character;          !!!next-input-character;
1625          redo A;          redo A;
1626        } elsif ($self->{next_char} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
1627          !!!cp (64);          !!!cp (64);
1628          $before_leave->();          $before_leave->();
1629          $self->{state} = SELF_CLOSING_START_TAG_STATE;          $self->{state} = SELF_CLOSING_START_TAG_STATE;
1630          !!!next-input-character;          !!!next-input-character;
1631          redo A;          redo A;
1632        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1633          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1634          $before_leave->();          $before_leave->();
1635          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1636            !!!cp (66);            !!!cp (66);
1637            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1638          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1639            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1640            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1641              !!!cp (67);              !!!cp (67);
1642              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1643            } else {            } else {
# Line 1443  sub _get_next_token ($) { Line 1645  sub _get_next_token ($) {
1645              !!!cp (68);              !!!cp (68);
1646            }            }
1647          } else {          } else {
1648            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1649          }          }
1650          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1651          # reconsume          # reconsume
1652    
1653          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1654    
1655          redo A;          redo A;
1656        } else {        } else {
1657          if ($self->{next_char} == 0x0022 or # "          if ($self->{nc} == 0x0022 or # "
1658              $self->{next_char} == 0x0027) { # '              $self->{nc} == 0x0027) { # '
1659            !!!cp (69);            !!!cp (69);
1660            !!!parse-error (type => 'bad attribute name');            !!!parse-error (type => 'bad attribute name');
1661          } else {          } else {
1662            !!!cp (70);            !!!cp (70);
1663          }          }
1664          $self->{current_attribute}->{name} .= chr ($self->{next_char});          $self->{ca}->{name} .= chr ($self->{nc});
1665          ## Stay in the state          ## Stay in the state
1666          !!!next-input-character;          !!!next-input-character;
1667          redo A;          redo A;
1668        }        }
1669      } elsif ($self->{state} == AFTER_ATTRIBUTE_NAME_STATE) {      } elsif ($self->{state} == AFTER_ATTRIBUTE_NAME_STATE) {
1670        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
1671          !!!cp (71);          !!!cp (71);
1672          ## Stay in the state          ## Stay in the state
1673          !!!next-input-character;          !!!next-input-character;
1674          redo A;          redo A;
1675        } elsif ($self->{next_char} == 0x003D) { # =        } elsif ($self->{nc} == 0x003D) { # =
1676          !!!cp (72);          !!!cp (72);
1677          $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;          $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;
1678          !!!next-input-character;          !!!next-input-character;
1679          redo A;          redo A;
1680        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1681          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1682            !!!cp (73);            !!!cp (73);
1683            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1684          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1685            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1686            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1687              !!!cp (74);              !!!cp (74);
1688              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1689            } else {            } else {
# Line 1493  sub _get_next_token ($) { Line 1691  sub _get_next_token ($) {
1691              !!!cp (75);              !!!cp (75);
1692            }            }
1693          } else {          } else {
1694            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1695          }          }
1696          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1697          !!!next-input-character;          !!!next-input-character;
1698    
1699          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1700    
1701          redo A;          redo A;
1702        } elsif (0x0041 <= $self->{next_char} and        } elsif (0x0041 <= $self->{nc} and
1703                 $self->{next_char} <= 0x005A) { # A..Z                 $self->{nc} <= 0x005A) { # A..Z
1704          !!!cp (76);          !!!cp (76);
1705          $self->{current_attribute}          $self->{ca}
1706              = {name => chr ($self->{next_char} + 0x0020),              = {name => chr ($self->{nc} + 0x0020),
1707                 value => '',                 value => '',
1708                 line => $self->{line}, column => $self->{column}};                 line => $self->{line}, column => $self->{column}};
1709          $self->{state} = ATTRIBUTE_NAME_STATE;          $self->{state} = ATTRIBUTE_NAME_STATE;
1710          !!!next-input-character;          !!!next-input-character;
1711          redo A;          redo A;
1712        } elsif ($self->{next_char} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
1713          !!!cp (77);          !!!cp (77);
1714          $self->{state} = SELF_CLOSING_START_TAG_STATE;          $self->{state} = SELF_CLOSING_START_TAG_STATE;
1715          !!!next-input-character;          !!!next-input-character;
1716          redo A;          redo A;
1717        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1718          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1719          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1720            !!!cp (79);            !!!cp (79);
1721            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1722          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1723            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1724            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1725              !!!cp (80);              !!!cp (80);
1726              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1727            } else {            } else {
# Line 1531  sub _get_next_token ($) { Line 1729  sub _get_next_token ($) {
1729              !!!cp (81);              !!!cp (81);
1730            }            }
1731          } else {          } else {
1732            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1733          }          }
1734          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1735          # reconsume          # reconsume
1736    
1737          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1738    
1739          redo A;          redo A;
1740        } else {        } else {
1741          !!!cp (82);          if ($self->{nc} == 0x0022 or # "
1742          $self->{current_attribute}              $self->{nc} == 0x0027) { # '
1743              = {name => chr ($self->{next_char}),            !!!cp (78);
1744              !!!parse-error (type => 'bad attribute name');
1745            } else {
1746              !!!cp (82);
1747            }
1748            $self->{ca}
1749                = {name => chr ($self->{nc}),
1750                 value => '',                 value => '',
1751                 line => $self->{line}, column => $self->{column}};                 line => $self->{line}, column => $self->{column}};
1752          $self->{state} = ATTRIBUTE_NAME_STATE;          $self->{state} = ATTRIBUTE_NAME_STATE;
# Line 1550  sub _get_next_token ($) { Line 1754  sub _get_next_token ($) {
1754          redo A;                  redo A;        
1755        }        }
1756      } elsif ($self->{state} == BEFORE_ATTRIBUTE_VALUE_STATE) {      } elsif ($self->{state} == BEFORE_ATTRIBUTE_VALUE_STATE) {
1757        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP        
1758          !!!cp (83);          !!!cp (83);
1759          ## Stay in the state          ## Stay in the state
1760          !!!next-input-character;          !!!next-input-character;
1761          redo A;          redo A;
1762        } elsif ($self->{next_char} == 0x0022) { # "        } elsif ($self->{nc} == 0x0022) { # "
1763          !!!cp (84);          !!!cp (84);
1764          $self->{state} = ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE;          $self->{state} = ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE;
1765          !!!next-input-character;          !!!next-input-character;
1766          redo A;          redo A;
1767        } elsif ($self->{next_char} == 0x0026) { # &        } elsif ($self->{nc} == 0x0026) { # &
1768          !!!cp (85);          !!!cp (85);
1769          $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;          $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;
1770          ## reconsume          ## reconsume
1771          redo A;          redo A;
1772        } elsif ($self->{next_char} == 0x0027) { # '        } elsif ($self->{nc} == 0x0027) { # '
1773          !!!cp (86);          !!!cp (86);
1774          $self->{state} = ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE;          $self->{state} = ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE;
1775          !!!next-input-character;          !!!next-input-character;
1776          redo A;          redo A;
1777        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1778          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          !!!parse-error (type => 'empty unquoted attribute value');
1779            if ($self->{ct}->{type} == START_TAG_TOKEN) {
1780            !!!cp (87);            !!!cp (87);
1781            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1782          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1783            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1784            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1785              !!!cp (88);              !!!cp (88);
1786              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1787            } else {            } else {
# Line 1588  sub _get_next_token ($) { Line 1789  sub _get_next_token ($) {
1789              !!!cp (89);              !!!cp (89);
1790            }            }
1791          } else {          } else {
1792            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1793          }          }
1794          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1795          !!!next-input-character;          !!!next-input-character;
1796    
1797          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1798    
1799          redo A;          redo A;
1800        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1801          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1802          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1803            !!!cp (90);            !!!cp (90);
1804            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1805          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1806            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1807            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1808              !!!cp (91);              !!!cp (91);
1809              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1810            } else {            } else {
# Line 1611  sub _get_next_token ($) { Line 1812  sub _get_next_token ($) {
1812              !!!cp (92);              !!!cp (92);
1813            }            }
1814          } else {          } else {
1815            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1816          }          }
1817          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1818          ## reconsume          ## reconsume
1819    
1820          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1821    
1822          redo A;          redo A;
1823        } else {        } else {
1824          if ($self->{next_char} == 0x003D) { # =          if ($self->{nc} == 0x003D) { # =
1825            !!!cp (93);            !!!cp (93);
1826            !!!parse-error (type => 'bad attribute value');            !!!parse-error (type => 'bad attribute value');
1827          } else {          } else {
1828            !!!cp (94);            !!!cp (94);
1829          }          }
1830          $self->{current_attribute}->{value} .= chr ($self->{next_char});          $self->{ca}->{value} .= chr ($self->{nc});
1831          $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;          $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;
1832          !!!next-input-character;          !!!next-input-character;
1833          redo A;          redo A;
1834        }        }
1835      } elsif ($self->{state} == ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE) {      } elsif ($self->{state} == ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE) {
1836        if ($self->{next_char} == 0x0022) { # "        if ($self->{nc} == 0x0022) { # "
1837          !!!cp (95);          !!!cp (95);
1838          $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;          $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;
1839          !!!next-input-character;          !!!next-input-character;
1840          redo A;          redo A;
1841        } elsif ($self->{next_char} == 0x0026) { # &        } elsif ($self->{nc} == 0x0026) { # &
1842          !!!cp (96);          !!!cp (96);
1843          $self->{last_attribute_value_state} = $self->{state};          ## NOTE: In the spec, the tokenizer is switched to the
1844          $self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE;          ## "entity in attribute value state".  In this implementation, the
1845            ## tokenizer is switched to the |ENTITY_STATE|, which is an
1846            ## implementation of the "consume a character reference" algorithm.
1847            $self->{prev_state} = $self->{state};
1848            $self->{entity_add} = 0x0022; # "
1849            $self->{state} = ENTITY_STATE;
1850          !!!next-input-character;          !!!next-input-character;
1851          redo A;          redo A;
1852        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1853          !!!parse-error (type => 'unclosed attribute value');          !!!parse-error (type => 'unclosed attribute value');
1854          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1855            !!!cp (97);            !!!cp (97);
1856            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1857          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1858            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1859            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1860              !!!cp (98);              !!!cp (98);
1861              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1862            } else {            } else {
# Line 1658  sub _get_next_token ($) { Line 1864  sub _get_next_token ($) {
1864              !!!cp (99);              !!!cp (99);
1865            }            }
1866          } else {          } else {
1867            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1868          }          }
1869          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1870          ## reconsume          ## reconsume
1871    
1872          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1873    
1874          redo A;          redo A;
1875        } else {        } else {
1876          !!!cp (100);          !!!cp (100);
1877          $self->{current_attribute}->{value} .= chr ($self->{next_char});          $self->{ca}->{value} .= chr ($self->{nc});
1878            $self->{read_until}->($self->{ca}->{value},
1879                                  q["&],
1880                                  length $self->{ca}->{value});
1881    
1882          ## Stay in the state          ## Stay in the state
1883          !!!next-input-character;          !!!next-input-character;
1884          redo A;          redo A;
1885        }        }
1886      } elsif ($self->{state} == ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE) {      } elsif ($self->{state} == ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE) {
1887        if ($self->{next_char} == 0x0027) { # '        if ($self->{nc} == 0x0027) { # '
1888          !!!cp (101);          !!!cp (101);
1889          $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;          $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;
1890          !!!next-input-character;          !!!next-input-character;
1891          redo A;          redo A;
1892        } elsif ($self->{next_char} == 0x0026) { # &        } elsif ($self->{nc} == 0x0026) { # &
1893          !!!cp (102);          !!!cp (102);
1894          $self->{last_attribute_value_state} = $self->{state};          ## NOTE: In the spec, the tokenizer is switched to the
1895          $self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE;          ## "entity in attribute value state".  In this implementation, the
1896            ## tokenizer is switched to the |ENTITY_STATE|, which is an
1897            ## implementation of the "consume a character reference" algorithm.
1898            $self->{entity_add} = 0x0027; # '
1899            $self->{prev_state} = $self->{state};
1900            $self->{state} = ENTITY_STATE;
1901          !!!next-input-character;          !!!next-input-character;
1902          redo A;          redo A;
1903        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1904          !!!parse-error (type => 'unclosed attribute value');          !!!parse-error (type => 'unclosed attribute value');
1905          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1906            !!!cp (103);            !!!cp (103);
1907            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1908          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1909            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1910            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1911              !!!cp (104);              !!!cp (104);
1912              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1913            } else {            } else {
# Line 1700  sub _get_next_token ($) { Line 1915  sub _get_next_token ($) {
1915              !!!cp (105);              !!!cp (105);
1916            }            }
1917          } else {          } else {
1918            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1919          }          }
1920          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1921          ## reconsume          ## reconsume
1922    
1923          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1924    
1925          redo A;          redo A;
1926        } else {        } else {
1927          !!!cp (106);          !!!cp (106);
1928          $self->{current_attribute}->{value} .= chr ($self->{next_char});          $self->{ca}->{value} .= chr ($self->{nc});
1929            $self->{read_until}->($self->{ca}->{value},
1930                                  q['&],
1931                                  length $self->{ca}->{value});
1932    
1933          ## Stay in the state          ## Stay in the state
1934          !!!next-input-character;          !!!next-input-character;
1935          redo A;          redo A;
1936        }        }
1937      } elsif ($self->{state} == ATTRIBUTE_VALUE_UNQUOTED_STATE) {      } elsif ($self->{state} == ATTRIBUTE_VALUE_UNQUOTED_STATE) {
1938        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # HT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
1939          !!!cp (107);          !!!cp (107);
1940          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1941          !!!next-input-character;          !!!next-input-character;
1942          redo A;          redo A;
1943        } elsif ($self->{next_char} == 0x0026) { # &        } elsif ($self->{nc} == 0x0026) { # &
1944          !!!cp (108);          !!!cp (108);
1945          $self->{last_attribute_value_state} = $self->{state};          ## NOTE: In the spec, the tokenizer is switched to the
1946          $self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE;          ## "entity in attribute value state".  In this implementation, the
1947            ## tokenizer is switched to the |ENTITY_STATE|, which is an
1948            ## implementation of the "consume a character reference" algorithm.
1949            $self->{entity_add} = -1;
1950            $self->{prev_state} = $self->{state};
1951            $self->{state} = ENTITY_STATE;
1952          !!!next-input-character;          !!!next-input-character;
1953          redo A;          redo A;
1954        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1955          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1956            !!!cp (109);            !!!cp (109);
1957            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1958          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1959            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1960            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1961              !!!cp (110);              !!!cp (110);
1962              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1963            } else {            } else {
# Line 1745  sub _get_next_token ($) { Line 1965  sub _get_next_token ($) {
1965              !!!cp (111);              !!!cp (111);
1966            }            }
1967          } else {          } else {
1968            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1969          }          }
1970          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1971          !!!next-input-character;          !!!next-input-character;
1972    
1973          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1974    
1975          redo A;          redo A;
1976        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1977          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1978          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1979            !!!cp (112);            !!!cp (112);
1980            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1981          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1982            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1983            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1984              !!!cp (113);              !!!cp (113);
1985              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1986            } else {            } else {
# Line 1768  sub _get_next_token ($) { Line 1988  sub _get_next_token ($) {
1988              !!!cp (114);              !!!cp (114);
1989            }            }
1990          } else {          } else {
1991            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1992          }          }
1993          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1994          ## reconsume          ## reconsume
1995    
1996          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1997    
1998          redo A;          redo A;
1999        } else {        } else {
# Line 1781  sub _get_next_token ($) { Line 2001  sub _get_next_token ($) {
2001               0x0022 => 1, # "               0x0022 => 1, # "
2002               0x0027 => 1, # '               0x0027 => 1, # '
2003               0x003D => 1, # =               0x003D => 1, # =
2004              }->{$self->{next_char}}) {              }->{$self->{nc}}) {
2005            !!!cp (115);            !!!cp (115);
2006            !!!parse-error (type => 'bad attribute value');            !!!parse-error (type => 'bad attribute value');
2007          } else {          } else {
2008            !!!cp (116);            !!!cp (116);
2009          }          }
2010          $self->{current_attribute}->{value} .= chr ($self->{next_char});          $self->{ca}->{value} .= chr ($self->{nc});
2011            $self->{read_until}->($self->{ca}->{value},
2012                                  q["'=& >],
2013                                  length $self->{ca}->{value});
2014    
2015          ## Stay in the state          ## Stay in the state
2016          !!!next-input-character;          !!!next-input-character;
2017          redo A;          redo A;
2018        }        }
     } elsif ($self->{state} == ENTITY_IN_ATTRIBUTE_VALUE_STATE) {  
       my $token = $self->_tokenize_attempt_to_consume_an_entity  
           (1,  
            $self->{last_attribute_value_state}  
              == ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE ? 0x0022 : # "  
            $self->{last_attribute_value_state}  
              == ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE ? 0x0027 : # '  
            -1);  
   
       unless (defined $token) {  
         !!!cp (117);  
         $self->{current_attribute}->{value} .= '&';  
       } else {  
         !!!cp (118);  
         $self->{current_attribute}->{value} .= $token->{data};  
         $self->{current_attribute}->{has_reference} = $token->{has_reference};  
         ## ISSUE: spec says "append the returned character token to the current attribute's value"  
       }  
   
       $self->{state} = $self->{last_attribute_value_state};  
       # next-input-character is already done  
       redo A;  
2019      } elsif ($self->{state} == AFTER_ATTRIBUTE_VALUE_QUOTED_STATE) {      } elsif ($self->{state} == AFTER_ATTRIBUTE_VALUE_QUOTED_STATE) {
2020        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
2021          !!!cp (118);          !!!cp (118);
2022          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
2023          !!!next-input-character;          !!!next-input-character;
2024          redo A;          redo A;
2025        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2026          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
2027            !!!cp (119);            !!!cp (119);
2028            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
2029          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
2030            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
2031            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
2032              !!!cp (120);              !!!cp (120);
2033              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
2034            } else {            } else {
# Line 1838  sub _get_next_token ($) { Line 2036  sub _get_next_token ($) {
2036              !!!cp (121);              !!!cp (121);
2037            }            }
2038          } else {          } else {
2039            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
2040          }          }
2041          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2042          !!!next-input-character;          !!!next-input-character;
2043    
2044          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
2045    
2046          redo A;          redo A;
2047        } elsif ($self->{next_char} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
2048          !!!cp (122);          !!!cp (122);
2049          $self->{state} = SELF_CLOSING_START_TAG_STATE;          $self->{state} = SELF_CLOSING_START_TAG_STATE;
2050          !!!next-input-character;          !!!next-input-character;
2051          redo A;          redo A;
2052        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2053          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
2054          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
2055            !!!cp (122.3);            !!!cp (122.3);
2056            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
2057          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
2058            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
2059              !!!cp (122.1);              !!!cp (122.1);
2060              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
2061            } else {            } else {
# Line 1865  sub _get_next_token ($) { Line 2063  sub _get_next_token ($) {
2063              !!!cp (122.2);              !!!cp (122.2);
2064            }            }
2065          } else {          } else {
2066            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
2067          }          }
2068          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2069          ## Reconsume.          ## Reconsume.
2070          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
2071          redo A;          redo A;
2072        } else {        } else {
2073          !!!cp ('124.1');          !!!cp ('124.1');
# Line 1879  sub _get_next_token ($) { Line 2077  sub _get_next_token ($) {
2077          redo A;          redo A;
2078        }        }
2079      } elsif ($self->{state} == SELF_CLOSING_START_TAG_STATE) {      } elsif ($self->{state} == SELF_CLOSING_START_TAG_STATE) {
2080        if ($self->{next_char} == 0x003E) { # >        if ($self->{nc} == 0x003E) { # >
2081          if ($self->{current_token}->{type} == END_TAG_TOKEN) {          if ($self->{ct}->{type} == END_TAG_TOKEN) {
2082            !!!cp ('124.2');            !!!cp ('124.2');
2083            !!!parse-error (type => 'nestc', token => $self->{current_token});            !!!parse-error (type => 'nestc', token => $self->{ct});
2084            ## TODO: Different type than slash in start tag            ## TODO: Different type than slash in start tag
2085            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
2086            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
2087              !!!cp ('124.4');              !!!cp ('124.4');
2088              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
2089            } else {            } else {
# Line 1900  sub _get_next_token ($) { Line 2098  sub _get_next_token ($) {
2098          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2099          !!!next-input-character;          !!!next-input-character;
2100    
2101          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
2102    
2103          redo A;          redo A;
2104        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2105          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
2106          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
2107            !!!cp (124.7);            !!!cp (124.7);
2108            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
2109          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
2110            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
2111              !!!cp (124.5);              !!!cp (124.5);
2112              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
2113            } else {            } else {
# Line 1917  sub _get_next_token ($) { Line 2115  sub _get_next_token ($) {
2115              !!!cp (124.6);              !!!cp (124.6);
2116            }            }
2117          } else {          } else {
2118            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
2119          }          }
2120          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2121          ## Reconsume.          ## Reconsume.
2122          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
2123          redo A;          redo A;
2124        } else {        } else {
2125          !!!cp ('124.4');          !!!cp ('124.4');
# Line 1933  sub _get_next_token ($) { Line 2131  sub _get_next_token ($) {
2131        }        }
2132      } elsif ($self->{state} == BOGUS_COMMENT_STATE) {      } elsif ($self->{state} == BOGUS_COMMENT_STATE) {
2133        ## (only happen if PCDATA state)        ## (only happen if PCDATA state)
         
       ## NOTE: Set by the previous state  
       #my $token = {type => COMMENT_TOKEN, data => ''};  
   
       BC: {  
         if ($self->{next_char} == 0x003E) { # >  
           !!!cp (124);  
           $self->{state} = DATA_STATE;  
           !!!next-input-character;  
2134    
2135            !!!emit ($self->{current_token}); # comment        ## NOTE: Unlike spec's "bogus comment state", this implementation
2136          ## consumes characters one-by-one basis.
2137            redo A;        
2138          } elsif ($self->{next_char} == -1) {        if ($self->{nc} == 0x003E) { # >
2139            !!!cp (125);          !!!cp (124);
2140            $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2141            ## reconsume          !!!next-input-character;
2142    
2143            !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2144            redo A;
2145          } elsif ($self->{nc} == -1) {
2146            !!!cp (125);
2147            $self->{state} = DATA_STATE;
2148            ## reconsume
2149    
2150            redo A;          !!!emit ($self->{ct}); # comment
2151          } else {          redo A;
2152            !!!cp (126);        } else {
2153            $self->{current_token}->{data} .= chr ($self->{next_char}); # comment          !!!cp (126);
2154            !!!next-input-character;          $self->{ct}->{data} .= chr ($self->{nc}); # comment
2155            redo BC;          $self->{read_until}->($self->{ct}->{data},
2156          }                                q[>],
2157        } # BC                                length $self->{ct}->{data});
2158    
2159        die "$0: _get_next_token: unexpected case [BC]";          ## Stay in the state.
2160            !!!next-input-character;
2161            redo A;
2162          }
2163      } elsif ($self->{state} == MARKUP_DECLARATION_OPEN_STATE) {      } elsif ($self->{state} == MARKUP_DECLARATION_OPEN_STATE) {
2164        ## (only happen if PCDATA state)        ## (only happen if PCDATA state)
   
       my ($l, $c) = ($self->{line_prev}, $self->{column_prev} - 1);  
   
       my @next_char;  
       push @next_char, $self->{next_char};  
2165                
2166        if ($self->{next_char} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2167            !!!cp (133);
2168            $self->{state} = MD_HYPHEN_STATE;
2169          !!!next-input-character;          !!!next-input-character;
2170          push @next_char, $self->{next_char};          redo A;
2171          if ($self->{next_char} == 0x002D) { # -        } elsif ($self->{nc} == 0x0044 or # D
2172            !!!cp (127);                 $self->{nc} == 0x0064) { # d
2173            $self->{current_token} = {type => COMMENT_TOKEN, data => '',          ## ASCII case-insensitive.
2174                                      line => $l, column => $c,          !!!cp (130);
2175                                     };          $self->{state} = MD_DOCTYPE_STATE;
2176            $self->{state} = COMMENT_START_STATE;          $self->{s_kwd} = chr $self->{nc};
           !!!next-input-character;  
           redo A;  
         } else {  
           !!!cp (128);  
         }  
       } elsif ($self->{next_char} == 0x0044 or # D  
                $self->{next_char} == 0x0064) { # d  
2177          !!!next-input-character;          !!!next-input-character;
2178          push @next_char, $self->{next_char};          redo A;
         if ($self->{next_char} == 0x004F or # O  
             $self->{next_char} == 0x006F) { # o  
           !!!next-input-character;  
           push @next_char, $self->{next_char};  
           if ($self->{next_char} == 0x0043 or # C  
               $self->{next_char} == 0x0063) { # c  
             !!!next-input-character;  
             push @next_char, $self->{next_char};  
             if ($self->{next_char} == 0x0054 or # T  
                 $self->{next_char} == 0x0074) { # t  
               !!!next-input-character;  
               push @next_char, $self->{next_char};  
               if ($self->{next_char} == 0x0059 or # Y  
                   $self->{next_char} == 0x0079) { # y  
                 !!!next-input-character;  
                 push @next_char, $self->{next_char};  
                 if ($self->{next_char} == 0x0050 or # P  
                     $self->{next_char} == 0x0070) { # p  
                   !!!next-input-character;  
                   push @next_char, $self->{next_char};  
                   if ($self->{next_char} == 0x0045 or # E  
                       $self->{next_char} == 0x0065) { # e  
                     !!!cp (129);  
                     ## TODO: What a stupid code this is!  
                     $self->{state} = DOCTYPE_STATE;  
                     $self->{current_token} = {type => DOCTYPE_TOKEN,  
                                               quirks => 1,  
                                               line => $l, column => $c,  
                                              };  
                     !!!next-input-character;  
                     redo A;  
                   } else {  
                     !!!cp (130);  
                   }  
                 } else {  
                   !!!cp (131);  
                 }  
               } else {  
                 !!!cp (132);  
               }  
             } else {  
               !!!cp (133);  
             }  
           } else {  
             !!!cp (134);  
           }  
         } else {  
           !!!cp (135);  
         }  
2179        } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM and        } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM and
2180                 $self->{open_elements}->[-1]->[1] & FOREIGN_EL and                 $self->{open_elements}->[-1]->[1] & FOREIGN_EL and
2181                 $self->{next_char} == 0x005B) { # [                 $self->{nc} == 0x005B) { # [
2182            !!!cp (135.4);                
2183            $self->{state} = MD_CDATA_STATE;
2184            $self->{s_kwd} = '[';
2185          !!!next-input-character;          !!!next-input-character;
2186          push @next_char, $self->{next_char};          redo A;
         if ($self->{next_char} == 0x0043) { # C  
           !!!next-input-character;  
           push @next_char, $self->{next_char};  
           if ($self->{next_char} == 0x0044) { # D  
             !!!next-input-character;  
             push @next_char, $self->{next_char};  
             if ($self->{next_char} == 0x0041) { # A  
               !!!next-input-character;  
               push @next_char, $self->{next_char};  
               if ($self->{next_char} == 0x0054) { # T  
                 !!!next-input-character;  
                 push @next_char, $self->{next_char};  
                 if ($self->{next_char} == 0x0041) { # A  
                   !!!next-input-character;  
                   push @next_char, $self->{next_char};  
                   if ($self->{next_char} == 0x005B) { # [  
                     !!!cp (135.1);  
                     $self->{state} = CDATA_BLOCK_STATE;  
                     !!!next-input-character;  
                     redo A;  
                   } else {  
                     !!!cp (135.2);  
                   }  
                 } else {  
                   !!!cp (135.3);  
                 }  
               } else {  
                 !!!cp (135.4);                  
               }  
             } else {  
               !!!cp (135.5);  
             }  
           } else {  
             !!!cp (135.6);  
           }  
         } else {  
           !!!cp (135.7);  
         }  
2187        } else {        } else {
2188          !!!cp (136);          !!!cp (136);
2189        }        }
2190    
2191        !!!parse-error (type => 'bogus comment');        !!!parse-error (type => 'bogus comment',
2192        $self->{next_char} = shift @next_char;                        line => $self->{line_prev},
2193        !!!back-next-input-character (@next_char);                        column => $self->{column_prev} - 1);
2194          ## Reconsume.
2195        $self->{state} = BOGUS_COMMENT_STATE;        $self->{state} = BOGUS_COMMENT_STATE;
2196        $self->{current_token} = {type => COMMENT_TOKEN, data => '',        $self->{ct} = {type => COMMENT_TOKEN, data => '',
2197                                  line => $l, column => $c,                                  line => $self->{line_prev},
2198                                    column => $self->{column_prev} - 1,
2199                                 };                                 };
2200        redo A;        redo A;
2201              } elsif ($self->{state} == MD_HYPHEN_STATE) {
2202        ## ISSUE: typos in spec: chacacters, is is a parse error        if ($self->{nc} == 0x002D) { # -
2203        ## ISSUE: spec is somewhat unclear on "is the first character that will be in the comment"; what is "that will be in the comment" is what the algorithm defines, isn't it?          !!!cp (127);
2204            $self->{ct} = {type => COMMENT_TOKEN, data => '',
2205                                      line => $self->{line_prev},
2206                                      column => $self->{column_prev} - 2,
2207                                     };
2208            $self->{state} = COMMENT_START_STATE;
2209            !!!next-input-character;
2210            redo A;
2211          } else {
2212            !!!cp (128);
2213            !!!parse-error (type => 'bogus comment',
2214                            line => $self->{line_prev},
2215                            column => $self->{column_prev} - 2);
2216            $self->{state} = BOGUS_COMMENT_STATE;
2217            ## Reconsume.
2218            $self->{ct} = {type => COMMENT_TOKEN,
2219                                      data => '-',
2220                                      line => $self->{line_prev},
2221                                      column => $self->{column_prev} - 2,
2222                                     };
2223            redo A;
2224          }
2225        } elsif ($self->{state} == MD_DOCTYPE_STATE) {
2226          ## ASCII case-insensitive.
2227          if ($self->{nc} == [
2228                undef,
2229                0x004F, # O
2230                0x0043, # C
2231                0x0054, # T
2232                0x0059, # Y
2233                0x0050, # P
2234              ]->[length $self->{s_kwd}] or
2235              $self->{nc} == [
2236                undef,
2237                0x006F, # o
2238                0x0063, # c
2239                0x0074, # t
2240                0x0079, # y
2241                0x0070, # p
2242              ]->[length $self->{s_kwd}]) {
2243            !!!cp (131);
2244            ## Stay in the state.
2245            $self->{s_kwd} .= chr $self->{nc};
2246            !!!next-input-character;
2247            redo A;
2248          } elsif ((length $self->{s_kwd}) == 6 and
2249                   ($self->{nc} == 0x0045 or # E
2250                    $self->{nc} == 0x0065)) { # e
2251            !!!cp (129);
2252            $self->{state} = DOCTYPE_STATE;
2253            $self->{ct} = {type => DOCTYPE_TOKEN,
2254                                      quirks => 1,
2255                                      line => $self->{line_prev},
2256                                      column => $self->{column_prev} - 7,
2257                                     };
2258            !!!next-input-character;
2259            redo A;
2260          } else {
2261            !!!cp (132);        
2262            !!!parse-error (type => 'bogus comment',
2263                            line => $self->{line_prev},
2264                            column => $self->{column_prev} - 1 - length $self->{s_kwd});
2265            $self->{state} = BOGUS_COMMENT_STATE;
2266            ## Reconsume.
2267            $self->{ct} = {type => COMMENT_TOKEN,
2268                                      data => $self->{s_kwd},
2269                                      line => $self->{line_prev},
2270                                      column => $self->{column_prev} - 1 - length $self->{s_kwd},
2271                                     };
2272            redo A;
2273          }
2274        } elsif ($self->{state} == MD_CDATA_STATE) {
2275          if ($self->{nc} == {
2276                '[' => 0x0043, # C
2277                '[C' => 0x0044, # D
2278                '[CD' => 0x0041, # A
2279                '[CDA' => 0x0054, # T
2280                '[CDAT' => 0x0041, # A
2281              }->{$self->{s_kwd}}) {
2282            !!!cp (135.1);
2283            ## Stay in the state.
2284            $self->{s_kwd} .= chr $self->{nc};
2285            !!!next-input-character;
2286            redo A;
2287          } elsif ($self->{s_kwd} eq '[CDATA' and
2288                   $self->{nc} == 0x005B) { # [
2289            !!!cp (135.2);
2290            $self->{ct} = {type => CHARACTER_TOKEN,
2291                                      data => '',
2292                                      line => $self->{line_prev},
2293                                      column => $self->{column_prev} - 7};
2294            $self->{state} = CDATA_SECTION_STATE;
2295            !!!next-input-character;
2296            redo A;
2297          } else {
2298            !!!cp (135.3);
2299            !!!parse-error (type => 'bogus comment',
2300                            line => $self->{line_prev},
2301                            column => $self->{column_prev} - 1 - length $self->{s_kwd});
2302            $self->{state} = BOGUS_COMMENT_STATE;
2303            ## Reconsume.
2304            $self->{ct} = {type => COMMENT_TOKEN,
2305                                      data => $self->{s_kwd},
2306                                      line => $self->{line_prev},
2307                                      column => $self->{column_prev} - 1 - length $self->{s_kwd},
2308                                     };
2309            redo A;
2310          }
2311      } elsif ($self->{state} == COMMENT_START_STATE) {      } elsif ($self->{state} == COMMENT_START_STATE) {
2312        if ($self->{next_char} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2313          !!!cp (137);          !!!cp (137);
2314          $self->{state} = COMMENT_START_DASH_STATE;          $self->{state} = COMMENT_START_DASH_STATE;
2315          !!!next-input-character;          !!!next-input-character;
2316          redo A;          redo A;
2317        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2318          !!!cp (138);          !!!cp (138);
2319          !!!parse-error (type => 'bogus comment');          !!!parse-error (type => 'bogus comment');
2320          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2321          !!!next-input-character;          !!!next-input-character;
2322    
2323          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2324    
2325          redo A;          redo A;
2326        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2327          !!!cp (139);          !!!cp (139);
2328          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2329          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2330          ## reconsume          ## reconsume
2331    
2332          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2333    
2334          redo A;          redo A;
2335        } else {        } else {
2336          !!!cp (140);          !!!cp (140);
2337          $self->{current_token}->{data} # comment          $self->{ct}->{data} # comment
2338              .= chr ($self->{next_char});              .= chr ($self->{nc});
2339          $self->{state} = COMMENT_STATE;          $self->{state} = COMMENT_STATE;
2340          !!!next-input-character;          !!!next-input-character;
2341          redo A;          redo A;
2342        }        }
2343      } elsif ($self->{state} == COMMENT_START_DASH_STATE) {      } elsif ($self->{state} == COMMENT_START_DASH_STATE) {
2344        if ($self->{next_char} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2345          !!!cp (141);          !!!cp (141);
2346          $self->{state} = COMMENT_END_STATE;          $self->{state} = COMMENT_END_STATE;
2347          !!!next-input-character;          !!!next-input-character;
2348          redo A;          redo A;
2349        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2350          !!!cp (142);          !!!cp (142);
2351          !!!parse-error (type => 'bogus comment');          !!!parse-error (type => 'bogus comment');
2352          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2353          !!!next-input-character;          !!!next-input-character;
2354    
2355          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2356    
2357          redo A;          redo A;
2358        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2359          !!!cp (143);          !!!cp (143);
2360          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2361          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2362          ## reconsume          ## reconsume
2363    
2364          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2365    
2366          redo A;          redo A;
2367        } else {        } else {
2368          !!!cp (144);          !!!cp (144);
2369          $self->{current_token}->{data} # comment          $self->{ct}->{data} # comment
2370              .= '-' . chr ($self->{next_char});              .= '-' . chr ($self->{nc});
2371          $self->{state} = COMMENT_STATE;          $self->{state} = COMMENT_STATE;
2372          !!!next-input-character;          !!!next-input-character;
2373          redo A;          redo A;
2374        }        }
2375      } elsif ($self->{state} == COMMENT_STATE) {      } elsif ($self->{state} == COMMENT_STATE) {
2376        if ($self->{next_char} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2377          !!!cp (145);          !!!cp (145);
2378          $self->{state} = COMMENT_END_DASH_STATE;          $self->{state} = COMMENT_END_DASH_STATE;
2379          !!!next-input-character;          !!!next-input-character;
2380          redo A;          redo A;
2381        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2382          !!!cp (146);          !!!cp (146);
2383          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2384          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2385          ## reconsume          ## reconsume
2386    
2387          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2388    
2389          redo A;          redo A;
2390        } else {        } else {
2391          !!!cp (147);          !!!cp (147);
2392          $self->{current_token}->{data} .= chr ($self->{next_char}); # comment          $self->{ct}->{data} .= chr ($self->{nc}); # comment
2393            $self->{read_until}->($self->{ct}->{data},
2394                                  q[-],
2395                                  length $self->{ct}->{data});
2396    
2397          ## Stay in the state          ## Stay in the state
2398          !!!next-input-character;          !!!next-input-character;
2399          redo A;          redo A;
2400        }        }
2401      } elsif ($self->{state} == COMMENT_END_DASH_STATE) {      } elsif ($self->{state} == COMMENT_END_DASH_STATE) {
2402        if ($self->{next_char} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2403          !!!cp (148);          !!!cp (148);
2404          $self->{state} = COMMENT_END_STATE;          $self->{state} = COMMENT_END_STATE;
2405          !!!next-input-character;          !!!next-input-character;
2406          redo A;          redo A;
2407        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2408          !!!cp (149);          !!!cp (149);
2409          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2410          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2411          ## reconsume          ## reconsume
2412    
2413          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2414    
2415          redo A;          redo A;
2416        } else {        } else {
2417          !!!cp (150);          !!!cp (150);
2418          $self->{current_token}->{data} .= '-' . chr ($self->{next_char}); # comment          $self->{ct}->{data} .= '-' . chr ($self->{nc}); # comment
2419          $self->{state} = COMMENT_STATE;          $self->{state} = COMMENT_STATE;
2420          !!!next-input-character;          !!!next-input-character;
2421          redo A;          redo A;
2422        }        }
2423      } elsif ($self->{state} == COMMENT_END_STATE) {      } elsif ($self->{state} == COMMENT_END_STATE) {
2424        if ($self->{next_char} == 0x003E) { # >        if ($self->{nc} == 0x003E) { # >
2425          !!!cp (151);          !!!cp (151);
2426          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2427          !!!next-input-character;          !!!next-input-character;
2428    
2429          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2430    
2431          redo A;          redo A;
2432        } elsif ($self->{next_char} == 0x002D) { # -        } elsif ($self->{nc} == 0x002D) { # -
2433          !!!cp (152);          !!!cp (152);
2434          !!!parse-error (type => 'dash in comment',          !!!parse-error (type => 'dash in comment',
2435                          line => $self->{line_prev},                          line => $self->{line_prev},
2436                          column => $self->{column_prev});                          column => $self->{column_prev});
2437          $self->{current_token}->{data} .= '-'; # comment          $self->{ct}->{data} .= '-'; # comment
2438          ## Stay in the state          ## Stay in the state
2439          !!!next-input-character;          !!!next-input-character;
2440          redo A;          redo A;
2441        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2442          !!!cp (153);          !!!cp (153);
2443          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2444          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2445          ## reconsume          ## reconsume
2446    
2447          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2448    
2449          redo A;          redo A;
2450        } else {        } else {
# Line 2236  sub _get_next_token ($) { Line 2452  sub _get_next_token ($) {
2452          !!!parse-error (type => 'dash in comment',          !!!parse-error (type => 'dash in comment',
2453                          line => $self->{line_prev},                          line => $self->{line_prev},
2454                          column => $self->{column_prev});                          column => $self->{column_prev});
2455          $self->{current_token}->{data} .= '--' . chr ($self->{next_char}); # comment          $self->{ct}->{data} .= '--' . chr ($self->{nc}); # comment
2456          $self->{state} = COMMENT_STATE;          $self->{state} = COMMENT_STATE;
2457          !!!next-input-character;          !!!next-input-character;
2458          redo A;          redo A;
2459        }        }
2460      } elsif ($self->{state} == DOCTYPE_STATE) {      } elsif ($self->{state} == DOCTYPE_STATE) {
2461        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
2462          !!!cp (155);          !!!cp (155);
2463          $self->{state} = BEFORE_DOCTYPE_NAME_STATE;          $self->{state} = BEFORE_DOCTYPE_NAME_STATE;
2464          !!!next-input-character;          !!!next-input-character;
# Line 2259  sub _get_next_token ($) { Line 2471  sub _get_next_token ($) {
2471          redo A;          redo A;
2472        }        }
2473      } elsif ($self->{state} == BEFORE_DOCTYPE_NAME_STATE) {      } elsif ($self->{state} == BEFORE_DOCTYPE_NAME_STATE) {
2474        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
2475          !!!cp (157);          !!!cp (157);
2476          ## Stay in the state          ## Stay in the state
2477          !!!next-input-character;          !!!next-input-character;
2478          redo A;          redo A;
2479        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2480          !!!cp (158);          !!!cp (158);
2481          !!!parse-error (type => 'no DOCTYPE name');          !!!parse-error (type => 'no DOCTYPE name');
2482          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2483          !!!next-input-character;          !!!next-input-character;
2484    
2485          !!!emit ($self->{current_token}); # DOCTYPE (quirks)          !!!emit ($self->{ct}); # DOCTYPE (quirks)
2486    
2487          redo A;          redo A;
2488        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2489          !!!cp (159);          !!!cp (159);
2490          !!!parse-error (type => 'no DOCTYPE name');          !!!parse-error (type => 'no DOCTYPE name');
2491          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2492          ## reconsume          ## reconsume
2493    
2494          !!!emit ($self->{current_token}); # DOCTYPE (quirks)          !!!emit ($self->{ct}); # DOCTYPE (quirks)
2495    
2496          redo A;          redo A;
2497        } else {        } else {
2498          !!!cp (160);          !!!cp (160);
2499          $self->{current_token}->{name} = chr $self->{next_char};          $self->{ct}->{name} = chr $self->{nc};
2500          delete $self->{current_token}->{quirks};          delete $self->{ct}->{quirks};
 ## ISSUE: "Set the token's name name to the" in the spec  
2501          $self->{state} = DOCTYPE_NAME_STATE;          $self->{state} = DOCTYPE_NAME_STATE;
2502          !!!next-input-character;          !!!next-input-character;
2503          redo A;          redo A;
2504        }        }
2505      } elsif ($self->{state} == DOCTYPE_NAME_STATE) {      } elsif ($self->{state} == DOCTYPE_NAME_STATE) {
2506  ## ISSUE: Redundant "First," in the spec.  ## ISSUE: Redundant "First," in the spec.
2507        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
2508          !!!cp (161);          !!!cp (161);
2509          $self->{state} = AFTER_DOCTYPE_NAME_STATE;          $self->{state} = AFTER_DOCTYPE_NAME_STATE;
2510          !!!next-input-character;          !!!next-input-character;
2511          redo A;          redo A;
2512        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2513          !!!cp (162);          !!!cp (162);
2514          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2515          !!!next-input-character;          !!!next-input-character;
2516    
2517          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2518    
2519          redo A;          redo A;
2520        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2521          !!!cp (163);          !!!cp (163);
2522          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2523          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2524          ## reconsume          ## reconsume
2525    
2526          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2527          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2528    
2529          redo A;          redo A;
2530        } else {        } else {
2531          !!!cp (164);          !!!cp (164);
2532          $self->{current_token}->{name}          $self->{ct}->{name}
2533            .= chr ($self->{next_char}); # DOCTYPE            .= chr ($self->{nc}); # DOCTYPE
2534          ## Stay in the state          ## Stay in the state
2535          !!!next-input-character;          !!!next-input-character;
2536          redo A;          redo A;
2537        }        }
2538      } elsif ($self->{state} == AFTER_DOCTYPE_NAME_STATE) {      } elsif ($self->{state} == AFTER_DOCTYPE_NAME_STATE) {
2539        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
2540          !!!cp (165);          !!!cp (165);
2541          ## Stay in the state          ## Stay in the state
2542          !!!next-input-character;          !!!next-input-character;
2543          redo A;          redo A;
2544        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2545          !!!cp (166);          !!!cp (166);
2546          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2547          !!!next-input-character;          !!!next-input-character;
2548    
2549          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2550    
2551          redo A;          redo A;
2552        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2553          !!!cp (167);          !!!cp (167);
2554          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2555          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2556          ## reconsume          ## reconsume
2557    
2558          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2559          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2560    
2561          redo A;          redo A;
2562        } elsif ($self->{next_char} == 0x0050 or # P        } elsif ($self->{nc} == 0x0050 or # P
2563                 $self->{next_char} == 0x0070) { # p                 $self->{nc} == 0x0070) { # p
2564            $self->{state} = PUBLIC_STATE;
2565            $self->{s_kwd} = chr $self->{nc};
2566          !!!next-input-character;          !!!next-input-character;
2567          if ($self->{next_char} == 0x0055 or # U          redo A;
2568              $self->{next_char} == 0x0075) { # u        } elsif ($self->{nc} == 0x0053 or # S
2569            !!!next-input-character;                 $self->{nc} == 0x0073) { # s
2570            if ($self->{next_char} == 0x0042 or # B          $self->{state} = SYSTEM_STATE;
2571                $self->{next_char} == 0x0062) { # b          $self->{s_kwd} = chr $self->{nc};
             !!!next-input-character;  
             if ($self->{next_char} == 0x004C or # L  
                 $self->{next_char} == 0x006C) { # l  
               !!!next-input-character;  
               if ($self->{next_char} == 0x0049 or # I  
                   $self->{next_char} == 0x0069) { # i  
                 !!!next-input-character;  
                 if ($self->{next_char} == 0x0043 or # C  
                     $self->{next_char} == 0x0063) { # c  
                   !!!cp (168);  
                   $self->{state} = BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE;  
                   !!!next-input-character;  
                   redo A;  
                 } else {  
                   !!!cp (169);  
                 }  
               } else {  
                 !!!cp (170);  
               }  
             } else {  
               !!!cp (171);  
             }  
           } else {  
             !!!cp (172);  
           }  
         } else {  
           !!!cp (173);  
         }  
   
         #  
       } elsif ($self->{next_char} == 0x0053 or # S  
                $self->{next_char} == 0x0073) { # s  
2572          !!!next-input-character;          !!!next-input-character;
2573          if ($self->{next_char} == 0x0059 or # Y          redo A;
             $self->{next_char} == 0x0079) { # y  
           !!!next-input-character;  
           if ($self->{next_char} == 0x0053 or # S  
               $self->{next_char} == 0x0073) { # s  
             !!!next-input-character;  
             if ($self->{next_char} == 0x0054 or # T  
                 $self->{next_char} == 0x0074) { # t  
               !!!next-input-character;  
               if ($self->{next_char} == 0x0045 or # E  
                   $self->{next_char} == 0x0065) { # e  
                 !!!next-input-character;  
                 if ($self->{next_char} == 0x004D or # M  
                     $self->{next_char} == 0x006D) { # m  
                   !!!cp (174);  
                   $self->{state} = BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE;  
                   !!!next-input-character;  
                   redo A;  
                 } else {  
                   !!!cp (175);  
                 }  
               } else {  
                 !!!cp (176);  
               }  
             } else {  
               !!!cp (177);  
             }  
           } else {  
             !!!cp (178);  
           }  
         } else {  
           !!!cp (179);  
         }  
   
         #  
2574        } else {        } else {
2575          !!!cp (180);          !!!cp (180);
2576            !!!parse-error (type => 'string after DOCTYPE name');
2577            $self->{ct}->{quirks} = 1;
2578    
2579            $self->{state} = BOGUS_DOCTYPE_STATE;
2580          !!!next-input-character;          !!!next-input-character;
2581          #          redo A;
2582        }        }
2583        } elsif ($self->{state} == PUBLIC_STATE) {
2584          ## ASCII case-insensitive
2585          if ($self->{nc} == [
2586                undef,
2587                0x0055, # U
2588                0x0042, # B
2589                0x004C, # L
2590                0x0049, # I
2591              ]->[length $self->{s_kwd}] or
2592              $self->{nc} == [
2593                undef,
2594                0x0075, # u
2595                0x0062, # b
2596                0x006C, # l
2597                0x0069, # i
2598              ]->[length $self->{s_kwd}]) {
2599            !!!cp (175);
2600            ## Stay in the state.
2601            $self->{s_kwd} .= chr $self->{nc};
2602            !!!next-input-character;
2603            redo A;
2604          } elsif ((length $self->{s_kwd}) == 5 and
2605                   ($self->{nc} == 0x0043 or # C
2606                    $self->{nc} == 0x0063)) { # c
2607            !!!cp (168);
2608            $self->{state} = BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2609            !!!next-input-character;
2610            redo A;
2611          } else {
2612            !!!cp (169);
2613            !!!parse-error (type => 'string after DOCTYPE name',
2614                            line => $self->{line_prev},
2615                            column => $self->{column_prev} + 1 - length $self->{s_kwd});
2616            $self->{ct}->{quirks} = 1;
2617    
2618        !!!parse-error (type => 'string after DOCTYPE name');          $self->{state} = BOGUS_DOCTYPE_STATE;
2619        $self->{current_token}->{quirks} = 1;          ## Reconsume.
2620            redo A;
2621          }
2622        } elsif ($self->{state} == SYSTEM_STATE) {
2623          ## ASCII case-insensitive
2624          if ($self->{nc} == [
2625                undef,
2626                0x0059, # Y
2627                0x0053, # S
2628                0x0054, # T
2629                0x0045, # E
2630              ]->[length $self->{s_kwd}] or
2631              $self->{nc} == [
2632                undef,
2633                0x0079, # y
2634                0x0073, # s
2635                0x0074, # t
2636                0x0065, # e
2637              ]->[length $self->{s_kwd}]) {
2638            !!!cp (170);
2639            ## Stay in the state.
2640            $self->{s_kwd} .= chr $self->{nc};
2641            !!!next-input-character;
2642            redo A;
2643          } elsif ((length $self->{s_kwd}) == 5 and
2644                   ($self->{nc} == 0x004D or # M
2645                    $self->{nc} == 0x006D)) { # m
2646            !!!cp (171);
2647            $self->{state} = BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2648            !!!next-input-character;
2649            redo A;
2650          } else {
2651            !!!cp (172);
2652            !!!parse-error (type => 'string after DOCTYPE name',
2653                            line => $self->{line_prev},
2654                            column => $self->{column_prev} + 1 - length $self->{s_kwd});
2655            $self->{ct}->{quirks} = 1;
2656    
2657        $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2658        # next-input-character is already done          ## Reconsume.
2659        redo A;          redo A;
2660          }
2661      } elsif ($self->{state} == BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {      } elsif ($self->{state} == BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {
2662        if ({        if ($is_space->{$self->{nc}}) {
             0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,  
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_char}}) {  
2663          !!!cp (181);          !!!cp (181);
2664          ## Stay in the state          ## Stay in the state
2665          !!!next-input-character;          !!!next-input-character;
2666          redo A;          redo A;
2667        } elsif ($self->{next_char} eq 0x0022) { # "        } elsif ($self->{nc} eq 0x0022) { # "
2668          !!!cp (182);          !!!cp (182);
2669          $self->{current_token}->{public_identifier} = ''; # DOCTYPE          $self->{ct}->{pubid} = ''; # DOCTYPE
2670          $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE;          $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE;
2671          !!!next-input-character;          !!!next-input-character;
2672          redo A;          redo A;
2673        } elsif ($self->{next_char} eq 0x0027) { # '        } elsif ($self->{nc} eq 0x0027) { # '
2674          !!!cp (183);          !!!cp (183);
2675          $self->{current_token}->{public_identifier} = ''; # DOCTYPE          $self->{ct}->{pubid} = ''; # DOCTYPE
2676          $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE;          $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE;
2677          !!!next-input-character;          !!!next-input-character;
2678          redo A;          redo A;
2679        } elsif ($self->{next_char} eq 0x003E) { # >        } elsif ($self->{nc} eq 0x003E) { # >
2680          !!!cp (184);          !!!cp (184);
2681          !!!parse-error (type => 'no PUBLIC literal');          !!!parse-error (type => 'no PUBLIC literal');
2682    
2683          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2684          !!!next-input-character;          !!!next-input-character;
2685    
2686          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2687          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2688    
2689          redo A;          redo A;
2690        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2691          !!!cp (185);          !!!cp (185);
2692          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2693    
2694          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2695          ## reconsume          ## reconsume
2696    
2697          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2698          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2699    
2700          redo A;          redo A;
2701        } else {        } else {
2702          !!!cp (186);          !!!cp (186);
2703          !!!parse-error (type => 'string after PUBLIC');          !!!parse-error (type => 'string after PUBLIC');
2704          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2705    
2706          $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2707          !!!next-input-character;          !!!next-input-character;
2708          redo A;          redo A;
2709        }        }
2710      } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE) {      } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE) {
2711        if ($self->{next_char} == 0x0022) { # "        if ($self->{nc} == 0x0022) { # "
2712          !!!cp (187);          !!!cp (187);
2713          $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;          $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2714          !!!next-input-character;          !!!next-input-character;
2715          redo A;          redo A;
2716        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2717          !!!cp (188);          !!!cp (188);
2718          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed PUBLIC literal');
2719    
2720          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2721          !!!next-input-character;          !!!next-input-character;
2722    
2723          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2724          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2725    
2726          redo A;          redo A;
2727        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2728          !!!cp (189);          !!!cp (189);
2729          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed PUBLIC literal');
2730    
2731          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2732          ## reconsume          ## reconsume
2733    
2734          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2735          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2736    
2737          redo A;          redo A;
2738        } else {        } else {
2739          !!!cp (190);          !!!cp (190);
2740          $self->{current_token}->{public_identifier} # DOCTYPE          $self->{ct}->{pubid} # DOCTYPE
2741              .= chr $self->{next_char};              .= chr $self->{nc};
2742            $self->{read_until}->($self->{ct}->{pubid}, q[">],
2743                                  length $self->{ct}->{pubid});
2744    
2745          ## Stay in the state          ## Stay in the state
2746          !!!next-input-character;          !!!next-input-character;
2747          redo A;          redo A;
2748        }        }
2749      } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE) {      } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE) {
2750        if ($self->{next_char} == 0x0027) { # '        if ($self->{nc} == 0x0027) { # '
2751          !!!cp (191);          !!!cp (191);
2752          $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;          $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2753          !!!next-input-character;          !!!next-input-character;
2754          redo A;          redo A;
2755        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2756          !!!cp (192);          !!!cp (192);
2757          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed PUBLIC literal');
2758    
2759          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2760          !!!next-input-character;          !!!next-input-character;
2761    
2762          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2763          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2764    
2765          redo A;          redo A;
2766        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2767          !!!cp (193);          !!!cp (193);
2768          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed PUBLIC literal');
2769    
2770          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2771          ## reconsume          ## reconsume
2772    
2773          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2774          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2775    
2776          redo A;          redo A;
2777        } else {        } else {
2778          !!!cp (194);          !!!cp (194);
2779          $self->{current_token}->{public_identifier} # DOCTYPE          $self->{ct}->{pubid} # DOCTYPE
2780              .= chr $self->{next_char};              .= chr $self->{nc};
2781            $self->{read_until}->($self->{ct}->{pubid}, q['>],
2782                                  length $self->{ct}->{pubid});
2783    
2784          ## Stay in the state          ## Stay in the state
2785          !!!next-input-character;          !!!next-input-character;
2786          redo A;          redo A;
2787        }        }
2788      } elsif ($self->{state} == AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {      } elsif ($self->{state} == AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {
2789        if ({        if ($is_space->{$self->{nc}}) {
             0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,  
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_char}}) {  
2790          !!!cp (195);          !!!cp (195);
2791          ## Stay in the state          ## Stay in the state
2792          !!!next-input-character;          !!!next-input-character;
2793          redo A;          redo A;
2794        } elsif ($self->{next_char} == 0x0022) { # "        } elsif ($self->{nc} == 0x0022) { # "
2795          !!!cp (196);          !!!cp (196);
2796          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          $self->{ct}->{sysid} = ''; # DOCTYPE
2797          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;
2798          !!!next-input-character;          !!!next-input-character;
2799          redo A;          redo A;
2800        } elsif ($self->{next_char} == 0x0027) { # '        } elsif ($self->{nc} == 0x0027) { # '
2801          !!!cp (197);          !!!cp (197);
2802          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          $self->{ct}->{sysid} = ''; # DOCTYPE
2803          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;
2804          !!!next-input-character;          !!!next-input-character;
2805          redo A;          redo A;
2806        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2807          !!!cp (198);          !!!cp (198);
2808          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2809          !!!next-input-character;          !!!next-input-character;
2810    
2811          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2812    
2813          redo A;          redo A;
2814        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2815          !!!cp (199);          !!!cp (199);
2816          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2817    
2818          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2819          ## reconsume          ## reconsume
2820    
2821          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2822          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2823    
2824          redo A;          redo A;
2825        } else {        } else {
2826          !!!cp (200);          !!!cp (200);
2827          !!!parse-error (type => 'string after PUBLIC literal');          !!!parse-error (type => 'string after PUBLIC literal');
2828          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2829    
2830          $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2831          !!!next-input-character;          !!!next-input-character;
2832          redo A;          redo A;
2833        }        }
2834      } elsif ($self->{state} == BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {      } elsif ($self->{state} == BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {
2835        if ({        if ($is_space->{$self->{nc}}) {
             0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,  
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_char}}) {  
2836          !!!cp (201);          !!!cp (201);
2837          ## Stay in the state          ## Stay in the state
2838          !!!next-input-character;          !!!next-input-character;
2839          redo A;          redo A;
2840        } elsif ($self->{next_char} == 0x0022) { # "        } elsif ($self->{nc} == 0x0022) { # "
2841          !!!cp (202);          !!!cp (202);
2842          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          $self->{ct}->{sysid} = ''; # DOCTYPE
2843          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;
2844          !!!next-input-character;          !!!next-input-character;
2845          redo A;          redo A;
2846        } elsif ($self->{next_char} == 0x0027) { # '        } elsif ($self->{nc} == 0x0027) { # '
2847          !!!cp (203);          !!!cp (203);
2848          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          $self->{ct}->{sysid} = ''; # DOCTYPE
2849          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;
2850          !!!next-input-character;          !!!next-input-character;
2851          redo A;          redo A;
2852        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2853          !!!cp (204);          !!!cp (204);
2854          !!!parse-error (type => 'no SYSTEM literal');          !!!parse-error (type => 'no SYSTEM literal');
2855          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2856          !!!next-input-character;          !!!next-input-character;
2857    
2858          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2859          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2860    
2861          redo A;          redo A;
2862        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2863          !!!cp (205);          !!!cp (205);
2864          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2865    
2866          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2867          ## reconsume          ## reconsume
2868    
2869          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2870          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2871    
2872          redo A;          redo A;
2873        } else {        } else {
2874          !!!cp (206);          !!!cp (206);
2875          !!!parse-error (type => 'string after SYSTEM');          !!!parse-error (type => 'string after SYSTEM');
2876          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2877    
2878          $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2879          !!!next-input-character;          !!!next-input-character;
2880          redo A;          redo A;
2881        }        }
2882      } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE) {      } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE) {
2883        if ($self->{next_char} == 0x0022) { # "        if ($self->{nc} == 0x0022) { # "
2884          !!!cp (207);          !!!cp (207);
2885          $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;          $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2886          !!!next-input-character;          !!!next-input-character;
2887          redo A;          redo A;
2888        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2889          !!!cp (208);          !!!cp (208);
2890          !!!parse-error (type => 'unclosed SYSTEM literal');          !!!parse-error (type => 'unclosed SYSTEM literal');
2891    
2892          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2893          !!!next-input-character;          !!!next-input-character;
2894    
2895          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2896          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2897    
2898          redo A;          redo A;
2899        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2900          !!!cp (209);          !!!cp (209);
2901          !!!parse-error (type => 'unclosed SYSTEM literal');          !!!parse-error (type => 'unclosed SYSTEM literal');
2902    
2903          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2904          ## reconsume          ## reconsume
2905    
2906          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2907          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2908    
2909          redo A;          redo A;
2910        } else {        } else {
2911          !!!cp (210);          !!!cp (210);
2912          $self->{current_token}->{system_identifier} # DOCTYPE          $self->{ct}->{sysid} # DOCTYPE
2913              .= chr $self->{next_char};              .= chr $self->{nc};
2914            $self->{read_until}->($self->{ct}->{sysid}, q[">],
2915                                  length $self->{ct}->{sysid});
2916    
2917          ## Stay in the state          ## Stay in the state
2918          !!!next-input-character;          !!!next-input-character;
2919          redo A;          redo A;
2920        }        }
2921      } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE) {      } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE) {
2922        if ($self->{next_char} == 0x0027) { # '        if ($self->{nc} == 0x0027) { # '
2923          !!!cp (211);          !!!cp (211);
2924          $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;          $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2925          !!!next-input-character;          !!!next-input-character;
2926          redo A;          redo A;
2927        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2928          !!!cp (212);          !!!cp (212);
2929          !!!parse-error (type => 'unclosed SYSTEM literal');          !!!parse-error (type => 'unclosed SYSTEM literal');
2930    
2931          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2932          !!!next-input-character;          !!!next-input-character;
2933    
2934          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2935          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2936    
2937          redo A;          redo A;
2938        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2939          !!!cp (213);          !!!cp (213);
2940          !!!parse-error (type => 'unclosed SYSTEM literal');          !!!parse-error (type => 'unclosed SYSTEM literal');
2941    
2942          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2943          ## reconsume          ## reconsume
2944    
2945          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2946          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2947    
2948          redo A;          redo A;
2949        } else {        } else {
2950          !!!cp (214);          !!!cp (214);
2951          $self->{current_token}->{system_identifier} # DOCTYPE          $self->{ct}->{sysid} # DOCTYPE
2952              .= chr $self->{next_char};              .= chr $self->{nc};
2953            $self->{read_until}->($self->{ct}->{sysid}, q['>],
2954                                  length $self->{ct}->{sysid});
2955    
2956          ## Stay in the state          ## Stay in the state
2957          !!!next-input-character;          !!!next-input-character;
2958          redo A;          redo A;
2959        }        }
2960      } elsif ($self->{state} == AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {      } elsif ($self->{state} == AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {
2961        if ({        if ($is_space->{$self->{nc}}) {
             0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,  
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_char}}) {  
2962          !!!cp (215);          !!!cp (215);
2963          ## Stay in the state          ## Stay in the state
2964          !!!next-input-character;          !!!next-input-character;
2965          redo A;          redo A;
2966        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2967          !!!cp (216);          !!!cp (216);
2968          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2969          !!!next-input-character;          !!!next-input-character;
2970    
2971          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2972    
2973          redo A;          redo A;
2974        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2975          !!!cp (217);          !!!cp (217);
2976          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2977          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2978          ## reconsume          ## reconsume
2979    
2980          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2981          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2982    
2983          redo A;          redo A;
2984        } else {        } else {
2985          !!!cp (218);          !!!cp (218);
2986          !!!parse-error (type => 'string after SYSTEM literal');          !!!parse-error (type => 'string after SYSTEM literal');
2987          #$self->{current_token}->{quirks} = 1;          #$self->{ct}->{quirks} = 1;
2988    
2989          $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2990          !!!next-input-character;          !!!next-input-character;
2991          redo A;          redo A;
2992        }        }
2993      } elsif ($self->{state} == BOGUS_DOCTYPE_STATE) {      } elsif ($self->{state} == BOGUS_DOCTYPE_STATE) {
2994        if ($self->{next_char} == 0x003E) { # >        if ($self->{nc} == 0x003E) { # >
2995          !!!cp (219);          !!!cp (219);
2996          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2997          !!!next-input-character;          !!!next-input-character;
2998    
2999          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
3000    
3001          redo A;          redo A;
3002        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
3003          !!!cp (220);          !!!cp (220);
         !!!parse-error (type => 'unclosed DOCTYPE');  
3004          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
3005          ## reconsume          ## reconsume
3006    
3007          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
3008    
3009          redo A;          redo A;
3010        } else {        } else {
3011          !!!cp (221);          !!!cp (221);
3012            my $s = '';
3013            $self->{read_until}->($s, q[>], 0);
3014    
3015          ## Stay in the state          ## Stay in the state
3016          !!!next-input-character;          !!!next-input-character;
3017          redo A;          redo A;
3018        }        }
3019      } elsif ($self->{state} == CDATA_BLOCK_STATE) {      } elsif ($self->{state} == CDATA_SECTION_STATE) {
3020        my $s = '';        ## NOTE: "CDATA section state" in the state is jointly implemented
3021          ## by three states, |CDATA_SECTION_STATE|, |CDATA_SECTION_MSE1_STATE|,
3022          ## and |CDATA_SECTION_MSE2_STATE|.
3023                
3024        my ($l, $c) = ($self->{line}, $self->{column});        if ($self->{nc} == 0x005D) { # ]
3025            !!!cp (221.1);
3026        CS: while ($self->{next_char} != -1) {          $self->{state} = CDATA_SECTION_MSE1_STATE;
3027          if ($self->{next_char} == 0x005D) { # ]          !!!next-input-character;
3028            !!!next-input-character;          redo A;
3029            if ($self->{next_char} == 0x005D) { # ]        } elsif ($self->{nc} == -1) {
3030              !!!next-input-character;          $self->{state} = DATA_STATE;
             MDC: {  
               if ($self->{next_char} == 0x003E) { # >  
                 !!!cp (221.1);  
                 !!!next-input-character;  
                 last CS;  
               } elsif ($self->{next_char} == 0x005D) { # ]  
                 !!!cp (221.2);  
                 $s .= ']';  
                 !!!next-input-character;  
                 redo MDC;  
               } else {  
                 !!!cp (221.3);  
                 $s .= ']]';  
                 #  
               }  
             } # MDC  
           } else {  
             !!!cp (221.4);  
             $s .= ']';  
             #  
           }  
         } else {  
           !!!cp (221.5);  
           #  
         }  
         $s .= chr $self->{next_char};  
3031          !!!next-input-character;          !!!next-input-character;
3032        } # CS          if (length $self->{ct}->{data}) { # character
3033              !!!cp (221.2);
3034              !!!emit ($self->{ct}); # character
3035            } else {
3036              !!!cp (221.3);
3037              ## No token to emit. $self->{ct} is discarded.
3038            }        
3039            redo A;
3040          } else {
3041            !!!cp (221.4);
3042            $self->{ct}->{data} .= chr $self->{nc};
3043            $self->{read_until}->($self->{ct}->{data},
3044                                  q<]>,
3045                                  length $self->{ct}->{data});
3046    
3047        $self->{state} = DATA_STATE;          ## Stay in the state.
3048        ## next-input-character done or EOF, which is reconsumed.          !!!next-input-character;
3049            redo A;
3050          }
3051    
3052        if (length $s) {        ## ISSUE: "text tokens" in spec.
3053        } elsif ($self->{state} == CDATA_SECTION_MSE1_STATE) {
3054          if ($self->{nc} == 0x005D) { # ]
3055            !!!cp (221.5);
3056            $self->{state} = CDATA_SECTION_MSE2_STATE;
3057            !!!next-input-character;
3058            redo A;
3059          } else {
3060          !!!cp (221.6);          !!!cp (221.6);
3061          !!!emit ({type => CHARACTER_TOKEN, data => $s,          $self->{ct}->{data} .= ']';
3062                    line => $l, column => $c});          $self->{state} = CDATA_SECTION_STATE;
3063            ## Reconsume.
3064            redo A;
3065          }
3066        } elsif ($self->{state} == CDATA_SECTION_MSE2_STATE) {
3067          if ($self->{nc} == 0x003E) { # >
3068            $self->{state} = DATA_STATE;
3069            !!!next-input-character;
3070            if (length $self->{ct}->{data}) { # character
3071              !!!cp (221.7);
3072              !!!emit ($self->{ct}); # character
3073            } else {
3074              !!!cp (221.8);
3075              ## No token to emit. $self->{ct} is discarded.
3076            }
3077            redo A;
3078          } elsif ($self->{nc} == 0x005D) { # ]
3079            !!!cp (221.9); # character
3080            $self->{ct}->{data} .= ']'; ## Add first "]" of "]]]".
3081            ## Stay in the state.
3082            !!!next-input-character;
3083            redo A;
3084        } else {        } else {
3085          !!!cp (221.7);          !!!cp (221.11);
3086            $self->{ct}->{data} .= ']]'; # character
3087            $self->{state} = CDATA_SECTION_STATE;
3088            ## Reconsume.
3089            redo A;
3090          }
3091        } elsif ($self->{state} == ENTITY_STATE) {
3092          if ($is_space->{$self->{nc}} or
3093              {
3094                0x003C => 1, 0x0026 => 1, -1 => 1, # <, &
3095                $self->{entity_add} => 1,
3096              }->{$self->{nc}}) {
3097            !!!cp (1001);
3098            ## Don't consume
3099            ## No error
3100            ## Return nothing.
3101            #
3102          } elsif ($self->{nc} == 0x0023) { # #
3103            !!!cp (999);
3104            $self->{state} = ENTITY_HASH_STATE;
3105            $self->{s_kwd} = '#';
3106            !!!next-input-character;
3107            redo A;
3108          } elsif ((0x0041 <= $self->{nc} and
3109                    $self->{nc} <= 0x005A) or # A..Z
3110                   (0x0061 <= $self->{nc} and
3111                    $self->{nc} <= 0x007A)) { # a..z
3112            !!!cp (998);
3113            require Whatpm::_NamedEntityList;
3114            $self->{state} = ENTITY_NAME_STATE;
3115            $self->{s_kwd} = chr $self->{nc};
3116            $self->{entity__value} = $self->{s_kwd};
3117            $self->{entity__match} = 0;
3118            !!!next-input-character;
3119            redo A;
3120          } else {
3121            !!!cp (1027);
3122            !!!parse-error (type => 'bare ero');
3123            ## Return nothing.
3124            #
3125        }        }
3126    
3127        redo A;        ## NOTE: No character is consumed by the "consume a character
3128          ## reference" algorithm.  In other word, there is an "&" character
3129        ## ISSUE: "text tokens" in spec.        ## that does not introduce a character reference, which would be
3130        ## TODO: Streaming support        ## appended to the parent element or the attribute value in later
3131      } else {        ## process of the tokenizer.
3132        die "$0: $self->{state}: Unknown state";  
3133      }        if ($self->{prev_state} == DATA_STATE) {
3134    } # A            !!!cp (997);
3135            $self->{state} = $self->{prev_state};
3136    die "$0: _get_next_token: unexpected case";          ## Reconsume.
3137  } # _get_next_token          !!!emit ({type => CHARACTER_TOKEN, data => '&',
3138                      line => $self->{line_prev},
3139  sub _tokenize_attempt_to_consume_an_entity ($$$) {                    column => $self->{column_prev},
3140    my ($self, $in_attr, $additional) = @_;                   });
3141            redo A;
3142    my ($l, $c) = ($self->{line_prev}, $self->{column_prev});        } else {
3143            !!!cp (996);
3144            $self->{ca}->{value} .= '&';
3145            $self->{state} = $self->{prev_state};
3146            ## Reconsume.
3147            redo A;
3148          }
3149        } elsif ($self->{state} == ENTITY_HASH_STATE) {
3150          if ($self->{nc} == 0x0078 or # x
3151              $self->{nc} == 0x0058) { # X
3152            !!!cp (995);
3153            $self->{state} = HEXREF_X_STATE;
3154            $self->{s_kwd} .= chr $self->{nc};
3155            !!!next-input-character;
3156            redo A;
3157          } elsif (0x0030 <= $self->{nc} and
3158                   $self->{nc} <= 0x0039) { # 0..9
3159            !!!cp (994);
3160            $self->{state} = NCR_NUM_STATE;
3161            $self->{s_kwd} = $self->{nc} - 0x0030;
3162            !!!next-input-character;
3163            redo A;
3164          } else {
3165            !!!parse-error (type => 'bare nero',
3166                            line => $self->{line_prev},
3167                            column => $self->{column_prev} - 1);
3168    
3169    if ({          ## NOTE: According to the spec algorithm, nothing is returned,
3170         0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, # HT, LF, VT, FF,          ## and then "&#" is appended to the parent element or the attribute
3171         0x0020 => 1, 0x003C => 1, 0x0026 => 1, -1 => 1, # SP, <, & # 0x000D # CR          ## value in the later processing.
3172         $additional => 1,  
3173        }->{$self->{next_char}}) {          if ($self->{prev_state} == DATA_STATE) {
3174      !!!cp (1001);            !!!cp (1019);
3175      ## Don't consume            $self->{state} = $self->{prev_state};
3176      ## No error            ## Reconsume.
3177      return undef;            !!!emit ({type => CHARACTER_TOKEN,
3178    } elsif ($self->{next_char} == 0x0023) { # #                      data => '&#',
3179      !!!next-input-character;                      line => $self->{line_prev},
3180      if ($self->{next_char} == 0x0078 or # x                      column => $self->{column_prev} - 1,
3181          $self->{next_char} == 0x0058) { # X                     });
3182        my $code;            redo A;
       X: {  
         my $x_char = $self->{next_char};  
         !!!next-input-character;  
         if (0x0030 <= $self->{next_char} and  
             $self->{next_char} <= 0x0039) { # 0..9  
           !!!cp (1002);  
           $code ||= 0;  
           $code *= 0x10;  
           $code += $self->{next_char} - 0x0030;  
           redo X;  
         } elsif (0x0061 <= $self->{next_char} and  
                  $self->{next_char} <= 0x0066) { # a..f  
           !!!cp (1003);  
           $code ||= 0;  
           $code *= 0x10;  
           $code += $self->{next_char} - 0x0060 + 9;  
           redo X;  
         } elsif (0x0041 <= $self->{next_char} and  
                  $self->{next_char} <= 0x0046) { # A..F  
           !!!cp (1004);  
           $code ||= 0;  
           $code *= 0x10;  
           $code += $self->{next_char} - 0x0040 + 9;  
           redo X;  
         } elsif (not defined $code) { # no hexadecimal digit  
           !!!cp (1005);  
           !!!parse-error (type => 'bare hcro', line => $l, column => $c);  
           !!!back-next-input-character ($x_char, $self->{next_char});  
           $self->{next_char} = 0x0023; # #  
           return undef;  
         } elsif ($self->{next_char} == 0x003B) { # ;  
           !!!cp (1006);  
           !!!next-input-character;  
3183          } else {          } else {
3184            !!!cp (1007);            !!!cp (993);
3185            !!!parse-error (type => 'no refc', line => $l, column => $c);            $self->{ca}->{value} .= '&#';
3186              $self->{state} = $self->{prev_state};
3187              ## Reconsume.
3188              redo A;
3189          }          }
3190          }
3191          if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) {      } elsif ($self->{state} == NCR_NUM_STATE) {
3192            !!!cp (1008);        if (0x0030 <= $self->{nc} and
3193            !!!parse-error (type => 'invalid character reference',            $self->{nc} <= 0x0039) { # 0..9
                           text => (sprintf 'U+%04X', $code),  
                           line => $l, column => $c);  
           $code = 0xFFFD;  
         } elsif ($code > 0x10FFFF) {  
           !!!cp (1009);  
           !!!parse-error (type => 'invalid character reference',  
                           text => (sprintf 'U-%08X', $code),  
                           line => $l, column => $c);  
           $code = 0xFFFD;  
         } elsif ($code == 0x000D) {  
           !!!cp (1010);  
           !!!parse-error (type => 'CR character reference', line => $l, column => $c);  
           $code = 0x000A;  
         } elsif (0x80 <= $code and $code <= 0x9F) {  
           !!!cp (1011);  
           !!!parse-error (type => 'C1 character reference', text => (sprintf 'U+%04X', $code), line => $l, column => $c);  
           $code = $c1_entity_char->{$code};  
         }  
   
         return {type => CHARACTER_TOKEN, data => chr $code,  
                 has_reference => 1,  
                 line => $l, column => $c,  
                };  
       } # X  
     } elsif (0x0030 <= $self->{next_char} and  
              $self->{next_char} <= 0x0039) { # 0..9  
       my $code = $self->{next_char} - 0x0030;  
       !!!next-input-character;  
         
       while (0x0030 <= $self->{next_char} and  
                 $self->{next_char} <= 0x0039) { # 0..9  
3194          !!!cp (1012);          !!!cp (1012);
3195          $code *= 10;          $self->{s_kwd} *= 10;
3196          $code += $self->{next_char} - 0x0030;          $self->{s_kwd} += $self->{nc} - 0x0030;
3197                    
3198            ## Stay in the state.
3199          !!!next-input-character;          !!!next-input-character;
3200        }          redo A;
3201          } elsif ($self->{nc} == 0x003B) { # ;
       if ($self->{next_char} == 0x003B) { # ;  
3202          !!!cp (1013);          !!!cp (1013);
3203          !!!next-input-character;          !!!next-input-character;
3204            #
3205        } else {        } else {
3206          !!!cp (1014);          !!!cp (1014);
3207          !!!parse-error (type => 'no refc', line => $l, column => $c);          !!!parse-error (type => 'no refc');
3208            ## Reconsume.
3209            #
3210        }        }
3211    
3212        if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) {        my $code = $self->{s_kwd};
3213          my $l = $self->{line_prev};
3214          my $c = $self->{column_prev};
3215          if ($charref_map->{$code}) {
3216          !!!cp (1015);          !!!cp (1015);
3217          !!!parse-error (type => 'invalid character reference',          !!!parse-error (type => 'invalid character reference',
3218                          text => (sprintf 'U+%04X', $code),                          text => (sprintf 'U+%04X', $code),
3219                          line => $l, column => $c);                          line => $l, column => $c);
3220          $code = 0xFFFD;          $code = $charref_map->{$code};
3221        } elsif ($code > 0x10FFFF) {        } elsif ($code > 0x10FFFF) {
3222          !!!cp (1016);          !!!cp (1016);
3223          !!!parse-error (type => 'invalid character reference',          !!!parse-error (type => 'invalid character reference',
3224                          text => (sprintf 'U-%08X', $code),                          text => (sprintf 'U-%08X', $code),
3225                          line => $l, column => $c);                          line => $l, column => $c);
3226          $code = 0xFFFD;          $code = 0xFFFD;
3227        } elsif ($code == 0x000D) {        }
3228          !!!cp (1017);  
3229          !!!parse-error (type => 'CR character reference',        if ($self->{prev_state} == DATA_STATE) {
3230                          line => $l, column => $c);          !!!cp (992);
3231          $code = 0x000A;          $self->{state} = $self->{prev_state};
3232        } elsif (0x80 <= $code and $code <= 0x9F) {          ## Reconsume.
3233          !!!cp (1018);          !!!emit ({type => CHARACTER_TOKEN, data => chr $code,
3234          !!!parse-error (type => 'C1 character reference',                    line => $l, column => $c,
3235                     });
3236            redo A;
3237          } else {
3238            !!!cp (991);
3239            $self->{ca}->{value} .= chr $code;
3240            $self->{ca}->{has_reference} = 1;
3241            $self->{state} = $self->{prev_state};
3242            ## Reconsume.
3243            redo A;
3244          }
3245        } elsif ($self->{state} == HEXREF_X_STATE) {
3246          if ((0x0030 <= $self->{nc} and $self->{nc} <= 0x0039) or
3247              (0x0041 <= $self->{nc} and $self->{nc} <= 0x0046) or
3248              (0x0061 <= $self->{nc} and $self->{nc} <= 0x0066)) {
3249            # 0..9, A..F, a..f
3250            !!!cp (990);
3251            $self->{state} = HEXREF_HEX_STATE;
3252            $self->{s_kwd} = 0;
3253            ## Reconsume.
3254            redo A;
3255          } else {
3256            !!!parse-error (type => 'bare hcro',
3257                            line => $self->{line_prev},
3258                            column => $self->{column_prev} - 2);
3259    
3260            ## NOTE: According to the spec algorithm, nothing is returned,
3261            ## and then "&#" followed by "X" or "x" is appended to the parent
3262            ## element or the attribute value in the later processing.
3263    
3264            if ($self->{prev_state} == DATA_STATE) {
3265              !!!cp (1005);
3266              $self->{state} = $self->{prev_state};
3267              ## Reconsume.
3268              !!!emit ({type => CHARACTER_TOKEN,
3269                        data => '&' . $self->{s_kwd},
3270                        line => $self->{line_prev},
3271                        column => $self->{column_prev} - length $self->{s_kwd},
3272                       });
3273              redo A;
3274            } else {
3275              !!!cp (989);
3276              $self->{ca}->{value} .= '&' . $self->{s_kwd};
3277              $self->{state} = $self->{prev_state};
3278              ## Reconsume.
3279              redo A;
3280            }
3281          }
3282        } elsif ($self->{state} == HEXREF_HEX_STATE) {
3283          if (0x0030 <= $self->{nc} and $self->{nc} <= 0x0039) {
3284            # 0..9
3285            !!!cp (1002);
3286            $self->{s_kwd} *= 0x10;
3287            $self->{s_kwd} += $self->{nc} - 0x0030;
3288            ## Stay in the state.
3289            !!!next-input-character;
3290            redo A;
3291          } elsif (0x0061 <= $self->{nc} and
3292                   $self->{nc} <= 0x0066) { # a..f
3293            !!!cp (1003);
3294            $self->{s_kwd} *= 0x10;
3295            $self->{s_kwd} += $self->{nc} - 0x0060 + 9;
3296            ## Stay in the state.
3297            !!!next-input-character;
3298            redo A;
3299          } elsif (0x0041 <= $self->{nc} and
3300                   $self->{nc} <= 0x0046) { # A..F
3301            !!!cp (1004);
3302            $self->{s_kwd} *= 0x10;
3303            $self->{s_kwd} += $self->{nc} - 0x0040 + 9;
3304            ## Stay in the state.
3305            !!!next-input-character;
3306            redo A;
3307          } elsif ($self->{nc} == 0x003B) { # ;
3308            !!!cp (1006);
3309            !!!next-input-character;
3310            #
3311          } else {
3312            !!!cp (1007);
3313            !!!parse-error (type => 'no refc',
3314                            line => $self->{line},
3315                            column => $self->{column});
3316            ## Reconsume.
3317            #
3318          }
3319    
3320          my $code = $self->{s_kwd};
3321          my $l = $self->{line_prev};
3322          my $c = $self->{column_prev};
3323          if ($charref_map->{$code}) {
3324            !!!cp (1008);
3325            !!!parse-error (type => 'invalid character reference',
3326                          text => (sprintf 'U+%04X', $code),                          text => (sprintf 'U+%04X', $code),
3327                          line => $l, column => $c);                          line => $l, column => $c);
3328          $code = $c1_entity_char->{$code};          $code = $charref_map->{$code};
3329          } elsif ($code > 0x10FFFF) {
3330            !!!cp (1009);
3331            !!!parse-error (type => 'invalid character reference',
3332                            text => (sprintf 'U-%08X', $code),
3333                            line => $l, column => $c);
3334            $code = 0xFFFD;
3335        }        }
3336          
3337        return {type => CHARACTER_TOKEN, data => chr $code, has_reference => 1,        if ($self->{prev_state} == DATA_STATE) {
3338                line => $l, column => $c,          !!!cp (988);
3339               };          $self->{state} = $self->{prev_state};
3340      } else {          ## Reconsume.
3341        !!!cp (1019);          !!!emit ({type => CHARACTER_TOKEN, data => chr $code,
3342        !!!parse-error (type => 'bare nero', line => $l, column => $c);                    line => $l, column => $c,
3343        !!!back-next-input-character ($self->{next_char});                   });
3344        $self->{next_char} = 0x0023; # #          redo A;
3345        return undef;        } else {
3346      }          !!!cp (987);
3347    } elsif ((0x0041 <= $self->{next_char} and          $self->{ca}->{value} .= chr $code;
3348              $self->{next_char} <= 0x005A) or          $self->{ca}->{has_reference} = 1;
3349             (0x0061 <= $self->{next_char} and          $self->{state} = $self->{prev_state};
3350              $self->{next_char} <= 0x007A)) {          ## Reconsume.
3351      my $entity_name = chr $self->{next_char};          redo A;
3352      !!!next-input-character;        }
3353        } elsif ($self->{state} == ENTITY_NAME_STATE) {
3354      my $value = $entity_name;        if (length $self->{s_kwd} < 30 and
3355      my $match = 0;            ## NOTE: Some number greater than the maximum length of entity name
3356      require Whatpm::_NamedEntityList;            ((0x0041 <= $self->{nc} and # a
3357      our $EntityChar;              $self->{nc} <= 0x005A) or # x
3358               (0x0061 <= $self->{nc} and # a
3359      while (length $entity_name < 30 and              $self->{nc} <= 0x007A) or # z
3360             ## NOTE: Some number greater than the maximum length of entity name             (0x0030 <= $self->{nc} and # 0
3361             ((0x0041 <= $self->{next_char} and # a              $self->{nc} <= 0x0039) or # 9
3362               $self->{next_char} <= 0x005A) or # x             $self->{nc} == 0x003B)) { # ;
3363              (0x0061 <= $self->{next_char} and # a          our $EntityChar;
3364               $self->{next_char} <= 0x007A) or # z          $self->{s_kwd} .= chr $self->{nc};
3365              (0x0030 <= $self->{next_char} and # 0          if (defined $EntityChar->{$self->{s_kwd}}) {
3366               $self->{next_char} <= 0x0039) or # 9            if ($self->{nc} == 0x003B) { # ;
3367              $self->{next_char} == 0x003B)) { # ;              !!!cp (1020);
3368        $entity_name .= chr $self->{next_char};              $self->{entity__value} = $EntityChar->{$self->{s_kwd}};
3369        if (defined $EntityChar->{$entity_name}) {              $self->{entity__match} = 1;
3370          if ($self->{next_char} == 0x003B) { # ;              !!!next-input-character;
3371            !!!cp (1020);              #
3372            $value = $EntityChar->{$entity_name};            } else {
3373            $match = 1;              !!!cp (1021);
3374            !!!next-input-character;              $self->{entity__value} = $EntityChar->{$self->{s_kwd}};
3375            last;              $self->{entity__match} = -1;
3376                ## Stay in the state.
3377                !!!next-input-character;
3378                redo A;
3379              }
3380          } else {          } else {
3381            !!!cp (1021);            !!!cp (1022);
3382            $value = $EntityChar->{$entity_name};            $self->{entity__value} .= chr $self->{nc};
3383            $match = -1;            $self->{entity__match} *= 2;
3384              ## Stay in the state.
3385            !!!next-input-character;            !!!next-input-character;
3386              redo A;
3387            }
3388          }
3389    
3390          my $data;
3391          my $has_ref;
3392          if ($self->{entity__match} > 0) {
3393            !!!cp (1023);
3394            $data = $self->{entity__value};
3395            $has_ref = 1;
3396            #
3397          } elsif ($self->{entity__match} < 0) {
3398            !!!parse-error (type => 'no refc');
3399            if ($self->{prev_state} != DATA_STATE and # in attribute
3400                $self->{entity__match} < -1) {
3401              !!!cp (1024);
3402              $data = '&' . $self->{s_kwd};
3403              #
3404            } else {
3405              !!!cp (1025);
3406              $data = $self->{entity__value};
3407              $has_ref = 1;
3408              #
3409          }          }
3410        } else {        } else {
3411          !!!cp (1022);          !!!cp (1026);
3412          $value .= chr $self->{next_char};          !!!parse-error (type => 'bare ero',
3413          $match *= 2;                          line => $self->{line_prev},
3414          !!!next-input-character;                          column => $self->{column_prev} - length $self->{s_kwd});
3415            $data = '&' . $self->{s_kwd};
3416            #
3417        }        }
3418      }    
3419              ## NOTE: In these cases, when a character reference is found,
3420      if ($match > 0) {        ## it is consumed and a character token is returned, or, otherwise,
3421        !!!cp (1023);        ## nothing is consumed and returned, according to the spec algorithm.
3422        return {type => CHARACTER_TOKEN, data => $value, has_reference => 1,        ## In this implementation, anything that has been examined by the
3423                line => $l, column => $c,        ## tokenizer is appended to the parent element or the attribute value
3424               };        ## as string, either literal string when no character reference or
3425      } elsif ($match < 0) {        ## entity-replaced string otherwise, in this stage, since any characters
3426        !!!parse-error (type => 'no refc', line => $l, column => $c);        ## that would not be consumed are appended in the data state or in an
3427        if ($in_attr and $match < -1) {        ## appropriate attribute value state anyway.
3428          !!!cp (1024);  
3429          return {type => CHARACTER_TOKEN, data => '&'.$entity_name,        if ($self->{prev_state} == DATA_STATE) {
3430                  line => $l, column => $c,          !!!cp (986);
3431                 };          $self->{state} = $self->{prev_state};
3432        } else {          ## Reconsume.
3433          !!!cp (1025);          !!!emit ({type => CHARACTER_TOKEN,
3434          return {type => CHARACTER_TOKEN, data => $value, has_reference => 1,                    data => $data,
3435                  line => $l, column => $c,                    line => $self->{line_prev},
3436                 };                    column => $self->{column_prev} + 1 - length $self->{s_kwd},
3437                     });
3438            redo A;
3439          } else {
3440            !!!cp (985);
3441            $self->{ca}->{value} .= $data;
3442            $self->{ca}->{has_reference} = 1 if $has_ref;
3443            $self->{state} = $self->{prev_state};
3444            ## Reconsume.
3445            redo A;
3446        }        }
3447      } else {      } else {
3448        !!!cp (1026);        die "$0: $self->{state}: Unknown state";
       !!!parse-error (type => 'bare ero', line => $l, column => $c);  
       ## NOTE: "No characters are consumed" in the spec.  
       return {type => CHARACTER_TOKEN, data => '&'.$value,  
               line => $l, column => $c,  
              };  
3449      }      }
3450    } else {    } # A  
3451      !!!cp (1027);  
3452      ## no characters are consumed    die "$0: _get_next_token: unexpected case";
3453      !!!parse-error (type => 'bare ero', line => $l, column => $c);  } # _get_next_token
     return undef;  
   }  
 } # _tokenize_attempt_to_consume_an_entity  
3454    
3455  sub _initialize_tree_constructor ($) {  sub _initialize_tree_constructor ($) {
3456    my $self = shift;    my $self = shift;
# Line 3091  sub _initialize_tree_constructor ($) { Line 3459  sub _initialize_tree_constructor ($) {
3459    ## TODO: Turn mutation events off # MUST    ## TODO: Turn mutation events off # MUST
3460    ## TODO: Turn loose Document option (manakai extension) on    ## TODO: Turn loose Document option (manakai extension) on
3461    $self->{document}->manakai_is_html (1); # MUST    $self->{document}->manakai_is_html (1); # MUST
3462      $self->{document}->set_user_data (manakai_source_line => 1);
3463      $self->{document}->set_user_data (manakai_source_column => 1);
3464  } # _initialize_tree_constructor  } # _initialize_tree_constructor
3465    
3466  sub _terminate_tree_constructor ($) {  sub _terminate_tree_constructor ($) {
# Line 3110  sub _construct_tree ($) { Line 3480  sub _construct_tree ($) {
3480    ## When an interactive UA render the $self->{document} available    ## When an interactive UA render the $self->{document} available
3481    ## to the user, or when it begin accepting user input, are    ## to the user, or when it begin accepting user input, are
3482    ## not defined.    ## not defined.
   
   ## Append a character: collect it and all subsequent consecutive  
   ## characters and insert one Text node whose data is concatenation  
   ## of all those characters. # MUST  
3483        
3484    !!!next-token;    !!!next-token;
3485    
3486    undef $self->{form_element};    undef $self->{form_element};
3487    undef $self->{head_element};    undef $self->{head_element};
3488      undef $self->{head_element_inserted};
3489    $self->{open_elements} = [];    $self->{open_elements} = [];
3490    undef $self->{inner_html_node};    undef $self->{inner_html_node};
3491      undef $self->{ignore_newline};
3492    
3493    ## NOTE: The "initial" insertion mode.    ## NOTE: The "initial" insertion mode.
3494    $self->_tree_construction_initial; # MUST    $self->_tree_construction_initial; # MUST
# Line 3145  sub _tree_construction_initial ($) { Line 3513  sub _tree_construction_initial ($) {
3513        ## language.        ## language.
3514        my $doctype_name = $token->{name};        my $doctype_name = $token->{name};
3515        $doctype_name = '' unless defined $doctype_name;        $doctype_name = '' unless defined $doctype_name;
3516        $doctype_name =~ tr/a-z/A-Z/;        $doctype_name =~ tr/a-z/A-Z/; # ASCII case-insensitive
3517        if (not defined $token->{name} or # <!DOCTYPE>        if (not defined $token->{name} or # <!DOCTYPE>
3518            defined $token->{public_identifier} or            defined $token->{sysid}) {
           defined $token->{system_identifier}) {  
3519          !!!cp ('t1');          !!!cp ('t1');
3520          !!!parse-error (type => 'not HTML5', token => $token);          !!!parse-error (type => 'not HTML5', token => $token);
3521        } elsif ($doctype_name ne 'HTML') {        } elsif ($doctype_name ne 'HTML') {
3522          !!!cp ('t2');          !!!cp ('t2');
         ## ISSUE: ASCII case-insensitive? (in fact it does not matter)  
3523          !!!parse-error (type => 'not HTML5', token => $token);          !!!parse-error (type => 'not HTML5', token => $token);
3524          } elsif (defined $token->{pubid}) {
3525            if ($token->{pubid} eq 'XSLT-compat') {
3526              !!!cp ('t1.2');
3527              !!!parse-error (type => 'XSLT-compat', token => $token,
3528                              level => $self->{level}->{should});
3529            } else {
3530              !!!parse-error (type => 'not HTML5', token => $token);
3531            }
3532        } else {        } else {
3533          !!!cp ('t3');          !!!cp ('t3');
3534            #
3535        }        }
3536                
3537        my $doctype = $self->{document}->create_document_type_definition        my $doctype = $self->{document}->create_document_type_definition
3538          ($token->{name}); ## ISSUE: If name is missing (e.g. <!DOCTYPE>)?          ($token->{name}); ## ISSUE: If name is missing (e.g. <!DOCTYPE>)?
3539        ## NOTE: Default value for both |public_id| and |system_id| attributes        ## NOTE: Default value for both |public_id| and |system_id| attributes
3540        ## are empty strings, so that we don't set any value in missing cases.        ## are empty strings, so that we don't set any value in missing cases.
3541        $doctype->public_id ($token->{public_identifier})        $doctype->public_id ($token->{pubid}) if defined $token->{pubid};
3542            if defined $token->{public_identifier};        $doctype->system_id ($token->{sysid}) if defined $token->{sysid};
       $doctype->system_id ($token->{system_identifier})  
           if defined $token->{system_identifier};  
3543        ## NOTE: Other DocumentType attributes are null or empty lists.        ## NOTE: Other DocumentType attributes are null or empty lists.
3544        ## ISSUE: internalSubset = null??        ## ISSUE: internalSubset = null??
3545        $self->{document}->append_child ($doctype);        $self->{document}->append_child ($doctype);
# Line 3174  sub _tree_construction_initial ($) { Line 3547  sub _tree_construction_initial ($) {
3547        if ($token->{quirks} or $doctype_name ne 'HTML') {        if ($token->{quirks} or $doctype_name ne 'HTML') {
3548          !!!cp ('t4');          !!!cp ('t4');
3549          $self->{document}->manakai_compat_mode ('quirks');          $self->{document}->manakai_compat_mode ('quirks');
3550        } elsif (defined $token->{public_identifier}) {        } elsif (defined $token->{pubid}) {
3551          my $pubid = $token->{public_identifier};          my $pubid = $token->{pubid};
3552          $pubid =~ tr/a-z/A-z/;          $pubid =~ tr/a-z/A-z/;
3553          my $prefix = [          my $prefix = [
3554            "+//SILMARIL//DTD HTML PRO V0R11 19970101//",            "+//SILMARIL//DTD HTML PRO V0R11 19970101//",
# Line 3249  sub _tree_construction_initial ($) { Line 3622  sub _tree_construction_initial ($) {
3622            $self->{document}->manakai_compat_mode ('quirks');            $self->{document}->manakai_compat_mode ('quirks');
3623          } elsif ($pubid =~ m[^-//W3C//DTD HTML 4.01 FRAMESET//] or          } elsif ($pubid =~ m[^-//W3C//DTD HTML 4.01 FRAMESET//] or
3624                   $pubid =~ m[^-//W3C//DTD HTML 4.01 TRANSITIONAL//]) {                   $pubid =~ m[^-//W3C//DTD HTML 4.01 TRANSITIONAL//]) {
3625            if (defined $token->{system_identifier}) {            if (defined $token->{sysid}) {
3626              !!!cp ('t6');              !!!cp ('t6');
3627              $self->{document}->manakai_compat_mode ('quirks');              $self->{document}->manakai_compat_mode ('quirks');
3628            } else {            } else {
# Line 3266  sub _tree_construction_initial ($) { Line 3639  sub _tree_construction_initial ($) {
3639        } else {        } else {
3640          !!!cp ('t10');          !!!cp ('t10');
3641        }        }
3642        if (defined $token->{system_identifier}) {        if (defined $token->{sysid}) {
3643          my $sysid = $token->{system_identifier};          my $sysid = $token->{sysid};
3644          $sysid =~ tr/A-Z/a-z/;          $sysid =~ tr/A-Z/a-z/;
3645          if ($sysid eq "http://www.ibm.com/data/dtd/v11/ibmxhtml1-transitional.dtd") {          if ($sysid eq "http://www.ibm.com/data/dtd/v11/ibmxhtml1-transitional.dtd") {
3646            ## NOTE: Ensure that |PUBLIC "(limited quirks)" "(quirks)"| is            ## NOTE: Ensure that |PUBLIC "(limited quirks)" "(quirks)"| is
# Line 3297  sub _tree_construction_initial ($) { Line 3670  sub _tree_construction_initial ($) {
3670        !!!ack-later;        !!!ack-later;
3671        return;        return;
3672      } elsif ($token->{type} == CHARACTER_TOKEN) {      } elsif ($token->{type} == CHARACTER_TOKEN) {
3673        if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D        if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
3674          ## Ignore the token          ## Ignore the token
3675    
3676          unless (length $token->{data}) {          unless (length $token->{data}) {
# Line 3354  sub _tree_construction_root_element ($) Line 3727  sub _tree_construction_root_element ($)
3727          !!!next-token;          !!!next-token;
3728          redo B;          redo B;
3729        } elsif ($token->{type} == CHARACTER_TOKEN) {        } elsif ($token->{type} == CHARACTER_TOKEN) {
3730          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D          if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
3731            ## Ignore the token.            ## Ignore the token.
3732    
3733            unless (length $token->{data}) {            unless (length $token->{data}) {
# Line 3421  sub _tree_construction_root_element ($) Line 3794  sub _tree_construction_root_element ($)
3794      ## NOTE: Reprocess the token.      ## NOTE: Reprocess the token.
3795      !!!ack-later;      !!!ack-later;
3796      return; ## Go to the "before head" insertion mode.      return; ## Go to the "before head" insertion mode.
   
     ## ISSUE: There is an issue in the spec  
3797    } # B    } # B
3798    
3799    die "$0: _tree_construction_root_element: This should never be reached";    die "$0: _tree_construction_root_element: This should never be reached";
# Line 3458  sub _reset_insertion_mode ($) { Line 3829  sub _reset_insertion_mode ($) {
3829          ## SVG elements.  Currently the HTML syntax supports only MathML and          ## SVG elements.  Currently the HTML syntax supports only MathML and
3830          ## SVG elements as foreigners.          ## SVG elements as foreigners.
3831          $new_mode = IN_BODY_IM | IN_FOREIGN_CONTENT_IM;          $new_mode = IN_BODY_IM | IN_FOREIGN_CONTENT_IM;
3832        } elsif ($node->[1] & TABLE_CELL_EL) {        } elsif ($node->[1] == TABLE_CELL_EL) {
3833          if ($last) {          if ($last) {
3834            !!!cp ('t28.2');            !!!cp ('t28.2');
3835            #            #
# Line 3487  sub _reset_insertion_mode ($) { Line 3858  sub _reset_insertion_mode ($) {
3858        $self->{insertion_mode} = $new_mode and return if defined $new_mode;        $self->{insertion_mode} = $new_mode and return if defined $new_mode;
3859                
3860        ## Step 15        ## Step 15
3861        if ($node->[1] & HTML_EL) {        if ($node->[1] == HTML_EL) {
3862          unless (defined $self->{head_element}) {          unless (defined $self->{head_element}) {
3863            !!!cp ('t29');            !!!cp ('t29');
3864            $self->{insertion_mode} = BEFORE_HEAD_IM;            $self->{insertion_mode} = BEFORE_HEAD_IM;
# Line 3619  sub _tree_construction_main ($) { Line 3990  sub _tree_construction_main ($) {
3990    
3991      ## Step 1      ## Step 1
3992      my $start_tag_name = $token->{tag_name};      my $start_tag_name = $token->{tag_name};
3993      my $el;      !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
     !!!create-element ($el, $HTML_NS, $start_tag_name, $token->{attributes}, $token);  
3994    
3995      ## Step 2      ## Step 2
     $insert->($el);  
   
     ## Step 3  
3996      $self->{content_model} = $content_model_flag; # CDATA or RCDATA      $self->{content_model} = $content_model_flag; # CDATA or RCDATA
3997      delete $self->{escape}; # MUST      delete $self->{escape}; # MUST
3998    
3999      ## Step 4      ## Step 3, 4
4000      my $text = '';      $self->{insertion_mode} |= IN_CDATA_RCDATA_IM;
     !!!nack ('t40.1');  
     !!!next-token;  
     while ($token->{type} == CHARACTER_TOKEN) { # or until stop tokenizing  
       !!!cp ('t40');  
       $text .= $token->{data};  
       !!!next-token;  
     }  
   
     ## Step 5  
     if (length $text) {  
       !!!cp ('t41');  
       my $text = $self->{document}->create_text_node ($text);  
       $el->append_child ($text);  
     }  
   
     ## Step 6  
     $self->{content_model} = PCDATA_CONTENT_MODEL;  
4001    
4002      ## Step 7      !!!nack ('t40.1');
     if ($token->{type} == END_TAG_TOKEN and  
         $token->{tag_name} eq $start_tag_name) {  
       !!!cp ('t42');  
       ## Ignore the token  
     } else {  
       ## NOTE: An end-of-file token.  
       if ($content_model_flag == CDATA_CONTENT_MODEL) {  
         !!!cp ('t43');  
         !!!parse-error (type => 'in CDATA:#eof', token => $token);  
       } elsif ($content_model_flag == RCDATA_CONTENT_MODEL) {  
         !!!cp ('t44');  
         !!!parse-error (type => 'in RCDATA:#eof', token => $token);  
       } else {  
         die "$0: $content_model_flag in parse_rcdata";  
       }  
     }  
4003      !!!next-token;      !!!next-token;
4004    }; # $parse_rcdata    }; # $parse_rcdata
4005    
4006    my $script_start_tag = sub () {    my $script_start_tag = sub () {
4007        ## Step 1
4008      my $script_el;      my $script_el;
4009      !!!create-element ($script_el, $HTML_NS, 'script', $token->{attributes}, $token);      !!!create-element ($script_el, $HTML_NS, 'script', $token->{attributes}, $token);
4010    
4011        ## Step 2
4012      ## TODO: mark as "parser-inserted"      ## TODO: mark as "parser-inserted"
4013    
4014        ## Step 3
4015        ## TODO: Mark as "already executed", if ...
4016    
4017        ## Step 4
4018        $insert->($script_el);
4019    
4020        ## ISSUE: $script_el is not put into the stack
4021        push @{$self->{open_elements}}, [$script_el, $el_category->{script}];
4022    
4023        ## Step 5
4024      $self->{content_model} = CDATA_CONTENT_MODEL;      $self->{content_model} = CDATA_CONTENT_MODEL;
4025      delete $self->{escape}; # MUST      delete $self->{escape}; # MUST
       
     my $text = '';  
     !!!nack ('t45.1');  
     !!!next-token;  
     while ($token->{type} == CHARACTER_TOKEN) {  
       !!!cp ('t45');  
       $text .= $token->{data};  
       !!!next-token;  
     } # stop if non-character token or tokenizer stops tokenising  
     if (length $text) {  
       !!!cp ('t46');  
       $script_el->manakai_append_text ($text);  
     }  
                 
     $self->{content_model} = PCDATA_CONTENT_MODEL;  
4026    
4027      if ($token->{type} == END_TAG_TOKEN and      ## Step 6-7
4028          $token->{tag_name} eq 'script') {      $self->{insertion_mode} |= IN_CDATA_RCDATA_IM;
       !!!cp ('t47');  
       ## Ignore the token  
     } else {  
       !!!cp ('t48');  
       !!!parse-error (type => 'in CDATA:#eof', token => $token);  
       ## ISSUE: And ignore?  
       ## TODO: mark as "already executed"  
     }  
       
     if (defined $self->{inner_html_node}) {  
       !!!cp ('t49');  
       ## TODO: mark as "already executed"  
     } else {  
       !!!cp ('t50');  
       ## TODO: $old_insertion_point = current insertion point  
       ## TODO: insertion point = just before the next input character  
4029    
4030        $insert->($script_el);      !!!nack ('t40.2');
         
       ## TODO: insertion point = $old_insertion_point (might be "undefined")  
         
       ## TODO: if there is a script that will execute as soon as the parser resume, then...  
     }  
       
4031      !!!next-token;      !!!next-token;
4032    }; # $script_start_tag    }; # $script_start_tag
4033    
4034    ## NOTE: $open_tables->[-1]->[0] is the "current table" element node.    ## NOTE: $open_tables->[-1]->[0] is the "current table" element node.
4035    ## NOTE: $open_tables->[-1]->[1] is the "tainted" flag.    ## NOTE: $open_tables->[-1]->[1] is the "tainted" flag.
4036      ## NOTE: $open_tables->[-1]->[2] is set false when non-Text node inserted.
4037    my $open_tables = [[$self->{open_elements}->[0]->[0]]];    my $open_tables = [[$self->{open_elements}->[0]->[0]]];
4038    
4039    my $formatting_end_tag = sub {    my $formatting_end_tag = sub {
# Line 3807  sub _tree_construction_main ($) { Line 4118  sub _tree_construction_main ($) {
4118            !!!cp ('t59');            !!!cp ('t59');
4119            $furthest_block = $node;            $furthest_block = $node;
4120            $furthest_block_i_in_open = $_;            $furthest_block_i_in_open = $_;
4121              ## NOTE: The topmost (eldest) node.
4122          } elsif ($node->[0] eq $formatting_element->[0]) {          } elsif ($node->[0] eq $formatting_element->[0]) {
4123            !!!cp ('t60');            !!!cp ('t60');
4124            last OE;            last OE;
# Line 3893  sub _tree_construction_main ($) { Line 4205  sub _tree_construction_main ($) {
4205          my $foster_parent_element;          my $foster_parent_element;
4206          my $next_sibling;          my $next_sibling;
4207          OE: for (reverse 0..$#{$self->{open_elements}}) {          OE: for (reverse 0..$#{$self->{open_elements}}) {
4208            if ($self->{open_elements}->[$_]->[1] & TABLE_EL) {            if ($self->{open_elements}->[$_]->[1] == TABLE_EL) {
4209                               my $parent = $self->{open_elements}->[$_]->[0]->parent_node;                               my $parent = $self->{open_elements}->[$_]->[0]->parent_node;
4210                               if (defined $parent and $parent->node_type == 1) {                               if (defined $parent and $parent->node_type == 1) {
4211                                 !!!cp ('t65.1');                                 !!!cp ('t65.1');
# Line 3953  sub _tree_construction_main ($) { Line 4265  sub _tree_construction_main ($) {
4265            $i = $_;            $i = $_;
4266          }          }
4267        } # OE        } # OE
4268        splice @{$self->{open_elements}}, $i + 1, 1, $clone;        splice @{$self->{open_elements}}, $i + 1, 0, $clone;
4269                
4270        ## Step 14        ## Step 14
4271        redo FET;        redo FET;
# Line 3971  sub _tree_construction_main ($) { Line 4283  sub _tree_construction_main ($) {
4283        my $foster_parent_element;        my $foster_parent_element;
4284        my $next_sibling;        my $next_sibling;
4285        OE: for (reverse 0..$#{$self->{open_elements}}) {        OE: for (reverse 0..$#{$self->{open_elements}}) {
4286          if ($self->{open_elements}->[$_]->[1] & TABLE_EL) {          if ($self->{open_elements}->[$_]->[1] == TABLE_EL) {
4287                               my $parent = $self->{open_elements}->[$_]->[0]->parent_node;                               my $parent = $self->{open_elements}->[$_]->[0]->parent_node;
4288                               if (defined $parent and $parent->node_type == 1) {                               if (defined $parent and $parent->node_type == 1) {
4289                                 !!!cp ('t70');                                 !!!cp ('t70');
# Line 3996  sub _tree_construction_main ($) { Line 4308  sub _tree_construction_main ($) {
4308      }      }
4309    }; # $insert_to_foster    }; # $insert_to_foster
4310    
4311      ## NOTE: Insert a character (MUST): When a character is inserted, if
4312      ## the last node that was inserted by the parser is a Text node and
4313      ## the character has to be inserted after that node, then the
4314      ## character is appended to the Text node.  However, if any other
4315      ## node is inserted by the parser, then a new Text node is created
4316      ## and the character is appended as that Text node.  If I'm not
4317      ## wrong, for a parser with scripting disabled, there are only two
4318      ## cases where this occurs.  One is the case where an element node
4319      ## is inserted to the |head| element.  This is covered by using the
4320      ## |$self->{head_element_inserted}| flag.  Another is the case where
4321      ## an element or comment is inserted into the |table| subtree while
4322      ## foster parenting happens.  This is covered by using the [2] flag
4323      ## of the |$open_tables| structure.  All other cases are handled
4324      ## simply by calling |manakai_append_text| method.
4325    
4326      ## TODO: |<body><script>document.write("a<br>");
4327      ## document.body.removeChild (document.body.lastChild);
4328      ## document.write ("b")</script>|
4329    
4330    B: while (1) {    B: while (1) {
4331      if ($token->{type} == DOCTYPE_TOKEN) {      if ($token->{type} == DOCTYPE_TOKEN) {
4332        !!!cp ('t73');        !!!cp ('t73');
# Line 4043  sub _tree_construction_main ($) { Line 4374  sub _tree_construction_main ($) {
4374        } else {        } else {
4375          !!!cp ('t87');          !!!cp ('t87');
4376          $self->{open_elements}->[-1]->[0]->append_child ($comment);          $self->{open_elements}->[-1]->[0]->append_child ($comment);
4377            $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
4378        }        }
4379        !!!next-token;        !!!next-token;
4380        next B;        next B;
4381        } elsif ($self->{insertion_mode} & IN_CDATA_RCDATA_IM) {
4382          if ($token->{type} == CHARACTER_TOKEN) {
4383            $token->{data} =~ s/^\x0A// if $self->{ignore_newline};
4384            delete $self->{ignore_newline};
4385    
4386            if (length $token->{data}) {
4387              !!!cp ('t43');
4388              $self->{open_elements}->[-1]->[0]->manakai_append_text
4389                  ($token->{data});
4390            } else {
4391              !!!cp ('t43.1');
4392            }
4393            !!!next-token;
4394            next B;
4395          } elsif ($token->{type} == END_TAG_TOKEN) {
4396            delete $self->{ignore_newline};
4397    
4398            if ($token->{tag_name} eq 'script') {
4399              !!!cp ('t50');
4400              
4401              ## Para 1-2
4402              my $script = pop @{$self->{open_elements}};
4403              
4404              ## Para 3
4405              $self->{insertion_mode} &= ~ IN_CDATA_RCDATA_IM;
4406    
4407              ## Para 4
4408              ## TODO: $old_insertion_point = $current_insertion_point;
4409              ## TODO: $current_insertion_point = just before $self->{nc};
4410    
4411              ## Para 5
4412              ## TODO: Run the $script->[0].
4413    
4414              ## Para 6
4415              ## TODO: $current_insertion_point = $old_insertion_point;
4416    
4417              ## Para 7
4418              ## TODO: if ($pending_external_script) {
4419                ## TODO: ...
4420              ## TODO: }
4421    
4422              !!!next-token;
4423              next B;
4424            } else {
4425              !!!cp ('t42');
4426    
4427              pop @{$self->{open_elements}};
4428    
4429              $self->{insertion_mode} &= ~ IN_CDATA_RCDATA_IM;
4430              !!!next-token;
4431              next B;
4432            }
4433          } elsif ($token->{type} == END_OF_FILE_TOKEN) {
4434            delete $self->{ignore_newline};
4435    
4436            !!!cp ('t44');
4437            !!!parse-error (type => 'not closed',
4438                            text => $self->{open_elements}->[-1]->[0]
4439                                ->manakai_local_name,
4440                            token => $token);
4441    
4442            #if ($self->{open_elements}->[-1]->[1] == SCRIPT_EL) {
4443            #  ## TODO: Mark as "already executed"
4444            #}
4445    
4446            pop @{$self->{open_elements}};
4447    
4448            $self->{insertion_mode} &= ~ IN_CDATA_RCDATA_IM;
4449            ## Reprocess.
4450            next B;
4451          } else {
4452            die "$0: $token->{type}: In CDATA/RCDATA: Unknown token type";        
4453          }
4454      } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM) {      } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM) {
4455        if ($token->{type} == CHARACTER_TOKEN) {        if ($token->{type} == CHARACTER_TOKEN) {
4456          !!!cp ('t87.1');          !!!cp ('t87.1');
# Line 4057  sub _tree_construction_main ($) { Line 4462  sub _tree_construction_main ($) {
4462               $self->{open_elements}->[-1]->[1] & FOREIGN_FLOW_CONTENT_EL) or               $self->{open_elements}->[-1]->[1] & FOREIGN_FLOW_CONTENT_EL) or
4463              not ($self->{open_elements}->[-1]->[1] & FOREIGN_EL) or              not ($self->{open_elements}->[-1]->[1] & FOREIGN_EL) or
4464              ($token->{tag_name} eq 'svg' and              ($token->{tag_name} eq 'svg' and
4465               $self->{open_elements}->[-1]->[1] & MML_AXML_EL)) {               $self->{open_elements}->[-1]->[1] == MML_AXML_EL)) {
4466            ## NOTE: "using the rules for secondary insertion mode"then"continue"            ## NOTE: "using the rules for secondary insertion mode"then"continue"
4467            !!!cp ('t87.2');            !!!cp ('t87.2');
4468            #            #
# Line 4158  sub _tree_construction_main ($) { Line 4563  sub _tree_construction_main ($) {
4563          pop @{$self->{open_elements}}          pop @{$self->{open_elements}}
4564              while $self->{open_elements}->[-1]->[1] & FOREIGN_EL;              while $self->{open_elements}->[-1]->[1] & FOREIGN_EL;
4565    
4566            ## NOTE: |<span><svg>| ... two parse errors, |<svg>| ... a parse error.
4567    
4568          $self->{insertion_mode} &= ~ IN_FOREIGN_CONTENT_IM;          $self->{insertion_mode} &= ~ IN_FOREIGN_CONTENT_IM;
4569          ## Reprocess.          ## Reprocess.
4570          next B;          next B;
# Line 4168  sub _tree_construction_main ($) { Line 4575  sub _tree_construction_main ($) {
4575    
4576      if ($self->{insertion_mode} & HEAD_IMS) {      if ($self->{insertion_mode} & HEAD_IMS) {
4577        if ($token->{type} == CHARACTER_TOKEN) {        if ($token->{type} == CHARACTER_TOKEN) {
4578          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {          if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
4579            unless ($self->{insertion_mode} == BEFORE_HEAD_IM) {            unless ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4580              !!!cp ('t88.2');              if ($self->{head_element_inserted}) {
4581              $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);                !!!cp ('t88.3');
4582                  $self->{open_elements}->[-1]->[0]->append_child
4583                    ($self->{document}->create_text_node ($1));
4584                  delete $self->{head_element_inserted};
4585                  ## NOTE: |</head> <link> |
4586                  #
4587                } else {
4588                  !!!cp ('t88.2');
4589                  $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
4590                  ## NOTE: |</head> &#x20;|
4591                  #
4592                }
4593            } else {            } else {
4594              !!!cp ('t88.1');              !!!cp ('t88.1');
4595              ## Ignore the token.              ## Ignore the token.
4596              !!!next-token;              #
             next B;  
4597            }            }
4598            unless (length $token->{data}) {            unless (length $token->{data}) {
4599              !!!cp ('t88');              !!!cp ('t88');
4600              !!!next-token;              !!!next-token;
4601              next B;              next B;
4602            }            }
4603    ## TODO: set $token->{column} appropriately
4604          }          }
4605    
4606          if ($self->{insertion_mode} == BEFORE_HEAD_IM) {          if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
# Line 4267  sub _tree_construction_main ($) { Line 4685  sub _tree_construction_main ($) {
4685            !!!cp ('t97');            !!!cp ('t97');
4686          }          }
4687    
4688              if ($token->{tag_name} eq 'base') {          if ($token->{tag_name} eq 'base') {
4689                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {            if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4690                  !!!cp ('t98');              !!!cp ('t98');
4691                  ## As if </noscript>              ## As if </noscript>
4692                  pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
4693                  !!!parse-error (type => 'in noscript', text => 'base',              !!!parse-error (type => 'in noscript', text => 'base',
4694                                  token => $token);                              token => $token);
4695                            
4696                  $self->{insertion_mode} = IN_HEAD_IM;              $self->{insertion_mode} = IN_HEAD_IM;
4697                  ## Reprocess in the "in head" insertion mode...              ## Reprocess in the "in head" insertion mode...
4698                } else {            } else {
4699                  !!!cp ('t99');              !!!cp ('t99');
4700                }            }
4701    
4702                ## NOTE: There is a "as if in head" code clone.            ## NOTE: There is a "as if in head" code clone.
4703                if ($self->{insertion_mode} == AFTER_HEAD_IM) {            if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4704                  !!!cp ('t100');              !!!cp ('t100');
4705                  !!!parse-error (type => 'after head',              !!!parse-error (type => 'after head',
4706                                  text => $token->{tag_name}, token => $token);                              text => $token->{tag_name}, token => $token);
4707                  push @{$self->{open_elements}},              push @{$self->{open_elements}},
4708                      [$self->{head_element}, $el_category->{head}];                  [$self->{head_element}, $el_category->{head}];
4709                } else {              $self->{head_element_inserted} = 1;
4710                  !!!cp ('t101');            } else {
4711                }              !!!cp ('t101');
4712                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);            }
4713                pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.            !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4714                pop @{$self->{open_elements}} # <head>            pop @{$self->{open_elements}};
4715                    if $self->{insertion_mode} == AFTER_HEAD_IM;            pop @{$self->{open_elements}} # <head>
4716                !!!nack ('t101.1');                if $self->{insertion_mode} == AFTER_HEAD_IM;
4717                !!!next-token;            !!!nack ('t101.1');
4718                next B;            !!!next-token;
4719              } elsif ($token->{tag_name} eq 'link') {            next B;
4720                ## NOTE: There is a "as if in head" code clone.          } elsif ($token->{tag_name} eq 'link') {
4721                if ($self->{insertion_mode} == AFTER_HEAD_IM) {            ## NOTE: There is a "as if in head" code clone.
4722                  !!!cp ('t102');            if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4723                  !!!parse-error (type => 'after head',              !!!cp ('t102');
4724                                  text => $token->{tag_name}, token => $token);              !!!parse-error (type => 'after head',
4725                  push @{$self->{open_elements}},                              text => $token->{tag_name}, token => $token);
4726                      [$self->{head_element}, $el_category->{head}];              push @{$self->{open_elements}},
4727                } else {                  [$self->{head_element}, $el_category->{head}];
4728                  !!!cp ('t103');              $self->{head_element_inserted} = 1;
4729                }            } else {
4730                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);              !!!cp ('t103');
4731                pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.            }
4732                pop @{$self->{open_elements}} # <head>            !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4733                    if $self->{insertion_mode} == AFTER_HEAD_IM;            pop @{$self->{open_elements}};
4734                !!!ack ('t103.1');            pop @{$self->{open_elements}} # <head>
4735                !!!next-token;                if $self->{insertion_mode} == AFTER_HEAD_IM;
4736                next B;            !!!ack ('t103.1');
4737              } elsif ($token->{tag_name} eq 'meta') {            !!!next-token;
4738                ## NOTE: There is a "as if in head" code clone.            next B;
4739                if ($self->{insertion_mode} == AFTER_HEAD_IM) {          } elsif ($token->{tag_name} eq 'command' or
4740                  !!!cp ('t104');                   $token->{tag_name} eq 'eventsource') {
4741                  !!!parse-error (type => 'after head',            if ($self->{insertion_mode} == IN_HEAD_IM) {
4742                                  text => $token->{tag_name}, token => $token);              ## NOTE: If the insertion mode at the time of the emission
4743                  push @{$self->{open_elements}},              ## of the token was "before head", $self->{insertion_mode}
4744                      [$self->{head_element}, $el_category->{head}];              ## is already changed to |IN_HEAD_IM|.
4745                } else {  
4746                  !!!cp ('t105');              ## NOTE: There is a "as if in head" code clone.
4747                }              !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4748                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);              pop @{$self->{open_elements}};
4749                my $meta_el = pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.              pop @{$self->{open_elements}} # <head>
4750                    if $self->{insertion_mode} == AFTER_HEAD_IM;
4751                !!!ack ('t103.2');
4752                !!!next-token;
4753                next B;
4754              } else {
4755                ## NOTE: "in head noscript" or "after head" insertion mode
4756                ## - in these cases, these tags are treated as same as
4757                ## normal in-body tags.
4758                !!!cp ('t103.3');
4759                #
4760              }
4761            } elsif ($token->{tag_name} eq 'meta') {
4762              ## NOTE: There is a "as if in head" code clone.
4763              if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4764                !!!cp ('t104');
4765                !!!parse-error (type => 'after head',
4766                                text => $token->{tag_name}, token => $token);
4767                push @{$self->{open_elements}},
4768                    [$self->{head_element}, $el_category->{head}];
4769                $self->{head_element_inserted} = 1;
4770              } else {
4771                !!!cp ('t105');
4772              }
4773              !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4774              my $meta_el = pop @{$self->{open_elements}};
4775    
4776                unless ($self->{confident}) {                unless ($self->{confident}) {
4777                  if ($token->{attributes}->{charset}) {                  if ($token->{attributes}->{charset}) {
# Line 4346  sub _tree_construction_main ($) { Line 4789  sub _tree_construction_main ($) {
4789                  } elsif ($token->{attributes}->{content}) {                  } elsif ($token->{attributes}->{content}) {
4790                    if ($token->{attributes}->{content}->{value}                    if ($token->{attributes}->{content}->{value}
4791                        =~ /[Cc][Hh][Aa][Rr][Ss][Ee][Tt]                        =~ /[Cc][Hh][Aa][Rr][Ss][Ee][Tt]
4792                            [\x09-\x0D\x20]*=                            [\x09\x0A\x0C\x0D\x20]*=
4793                            [\x09-\x0D\x20]*(?>"([^"]*)"|'([^']*)'|                            [\x09\x0A\x0C\x0D\x20]*(?>"([^"]*)"|'([^']*)'|
4794                            ([^"'\x09-\x0D\x20][^\x09-\x0D\x20\x3B]*))/x) {                            ([^"'\x09\x0A\x0C\x0D\x20]
4795                               [^\x09\x0A\x0C\x0D\x20\x3B]*))/x) {
4796                      !!!cp ('t107');                      !!!cp ('t107');
4797                      ## NOTE: Whether the encoding is supported or not is handled                      ## NOTE: Whether the encoding is supported or not is handled
4798                      ## in the {change_encoding} callback.                      ## in the {change_encoding} callback.
# Line 4385  sub _tree_construction_main ($) { Line 4829  sub _tree_construction_main ($) {
4829                !!!ack ('t110.1');                !!!ack ('t110.1');
4830                !!!next-token;                !!!next-token;
4831                next B;                next B;
4832              } elsif ($token->{tag_name} eq 'title') {          } elsif ($token->{tag_name} eq 'title') {
4833                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {            if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4834                  !!!cp ('t111');              !!!cp ('t111');
4835                  ## As if </noscript>              ## As if </noscript>
4836                  pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
4837                  !!!parse-error (type => 'in noscript', text => 'title',              !!!parse-error (type => 'in noscript', text => 'title',
4838                                  token => $token);                              token => $token);
4839                            
4840                  $self->{insertion_mode} = IN_HEAD_IM;              $self->{insertion_mode} = IN_HEAD_IM;
4841                  ## Reprocess in the "in head" insertion mode...              ## Reprocess in the "in head" insertion mode...
4842                } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {            } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4843                  !!!cp ('t112');              !!!cp ('t112');
4844                  !!!parse-error (type => 'after head',              !!!parse-error (type => 'after head',
4845                                  text => $token->{tag_name}, token => $token);                              text => $token->{tag_name}, token => $token);
4846                  push @{$self->{open_elements}},              push @{$self->{open_elements}},
4847                      [$self->{head_element}, $el_category->{head}];                  [$self->{head_element}, $el_category->{head}];
4848                } else {              $self->{head_element_inserted} = 1;
4849                  !!!cp ('t113');            } else {
4850                }              !!!cp ('t113');
4851              }
4852    
4853                ## NOTE: There is a "as if in head" code clone.            ## NOTE: There is a "as if in head" code clone.
4854                my $parent = defined $self->{head_element} ? $self->{head_element}            $parse_rcdata->(RCDATA_CONTENT_MODEL);
4855                    : $self->{open_elements}->[-1]->[0];            ## ISSUE: A spec bug [Bug 6038]
4856                $parse_rcdata->(RCDATA_CONTENT_MODEL);            splice @{$self->{open_elements}}, -2, 1, () # <head>
4857                pop @{$self->{open_elements}} # <head>                if ($self->{insertion_mode} & AFTER_HEAD_IM) == AFTER_HEAD_IM;
4858                    if $self->{insertion_mode} == AFTER_HEAD_IM;            next B;
4859                next B;          } elsif ($token->{tag_name} eq 'style' or
4860              } elsif ($token->{tag_name} eq 'style' or                   $token->{tag_name} eq 'noframes') {
4861                       $token->{tag_name} eq 'noframes') {            ## NOTE: Or (scripting is enabled and tag_name eq 'noscript' and
4862                ## NOTE: Or (scripting is enabled and tag_name eq 'noscript' and            ## insertion mode IN_HEAD_IM)
4863                ## insertion mode IN_HEAD_IM)            ## NOTE: There is a "as if in head" code clone.
4864                ## NOTE: There is a "as if in head" code clone.            if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4865                if ($self->{insertion_mode} == AFTER_HEAD_IM) {              !!!cp ('t114');
4866                  !!!cp ('t114');              !!!parse-error (type => 'after head',
4867                  !!!parse-error (type => 'after head',                              text => $token->{tag_name}, token => $token);
4868                                  text => $token->{tag_name}, token => $token);              push @{$self->{open_elements}},
4869                  push @{$self->{open_elements}},                  [$self->{head_element}, $el_category->{head}];
4870                      [$self->{head_element}, $el_category->{head}];              $self->{head_element_inserted} = 1;
4871                } else {            } else {
4872                  !!!cp ('t115');              !!!cp ('t115');
4873                }            }
4874                $parse_rcdata->(CDATA_CONTENT_MODEL);            $parse_rcdata->(CDATA_CONTENT_MODEL);
4875                pop @{$self->{open_elements}} # <head>            ## ISSUE: A spec bug [Bug 6038]
4876                    if $self->{insertion_mode} == AFTER_HEAD_IM;            splice @{$self->{open_elements}}, -2, 1, () # <head>
4877                next B;                if ($self->{insertion_mode} & AFTER_HEAD_IM) == AFTER_HEAD_IM;
4878              } elsif ($token->{tag_name} eq 'noscript') {            next B;
4879            } elsif ($token->{tag_name} eq 'noscript') {
4880                if ($self->{insertion_mode} == IN_HEAD_IM) {                if ($self->{insertion_mode} == IN_HEAD_IM) {
4881                  !!!cp ('t116');                  !!!cp ('t116');
4882                  ## NOTE: and scripting is disalbed                  ## NOTE: and scripting is disalbed
# Line 4451  sub _tree_construction_main ($) { Line 4897  sub _tree_construction_main ($) {
4897                  !!!cp ('t118');                  !!!cp ('t118');
4898                  #                  #
4899                }                }
4900              } elsif ($token->{tag_name} eq 'script') {          } elsif ($token->{tag_name} eq 'script') {
4901                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {            if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4902                  !!!cp ('t119');              !!!cp ('t119');
4903                  ## As if </noscript>              ## As if </noscript>
4904                  pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
4905                  !!!parse-error (type => 'in noscript', text => 'script',              !!!parse-error (type => 'in noscript', text => 'script',
4906                                  token => $token);                              token => $token);
4907                            
4908                  $self->{insertion_mode} = IN_HEAD_IM;              $self->{insertion_mode} = IN_HEAD_IM;
4909                  ## Reprocess in the "in head" insertion mode...              ## Reprocess in the "in head" insertion mode...
4910                } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {            } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4911                  !!!cp ('t120');              !!!cp ('t120');
4912                  !!!parse-error (type => 'after head',              !!!parse-error (type => 'after head',
4913                                  text => $token->{tag_name}, token => $token);                              text => $token->{tag_name}, token => $token);
4914                  push @{$self->{open_elements}},              push @{$self->{open_elements}},
4915                      [$self->{head_element}, $el_category->{head}];                  [$self->{head_element}, $el_category->{head}];
4916                } else {              $self->{head_element_inserted} = 1;
4917                  !!!cp ('t121');            } else {
4918                }              !!!cp ('t121');
4919              }
4920    
4921                ## NOTE: There is a "as if in head" code clone.            ## NOTE: There is a "as if in head" code clone.
4922                $script_start_tag->();            $script_start_tag->();
4923                pop @{$self->{open_elements}} # <head>            ## ISSUE: A spec bug  [Bug 6038]
4924                    if $self->{insertion_mode} == AFTER_HEAD_IM;            splice @{$self->{open_elements}}, -2, 1 # <head>
4925                next B;                if ($self->{insertion_mode} & AFTER_HEAD_IM) == AFTER_HEAD_IM;
4926              } elsif ($token->{tag_name} eq 'body' or            next B;
4927                       $token->{tag_name} eq 'frameset') {          } elsif ($token->{tag_name} eq 'body' or
4928                     $token->{tag_name} eq 'frameset') {
4929                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4930                  !!!cp ('t122');                  !!!cp ('t122');
4931                  ## As if </noscript>                  ## As if </noscript>
# Line 4612  sub _tree_construction_main ($) { Line 5060  sub _tree_construction_main ($) {
5060              } elsif ({              } elsif ({
5061                        body => 1, html => 1,                        body => 1, html => 1,
5062                       }->{$token->{tag_name}}) {                       }->{$token->{tag_name}}) {
5063                if ($self->{insertion_mode} == BEFORE_HEAD_IM or                ## TODO: This branch is entirely redundant.
5064                  if ($self->{insertion_mode} == BEFORE_HEAD_IM or
5065                    $self->{insertion_mode} == IN_HEAD_IM or                    $self->{insertion_mode} == IN_HEAD_IM or
5066                    $self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {                    $self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
5067                  !!!cp ('t140');                  !!!cp ('t140');
# Line 4784  sub _tree_construction_main ($) { Line 5233  sub _tree_construction_main ($) {
5233        } else {        } else {
5234          die "$0: $token->{type}: Unknown token type";          die "$0: $token->{type}: Unknown token type";
5235        }        }
   
           ## ISSUE: An issue in the spec.  
5236      } elsif ($self->{insertion_mode} & BODY_IMS) {      } elsif ($self->{insertion_mode} & BODY_IMS) {
5237            if ($token->{type} == CHARACTER_TOKEN) {            if ($token->{type} == CHARACTER_TOKEN) {
5238              !!!cp ('t150');              !!!cp ('t150');
# Line 4805  sub _tree_construction_main ($) { Line 5252  sub _tree_construction_main ($) {
5252                  ## have an element in table scope                  ## have an element in table scope
5253                  for (reverse 0..$#{$self->{open_elements}}) {                  for (reverse 0..$#{$self->{open_elements}}) {
5254                    my $node = $self->{open_elements}->[$_];                    my $node = $self->{open_elements}->[$_];
5255                    if ($node->[1] & TABLE_CELL_EL) {                    if ($node->[1] == TABLE_CELL_EL) {
5256                      !!!cp ('t151');                      !!!cp ('t151');
5257    
5258                      ## Close the cell                      ## Close the cell
# Line 4839  sub _tree_construction_main ($) { Line 5286  sub _tree_construction_main ($) {
5286                  INSCOPE: {                  INSCOPE: {
5287                    for (reverse 0..$#{$self->{open_elements}}) {                    for (reverse 0..$#{$self->{open_elements}}) {
5288                      my $node = $self->{open_elements}->[$_];                      my $node = $self->{open_elements}->[$_];
5289                      if ($node->[1] & CAPTION_EL) {                      if ($node->[1] == CAPTION_EL) {
5290                        !!!cp ('t155');                        !!!cp ('t155');
5291                        $i = $_;                        $i = $_;
5292                        last INSCOPE;                        last INSCOPE;
# Line 4865  sub _tree_construction_main ($) { Line 5312  sub _tree_construction_main ($) {
5312                    pop @{$self->{open_elements}};                    pop @{$self->{open_elements}};
5313                  }                  }
5314    
5315                  unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {                  unless ($self->{open_elements}->[-1]->[1] == CAPTION_EL) {
5316                    !!!cp ('t159');                    !!!cp ('t159');
5317                    !!!parse-error (type => 'not closed',                    !!!parse-error (type => 'not closed',
5318                                    text => $self->{open_elements}->[-1]->[0]                                    text => $self->{open_elements}->[-1]->[0]
# Line 4962  sub _tree_construction_main ($) { Line 5409  sub _tree_construction_main ($) {
5409                  INSCOPE: {                  INSCOPE: {
5410                    for (reverse 0..$#{$self->{open_elements}}) {                    for (reverse 0..$#{$self->{open_elements}}) {
5411                      my $node = $self->{open_elements}->[$_];                      my $node = $self->{open_elements}->[$_];
5412                      if ($node->[1] & CAPTION_EL) {                      if ($node->[1] == CAPTION_EL) {
5413                        !!!cp ('t171');                        !!!cp ('t171');
5414                        $i = $_;                        $i = $_;
5415                        last INSCOPE;                        last INSCOPE;
# Line 4987  sub _tree_construction_main ($) { Line 5434  sub _tree_construction_main ($) {
5434                    pop @{$self->{open_elements}};                    pop @{$self->{open_elements}};
5435                  }                  }
5436                                    
5437                  unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {                  unless ($self->{open_elements}->[-1]->[1] == CAPTION_EL) {
5438                    !!!cp ('t175');                    !!!cp ('t175');
5439                    !!!parse-error (type => 'not closed',                    !!!parse-error (type => 'not closed',
5440                                    text => $self->{open_elements}->[-1]->[0]                                    text => $self->{open_elements}->[-1]->[0]
# Line 5037  sub _tree_construction_main ($) { Line 5484  sub _tree_construction_main ($) {
5484                                line => $token->{line},                                line => $token->{line},
5485                                column => $token->{column}};                                column => $token->{column}};
5486                      next B;                      next B;
5487                    } elsif ($node->[1] & TABLE_CELL_EL) {                    } elsif ($node->[1] == TABLE_CELL_EL) {
5488                      !!!cp ('t180');                      !!!cp ('t180');
5489                      $tn = $node->[0]->manakai_local_name;                      $tn = $node->[0]->manakai_local_name;
5490                      ## NOTE: There is exactly one |td| or |th| element                      ## NOTE: There is exactly one |td| or |th| element
# Line 5066  sub _tree_construction_main ($) { Line 5513  sub _tree_construction_main ($) {
5513                my $i;                my $i;
5514                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5515                  my $node = $self->{open_elements}->[$_];                  my $node = $self->{open_elements}->[$_];
5516                  if ($node->[1] & CAPTION_EL) {                  if ($node->[1] == CAPTION_EL) {
5517                    !!!cp ('t184');                    !!!cp ('t184');
5518                    $i = $_;                    $i = $_;
5519                    last INSCOPE;                    last INSCOPE;
# Line 5090  sub _tree_construction_main ($) { Line 5537  sub _tree_construction_main ($) {
5537                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
5538                }                }
5539    
5540                unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {                unless ($self->{open_elements}->[-1]->[1] == CAPTION_EL) {
5541                  !!!cp ('t188');                  !!!cp ('t188');
5542                  !!!parse-error (type => 'not closed',                  !!!parse-error (type => 'not closed',
5543                                  text => $self->{open_elements}->[-1]->[0]                                  text => $self->{open_elements}->[-1]->[0]
# Line 5157  sub _tree_construction_main ($) { Line 5604  sub _tree_construction_main ($) {
5604      } elsif ($self->{insertion_mode} & TABLE_IMS) {      } elsif ($self->{insertion_mode} & TABLE_IMS) {
5605        if ($token->{type} == CHARACTER_TOKEN) {        if ($token->{type} == CHARACTER_TOKEN) {
5606          if (not $open_tables->[-1]->[1] and # tainted          if (not $open_tables->[-1]->[1] and # tainted
5607              $token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {              $token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
5608            $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);            $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
5609                                
5610            unless (length $token->{data}) {            unless (length $token->{data}) {
# Line 5171  sub _tree_construction_main ($) { Line 5618  sub _tree_construction_main ($) {
5618    
5619          !!!parse-error (type => 'in table:#text', token => $token);          !!!parse-error (type => 'in table:#text', token => $token);
5620    
5621              ## As if in body, but insert into foster parent element          ## NOTE: As if in body, but insert into the foster parent element.
5622              ## ISSUE: Spec says that "whenever a node would be inserted          $reconstruct_active_formatting_elements->($insert_to_foster);
             ## into the current node" while characters might not be  
             ## result in a new Text node.  
             $reconstruct_active_formatting_elements->($insert_to_foster);  
5623                            
5624              if ($self->{open_elements}->[-1]->[1] & TABLE_ROWS_EL) {          if ($self->{open_elements}->[-1]->[1] & TABLE_ROWS_EL) {
5625                # MUST            # MUST
5626                my $foster_parent_element;            my $foster_parent_element;
5627                my $next_sibling;            my $next_sibling;
5628                my $prev_sibling;            my $prev_sibling;
5629                OE: for (reverse 0..$#{$self->{open_elements}}) {            OE: for (reverse 0..$#{$self->{open_elements}}) {
5630                  if ($self->{open_elements}->[$_]->[1] & TABLE_EL) {              if ($self->{open_elements}->[$_]->[1] == TABLE_EL) {
5631                    my $parent = $self->{open_elements}->[$_]->[0]->parent_node;                my $parent = $self->{open_elements}->[$_]->[0]->parent_node;
5632                    if (defined $parent and $parent->node_type == 1) {                if (defined $parent and $parent->node_type == 1) {
5633                      !!!cp ('t196');                  $foster_parent_element = $parent;
5634                      $foster_parent_element = $parent;                  !!!cp ('t196');
5635                      $next_sibling = $self->{open_elements}->[$_]->[0];                  $next_sibling = $self->{open_elements}->[$_]->[0];
5636                      $prev_sibling = $next_sibling->previous_sibling;                  $prev_sibling = $next_sibling->previous_sibling;
5637                    } else {                  #
                     !!!cp ('t197');  
                     $foster_parent_element = $self->{open_elements}->[$_ - 1]->[0];  
                     $prev_sibling = $foster_parent_element->last_child;  
                   }  
                   last OE;  
                 }  
               } # OE  
               $foster_parent_element = $self->{open_elements}->[0]->[0] and  
               $prev_sibling = $foster_parent_element->last_child  
                 unless defined $foster_parent_element;  
               if (defined $prev_sibling and  
                   $prev_sibling->node_type == 3) {  
                 !!!cp ('t198');  
                 $prev_sibling->manakai_append_text ($token->{data});  
5638                } else {                } else {
5639                  !!!cp ('t199');                  !!!cp ('t197');
5640                  $foster_parent_element->insert_before                  $foster_parent_element = $self->{open_elements}->[$_ - 1]->[0];
5641                    ($self->{document}->create_text_node ($token->{data}),                  $prev_sibling = $foster_parent_element->last_child;
5642                     $next_sibling);                  #
5643                }                }
5644                  last OE;
5645                }
5646              } # OE
5647              $foster_parent_element = $self->{open_elements}->[0]->[0] and
5648              $prev_sibling = $foster_parent_element->last_child
5649                  unless defined $foster_parent_element;
5650              undef $prev_sibling unless $open_tables->[-1]->[2]; # ~node inserted
5651              if (defined $prev_sibling and
5652                  $prev_sibling->node_type == 3) {
5653                !!!cp ('t198');
5654                $prev_sibling->manakai_append_text ($token->{data});
5655              } else {
5656                !!!cp ('t199');
5657                $foster_parent_element->insert_before
5658                    ($self->{document}->create_text_node ($token->{data}),
5659                     $next_sibling);
5660              }
5661            $open_tables->[-1]->[1] = 1; # tainted            $open_tables->[-1]->[1] = 1; # tainted
5662              $open_tables->[-1]->[2] = 1; # ~node inserted
5663          } else {          } else {
5664              ## NOTE: Fragment case or in a foster parent'ed element
5665              ## (e.g. |<table><span>a|).  In fragment case, whether the
5666              ## character is appended to existing node or a new node is
5667              ## created is irrelevant, since the foster parent'ed nodes
5668              ## are discarded and fragment parsing does not invoke any
5669              ## script.
5670            !!!cp ('t200');            !!!cp ('t200');
5671            $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});            $self->{open_elements}->[-1]->[0]->manakai_append_text
5672                  ($token->{data});
5673          }          }
5674                            
5675          !!!next-token;          !!!next-token;
# Line 5251  sub _tree_construction_main ($) { Line 5706  sub _tree_construction_main ($) {
5706                pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
5707              }              }
5708                                    
5709                  $self->{insertion_mode} = IN_ROW_IM;              $self->{insertion_mode} = IN_ROW_IM;
5710                  if ($token->{tag_name} eq 'tr') {              if ($token->{tag_name} eq 'tr') {
5711                    !!!cp ('t204');                !!!cp ('t204');
5712                    !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5713                    !!!nack ('t204');                $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5714                    !!!next-token;                !!!nack ('t204');
5715                    next B;                !!!next-token;
5716                  } else {                next B;
5717                    !!!cp ('t205');              } else {
5718                    !!!insert-element ('tr',, $token);                !!!cp ('t205');
5719                    ## reprocess in the "in row" insertion mode                !!!insert-element ('tr',, $token);
5720                  }                ## reprocess in the "in row" insertion mode
5721                } else {              }
5722                  !!!cp ('t206');            } else {
5723                }              !!!cp ('t206');
5724              }
5725    
5726                ## Clear back to table row context                ## Clear back to table row context
5727                while (not ($self->{open_elements}->[-1]->[1]                while (not ($self->{open_elements}->[-1]->[1]
# Line 5274  sub _tree_construction_main ($) { Line 5730  sub _tree_construction_main ($) {
5730                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
5731                }                }
5732                                
5733                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);            !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5734                $self->{insertion_mode} = IN_CELL_IM;            $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5735              $self->{insertion_mode} = IN_CELL_IM;
5736    
5737                push @$active_formatting_elements, ['#marker', ''];            push @$active_formatting_elements, ['#marker', ''];
5738                                
5739                !!!nack ('t207.1');            !!!nack ('t207.1');
5740              !!!next-token;
5741              next B;
5742            } elsif ({
5743                      caption => 1, col => 1, colgroup => 1,
5744                      tbody => 1, tfoot => 1, thead => 1,
5745                      tr => 1, # $self->{insertion_mode} == IN_ROW_IM
5746                     }->{$token->{tag_name}}) {
5747              if ($self->{insertion_mode} == IN_ROW_IM) {
5748                ## As if </tr>
5749                ## have an element in table scope
5750                my $i;
5751                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5752                  my $node = $self->{open_elements}->[$_];
5753                  if ($node->[1] == TABLE_ROW_EL) {
5754                    !!!cp ('t208');
5755                    $i = $_;
5756                    last INSCOPE;
5757                  } elsif ($node->[1] & TABLE_SCOPING_EL) {
5758                    !!!cp ('t209');
5759                    last INSCOPE;
5760                  }
5761                } # INSCOPE
5762                unless (defined $i) {
5763                  !!!cp ('t210');
5764                  ## TODO: This type is wrong.
5765                  !!!parse-error (type => 'unmacthed end tag',
5766                                  text => $token->{tag_name}, token => $token);
5767                  ## Ignore the token
5768                  !!!nack ('t210.1');
5769                !!!next-token;                !!!next-token;
5770                next B;                next B;
5771              } elsif ({              }
                       caption => 1, col => 1, colgroup => 1,  
                       tbody => 1, tfoot => 1, thead => 1,  
                       tr => 1, # $self->{insertion_mode} == IN_ROW_IM  
                      }->{$token->{tag_name}}) {  
               if ($self->{insertion_mode} == IN_ROW_IM) {  
                 ## As if </tr>  
                 ## have an element in table scope  
                 my $i;  
                 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
                   my $node = $self->{open_elements}->[$_];  
                   if ($node->[1] & TABLE_ROW_EL) {  
                     !!!cp ('t208');  
                     $i = $_;  
                     last INSCOPE;  
                   } elsif ($node->[1] & TABLE_SCOPING_EL) {  
                     !!!cp ('t209');  
                     last INSCOPE;  
                   }  
                 } # INSCOPE  
                 unless (defined $i) {  
                   !!!cp ('t210');  
 ## TODO: This type is wrong.  
                   !!!parse-error (type => 'unmacthed end tag',  
                                   text => $token->{tag_name}, token => $token);  
                   ## Ignore the token  
                   !!!nack ('t210.1');  
                   !!!next-token;  
                   next B;  
                 }  
5772                                    
5773                  ## Clear back to table row context                  ## Clear back to table row context
5774                  while (not ($self->{open_elements}->[-1]->[1]                  while (not ($self->{open_elements}->[-1]->[1]
# Line 5339  sub _tree_construction_main ($) { Line 5796  sub _tree_construction_main ($) {
5796                  my $i;                  my $i;
5797                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5798                    my $node = $self->{open_elements}->[$_];                    my $node = $self->{open_elements}->[$_];
5799                    if ($node->[1] & TABLE_ROW_GROUP_EL) {                    if ($node->[1] == TABLE_ROW_GROUP_EL) {
5800                      !!!cp ('t214');                      !!!cp ('t214');
5801                      $i = $_;                      $i = $_;
5802                      last INSCOPE;                      last INSCOPE;
# Line 5381  sub _tree_construction_main ($) { Line 5838  sub _tree_construction_main ($) {
5838                  !!!cp ('t218');                  !!!cp ('t218');
5839                }                }
5840    
5841                if ($token->{tag_name} eq 'col') {            if ($token->{tag_name} eq 'col') {
5842                  ## Clear back to table context              ## Clear back to table context
5843                  while (not ($self->{open_elements}->[-1]->[1]              while (not ($self->{open_elements}->[-1]->[1]
5844                                  & TABLE_SCOPING_EL)) {                              & TABLE_SCOPING_EL)) {
5845                    !!!cp ('t219');                !!!cp ('t219');
5846                    ## ISSUE: Can this state be reached?                ## ISSUE: Can this state be reached?
5847                    pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
5848                  }              }
5849                                
5850                  !!!insert-element ('colgroup',, $token);              !!!insert-element ('colgroup',, $token);
5851                  $self->{insertion_mode} = IN_COLUMN_GROUP_IM;              $self->{insertion_mode} = IN_COLUMN_GROUP_IM;
5852                  ## reprocess              ## reprocess
5853                  !!!ack-later;              $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5854                  next B;              !!!ack-later;
5855                } elsif ({              next B;
5856                          caption => 1,            } elsif ({
5857                          colgroup => 1,                      caption => 1,
5858                          tbody => 1, tfoot => 1, thead => 1,                      colgroup => 1,
5859                         }->{$token->{tag_name}}) {                      tbody => 1, tfoot => 1, thead => 1,
5860                  ## Clear back to table context                     }->{$token->{tag_name}}) {
5861                ## Clear back to table context
5862                  while (not ($self->{open_elements}->[-1]->[1]                  while (not ($self->{open_elements}->[-1]->[1]
5863                                  & TABLE_SCOPING_EL)) {                                  & TABLE_SCOPING_EL)) {
5864                    !!!cp ('t220');                    !!!cp ('t220');
# Line 5408  sub _tree_construction_main ($) { Line 5866  sub _tree_construction_main ($) {
5866                    pop @{$self->{open_elements}};                    pop @{$self->{open_elements}};
5867                  }                  }
5868                                    
5869                  push @$active_formatting_elements, ['#marker', '']              push @$active_formatting_elements, ['#marker', '']
5870                      if $token->{tag_name} eq 'caption';                  if $token->{tag_name} eq 'caption';
5871                                    
5872                  !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);              !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5873                  $self->{insertion_mode} = {              $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5874                                             caption => IN_CAPTION_IM,              $self->{insertion_mode} = {
5875                                             colgroup => IN_COLUMN_GROUP_IM,                                         caption => IN_CAPTION_IM,
5876                                             tbody => IN_TABLE_BODY_IM,                                         colgroup => IN_COLUMN_GROUP_IM,
5877                                             tfoot => IN_TABLE_BODY_IM,                                         tbody => IN_TABLE_BODY_IM,
5878                                             thead => IN_TABLE_BODY_IM,                                         tfoot => IN_TABLE_BODY_IM,
5879                                            }->{$token->{tag_name}};                                         thead => IN_TABLE_BODY_IM,
5880                  !!!next-token;                                        }->{$token->{tag_name}};
5881                  !!!nack ('t220.1');              !!!next-token;
5882                  next B;              !!!nack ('t220.1');
5883                } else {              next B;
5884                  die "$0: in table: <>: $token->{tag_name}";            } else {
5885                }              die "$0: in table: <>: $token->{tag_name}";
5886              }
5887              } elsif ($token->{tag_name} eq 'table') {              } elsif ($token->{tag_name} eq 'table') {
5888                !!!parse-error (type => 'not closed',                !!!parse-error (type => 'not closed',
5889                                text => $self->{open_elements}->[-1]->[0]                                text => $self->{open_elements}->[-1]->[0]
# Line 5436  sub _tree_construction_main ($) { Line 5895  sub _tree_construction_main ($) {
5895                my $i;                my $i;
5896                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5897                  my $node = $self->{open_elements}->[$_];                  my $node = $self->{open_elements}->[$_];
5898                  if ($node->[1] & TABLE_EL) {                  if ($node->[1] == TABLE_EL) {
5899                    !!!cp ('t221');                    !!!cp ('t221');
5900                    $i = $_;                    $i = $_;
5901                    last INSCOPE;                    last INSCOPE;
# Line 5463  sub _tree_construction_main ($) { Line 5922  sub _tree_construction_main ($) {
5922                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
5923                }                }
5924    
5925                unless ($self->{open_elements}->[-1]->[1] & TABLE_EL) {                unless ($self->{open_elements}->[-1]->[1] == TABLE_EL) {
5926                  !!!cp ('t225');                  !!!cp ('t225');
5927                  ## NOTE: |<table><tr><table>|                  ## NOTE: |<table><tr><table>|
5928                  !!!parse-error (type => 'not closed',                  !!!parse-error (type => 'not closed',
# Line 5487  sub _tree_construction_main ($) { Line 5946  sub _tree_construction_main ($) {
5946              !!!cp ('t227.8');              !!!cp ('t227.8');
5947              ## NOTE: This is a "as if in head" code clone.              ## NOTE: This is a "as if in head" code clone.
5948              $parse_rcdata->(CDATA_CONTENT_MODEL);              $parse_rcdata->(CDATA_CONTENT_MODEL);
5949                $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5950              next B;              next B;
5951            } else {            } else {
5952              !!!cp ('t227.7');              !!!cp ('t227.7');
# Line 5497  sub _tree_construction_main ($) { Line 5957  sub _tree_construction_main ($) {
5957              !!!cp ('t227.6');              !!!cp ('t227.6');
5958              ## NOTE: This is a "as if in head" code clone.              ## NOTE: This is a "as if in head" code clone.
5959              $script_start_tag->();              $script_start_tag->();
5960                $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5961              next B;              next B;
5962            } else {            } else {
5963              !!!cp ('t227.5');              !!!cp ('t227.5');
# Line 5512  sub _tree_construction_main ($) { Line 5973  sub _tree_construction_main ($) {
5973                                  text => $token->{tag_name}, token => $token);                                  text => $token->{tag_name}, token => $token);
5974    
5975                  !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);                  !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5976                    $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5977    
5978                  ## TODO: form element pointer                  ## TODO: form element pointer
5979    
# Line 5549  sub _tree_construction_main ($) { Line 6011  sub _tree_construction_main ($) {
6011                my $i;                my $i;
6012                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6013                  my $node = $self->{open_elements}->[$_];                  my $node = $self->{open_elements}->[$_];
6014                  if ($node->[1] & TABLE_ROW_EL) {                  if ($node->[1] == TABLE_ROW_EL) {
6015                    !!!cp ('t228');                    !!!cp ('t228');
6016                    $i = $_;                    $i = $_;
6017                    last INSCOPE;                    last INSCOPE;
# Line 5590  sub _tree_construction_main ($) { Line 6052  sub _tree_construction_main ($) {
6052                  my $i;                  my $i;
6053                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6054                    my $node = $self->{open_elements}->[$_];                    my $node = $self->{open_elements}->[$_];
6055                    if ($node->[1] & TABLE_ROW_EL) {                    if ($node->[1] == TABLE_ROW_EL) {
6056                      !!!cp ('t233');                      !!!cp ('t233');
6057                      $i = $_;                      $i = $_;
6058                      last INSCOPE;                      last INSCOPE;
# Line 5628  sub _tree_construction_main ($) { Line 6090  sub _tree_construction_main ($) {
6090                  my $i;                  my $i;
6091                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6092                    my $node = $self->{open_elements}->[$_];                    my $node = $self->{open_elements}->[$_];
6093                    if ($node->[1] & TABLE_ROW_GROUP_EL) {                    if ($node->[1] == TABLE_ROW_GROUP_EL) {
6094                      !!!cp ('t237');                      !!!cp ('t237');
6095                      $i = $_;                      $i = $_;
6096                      last INSCOPE;                      last INSCOPE;
# Line 5675  sub _tree_construction_main ($) { Line 6137  sub _tree_construction_main ($) {
6137                my $i;                my $i;
6138                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6139                  my $node = $self->{open_elements}->[$_];                  my $node = $self->{open_elements}->[$_];
6140                  if ($node->[1] & TABLE_EL) {                  if ($node->[1] == TABLE_EL) {
6141                    !!!cp ('t241');                    !!!cp ('t241');
6142                    $i = $_;                    $i = $_;
6143                    last INSCOPE;                    last INSCOPE;
# Line 5734  sub _tree_construction_main ($) { Line 6196  sub _tree_construction_main ($) {
6196                  my $i;                  my $i;
6197                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6198                    my $node = $self->{open_elements}->[$_];                    my $node = $self->{open_elements}->[$_];
6199                    if ($node->[1] & TABLE_ROW_EL) {                    if ($node->[1] == TABLE_ROW_EL) {
6200                      !!!cp ('t250');                      !!!cp ('t250');
6201                      $i = $_;                      $i = $_;
6202                      last INSCOPE;                      last INSCOPE;
# Line 5824  sub _tree_construction_main ($) { Line 6286  sub _tree_construction_main ($) {
6286            #            #
6287          }          }
6288        } elsif ($token->{type} == END_OF_FILE_TOKEN) {        } elsif ($token->{type} == END_OF_FILE_TOKEN) {
6289          unless ($self->{open_elements}->[-1]->[1] & HTML_EL and          unless ($self->{open_elements}->[-1]->[1] == HTML_EL and
6290                  @{$self->{open_elements}} == 1) { # redundant, maybe                  @{$self->{open_elements}} == 1) { # redundant, maybe
6291            !!!parse-error (type => 'in body:#eof', token => $token);            !!!parse-error (type => 'in body:#eof', token => $token);
6292            !!!cp ('t259.1');            !!!cp ('t259.1');
# Line 5841  sub _tree_construction_main ($) { Line 6303  sub _tree_construction_main ($) {
6303        }        }
6304      } elsif ($self->{insertion_mode} == IN_COLUMN_GROUP_IM) {      } elsif ($self->{insertion_mode} == IN_COLUMN_GROUP_IM) {
6305            if ($token->{type} == CHARACTER_TOKEN) {            if ($token->{type} == CHARACTER_TOKEN) {
6306              if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {              if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
6307                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
6308                unless (length $token->{data}) {                unless (length $token->{data}) {
6309                  !!!cp ('t260');                  !!!cp ('t260');
# Line 5866  sub _tree_construction_main ($) { Line 6328  sub _tree_construction_main ($) {
6328              }              }
6329            } elsif ($token->{type} == END_TAG_TOKEN) {            } elsif ($token->{type} == END_TAG_TOKEN) {
6330              if ($token->{tag_name} eq 'colgroup') {              if ($token->{tag_name} eq 'colgroup') {
6331                if ($self->{open_elements}->[-1]->[1] & HTML_EL) {                if ($self->{open_elements}->[-1]->[1] == HTML_EL) {
6332                  !!!cp ('t264');                  !!!cp ('t264');
6333                  !!!parse-error (type => 'unmatched end tag',                  !!!parse-error (type => 'unmatched end tag',
6334                                  text => 'colgroup', token => $token);                                  text => 'colgroup', token => $token);
# Line 5892  sub _tree_construction_main ($) { Line 6354  sub _tree_construction_main ($) {
6354                #                #
6355              }              }
6356        } elsif ($token->{type} == END_OF_FILE_TOKEN) {        } elsif ($token->{type} == END_OF_FILE_TOKEN) {
6357          if ($self->{open_elements}->[-1]->[1] & HTML_EL and          if ($self->{open_elements}->[-1]->[1] == HTML_EL and
6358              @{$self->{open_elements}} == 1) { # redundant, maybe              @{$self->{open_elements}} == 1) { # redundant, maybe
6359            !!!cp ('t270.2');            !!!cp ('t270.2');
6360            ## Stop parsing.            ## Stop parsing.
# Line 5910  sub _tree_construction_main ($) { Line 6372  sub _tree_construction_main ($) {
6372        }        }
6373    
6374            ## As if </colgroup>            ## As if </colgroup>
6375            if ($self->{open_elements}->[-1]->[1] & HTML_EL) {            if ($self->{open_elements}->[-1]->[1] == HTML_EL) {
6376              !!!cp ('t269');              !!!cp ('t269');
6377  ## TODO: Wrong error type?  ## TODO: Wrong error type?
6378              !!!parse-error (type => 'unmatched end tag',              !!!parse-error (type => 'unmatched end tag',
# Line 5935  sub _tree_construction_main ($) { Line 6397  sub _tree_construction_main ($) {
6397          next B;          next B;
6398        } elsif ($token->{type} == START_TAG_TOKEN) {        } elsif ($token->{type} == START_TAG_TOKEN) {
6399          if ($token->{tag_name} eq 'option') {          if ($token->{tag_name} eq 'option') {
6400            if ($self->{open_elements}->[-1]->[1] & OPTION_EL) {            if ($self->{open_elements}->[-1]->[1] == OPTION_EL) {
6401              !!!cp ('t272');              !!!cp ('t272');
6402              ## As if </option>              ## As if </option>
6403              pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
# Line 5948  sub _tree_construction_main ($) { Line 6410  sub _tree_construction_main ($) {
6410            !!!next-token;            !!!next-token;
6411            next B;            next B;
6412          } elsif ($token->{tag_name} eq 'optgroup') {          } elsif ($token->{tag_name} eq 'optgroup') {
6413            if ($self->{open_elements}->[-1]->[1] & OPTION_EL) {            if ($self->{open_elements}->[-1]->[1] == OPTION_EL) {
6414              !!!cp ('t274');              !!!cp ('t274');
6415              ## As if </option>              ## As if </option>
6416              pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
# Line 5956  sub _tree_construction_main ($) { Line 6418  sub _tree_construction_main ($) {
6418              !!!cp ('t275');              !!!cp ('t275');
6419            }            }
6420    
6421            if ($self->{open_elements}->[-1]->[1] & OPTGROUP_EL) {            if ($self->{open_elements}->[-1]->[1] == OPTGROUP_EL) {
6422              !!!cp ('t276');              !!!cp ('t276');
6423              ## As if </optgroup>              ## As if </optgroup>
6424              pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
# Line 5986  sub _tree_construction_main ($) { Line 6448  sub _tree_construction_main ($) {
6448            my $i;            my $i;
6449            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6450              my $node = $self->{open_elements}->[$_];              my $node = $self->{open_elements}->[$_];
6451              if ($node->[1] & SELECT_EL) {              if ($node->[1] == SELECT_EL) {
6452                !!!cp ('t278');                !!!cp ('t278');
6453                $i = $_;                $i = $_;
6454                last INSCOPE;                last INSCOPE;
# Line 6031  sub _tree_construction_main ($) { Line 6493  sub _tree_construction_main ($) {
6493          }          }
6494        } elsif ($token->{type} == END_TAG_TOKEN) {        } elsif ($token->{type} == END_TAG_TOKEN) {
6495          if ($token->{tag_name} eq 'optgroup') {          if ($token->{tag_name} eq 'optgroup') {
6496            if ($self->{open_elements}->[-1]->[1] & OPTION_EL and            if ($self->{open_elements}->[-1]->[1] == OPTION_EL and
6497                $self->{open_elements}->[-2]->[1] & OPTGROUP_EL) {                $self->{open_elements}->[-2]->[1] == OPTGROUP_EL) {
6498              !!!cp ('t283');              !!!cp ('t283');
6499              ## As if </option>              ## As if </option>
6500              splice @{$self->{open_elements}}, -2;              splice @{$self->{open_elements}}, -2;
6501            } elsif ($self->{open_elements}->[-1]->[1] & OPTGROUP_EL) {            } elsif ($self->{open_elements}->[-1]->[1] == OPTGROUP_EL) {
6502              !!!cp ('t284');              !!!cp ('t284');
6503              pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
6504            } else {            } else {
# Line 6049  sub _tree_construction_main ($) { Line 6511  sub _tree_construction_main ($) {
6511            !!!next-token;            !!!next-token;
6512            next B;            next B;
6513          } elsif ($token->{tag_name} eq 'option') {          } elsif ($token->{tag_name} eq 'option') {
6514            if ($self->{open_elements}->[-1]->[1] & OPTION_EL) {            if ($self->{open_elements}->[-1]->[1] == OPTION_EL) {
6515              !!!cp ('t286');              !!!cp ('t286');
6516              pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
6517            } else {            } else {
# Line 6066  sub _tree_construction_main ($) { Line 6528  sub _tree_construction_main ($) {
6528            my $i;            my $i;
6529            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6530              my $node = $self->{open_elements}->[$_];              my $node = $self->{open_elements}->[$_];
6531              if ($node->[1] & SELECT_EL) {              if ($node->[1] == SELECT_EL) {
6532                !!!cp ('t288');                !!!cp ('t288');
6533                $i = $_;                $i = $_;
6534                last INSCOPE;                last INSCOPE;
# Line 6128  sub _tree_construction_main ($) { Line 6590  sub _tree_construction_main ($) {
6590            undef $i;            undef $i;
6591            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6592              my $node = $self->{open_elements}->[$_];              my $node = $self->{open_elements}->[$_];
6593              if ($node->[1] & SELECT_EL) {              if ($node->[1] == SELECT_EL) {
6594                !!!cp ('t295');                !!!cp ('t295');
6595                $i = $_;                $i = $_;
6596                last INSCOPE;                last INSCOPE;
# Line 6167  sub _tree_construction_main ($) { Line 6629  sub _tree_construction_main ($) {
6629            next B;            next B;
6630          }          }
6631        } elsif ($token->{type} == END_OF_FILE_TOKEN) {        } elsif ($token->{type} == END_OF_FILE_TOKEN) {
6632          unless ($self->{open_elements}->[-1]->[1] & HTML_EL and          unless ($self->{open_elements}->[-1]->[1] == HTML_EL and
6633                  @{$self->{open_elements}} == 1) { # redundant, maybe                  @{$self->{open_elements}} == 1) { # redundant, maybe
6634            !!!cp ('t299.1');            !!!cp ('t299.1');
6635            !!!parse-error (type => 'in body:#eof', token => $token);            !!!parse-error (type => 'in body:#eof', token => $token);
# Line 6182  sub _tree_construction_main ($) { Line 6644  sub _tree_construction_main ($) {
6644        }        }
6645      } elsif ($self->{insertion_mode} & BODY_AFTER_IMS) {      } elsif ($self->{insertion_mode} & BODY_AFTER_IMS) {
6646        if ($token->{type} == CHARACTER_TOKEN) {        if ($token->{type} == CHARACTER_TOKEN) {
6647          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {          if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
6648            my $data = $1;            my $data = $1;
6649            ## As if in body            ## As if in body
6650            $reconstruct_active_formatting_elements->($insert_to_current);            $reconstruct_active_formatting_elements->($insert_to_current);
# Line 6199  sub _tree_construction_main ($) { Line 6661  sub _tree_construction_main ($) {
6661          if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {          if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
6662            !!!cp ('t301');            !!!cp ('t301');
6663            !!!parse-error (type => 'after html:#text', token => $token);            !!!parse-error (type => 'after html:#text', token => $token);
6664              #
           ## Reprocess in the "after body" insertion mode.  
6665          } else {          } else {
6666            !!!cp ('t302');            !!!cp ('t302');
6667              ## "after body" insertion mode
6668              !!!parse-error (type => 'after body:#text', token => $token);
6669              #
6670          }          }
           
         ## "after body" insertion mode  
         !!!parse-error (type => 'after body:#text', token => $token);  
6671    
6672          $self->{insertion_mode} = IN_BODY_IM;          $self->{insertion_mode} = IN_BODY_IM;
6673          ## reprocess          ## reprocess
# Line 6216  sub _tree_construction_main ($) { Line 6677  sub _tree_construction_main ($) {
6677            !!!cp ('t303');            !!!cp ('t303');
6678            !!!parse-error (type => 'after html',            !!!parse-error (type => 'after html',
6679                            text => $token->{tag_name}, token => $token);                            text => $token->{tag_name}, token => $token);
6680                        #
           ## Reprocess in the "after body" insertion mode.  
6681          } else {          } else {
6682            !!!cp ('t304');            !!!cp ('t304');
6683              ## "after body" insertion mode
6684              !!!parse-error (type => 'after body',
6685                              text => $token->{tag_name}, token => $token);
6686              #
6687          }          }
6688    
         ## "after body" insertion mode  
         !!!parse-error (type => 'after body',  
                         text => $token->{tag_name}, token => $token);  
   
6689          $self->{insertion_mode} = IN_BODY_IM;          $self->{insertion_mode} = IN_BODY_IM;
6690          !!!ack-later;          !!!ack-later;
6691          ## reprocess          ## reprocess
# Line 6236  sub _tree_construction_main ($) { Line 6696  sub _tree_construction_main ($) {
6696            !!!parse-error (type => 'after html:/',            !!!parse-error (type => 'after html:/',
6697                            text => $token->{tag_name}, token => $token);                            text => $token->{tag_name}, token => $token);
6698                        
6699            $self->{insertion_mode} = AFTER_BODY_IM;            $self->{insertion_mode} = IN_BODY_IM;
6700            ## Reprocess in the "after body" insertion mode.            ## Reprocess.
6701              next B;
6702          } else {          } else {
6703            !!!cp ('t306');            !!!cp ('t306');
6704          }          }
# Line 6275  sub _tree_construction_main ($) { Line 6736  sub _tree_construction_main ($) {
6736        }        }
6737      } elsif ($self->{insertion_mode} & FRAME_IMS) {      } elsif ($self->{insertion_mode} & FRAME_IMS) {
6738        if ($token->{type} == CHARACTER_TOKEN) {        if ($token->{type} == CHARACTER_TOKEN) {
6739          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {          if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
6740            $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);            $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
6741                        
6742            unless (length $token->{data}) {            unless (length $token->{data}) {
# Line 6285  sub _tree_construction_main ($) { Line 6746  sub _tree_construction_main ($) {
6746            }            }
6747          }          }
6748                    
6749          if ($token->{data} =~ s/^[^\x09\x0A\x0B\x0C\x20]+//) {          if ($token->{data} =~ s/^[^\x09\x0A\x0C\x20]+//) {
6750            if ($self->{insertion_mode} == IN_FRAMESET_IM) {            if ($self->{insertion_mode} == IN_FRAMESET_IM) {
6751              !!!cp ('t311');              !!!cp ('t311');
6752              !!!parse-error (type => 'in frameset:#text', token => $token);              !!!parse-error (type => 'in frameset:#text', token => $token);
6753            } elsif ($self->{insertion_mode} == AFTER_FRAMESET_IM) {            } elsif ($self->{insertion_mode} == AFTER_FRAMESET_IM) {
6754              !!!cp ('t312');              !!!cp ('t312');
6755              !!!parse-error (type => 'after frameset:#text', token => $token);              !!!parse-error (type => 'after frameset:#text', token => $token);
6756            } else { # "after html frameset"            } else { # "after after frameset"
6757              !!!cp ('t313');              !!!cp ('t313');
6758              !!!parse-error (type => 'after html:#text', token => $token);              !!!parse-error (type => 'after html:#text', token => $token);
   
             $self->{insertion_mode} = AFTER_FRAMESET_IM;  
             ## Reprocess in the "after frameset" insertion mode.  
             !!!parse-error (type => 'after frameset:#text', token => $token);  
6759            }            }
6760                        
6761            ## Ignore the token.            ## Ignore the token.
# Line 6314  sub _tree_construction_main ($) { Line 6771  sub _tree_construction_main ($) {
6771                    
6772          die qq[$0: Character "$token->{data}"];          die qq[$0: Character "$token->{data}"];
6773        } elsif ($token->{type} == START_TAG_TOKEN) {        } elsif ($token->{type} == START_TAG_TOKEN) {
         if ($self->{insertion_mode} == AFTER_HTML_FRAMESET_IM) {  
           !!!cp ('t316');  
           !!!parse-error (type => 'after html',  
                           text => $token->{tag_name}, token => $token);  
   
           $self->{insertion_mode} = AFTER_FRAMESET_IM;  
           ## Process in the "after frameset" insertion mode.  
         } else {  
           !!!cp ('t317');  
         }  
   
6774          if ($token->{tag_name} eq 'frameset' and          if ($token->{tag_name} eq 'frameset' and
6775              $self->{insertion_mode} == IN_FRAMESET_IM) {              $self->{insertion_mode} == IN_FRAMESET_IM) {
6776            !!!cp ('t318');            !!!cp ('t318');
# Line 6345  sub _tree_construction_main ($) { Line 6791  sub _tree_construction_main ($) {
6791            ## NOTE: As if in head.            ## NOTE: As if in head.
6792            $parse_rcdata->(CDATA_CONTENT_MODEL);            $parse_rcdata->(CDATA_CONTENT_MODEL);
6793            next B;            next B;
6794    
6795              ## NOTE: |<!DOCTYPE HTML><frameset></frameset></html><noframes></noframes>|
6796              ## has no parse error.
6797          } else {          } else {
6798            if ($self->{insertion_mode} == IN_FRAMESET_IM) {            if ($self->{insertion_mode} == IN_FRAMESET_IM) {
6799              !!!cp ('t321');              !!!cp ('t321');
6800              !!!parse-error (type => 'in frameset',              !!!parse-error (type => 'in frameset',
6801                              text => $token->{tag_name}, token => $token);                              text => $token->{tag_name}, token => $token);
6802            } else {            } elsif ($self->{insertion_mode} == AFTER_FRAMESET_IM) {
6803              !!!cp ('t322');              !!!cp ('t322');
6804              !!!parse-error (type => 'after frameset',              !!!parse-error (type => 'after frameset',
6805                              text => $token->{tag_name}, token => $token);                              text => $token->{tag_name}, token => $token);
6806              } else { # "after after frameset"
6807                !!!cp ('t322.2');
6808                !!!parse-error (type => 'after after frameset',
6809                                text => $token->{tag_name}, token => $token);
6810            }            }
6811            ## Ignore the token            ## Ignore the token
6812            !!!nack ('t322.1');            !!!nack ('t322.1');
# Line 6361  sub _tree_construction_main ($) { Line 6814  sub _tree_construction_main ($) {
6814            next B;            next B;
6815          }          }
6816        } elsif ($token->{type} == END_TAG_TOKEN) {        } elsif ($token->{type} == END_TAG_TOKEN) {
         if ($self->{insertion_mode} == AFTER_HTML_FRAMESET_IM) {  
           !!!cp ('t323');  
           !!!parse-error (type => 'after html:/',  
                           text => $token->{tag_name}, token => $token);  
   
           $self->{insertion_mode} = AFTER_FRAMESET_IM;  
           ## Process in the "after frameset" insertion mode.  
         } else {  
           !!!cp ('t324');  
         }  
   
6817          if ($token->{tag_name} eq 'frameset' and          if ($token->{tag_name} eq 'frameset' and
6818              $self->{insertion_mode} == IN_FRAMESET_IM) {              $self->{insertion_mode} == IN_FRAMESET_IM) {
6819            if ($self->{open_elements}->[-1]->[1] & HTML_EL and            if ($self->{open_elements}->[-1]->[1] == HTML_EL and
6820                @{$self->{open_elements}} == 1) {                @{$self->{open_elements}} == 1) {
6821              !!!cp ('t325');              !!!cp ('t325');
6822              !!!parse-error (type => 'unmatched end tag',              !!!parse-error (type => 'unmatched end tag',
# Line 6388  sub _tree_construction_main ($) { Line 6830  sub _tree_construction_main ($) {
6830            }            }
6831    
6832            if (not defined $self->{inner_html_node} and            if (not defined $self->{inner_html_node} and
6833                not ($self->{open_elements}->[-1]->[1] & FRAMESET_EL)) {                not ($self->{open_elements}->[-1]->[1] == FRAMESET_EL)) {
6834              !!!cp ('t327');              !!!cp ('t327');
6835              $self->{insertion_mode} = AFTER_FRAMESET_IM;              $self->{insertion_mode} = AFTER_FRAMESET_IM;
6836            } else {            } else {
# Line 6406  sub _tree_construction_main ($) { Line 6848  sub _tree_construction_main ($) {
6848              !!!cp ('t330');              !!!cp ('t330');
6849              !!!parse-error (type => 'in frameset:/',              !!!parse-error (type => 'in frameset:/',
6850                              text => $token->{tag_name}, token => $token);                              text => $token->{tag_name}, token => $token);
6851            } else {            } elsif ($self->{insertion_mode} == AFTER_FRAMESET_IM) {
6852              !!!cp ('t331');              !!!cp ('t330.1');
6853              !!!parse-error (type => 'after frameset:/',              !!!parse-error (type => 'after frameset:/',
6854                              text => $token->{tag_name}, token => $token);                              text => $token->{tag_name}, token => $token);
6855              } else { # "after after html"
6856                !!!cp ('t331');
6857                !!!parse-error (type => 'after after frameset:/',
6858                                text => $token->{tag_name}, token => $token);
6859            }            }
6860            ## Ignore the token            ## Ignore the token
6861            !!!next-token;            !!!next-token;
6862            next B;            next B;
6863          }          }
6864        } elsif ($token->{type} == END_OF_FILE_TOKEN) {        } elsif ($token->{type} == END_OF_FILE_TOKEN) {
6865          unless ($self->{open_elements}->[-1]->[1] & HTML_EL and          unless ($self->{open_elements}->[-1]->[1] == HTML_EL and
6866                  @{$self->{open_elements}} == 1) { # redundant, maybe                  @{$self->{open_elements}} == 1) { # redundant, maybe
6867            !!!cp ('t331.1');            !!!cp ('t331.1');
6868            !!!parse-error (type => 'in body:#eof', token => $token);            !!!parse-error (type => 'in body:#eof', token => $token);
# Line 6429  sub _tree_construction_main ($) { Line 6875  sub _tree_construction_main ($) {
6875        } else {        } else {
6876          die "$0: $token->{type}: Unknown token type";          die "$0: $token->{type}: Unknown token type";
6877        }        }
   
       ## ISSUE: An issue in spec here  
6878      } else {      } else {
6879        die "$0: $self->{insertion_mode}: Unknown insertion mode";        die "$0: $self->{insertion_mode}: Unknown insertion mode";
6880      }      }
# Line 6448  sub _tree_construction_main ($) { Line 6892  sub _tree_construction_main ($) {
6892          $parse_rcdata->(CDATA_CONTENT_MODEL);          $parse_rcdata->(CDATA_CONTENT_MODEL);
6893          next B;          next B;
6894        } elsif ({        } elsif ({
6895                  base => 1, link => 1,                  base => 1, command => 1, eventsource => 1, link => 1,
6896                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
6897          !!!cp ('t334');          !!!cp ('t334');
6898          ## NOTE: This is an "as if in head" code clone, only "-t" differs          ## NOTE: This is an "as if in head" code clone, only "-t" differs
6899          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
6900          pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.          pop @{$self->{open_elements}};
6901          !!!ack ('t334.1');          !!!ack ('t334.1');
6902          !!!next-token;          !!!next-token;
6903          next B;          next B;
6904        } elsif ($token->{tag_name} eq 'meta') {        } elsif ($token->{tag_name} eq 'meta') {
6905          ## NOTE: This is an "as if in head" code clone, only "-t" differs          ## NOTE: This is an "as if in head" code clone, only "-t" differs
6906          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
6907          my $meta_el = pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.          my $meta_el = pop @{$self->{open_elements}};
6908    
6909          unless ($self->{confident}) {          unless ($self->{confident}) {
6910            if ($token->{attributes}->{charset}) {            if ($token->{attributes}->{charset}) {
# Line 6477  sub _tree_construction_main ($) { Line 6921  sub _tree_construction_main ($) {
6921            } elsif ($token->{attributes}->{content}) {            } elsif ($token->{attributes}->{content}) {
6922              if ($token->{attributes}->{content}->{value}              if ($token->{attributes}->{content}->{value}
6923                  =~ /[Cc][Hh][Aa][Rr][Ss][Ee][Tt]                  =~ /[Cc][Hh][Aa][Rr][Ss][Ee][Tt]
6924                      [\x09-\x0D\x20]*=                      [\x09\x0A\x0C\x0D\x20]*=
6925                      [\x09-\x0D\x20]*(?>"([^"]*)"|'([^']*)'|                      [\x09\x0A\x0C\x0D\x20]*(?>"([^"]*)"|'([^']*)'|
6926                      ([^"'\x09-\x0D\x20][^\x09-\x0D\x20\x3B]*))/x) {                      ([^"'\x09\x0A\x0C\x0D\x20][^\x09\x0A\x0C\x0D\x20\x3B]*))
6927                       /x) {
6928                !!!cp ('t336');                !!!cp ('t336');
6929                ## NOTE: Whether the encoding is supported or not is handled                ## NOTE: Whether the encoding is supported or not is handled
6930                ## in the {change_encoding} callback.                ## in the {change_encoding} callback.
# Line 6520  sub _tree_construction_main ($) { Line 6965  sub _tree_construction_main ($) {
6965          !!!parse-error (type => 'in body', text => 'body', token => $token);          !!!parse-error (type => 'in body', text => 'body', token => $token);
6966                                
6967          if (@{$self->{open_elements}} == 1 or          if (@{$self->{open_elements}} == 1 or
6968              not ($self->{open_elements}->[1]->[1] & BODY_EL)) {              not ($self->{open_elements}->[1]->[1] == BODY_EL)) {
6969            !!!cp ('t342');            !!!cp ('t342');
6970            ## Ignore the token            ## Ignore the token
6971          } else {          } else {
# Line 6538  sub _tree_construction_main ($) { Line 6983  sub _tree_construction_main ($) {
6983          !!!next-token;          !!!next-token;
6984          next B;          next B;
6985        } elsif ({        } elsif ({
6986                  address => 1, blockquote => 1, center => 1, dir => 1,                  ## NOTE: Start tags for non-phrasing flow content elements
6987                  div => 1, dl => 1, fieldset => 1,  
6988                  h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,                  ## NOTE: The normal one
6989                  menu => 1, ol => 1, p => 1, ul => 1,                  address => 1, article => 1, aside => 1, blockquote => 1,
6990                    center => 1, datagrid => 1, details => 1, dialog => 1,
6991                    dir => 1, div => 1, dl => 1, fieldset => 1, figure => 1,
6992                    footer => 1, h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1,
6993                    h6 => 1, header => 1, menu => 1, nav => 1, ol => 1, p => 1,
6994                    section => 1, ul => 1,
6995                    ## NOTE: As normal, but drops leading newline
6996                  pre => 1, listing => 1,                  pre => 1, listing => 1,
6997                    ## NOTE: As normal, but interacts with the form element pointer
6998                  form => 1,                  form => 1,
6999                    
7000                  table => 1,                  table => 1,
7001                  hr => 1,                  hr => 1,
7002                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
# Line 6558  sub _tree_construction_main ($) { Line 7011  sub _tree_construction_main ($) {
7011    
7012          ## has a p element in scope          ## has a p element in scope
7013          INSCOPE: for (reverse @{$self->{open_elements}}) {          INSCOPE: for (reverse @{$self->{open_elements}}) {
7014            if ($_->[1] & P_EL) {            if ($_->[1] == P_EL) {
7015              !!!cp ('t344');              !!!cp ('t344');
7016              !!!back-token; # <form>              !!!back-token; # <form>
7017              $token = {type => END_TAG_TOKEN, tag_name => 'p',              $token = {type => END_TAG_TOKEN, tag_name => 'p',
# Line 6610  sub _tree_construction_main ($) { Line 7063  sub _tree_construction_main ($) {
7063            !!!next-token;            !!!next-token;
7064          }          }
7065          next B;          next B;
7066        } elsif ({li => 1, dt => 1, dd => 1}->{$token->{tag_name}}) {        } elsif ($token->{tag_name} eq 'li') {
7067          ## has a p element in scope          ## NOTE: As normal, but imply </li> when there's another <li> ...
7068    
7069            ## NOTE: Special, Scope (<li><foo><li> == <li><foo><li/></foo></li>)
7070              ## Interpreted as <li><foo/></li><li/> (non-conforming)
7071              ## blockquote (O9.27), center (O), dd (Fx3, O, S3.1.2, IE7),
7072              ## dt (Fx, O, S, IE), dl (O), fieldset (O, S, IE), form (Fx, O, S),
7073              ## hn (O), pre (O), applet (O, S), button (O, S), marquee (Fx, O, S),
7074              ## object (Fx)
7075              ## Generate non-tree (non-conforming)
7076              ## basefont (IE7 (where basefont is non-void)), center (IE),
7077              ## form (IE), hn (IE)
7078            ## address, div, p (<li><foo><li> == <li><foo/></li><li/>)
7079              ## Interpreted as <li><foo><li/></foo></li> (non-conforming)
7080              ## div (Fx, S)
7081    
7082            my $non_optional;
7083            my $i = -1;
7084    
7085            ## 1.
7086            for my $node (reverse @{$self->{open_elements}}) {
7087              if ($node->[1] == LI_EL) {
7088                ## 2. (a) As if </li>
7089                {
7090                  ## If no </li> - not applied
7091                  #
7092    
7093                  ## Otherwise
7094    
7095                  ## 1. generate implied end tags, except for </li>
7096                  #
7097    
7098                  ## 2. If current node != "li", parse error
7099                  if ($non_optional) {
7100                    !!!parse-error (type => 'not closed',
7101                                    text => $non_optional->[0]->manakai_local_name,
7102                                    token => $token);
7103                    !!!cp ('t355');
7104                  } else {
7105                    !!!cp ('t356');
7106                  }
7107    
7108                  ## 3. Pop
7109                  splice @{$self->{open_elements}}, $i;
7110                }
7111    
7112                last; ## 2. (b) goto 5.
7113              } elsif (
7114                       ## NOTE: not "formatting" and not "phrasing"
7115                       ($node->[1] & SPECIAL_EL or
7116                        $node->[1] & SCOPING_EL) and
7117                       ## NOTE: "li", "dt", and "dd" are in |SPECIAL_EL|.
7118                       (not $node->[1] & ADDRESS_DIV_P_EL)
7119                      ) {
7120                ## 3.
7121                !!!cp ('t357');
7122                last; ## goto 5.
7123              } elsif ($node->[1] & END_TAG_OPTIONAL_EL) {
7124                !!!cp ('t358');
7125                #
7126              } else {
7127                !!!cp ('t359');
7128                $non_optional ||= $node;
7129                #
7130              }
7131              ## 4.
7132              ## goto 2.
7133              $i--;
7134            }
7135    
7136            ## 5. (a) has a |p| element in scope
7137          INSCOPE: for (reverse @{$self->{open_elements}}) {          INSCOPE: for (reverse @{$self->{open_elements}}) {
7138            if ($_->[1] & P_EL) {            if ($_->[1] == P_EL) {
7139              !!!cp ('t353');              !!!cp ('t353');
7140    
7141                ## NOTE: |<p><li>|, for example.
7142    
7143              !!!back-token; # <x>              !!!back-token; # <x>
7144              $token = {type => END_TAG_TOKEN, tag_name => 'p',              $token = {type => END_TAG_TOKEN, tag_name => 'p',
7145                        line => $token->{line}, column => $token->{column}};                        line => $token->{line}, column => $token->{column}};
# Line 6624  sub _tree_construction_main ($) { Line 7149  sub _tree_construction_main ($) {
7149              last INSCOPE;              last INSCOPE;
7150            }            }
7151          } # INSCOPE          } # INSCOPE
7152              
7153          ## Step 1          ## 5. (b) insert
7154            !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7155            !!!nack ('t359.1');
7156            !!!next-token;
7157            next B;
7158          } elsif ($token->{tag_name} eq 'dt' or
7159                   $token->{tag_name} eq 'dd') {
7160            ## NOTE: As normal, but imply </dt> or </dd> when ...
7161    
7162            my $non_optional;
7163          my $i = -1;          my $i = -1;
7164          my $node = $self->{open_elements}->[$i];  
7165          my $li_or_dtdd = {li => {li => 1},          ## 1.
7166                            dt => {dt => 1, dd => 1},          for my $node (reverse @{$self->{open_elements}}) {
7167                            dd => {dt => 1, dd => 1}}->{$token->{tag_name}};            if ($node->[1] == DT_EL or $node->[1] == DD_EL) {
7168          LI: {              ## 2. (a) As if </li>
7169            ## Step 2              {
7170            if ($li_or_dtdd->{$node->[0]->manakai_local_name}) {                ## If no </li> - not applied
7171              if ($i != -1) {                #
7172                !!!cp ('t355');  
7173                !!!parse-error (type => 'not closed',                ## Otherwise
7174                                text => $self->{open_elements}->[-1]->[0]  
7175                                    ->manakai_local_name,                ## 1. generate implied end tags, except for </dt> or </dd>
7176                                token => $token);                #
7177              } else {  
7178                !!!cp ('t356');                ## 2. If current node != "dt"|"dd", parse error
7179                  if ($non_optional) {
7180                    !!!parse-error (type => 'not closed',
7181                                    text => $non_optional->[0]->manakai_local_name,
7182                                    token => $token);
7183                    !!!cp ('t355.1');
7184                  } else {
7185                    !!!cp ('t356.1');
7186                  }
7187    
7188                  ## 3. Pop
7189                  splice @{$self->{open_elements}}, $i;
7190              }              }
7191              splice @{$self->{open_elements}}, $i;  
7192              last LI;              last; ## 2. (b) goto 5.
7193              } elsif (
7194                       ## NOTE: not "formatting" and not "phrasing"
7195                       ($node->[1] & SPECIAL_EL or
7196                        $node->[1] & SCOPING_EL) and
7197                       ## NOTE: "li", "dt", and "dd" are in |SPECIAL_EL|.
7198    
7199                       (not $node->[1] & ADDRESS_DIV_P_EL)
7200                      ) {
7201                ## 3.
7202                !!!cp ('t357.1');
7203                last; ## goto 5.
7204              } elsif ($node->[1] & END_TAG_OPTIONAL_EL) {
7205                !!!cp ('t358.1');
7206                #
7207            } else {            } else {
7208              !!!cp ('t357');              !!!cp ('t359.1');
7209            }              $non_optional ||= $node;
7210                          #
           ## Step 3  
           if (not ($node->[1] & FORMATTING_EL) and  
               #not $phrasing_category->{$node->[1]} and  
               ($node->[1] & SPECIAL_EL or  
                $node->[1] & SCOPING_EL) and  
               not ($node->[1] & ADDRESS_EL) and  
               not ($node->[1] & DIV_EL)) {  
             !!!cp ('t358');  
             last LI;  
7211            }            }
7212                        ## 4.
7213            !!!cp ('t359');            ## goto 2.
           ## Step 4  
7214            $i--;            $i--;
7215            $node = $self->{open_elements}->[$i];          }
7216            redo LI;  
7217          } # LI          ## 5. (a) has a |p| element in scope
7218                      INSCOPE: for (reverse @{$self->{open_elements}}) {
7219              if ($_->[1] == P_EL) {
7220                !!!cp ('t353.1');
7221                !!!back-token; # <x>
7222                $token = {type => END_TAG_TOKEN, tag_name => 'p',
7223                          line => $token->{line}, column => $token->{column}};
7224                next B;
7225              } elsif ($_->[1] & SCOPING_EL) {
7226                !!!cp ('t354.1');
7227                last INSCOPE;
7228              }
7229            } # INSCOPE
7230    
7231            ## 5. (b) insert
7232          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7233          !!!nack ('t359.1');          !!!nack ('t359.2');
7234          !!!next-token;          !!!next-token;
7235          next B;          next B;
7236        } elsif ($token->{tag_name} eq 'plaintext') {        } elsif ($token->{tag_name} eq 'plaintext') {
7237            ## NOTE: As normal, but effectively ends parsing
7238    
7239          ## has a p element in scope          ## has a p element in scope
7240          INSCOPE: for (reverse @{$self->{open_elements}}) {          INSCOPE: for (reverse @{$self->{open_elements}}) {
7241            if ($_->[1] & P_EL) {            if ($_->[1] == P_EL) {
7242              !!!cp ('t367');              !!!cp ('t367');
7243              !!!back-token; # <plaintext>              !!!back-token; # <plaintext>
7244              $token = {type => END_TAG_TOKEN, tag_name => 'p',              $token = {type => END_TAG_TOKEN, tag_name => 'p',
# Line 6696  sub _tree_construction_main ($) { Line 7260  sub _tree_construction_main ($) {
7260        } elsif ($token->{tag_name} eq 'a') {        } elsif ($token->{tag_name} eq 'a') {
7261          AFE: for my $i (reverse 0..$#$active_formatting_elements) {          AFE: for my $i (reverse 0..$#$active_formatting_elements) {
7262            my $node = $active_formatting_elements->[$i];            my $node = $active_formatting_elements->[$i];
7263            if ($node->[1] & A_EL) {            if ($node->[1] == A_EL) {
7264              !!!cp ('t371');              !!!cp ('t371');
7265              !!!parse-error (type => 'in a:a', token => $token);              !!!parse-error (type => 'in a:a', token => $token);
7266                            
# Line 6740  sub _tree_construction_main ($) { Line 7304  sub _tree_construction_main ($) {
7304          ## has a |nobr| element in scope          ## has a |nobr| element in scope
7305          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7306            my $node = $self->{open_elements}->[$_];            my $node = $self->{open_elements}->[$_];
7307            if ($node->[1] & NOBR_EL) {            if ($node->[1] == NOBR_EL) {
7308              !!!cp ('t376');              !!!cp ('t376');
7309              !!!parse-error (type => 'in nobr:nobr', token => $token);              !!!parse-error (type => 'in nobr:nobr', token => $token);
7310              !!!back-token; # <nobr>              !!!back-token; # <nobr>
# Line 6763  sub _tree_construction_main ($) { Line 7327  sub _tree_construction_main ($) {
7327          ## has a button element in scope          ## has a button element in scope
7328          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7329            my $node = $self->{open_elements}->[$_];            my $node = $self->{open_elements}->[$_];
7330            if ($node->[1] & BUTTON_EL) {            if ($node->[1] == BUTTON_EL) {
7331              !!!cp ('t378');              !!!cp ('t378');
7332              !!!parse-error (type => 'in button:button', token => $token);              !!!parse-error (type => 'in button:button', token => $token);
7333              !!!back-token; # <button>              !!!back-token; # <button>
# Line 6863  sub _tree_construction_main ($) { Line 7427  sub _tree_construction_main ($) {
7427            next B;            next B;
7428          }          }
7429        } elsif ($token->{tag_name} eq 'textarea') {        } elsif ($token->{tag_name} eq 'textarea') {
7430          my $tag_name = $token->{tag_name};          ## Step 1
7431          my $el;          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
         !!!create-element ($el, $HTML_NS, $token->{tag_name}, $token->{attributes}, $token);  
7432                    
7433            ## Step 2
7434          ## TODO: $self->{form_element} if defined          ## TODO: $self->{form_element} if defined
7435    
7436            ## Step 3
7437            $self->{ignore_newline} = 1;
7438    
7439            ## Step 4
7440            ## ISSUE: This step is wrong. (r2302 enbugged)
7441    
7442            ## Step 5
7443          $self->{content_model} = RCDATA_CONTENT_MODEL;          $self->{content_model} = RCDATA_CONTENT_MODEL;
7444          delete $self->{escape}; # MUST          delete $self->{escape}; # MUST
7445            
7446          $insert->($el);          ## Step 6-7
7447                    $self->{insertion_mode} |= IN_CDATA_RCDATA_IM;
7448          my $text = '';  
7449          !!!nack ('t392.1');          !!!nack ('t392.1');
7450          !!!next-token;          !!!next-token;
7451          if ($token->{type} == CHARACTER_TOKEN) {          next B;
7452            $token->{data} =~ s/^\x0A//;        } elsif ($token->{tag_name} eq 'optgroup' or
7453            unless (length $token->{data}) {                 $token->{tag_name} eq 'option') {
7454              !!!cp ('t392');          ## has an |option| element in scope
7455              !!!next-token;          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7456            } else {            my $node = $self->{open_elements}->[$_];
7457              !!!cp ('t393');            if ($node->[1] == OPTION_EL) {
7458                !!!cp ('t397.1');
7459                ## NOTE: As if </option>
7460                !!!back-token; # <option> or <optgroup>
7461                $token = {type => END_TAG_TOKEN, tag_name => 'option',
7462                          line => $token->{line}, column => $token->{column}};
7463                next B;
7464              } elsif ($node->[1] & SCOPING_EL) {
7465                !!!cp ('t397.2');
7466                last INSCOPE;
7467            }            }
7468          } else {          } # INSCOPE
7469            !!!cp ('t394');  
7470          }          $reconstruct_active_formatting_elements->($insert_to_current);
7471          while ($token->{type} == CHARACTER_TOKEN) {  
7472            !!!cp ('t395');          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7473            $text .= $token->{data};  
7474            !!!next-token;          !!!nack ('t397.3');
         }  
         if (length $text) {  
           !!!cp ('t396');  
           $el->manakai_append_text ($text);  
         }  
           
         $self->{content_model} = PCDATA_CONTENT_MODEL;  
           
         if ($token->{type} == END_TAG_TOKEN and  
             $token->{tag_name} eq $tag_name) {  
           !!!cp ('t397');  
           ## Ignore the token  
         } else {  
           !!!cp ('t398');  
           !!!parse-error (type => 'in RCDATA:#eof', token => $token);  
         }  
7475          !!!next-token;          !!!next-token;
7476          next B;          redo B;
7477        } elsif ($token->{tag_name} eq 'rt' or        } elsif ($token->{tag_name} eq 'rt' or
7478                 $token->{tag_name} eq 'rp') {                 $token->{tag_name} eq 'rp') {
7479          ## has a |ruby| element in scope          ## has a |ruby| element in scope
7480          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7481            my $node = $self->{open_elements}->[$_];            my $node = $self->{open_elements}->[$_];
7482            if ($node->[1] & RUBY_EL) {            if ($node->[1] == RUBY_EL) {
7483              !!!cp ('t398.1');              !!!cp ('t398.1');
7484              ## generate implied end tags              ## generate implied end tags
7485              while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {              while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
7486                !!!cp ('t398.2');                !!!cp ('t398.2');
7487                pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
7488              }              }
7489              unless ($self->{open_elements}->[-1]->[1] & RUBY_EL) {              unless ($self->{open_elements}->[-1]->[1] == RUBY_EL) {
7490                !!!cp ('t398.3');                !!!cp ('t398.3');
7491                !!!parse-error (type => 'not closed',                !!!parse-error (type => 'not closed',
7492                                text => $self->{open_elements}->[-1]->[0]                                text => $self->{open_elements}->[-1]->[0]
7493                                    ->manakai_local_name,                                    ->manakai_local_name,
7494                                token => $token);                                token => $token);
7495                pop @{$self->{open_elements}}                pop @{$self->{open_elements}}
7496                    while not $self->{open_elements}->[-1]->[1] & RUBY_EL;                    while not $self->{open_elements}->[-1]->[1] == RUBY_EL;
7497              }              }
7498              last INSCOPE;              last INSCOPE;
7499            } elsif ($node->[1] & SCOPING_EL) {            } elsif ($node->[1] & SCOPING_EL) {
# Line 6946  sub _tree_construction_main ($) { Line 7511  sub _tree_construction_main ($) {
7511                 $token->{tag_name} eq 'svg') {                 $token->{tag_name} eq 'svg') {
7512          $reconstruct_active_formatting_elements->($insert_to_current);          $reconstruct_active_formatting_elements->($insert_to_current);
7513    
7514            ## "Adjust MathML attributes" ('math' only) - done in insert-element-f
7515    
7516          ## "adjust SVG attributes" ('svg' only) - done in insert-element-f          ## "adjust SVG attributes" ('svg' only) - done in insert-element-f
7517    
7518          ## "adjust foreign attributes" - done in insert-element-f          ## "adjust foreign attributes" - done in insert-element-f
# Line 6954  sub _tree_construction_main ($) { Line 7521  sub _tree_construction_main ($) {
7521                    
7522          if ($self->{self_closing}) {          if ($self->{self_closing}) {
7523            pop @{$self->{open_elements}};            pop @{$self->{open_elements}};
7524            !!!ack ('t398.1');            !!!ack ('t398.6');
7525          } else {          } else {
7526            !!!cp ('t398.2');            !!!cp ('t398.7');
7527            $self->{insertion_mode} |= IN_FOREIGN_CONTENT_IM;            $self->{insertion_mode} |= IN_FOREIGN_CONTENT_IM;
7528            ## NOTE: |<body><math><mi><svg>| -> "in foreign content" insertion            ## NOTE: |<body><math><mi><svg>| -> "in foreign content" insertion
7529            ## mode, "in body" (not "in foreign content") secondary insertion            ## mode, "in body" (not "in foreign content") secondary insertion
# Line 6967  sub _tree_construction_main ($) { Line 7534  sub _tree_construction_main ($) {
7534          next B;          next B;
7535        } elsif ({        } elsif ({
7536                  caption => 1, col => 1, colgroup => 1, frame => 1,                  caption => 1, col => 1, colgroup => 1, frame => 1,
7537                  frameset => 1, head => 1, option => 1, optgroup => 1,                  frameset => 1, head => 1,
7538                  tbody => 1, td => 1, tfoot => 1, th => 1,                  tbody => 1, td => 1, tfoot => 1, th => 1,
7539                  thead => 1, tr => 1,                  thead => 1, tr => 1,
7540                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
# Line 6978  sub _tree_construction_main ($) { Line 7545  sub _tree_construction_main ($) {
7545          !!!nack ('t401.1'); ## NOTE: |<col/>| or |<frame/>| here is an error.          !!!nack ('t401.1'); ## NOTE: |<col/>| or |<frame/>| here is an error.
7546          !!!next-token;          !!!next-token;
7547          next B;          next B;
7548                  } elsif ($token->{tag_name} eq 'param' or
7549          ## ISSUE: An issue on HTML5 new elements in the spec.                 $token->{tag_name} eq 'source') {
7550            !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7551            pop @{$self->{open_elements}};
7552    
7553            !!!ack ('t398.5');
7554            !!!next-token;
7555            redo B;
7556        } else {        } else {
7557          if ($token->{tag_name} eq 'image') {          if ($token->{tag_name} eq 'image') {
7558            !!!cp ('t384');            !!!cp ('t384');
# Line 7002  sub _tree_construction_main ($) { Line 7575  sub _tree_construction_main ($) {
7575            !!!nack ('t380.1');            !!!nack ('t380.1');
7576          } elsif ({          } elsif ({
7577                    b => 1, big => 1, em => 1, font => 1, i => 1,                    b => 1, big => 1, em => 1, font => 1, i => 1,
7578                    s => 1, small => 1, strile => 1,                    s => 1, small => 1, strike => 1,
7579                    strong => 1, tt => 1, u => 1,                    strong => 1, tt => 1, u => 1,
7580                   }->{$token->{tag_name}}) {                   }->{$token->{tag_name}}) {
7581            !!!cp ('t375');            !!!cp ('t375');
# Line 7015  sub _tree_construction_main ($) { Line 7588  sub _tree_construction_main ($) {
7588            !!!ack ('t388.2');            !!!ack ('t388.2');
7589          } elsif ({          } elsif ({
7590                    area => 1, basefont => 1, bgsound => 1, br => 1,                    area => 1, basefont => 1, bgsound => 1, br => 1,
7591                    embed => 1, img => 1, param => 1, spacer => 1, wbr => 1,                    embed => 1, img => 1, spacer => 1, wbr => 1,
                   #image => 1,  
7592                   }->{$token->{tag_name}}) {                   }->{$token->{tag_name}}) {
7593            !!!cp ('t388.1');            !!!cp ('t388.1');
7594            pop @{$self->{open_elements}};            pop @{$self->{open_elements}};
# Line 7047  sub _tree_construction_main ($) { Line 7619  sub _tree_construction_main ($) {
7619          my $i;          my $i;
7620          INSCOPE: {          INSCOPE: {
7621            for (reverse @{$self->{open_elements}}) {            for (reverse @{$self->{open_elements}}) {
7622              if ($_->[1] & BODY_EL) {              if ($_->[1] == BODY_EL) {
7623                !!!cp ('t405');                !!!cp ('t405');
7624                $i = $_;                $i = $_;
7625                last INSCOPE;                last INSCOPE;
# Line 7057  sub _tree_construction_main ($) { Line 7629  sub _tree_construction_main ($) {
7629              }              }
7630            }            }
7631    
7632            !!!parse-error (type => 'start tag not allowed',            ## NOTE: |<marquee></body>|, |<svg><foreignobject></body>|
7633    
7634              !!!parse-error (type => 'unmatched end tag',
7635                            text => $token->{tag_name}, token => $token);                            text => $token->{tag_name}, token => $token);
7636            ## NOTE: Ignore the token.            ## NOTE: Ignore the token.
7637            !!!next-token;            !!!next-token;
# Line 7083  sub _tree_construction_main ($) { Line 7657  sub _tree_construction_main ($) {
7657          ## TODO: Update this code.  It seems that the code below is not          ## TODO: Update this code.  It seems that the code below is not
7658          ## up-to-date, though it has same effect as speced.          ## up-to-date, though it has same effect as speced.
7659          if (@{$self->{open_elements}} > 1 and          if (@{$self->{open_elements}} > 1 and
7660              $self->{open_elements}->[1]->[1] & BODY_EL) {              $self->{open_elements}->[1]->[1] == BODY_EL) {
7661            ## ISSUE: There is an issue in the spec.            unless ($self->{open_elements}->[-1]->[1] == BODY_EL) {
           unless ($self->{open_elements}->[-1]->[1] & BODY_EL) {  
7662              !!!cp ('t406');              !!!cp ('t406');
7663              !!!parse-error (type => 'not closed',              !!!parse-error (type => 'not closed',
7664                              text => $self->{open_elements}->[1]->[0]                              text => $self->{open_elements}->[1]->[0]
# Line 7106  sub _tree_construction_main ($) { Line 7679  sub _tree_construction_main ($) {
7679            next B;            next B;
7680          }          }
7681        } elsif ({        } elsif ({
7682                  address => 1, blockquote => 1, center => 1, dir => 1,                  ## NOTE: End tags for non-phrasing flow content elements
7683                  div => 1, dl => 1, fieldset => 1, listing => 1,  
7684                  menu => 1, ol => 1, pre => 1, ul => 1,                  ## NOTE: The normal ones
7685                    address => 1, article => 1, aside => 1, blockquote => 1,
7686                    center => 1, datagrid => 1, details => 1, dialog => 1,
7687                    dir => 1, div => 1, dl => 1, fieldset => 1, figure => 1,
7688                    footer => 1, header => 1, listing => 1, menu => 1, nav => 1,
7689                    ol => 1, pre => 1, section => 1, ul => 1,
7690    
7691                    ## NOTE: As normal, but ... optional tags
7692                  dd => 1, dt => 1, li => 1,                  dd => 1, dt => 1, li => 1,
7693    
7694                  applet => 1, button => 1, marquee => 1, object => 1,                  applet => 1, button => 1, marquee => 1, object => 1,
7695                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
7696            ## NOTE: Code for <li> start tags includes "as if </li>" code.
7697            ## Code for <dt> or <dd> start tags includes "as if </dt> or
7698            ## </dd>" code.
7699    
7700          ## has an element in scope          ## has an element in scope
7701          my $i;          my $i;
7702          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
# Line 7130  sub _tree_construction_main ($) { Line 7715  sub _tree_construction_main ($) {
7715            !!!cp ('t413');            !!!cp ('t413');
7716            !!!parse-error (type => 'unmatched end tag',            !!!parse-error (type => 'unmatched end tag',
7717                            text => $token->{tag_name}, token => $token);                            text => $token->{tag_name}, token => $token);
7718              ## NOTE: Ignore the token.
7719          } else {          } else {
7720            ## Step 1. generate implied end tags            ## Step 1. generate implied end tags
7721            while ({            while ({
# Line 7137  sub _tree_construction_main ($) { Line 7723  sub _tree_construction_main ($) {
7723                    dd => ($token->{tag_name} ne 'dd'),                    dd => ($token->{tag_name} ne 'dd'),
7724                    dt => ($token->{tag_name} ne 'dt'),                    dt => ($token->{tag_name} ne 'dt'),
7725                    li => ($token->{tag_name} ne 'li'),                    li => ($token->{tag_name} ne 'li'),
7726                      option => 1,
7727                      optgroup => 1,
7728                    p => 1,                    p => 1,
7729                    rt => 1,                    rt => 1,
7730                    rp => 1,                    rp => 1,
# Line 7169  sub _tree_construction_main ($) { Line 7757  sub _tree_construction_main ($) {
7757          !!!next-token;          !!!next-token;
7758          next B;          next B;
7759        } elsif ($token->{tag_name} eq 'form') {        } elsif ($token->{tag_name} eq 'form') {
7760            ## NOTE: As normal, but interacts with the form element pointer
7761    
7762          undef $self->{form_element};          undef $self->{form_element};
7763    
7764          ## has an element in scope          ## has an element in scope
7765          my $i;          my $i;
7766          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7767            my $node = $self->{open_elements}->[$_];            my $node = $self->{open_elements}->[$_];
7768            if ($node->[1] & FORM_EL) {            if ($node->[1] == FORM_EL) {
7769              !!!cp ('t418');              !!!cp ('t418');
7770              $i = $_;              $i = $_;
7771              last INSCOPE;              last INSCOPE;
# Line 7189  sub _tree_construction_main ($) { Line 7779  sub _tree_construction_main ($) {
7779            !!!cp ('t421');            !!!cp ('t421');
7780            !!!parse-error (type => 'unmatched end tag',            !!!parse-error (type => 'unmatched end tag',
7781                            text => $token->{tag_name}, token => $token);                            text => $token->{tag_name}, token => $token);
7782              ## NOTE: Ignore the token.
7783          } else {          } else {
7784            ## Step 1. generate implied end tags            ## Step 1. generate implied end tags
7785            while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {            while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
# Line 7215  sub _tree_construction_main ($) { Line 7806  sub _tree_construction_main ($) {
7806          !!!next-token;          !!!next-token;
7807          next B;          next B;
7808        } elsif ({        } elsif ({
7809                    ## NOTE: As normal, except acts as a closer for any ...
7810                  h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,                  h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,
7811                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
7812          ## has an element in scope          ## has an element in scope
7813          my $i;          my $i;
7814          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7815            my $node = $self->{open_elements}->[$_];            my $node = $self->{open_elements}->[$_];
7816            if ($node->[1] & HEADING_EL) {            if ($node->[1] == HEADING_EL) {
7817              !!!cp ('t423');              !!!cp ('t423');
7818              $i = $_;              $i = $_;
7819              last INSCOPE;              last INSCOPE;
# Line 7235  sub _tree_construction_main ($) { Line 7827  sub _tree_construction_main ($) {
7827            !!!cp ('t425.1');            !!!cp ('t425.1');
7828            !!!parse-error (type => 'unmatched end tag',            !!!parse-error (type => 'unmatched end tag',
7829                            text => $token->{tag_name}, token => $token);                            text => $token->{tag_name}, token => $token);
7830              ## NOTE: Ignore the token.
7831          } else {          } else {
7832            ## Step 1. generate implied end tags            ## Step 1. generate implied end tags
7833            while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {            while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
# Line 7259  sub _tree_construction_main ($) { Line 7852  sub _tree_construction_main ($) {
7852          !!!next-token;          !!!next-token;
7853          next B;          next B;
7854        } elsif ($token->{tag_name} eq 'p') {        } elsif ($token->{tag_name} eq 'p') {
7855            ## NOTE: As normal, except </p> implies <p> and ...
7856    
7857          ## has an element in scope          ## has an element in scope
7858            my $non_optional;
7859          my $i;          my $i;
7860          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7861            my $node = $self->{open_elements}->[$_];            my $node = $self->{open_elements}->[$_];
7862            if ($node->[1] & P_EL) {            if ($node->[1] == P_EL) {
7863              !!!cp ('t410.1');              !!!cp ('t410.1');
7864              $i = $_;              $i = $_;
7865              last INSCOPE;              last INSCOPE;
7866            } elsif ($node->[1] & SCOPING_EL) {            } elsif ($node->[1] & SCOPING_EL) {
7867              !!!cp ('t411.1');              !!!cp ('t411.1');
7868              last INSCOPE;              last INSCOPE;
7869              } elsif ($node->[1] & END_TAG_OPTIONAL_EL) {
7870                ## NOTE: |END_TAG_OPTIONAL_EL| includes "p"
7871                !!!cp ('t411.2');
7872                #
7873              } else {
7874                !!!cp ('t411.3');
7875                $non_optional ||= $node;
7876                #
7877            }            }
7878          } # INSCOPE          } # INSCOPE
7879    
7880          if (defined $i) {          if (defined $i) {
7881            if ($self->{open_elements}->[-1]->[0]->manakai_local_name            ## 1. Generate implied end tags
7882                    ne $token->{tag_name}) {            #
7883    
7884              ## 2. If current node != "p", parse error
7885              if ($non_optional) {
7886              !!!cp ('t412.1');              !!!cp ('t412.1');
7887              !!!parse-error (type => 'not closed',              !!!parse-error (type => 'not closed',
7888                              text => $self->{open_elements}->[-1]->[0]                              text => $non_optional->[0]->manakai_local_name,
                                 ->manakai_local_name,  
7889                              token => $token);                              token => $token);
7890            } else {            } else {
7891              !!!cp ('t414.1');              !!!cp ('t414.1');
7892            }            }
7893    
7894              ## 3. Pop
7895            splice @{$self->{open_elements}}, $i;            splice @{$self->{open_elements}}, $i;
7896          } else {          } else {
7897            !!!cp ('t413.1');            !!!cp ('t413.1');
# Line 7304  sub _tree_construction_main ($) { Line 7911  sub _tree_construction_main ($) {
7911        } elsif ({        } elsif ({
7912                  a => 1,                  a => 1,
7913                  b => 1, big => 1, em => 1, font => 1, i => 1,                  b => 1, big => 1, em => 1, font => 1, i => 1,
7914                  nobr => 1, s => 1, small => 1, strile => 1,                  nobr => 1, s => 1, small => 1, strike => 1,
7915                  strong => 1, tt => 1, u => 1,                  strong => 1, tt => 1, u => 1,
7916                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
7917          !!!cp ('t427');          !!!cp ('t427');
# Line 7325  sub _tree_construction_main ($) { Line 7932  sub _tree_construction_main ($) {
7932          ## Ignore the token.          ## Ignore the token.
7933          !!!next-token;          !!!next-token;
7934          next B;          next B;
       } elsif ({  
                 caption => 1, col => 1, colgroup => 1, frame => 1,  
                 frameset => 1, head => 1, option => 1, optgroup => 1,  
                 tbody => 1, td => 1, tfoot => 1, th => 1,  
                 thead => 1, tr => 1,  
                 area => 1, basefont => 1, bgsound => 1,  
                 embed => 1, hr => 1, iframe => 1, image => 1,  
                 img => 1, input => 1, isindex => 1, noembed => 1,  
                 noframes => 1, param => 1, select => 1, spacer => 1,  
                 table => 1, textarea => 1, wbr => 1,  
                 noscript => 0, ## TODO: if scripting is enabled  
                }->{$token->{tag_name}}) {  
         !!!cp ('t429');  
         !!!parse-error (type => 'unmatched end tag',  
                         text => $token->{tag_name}, token => $token);  
         ## Ignore the token  
         !!!next-token;  
         next B;  
           
         ## ISSUE: Issue on HTML5 new elements in spec  
           
7935        } else {        } else {
7936            if ($token->{tag_name} eq 'sarcasm') {
7937              sleep 0.001; # take a deep breath
7938            }
7939    
7940          ## Step 1          ## Step 1
7941          my $node_i = -1;          my $node_i = -1;
7942          my $node = $self->{open_elements}->[$node_i];          my $node = $self->{open_elements}->[$node_i];
7943    
7944          ## Step 2          ## Step 2
7945          S2: {          S2: {
7946            if ($node->[0]->manakai_local_name eq $token->{tag_name}) {            my $node_tag_name = $node->[0]->manakai_local_name;
7947              $node_tag_name =~ tr/A-Z/a-z/; # for SVG camelCase tag names
7948              if ($node_tag_name eq $token->{tag_name}) {
7949              ## Step 1              ## Step 1
7950              ## generate implied end tags              ## generate implied end tags
7951              while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {              while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
# Line 7366  sub _tree_construction_main ($) { Line 7958  sub _tree_construction_main ($) {
7958              }              }
7959                    
7960              ## Step 2              ## Step 2
7961              if ($self->{open_elements}->[-1]->[0]->manakai_local_name              my $current_tag_name
7962                      ne $token->{tag_name}) {                  = $self->{open_elements}->[-1]->[0]->manakai_local_name;
7963                $current_tag_name =~ tr/A-Z/a-z/;
7964                if ($current_tag_name ne $token->{tag_name}) {
7965                !!!cp ('t431');                !!!cp ('t431');
7966                ## NOTE: <x><y></x>                ## NOTE: <x><y></x>
7967                !!!parse-error (type => 'not closed',                !!!parse-error (type => 'not closed',
# Line 7395  sub _tree_construction_main ($) { Line 7989  sub _tree_construction_main ($) {
7989                ## Ignore the token                ## Ignore the token
7990                !!!next-token;                !!!next-token;
7991                last S2;                last S2;
             }  
7992    
7993                  ## NOTE: |<span><dd></span>a|: In Safari 3.1.2 and Opera
7994                  ## 9.27, "a" is a child of <dd> (conforming).  In
7995                  ## Firefox 3.0.2, "a" is a child of <body>.  In WinIE 7,
7996                  ## "a" is a child of both <body> and <dd>.
7997                }
7998                
7999              !!!cp ('t434');              !!!cp ('t434');
8000            }            }
8001                        
# Line 7437  sub _tree_construction_main ($) { Line 8036  sub _tree_construction_main ($) {
8036    ## TODO: script stuffs    ## TODO: script stuffs
8037  } # _tree_construct_main  } # _tree_construct_main
8038    
8039  sub set_inner_html ($$$) {  sub set_inner_html ($$$$;$) {
8040    my $class = shift;    my $class = shift;
8041    my $node = shift;    my $node = shift;
8042    my $s = \$_[0];    #my $s = \$_[0];
8043    my $onerror = $_[1];    my $onerror = $_[1];
8044      my $get_wrapper = $_[2] || sub ($) { return $_[0] };
8045    
8046    ## ISSUE: Should {confident} be true?    ## ISSUE: Should {confident} be true?
8047    
# Line 7460  sub set_inner_html ($$$) { Line 8060  sub set_inner_html ($$$) {
8060      }      }
8061    
8062      ## Step 3, 4, 5 # MUST      ## Step 3, 4, 5 # MUST
8063      $class->parse_string ($$s => $node, $onerror);      $class->parse_char_string ($_[0] => $node, $onerror, $get_wrapper);
8064    } elsif ($nt == 1) {    } elsif ($nt == 1) {
8065      ## TODO: If non-html element      ## TODO: If non-html element
8066    
8067      ## NOTE: Most of this code is copied from |parse_string|      ## NOTE: Most of this code is copied from |parse_string|
8068    
8069    ## TODO: Support for $get_wrapper
8070    
8071      ## Step 1 # MUST      ## Step 1 # MUST
8072      my $this_doc = $node->owner_document;      my $this_doc = $node->owner_document;
8073      my $doc = $this_doc->implementation->create_document;      my $doc = $this_doc->implementation->create_document;
# Line 7477  sub set_inner_html ($$$) { Line 8079  sub set_inner_html ($$$) {
8079      my $i = 0;      my $i = 0;
8080      $p->{line_prev} = $p->{line} = 1;      $p->{line_prev} = $p->{line} = 1;
8081      $p->{column_prev} = $p->{column} = 0;      $p->{column_prev} = $p->{column} = 0;
8082      $p->{set_next_char} = sub {      require Whatpm::Charset::DecodeHandle;
8083        my $input = Whatpm::Charset::DecodeHandle::CharString->new (\($_[0]));
8084        $input = $get_wrapper->($input);
8085        $p->{set_nc} = sub {
8086        my $self = shift;        my $self = shift;
8087    
8088        pop @{$self->{prev_char}};        my $char = '';
8089        unshift @{$self->{prev_char}}, $self->{next_char};        if (defined $self->{next_nc}) {
8090            $char = $self->{next_nc};
8091        $self->{next_char} = -1 and return if $i >= length $$s;          delete $self->{next_nc};
8092        $self->{next_char} = ord substr $$s, $i++, 1;          $self->{nc} = ord $char;
8093          } else {
8094            $self->{char_buffer} = '';
8095            $self->{char_buffer_pos} = 0;
8096            
8097            my $count = $input->manakai_read_until
8098                ($self->{char_buffer}, qr/[^\x00\x0A\x0D]/,
8099                 $self->{char_buffer_pos});
8100            if ($count) {
8101              $self->{line_prev} = $self->{line};
8102              $self->{column_prev} = $self->{column};
8103              $self->{column}++;
8104              $self->{nc}
8105                  = ord substr ($self->{char_buffer},
8106                                $self->{char_buffer_pos}++, 1);
8107              return;
8108            }
8109            
8110            if ($input->read ($char, 1)) {
8111              $self->{nc} = ord $char;
8112            } else {
8113              $self->{nc} = -1;
8114              return;
8115            }
8116          }
8117    
8118        ($p->{line_prev}, $p->{column_prev}) = ($p->{line}, $p->{column});        ($p->{line_prev}, $p->{column_prev}) = ($p->{line}, $p->{column});
8119        $p->{column}++;        $p->{column}++;
8120    
8121        if ($self->{next_char} == 0x000A) { # LF        if ($self->{nc} == 0x000A) { # LF
8122          $p->{line}++;          $p->{line}++;
8123          $p->{column} = 0;          $p->{column} = 0;
8124          !!!cp ('i1');          !!!cp ('i1');
8125        } elsif ($self->{next_char} == 0x000D) { # CR        } elsif ($self->{nc} == 0x000D) { # CR
8126          $i++ if substr ($$s, $i, 1) eq "\x0A";  ## TODO: support for abort/streaming
8127          $self->{next_char} = 0x000A; # LF # MUST          my $next = '';
8128            if ($input->read ($next, 1) and $next ne "\x0A") {
8129              $self->{next_nc} = $next;
8130            }
8131            $self->{nc} = 0x000A; # LF # MUST
8132          $p->{line}++;          $p->{line}++;
8133          $p->{column} = 0;          $p->{column} = 0;
8134          !!!cp ('i2');          !!!cp ('i2');
8135        } elsif ($self->{next_char} > 0x10FFFF) {        } elsif ($self->{nc} == 0x0000) { # NULL
         $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST  
         !!!cp ('i3');  
       } elsif ($self->{next_char} == 0x0000) { # NULL  
8136          !!!cp ('i4');          !!!cp ('i4');
8137          !!!parse-error (type => 'NULL');          !!!parse-error (type => 'NULL');
8138          $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST          $self->{nc} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
       } elsif ($self->{next_char} <= 0x0008 or  
                (0x000E <= $self->{next_char} and  
                 $self->{next_char} <= 0x001F) or  
                (0x007F <= $self->{next_char} and  
                 $self->{next_char} <= 0x009F) or  
                (0xD800 <= $self->{next_char} and  
                 $self->{next_char} <= 0xDFFF) or  
                (0xFDD0 <= $self->{next_char} and  
                 $self->{next_char} <= 0xFDDF) or  
                {  
                 0xFFFE => 1, 0xFFFF => 1, 0x1FFFE => 1, 0x1FFFF => 1,  
                 0x2FFFE => 1, 0x2FFFF => 1, 0x3FFFE => 1, 0x3FFFF => 1,  
                 0x4FFFE => 1, 0x4FFFF => 1, 0x5FFFE => 1, 0x5FFFF => 1,  
                 0x6FFFE => 1, 0x6FFFF => 1, 0x7FFFE => 1, 0x7FFFF => 1,  
                 0x8FFFE => 1, 0x8FFFF => 1, 0x9FFFE => 1, 0x9FFFF => 1,  
                 0xAFFFE => 1, 0xAFFFF => 1, 0xBFFFE => 1, 0xBFFFF => 1,  
                 0xCFFFE => 1, 0xCFFFF => 1, 0xDFFFE => 1, 0xDFFFF => 1,  
                 0xEFFFE => 1, 0xEFFFF => 1, 0xFFFFE => 1, 0xFFFFF => 1,  
                 0x10FFFE => 1, 0x10FFFF => 1,  
                }->{$self->{next_char}}) {  
         !!!cp ('i4.1');  
         if ($self->{next_char} < 0x10000) {  
           !!!parse-error (type => 'control char',  
                           text => (sprintf 'U+%04X', $self->{next_char}));  
         } else {  
           !!!parse-error (type => 'control char',  
                           text => (sprintf 'U-%08X', $self->{next_char}));  
         }  
8139        }        }
8140      };      };
8141      $p->{prev_char} = [-1, -1, -1];  
8142      $p->{next_char} = -1;      $p->{read_until} = sub {
8143              #my ($scalar, $specials_range, $offset) = @_;
8144          return 0 if defined $p->{next_nc};
8145    
8146          my $pattern = qr/[^$_[1]\x00\x0A\x0D]/;
8147          my $offset = $_[2] || 0;
8148          
8149          if ($p->{char_buffer_pos} < length $p->{char_buffer}) {
8150            pos ($p->{char_buffer}) = $p->{char_buffer_pos};
8151            if ($p->{char_buffer} =~ /\G(?>$pattern)+/) {
8152              substr ($_[0], $offset)
8153                  = substr ($p->{char_buffer}, $-[0], $+[0] - $-[0]);
8154              my $count = $+[0] - $-[0];
8155              if ($count) {
8156                $p->{column} += $count;
8157                $p->{char_buffer_pos} += $count;
8158                $p->{line_prev} = $p->{line};
8159                $p->{column_prev} = $p->{column} - 1;
8160                $p->{nc} = -1;
8161              }
8162              return $count;
8163            } else {
8164              return 0;
8165            }
8166          } else {
8167            my $count = $input->manakai_read_until ($_[0], $pattern, $_[2]);
8168            if ($count) {
8169              $p->{column} += $count;
8170              $p->{column_prev} += $count;
8171              $p->{nc} = -1;
8172            }
8173            return $count;
8174          }
8175        }; # $p->{read_until}
8176    
8177      my $ponerror = $onerror || sub {      my $ponerror = $onerror || sub {
8178        my (%opt) = @_;        my (%opt) = @_;
8179        my $line = $opt{line};        my $line = $opt{line};
# Line 7553  sub set_inner_html ($$$) { Line 8188  sub set_inner_html ($$$) {
8188        $ponerror->(line => $p->{line}, column => $p->{column}, @_);        $ponerror->(line => $p->{line}, column => $p->{column}, @_);
8189      };      };
8190            
8191        my $char_onerror = sub {
8192          my (undef, $type, %opt) = @_;
8193          $ponerror->(layer => 'encode',
8194                      line => $p->{line}, column => $p->{column} + 1,
8195                      %opt, type => $type);
8196        }; # $char_onerror
8197        $input->onerror ($char_onerror);
8198    
8199      $p->_initialize_tokenizer;      $p->_initialize_tokenizer;
8200      $p->_initialize_tree_constructor;      $p->_initialize_tree_constructor;
8201    
# Line 7588  sub set_inner_html ($$$) { Line 8231  sub set_inner_html ($$$) {
8231      push @{$p->{open_elements}}, [$root, $el_category->{html}];      push @{$p->{open_elements}}, [$root, $el_category->{html}];
8232    
8233      undef $p->{head_element};      undef $p->{head_element};
8234        undef $p->{head_element_inserted};
8235    
8236      ## Step 6 # MUST      ## Step 6 # MUST
8237      $p->_reset_insertion_mode;      $p->_reset_insertion_mode;

Legend:
Removed from v.1.153  
changed lines
  Added in v.1.206

[email protected]
ViewVC Help
Powered by ViewVC 1.1.24