/[suikacvs]/markup/html/whatpm/Whatpm/HTML.pm.src
Suika

Diff of /markup/html/whatpm/Whatpm/HTML.pm.src

Parent Directory Parent Directory | Revision Log Revision Log | View Patch Patch

revision 1.167 by wakaba, Sat Sep 13 09:02:28 2008 UTC revision 1.206 by wakaba, Mon Oct 13 08:22:30 2008 UTC
# Line 3  use strict; Line 3  use strict;
3  our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r};  our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r};
4  use Error qw(:try);  use Error qw(:try);
5    
6    ## NOTE: This module don't check all HTML5 parse errors; character
7    ## encoding related parse errors are expected to be handled by relevant
8    ## modules.
9    ## Parse errors for control characters that are not allowed in HTML5
10    ## documents, for surrogate code points, and for noncharacter code
11    ## points, as well as U+FFFD substitions for characters whose code points
12    ## is higher than U+10FFFF may be detected by combining the parser with
13    ## the checker implemented by Whatpm::Charset::UnicodeChecker (for its
14    ## usage example, see |t/HTML-tree.t| in the Whatpm package or the
15    ## WebHACC::Language::HTML module in the WebHACC package).
16    
17  ## ISSUE:  ## ISSUE:
18  ## var doc = implementation.createDocument (null, null, null);  ## var doc = implementation.createDocument (null, null, null);
19  ## doc.write ('');  ## doc.write ('');
# Line 17  my $XLINK_NS = q<http://www.w3.org/1999/ Line 28  my $XLINK_NS = q<http://www.w3.org/1999/
28  my $XML_NS = q<http://www.w3.org/XML/1998/namespace>;  my $XML_NS = q<http://www.w3.org/XML/1998/namespace>;
29  my $XMLNS_NS = q<http://www.w3.org/2000/xmlns/>;  my $XMLNS_NS = q<http://www.w3.org/2000/xmlns/>;
30    
31  sub A_EL () { 0b1 }  ## Bits 12-15
32  sub ADDRESS_EL () { 0b10 }  sub SPECIAL_EL () { 0b1_000000000000000 }
33  sub BODY_EL () { 0b100 }  sub SCOPING_EL () { 0b1_00000000000000 }
34  sub BUTTON_EL () { 0b1000 }  sub FORMATTING_EL () { 0b1_0000000000000 }
35  sub CAPTION_EL () { 0b10000 }  sub PHRASING_EL () { 0b1_000000000000 }
36  sub DD_EL () { 0b100000 }  
37  sub DIV_EL () { 0b1000000 }  ## Bits 10-11
38  sub DT_EL () { 0b10000000 }  sub FOREIGN_EL () { 0b1_00000000000 }
39  sub FORM_EL () { 0b100000000 }  sub FOREIGN_FLOW_CONTENT_EL () { 0b1_0000000000 }
40  sub FORMATTING_EL () { 0b1000000000 }  
41  sub FRAMESET_EL () { 0b10000000000 }  ## Bits 6-9
42  sub HEADING_EL () { 0b100000000000 }  sub TABLE_SCOPING_EL () { 0b1_000000000 }
43  sub HTML_EL () { 0b1000000000000 }  sub TABLE_ROWS_SCOPING_EL () { 0b1_00000000 }
44  sub LI_EL () { 0b10000000000000 }  sub TABLE_ROW_SCOPING_EL () { 0b1_0000000 }
45  sub NOBR_EL () { 0b100000000000000 }  sub TABLE_ROWS_EL () { 0b1_000000 }
 sub OPTION_EL () { 0b1000000000000000 }  
 sub OPTGROUP_EL () { 0b10000000000000000 }  
 sub P_EL () { 0b100000000000000000 }  
 sub SELECT_EL () { 0b1000000000000000000 }  
 sub TABLE_EL () { 0b10000000000000000000 }  
 sub TABLE_CELL_EL () { 0b100000000000000000000 }  
 sub TABLE_ROW_EL () { 0b1000000000000000000000 }  
 sub TABLE_ROW_GROUP_EL () { 0b10000000000000000000000 }  
 sub MISC_SCOPING_EL () { 0b100000000000000000000000 }  
 sub MISC_SPECIAL_EL () { 0b1000000000000000000000000 }  
 sub FOREIGN_EL () { 0b10000000000000000000000000 }  
 sub FOREIGN_FLOW_CONTENT_EL () { 0b100000000000000000000000000 }  
 sub MML_AXML_EL () { 0b1000000000000000000000000000 }  
 sub RUBY_EL () { 0b10000000000000000000000000000 }  
 sub RUBY_COMPONENT_EL () { 0b100000000000000000000000000000 }  
   
 sub TABLE_ROWS_EL () {  
   TABLE_EL |  
   TABLE_ROW_EL |  
   TABLE_ROW_GROUP_EL  
 }  
46    
47  ## NOTE: Used in "generate implied end tags" algorithm.  ## Bit 5
48  ## NOTE: There is a code where a modified version of END_TAG_OPTIONAL_EL  sub ADDRESS_DIV_P_EL () { 0b1_00000 }
 ## is used in "generate implied end tags" implementation (search for the  
 ## function mae).  
 sub END_TAG_OPTIONAL_EL () {  
   DD_EL |  
   DT_EL |  
   LI_EL |  
   P_EL |  
   RUBY_COMPONENT_EL  
 }  
49    
50  ## NOTE: Used in </body> and EOF algorithms.  ## NOTE: Used in </body> and EOF algorithms.
51  sub ALL_END_TAG_OPTIONAL_EL () {  ## Bit 4
52    DD_EL |  sub ALL_END_TAG_OPTIONAL_EL () { 0b1_0000 }
   DT_EL |  
   LI_EL |  
   P_EL |  
   
   BODY_EL |  
   HTML_EL |  
   TABLE_CELL_EL |  
   TABLE_ROW_EL |  
   TABLE_ROW_GROUP_EL  
 }  
53    
54  sub SCOPING_EL () {  ## NOTE: Used in "generate implied end tags" algorithm.
55    BUTTON_EL |  ## NOTE: There is a code where a modified version of
56    CAPTION_EL |  ## END_TAG_OPTIONAL_EL is used in "generate implied end tags"
57    HTML_EL |  ## implementation (search for the algorithm name).
58    TABLE_EL |  ## Bit 3
59    TABLE_CELL_EL |  sub END_TAG_OPTIONAL_EL () { 0b1_000 }
60    MISC_SCOPING_EL  
61    ## Bits 0-2
62    
63    sub MISC_SPECIAL_EL () { SPECIAL_EL | 0b000 }
64    sub FORM_EL () { SPECIAL_EL | 0b001 }
65    sub FRAMESET_EL () { SPECIAL_EL | 0b010 }
66    sub HEADING_EL () { SPECIAL_EL | 0b011 }
67    sub SELECT_EL () { SPECIAL_EL | 0b100 }
68    sub SCRIPT_EL () { SPECIAL_EL | 0b101 }
69    
70    sub ADDRESS_DIV_EL () { SPECIAL_EL | ADDRESS_DIV_P_EL | 0b001 }
71    sub BODY_EL () { SPECIAL_EL | ALL_END_TAG_OPTIONAL_EL | 0b001 }
72    
73    sub DD_EL () {
74      SPECIAL_EL |
75      END_TAG_OPTIONAL_EL |
76      ALL_END_TAG_OPTIONAL_EL |
77      0b001
78  }  }
79    sub DT_EL () {
80  sub TABLE_SCOPING_EL () {    SPECIAL_EL |
81    HTML_EL |    END_TAG_OPTIONAL_EL |
82    TABLE_EL    ALL_END_TAG_OPTIONAL_EL |
83      0b010
84  }  }
85    sub LI_EL () {
86  sub TABLE_ROWS_SCOPING_EL () {    SPECIAL_EL |
87    HTML_EL |    END_TAG_OPTIONAL_EL |
88    TABLE_ROW_GROUP_EL    ALL_END_TAG_OPTIONAL_EL |
89      0b100
90    }
91    sub P_EL () {
92      SPECIAL_EL |
93      ADDRESS_DIV_P_EL |
94      END_TAG_OPTIONAL_EL |
95      ALL_END_TAG_OPTIONAL_EL |
96      0b001
97  }  }
98    
99  sub TABLE_ROW_SCOPING_EL () {  sub TABLE_ROW_EL () {
100    HTML_EL |    SPECIAL_EL |
101    TABLE_ROW_EL    TABLE_ROWS_EL |
102      TABLE_ROW_SCOPING_EL |
103      ALL_END_TAG_OPTIONAL_EL |
104      0b001
105    }
106    sub TABLE_ROW_GROUP_EL () {
107      SPECIAL_EL |
108      TABLE_ROWS_EL |
109      TABLE_ROWS_SCOPING_EL |
110      ALL_END_TAG_OPTIONAL_EL |
111      0b001
112  }  }
113    
114  sub SPECIAL_EL () {  sub MISC_SCOPING_EL () { SCOPING_EL | 0b000 }
115    ADDRESS_EL |  sub BUTTON_EL () { SCOPING_EL | 0b001 }
116    BODY_EL |  sub CAPTION_EL () { SCOPING_EL | 0b010 }
117    DIV_EL |  sub HTML_EL () {
118      SCOPING_EL |
119    DD_EL |    TABLE_SCOPING_EL |
120    DT_EL |    TABLE_ROWS_SCOPING_EL |
121    LI_EL |    TABLE_ROW_SCOPING_EL |
122    P_EL |    ALL_END_TAG_OPTIONAL_EL |
123      0b001
   FORM_EL |  
   FRAMESET_EL |  
   HEADING_EL |  
   OPTION_EL |  
   OPTGROUP_EL |  
   SELECT_EL |  
   TABLE_ROW_EL |  
   TABLE_ROW_GROUP_EL |  
   MISC_SPECIAL_EL  
124  }  }
125    sub TABLE_EL () {
126      SCOPING_EL |
127      TABLE_ROWS_EL |
128      TABLE_SCOPING_EL |
129      0b001
130    }
131    sub TABLE_CELL_EL () {
132      SCOPING_EL |
133      TABLE_ROW_SCOPING_EL |
134      ALL_END_TAG_OPTIONAL_EL |
135      0b001
136    }
137    
138    sub MISC_FORMATTING_EL () { FORMATTING_EL | 0b000 }
139    sub A_EL () { FORMATTING_EL | 0b001 }
140    sub NOBR_EL () { FORMATTING_EL | 0b010 }
141    
142    sub RUBY_EL () { PHRASING_EL | 0b001 }
143    
144    ## ISSUE: ALL_END_TAG_OPTIONAL_EL?
145    sub OPTGROUP_EL () { PHRASING_EL | END_TAG_OPTIONAL_EL | 0b001 }
146    sub OPTION_EL () { PHRASING_EL | END_TAG_OPTIONAL_EL | 0b010 }
147    sub RUBY_COMPONENT_EL () { PHRASING_EL | END_TAG_OPTIONAL_EL | 0b100 }
148    
149    sub MML_AXML_EL () { PHRASING_EL | FOREIGN_EL | 0b001 }
150    
151  my $el_category = {  my $el_category = {
152    a => A_EL | FORMATTING_EL,    a => A_EL,
153    address => ADDRESS_EL,    address => ADDRESS_DIV_EL,
154    applet => MISC_SCOPING_EL,    applet => MISC_SCOPING_EL,
155    area => MISC_SPECIAL_EL,    area => MISC_SPECIAL_EL,
156      article => MISC_SPECIAL_EL,
157      aside => MISC_SPECIAL_EL,
158    b => FORMATTING_EL,    b => FORMATTING_EL,
159    base => MISC_SPECIAL_EL,    base => MISC_SPECIAL_EL,
160    basefont => MISC_SPECIAL_EL,    basefont => MISC_SPECIAL_EL,
# Line 143  my $el_category = { Line 168  my $el_category = {
168    center => MISC_SPECIAL_EL,    center => MISC_SPECIAL_EL,
169    col => MISC_SPECIAL_EL,    col => MISC_SPECIAL_EL,
170    colgroup => MISC_SPECIAL_EL,    colgroup => MISC_SPECIAL_EL,
171      command => MISC_SPECIAL_EL,
172      datagrid => MISC_SPECIAL_EL,
173    dd => DD_EL,    dd => DD_EL,
174      details => MISC_SPECIAL_EL,
175      dialog => MISC_SPECIAL_EL,
176    dir => MISC_SPECIAL_EL,    dir => MISC_SPECIAL_EL,
177    div => DIV_EL,    div => ADDRESS_DIV_EL,
178    dl => MISC_SPECIAL_EL,    dl => MISC_SPECIAL_EL,
179    dt => DT_EL,    dt => DT_EL,
180    em => FORMATTING_EL,    em => FORMATTING_EL,
181    embed => MISC_SPECIAL_EL,    embed => MISC_SPECIAL_EL,
182      eventsource => MISC_SPECIAL_EL,
183    fieldset => MISC_SPECIAL_EL,    fieldset => MISC_SPECIAL_EL,
184      figure => MISC_SPECIAL_EL,
185    font => FORMATTING_EL,    font => FORMATTING_EL,
186      footer => MISC_SPECIAL_EL,
187    form => FORM_EL,    form => FORM_EL,
188    frame => MISC_SPECIAL_EL,    frame => MISC_SPECIAL_EL,
189    frameset => FRAMESET_EL,    frameset => FRAMESET_EL,
# Line 162  my $el_category = { Line 194  my $el_category = {
194    h5 => HEADING_EL,    h5 => HEADING_EL,
195    h6 => HEADING_EL,    h6 => HEADING_EL,
196    head => MISC_SPECIAL_EL,    head => MISC_SPECIAL_EL,
197      header => MISC_SPECIAL_EL,
198    hr => MISC_SPECIAL_EL,    hr => MISC_SPECIAL_EL,
199    html => HTML_EL,    html => HTML_EL,
200    i => FORMATTING_EL,    i => FORMATTING_EL,
201    iframe => MISC_SPECIAL_EL,    iframe => MISC_SPECIAL_EL,
202    img => MISC_SPECIAL_EL,    img => MISC_SPECIAL_EL,
203      #image => MISC_SPECIAL_EL, ## NOTE: Commented out in the spec.
204    input => MISC_SPECIAL_EL,    input => MISC_SPECIAL_EL,
205    isindex => MISC_SPECIAL_EL,    isindex => MISC_SPECIAL_EL,
206    li => LI_EL,    li => LI_EL,
# Line 175  my $el_category = { Line 209  my $el_category = {
209    marquee => MISC_SCOPING_EL,    marquee => MISC_SCOPING_EL,
210    menu => MISC_SPECIAL_EL,    menu => MISC_SPECIAL_EL,
211    meta => MISC_SPECIAL_EL,    meta => MISC_SPECIAL_EL,
212    nobr => NOBR_EL | FORMATTING_EL,    nav => MISC_SPECIAL_EL,
213      nobr => NOBR_EL,
214    noembed => MISC_SPECIAL_EL,    noembed => MISC_SPECIAL_EL,
215    noframes => MISC_SPECIAL_EL,    noframes => MISC_SPECIAL_EL,
216    noscript => MISC_SPECIAL_EL,    noscript => MISC_SPECIAL_EL,
# Line 193  my $el_category = { Line 228  my $el_category = {
228    s => FORMATTING_EL,    s => FORMATTING_EL,
229    script => MISC_SPECIAL_EL,    script => MISC_SPECIAL_EL,
230    select => SELECT_EL,    select => SELECT_EL,
231      section => MISC_SPECIAL_EL,
232    small => FORMATTING_EL,    small => FORMATTING_EL,
233    spacer => MISC_SPECIAL_EL,    spacer => MISC_SPECIAL_EL,
234    strike => FORMATTING_EL,    strike => FORMATTING_EL,
# Line 216  my $el_category = { Line 252  my $el_category = {
252  my $el_category_f = {  my $el_category_f = {
253    $MML_NS => {    $MML_NS => {
254      'annotation-xml' => MML_AXML_EL,      'annotation-xml' => MML_AXML_EL,
255      mi => FOREIGN_FLOW_CONTENT_EL,      mi => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
256      mo => FOREIGN_FLOW_CONTENT_EL,      mo => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
257      mn => FOREIGN_FLOW_CONTENT_EL,      mn => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
258      ms => FOREIGN_FLOW_CONTENT_EL,      ms => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
259      mtext => FOREIGN_FLOW_CONTENT_EL,      mtext => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
260    },    },
261    $SVG_NS => {    $SVG_NS => {
262      foreignObject => FOREIGN_FLOW_CONTENT_EL,      foreignObject => SCOPING_EL | FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
263      desc => FOREIGN_FLOW_CONTENT_EL,      desc => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
264      title => FOREIGN_FLOW_CONTENT_EL,      title => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
265    },    },
266    ## NOTE: In addition, FOREIGN_EL is set to non-HTML elements.    ## NOTE: In addition, FOREIGN_EL is set to non-HTML elements.
267  };  };
# Line 312  my $foreign_attr_xname = { Line 348  my $foreign_attr_xname = {
348    
349  ## ISSUE: xmlns:xlink="non-xlink-ns" is not an error.  ## ISSUE: xmlns:xlink="non-xlink-ns" is not an error.
350    
351  my $c1_entity_char = {  my $charref_map = {
352      0x0D => 0x000A,
353    0x80 => 0x20AC,    0x80 => 0x20AC,
354    0x81 => 0xFFFD,    0x81 => 0xFFFD,
355    0x82 => 0x201A,    0x82 => 0x201A,
# Line 345  my $c1_entity_char = { Line 382  my $c1_entity_char = {
382    0x9D => 0xFFFD,    0x9D => 0xFFFD,
383    0x9E => 0x017E,    0x9E => 0x017E,
384    0x9F => 0x0178,    0x9F => 0x0178,
385  }; # $c1_entity_char  }; # $charref_map
386    $charref_map->{$_} = 0xFFFD
387        for 0x0000..0x0008, 0x000B, 0x000E..0x001F, 0x007F,
388            0xD800..0xDFFF, 0xFDD0..0xFDDF, ## ISSUE: 0xFDEF
389            0xFFFE, 0xFFFF, 0x1FFFE, 0x1FFFF, 0x2FFFE, 0x2FFFF, 0x3FFFE, 0x3FFFF,
390            0x4FFFE, 0x4FFFF, 0x5FFFE, 0x5FFFF, 0x6FFFE, 0x6FFFF, 0x7FFFE,
391            0x7FFFF, 0x8FFFE, 0x8FFFF, 0x9FFFE, 0x9FFFF, 0xAFFFE, 0xAFFFF,
392            0xBFFFE, 0xBFFFF, 0xCFFFE, 0xCFFFF, 0xDFFFE, 0xDFFFF, 0xEFFFE,
393            0xEFFFF, 0xFFFFE, 0xFFFFF, 0x10FFFE, 0x10FFFF;
394    
395    ## TODO: Invoke the reset algorithm when a resettable element is
396    ## created (cf. HTML5 revision 2259).
397    
398  sub parse_byte_string ($$$$;$) {  sub parse_byte_string ($$$$;$) {
399    my $self = shift;    my $self = shift;
# Line 390  sub parse_byte_stream ($$$$;$$) { Line 438  sub parse_byte_stream ($$$$;$$) {
438            ## TODO: Is this ok?  Transfer protocol's parameter should be            ## TODO: Is this ok?  Transfer protocol's parameter should be
439            ## interpreted in its semantics?            ## interpreted in its semantics?
440    
       ## ISSUE: Unsupported encoding is not ignored according to the spec.  
441        ($char_stream, $e_status) = $charset->get_decode_handle        ($char_stream, $e_status) = $charset->get_decode_handle
442            ($byte_stream, allow_error_reporting => 1,            ($byte_stream, allow_error_reporting => 1,
443             allow_fallback => 1);             allow_fallback => 1);
# Line 398  sub parse_byte_stream ($$$$;$$) { Line 445  sub parse_byte_stream ($$$$;$$) {
445          $self->{confident} = 1;          $self->{confident} = 1;
446          last SNIFFING;          last SNIFFING;
447        } else {        } else {
448          ## TODO: unsupported error          !!!parse-error (type => 'charset:not supported',
449                            layer => 'encode',
450                            line => 1, column => 1,
451                            value => $charset_name,
452                            level => $self->{level}->{uncertain});
453        }        }
454      }      }
455    
# Line 447  sub parse_byte_stream ($$$$;$$) { Line 498  sub parse_byte_stream ($$$$;$$) {
498      if (defined $charset_name) {      if (defined $charset_name) {
499        $charset = Message::Charset::Info->get_by_html_name ($charset_name);        $charset = Message::Charset::Info->get_by_html_name ($charset_name);
500    
       ## ISSUE: Unsupported encoding is not ignored according to the spec.  
501        require Whatpm::Charset::DecodeHandle;        require Whatpm::Charset::DecodeHandle;
502        $buffer = Whatpm::Charset::DecodeHandle::ByteBuffer->new        $buffer = Whatpm::Charset::DecodeHandle::ByteBuffer->new
503            ($byte_stream);            ($byte_stream);
# Line 496  sub parse_byte_stream ($$$$;$$) { Line 546  sub parse_byte_stream ($$$$;$$) {
546                      line => 1, column => 1,                      line => 1, column => 1,
547                      layer => 'encode');                      layer => 'encode');
548    } elsif (not ($e_status &    } elsif (not ($e_status &
549                  Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL())) {                  Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL ())) {
550      $self->{input_encoding} = $charset->get_iana_name;      $self->{input_encoding} = $charset->get_iana_name;
551      !!!parse-error (type => 'chardecode:no error',      !!!parse-error (type => 'chardecode:no error',
552                      text => $self->{input_encoding},                      text => $self->{input_encoding},
# Line 561  sub parse_byte_stream ($$$$;$$) { Line 611  sub parse_byte_stream ($$$$;$$) {
611    my $char_onerror = sub {    my $char_onerror = sub {
612      my (undef, $type, %opt) = @_;      my (undef, $type, %opt) = @_;
613      !!!parse-error (layer => 'encode',      !!!parse-error (layer => 'encode',
614                      %opt, type => $type,                      line => $self->{line}, column => $self->{column} + 1,
615                      line => $self->{line}, column => $self->{column} + 1);                      %opt, type => $type);
616      if ($opt{octets}) {      if ($opt{octets}) {
617        ${$opt{octets}} = "\x{FFFD}"; # relacement character        ${$opt{octets}} = "\x{FFFD}"; # relacement character
618      }      }
# Line 571  sub parse_byte_stream ($$$$;$$) { Line 621  sub parse_byte_stream ($$$$;$$) {
621    my $wrapped_char_stream = $get_wrapper->($char_stream);    my $wrapped_char_stream = $get_wrapper->($char_stream);
622    $wrapped_char_stream->onerror ($char_onerror);    $wrapped_char_stream->onerror ($char_onerror);
623    
624    my @args = @_; shift @args; # $s    my @args = ($_[1], $_[2]); # $doc, $onerror - $get_wrapper = undef;
625    my $return;    my $return;
626    try {    try {
627      $return = $self->parse_char_stream ($wrapped_char_stream, @args);        $return = $self->parse_char_stream ($wrapped_char_stream, @args);  
# Line 586  sub parse_byte_stream ($$$$;$$) { Line 636  sub parse_byte_stream ($$$$;$$) {
636                        line => 1, column => 1,                        line => 1, column => 1,
637                        layer => 'encode');                        layer => 'encode');
638      } elsif (not ($e_status &      } elsif (not ($e_status &
639                    Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL())) {                    Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL ())) {
640        $self->{input_encoding} = $charset->get_iana_name;        $self->{input_encoding} = $charset->get_iana_name;
641        !!!parse-error (type => 'chardecode:no error',        !!!parse-error (type => 'chardecode:no error',
642                        text => $self->{input_encoding},                        text => $self->{input_encoding},
# Line 618  sub parse_byte_stream ($$$$;$$) { Line 668  sub parse_byte_stream ($$$$;$$) {
668  sub parse_char_string ($$$;$$) {  sub parse_char_string ($$$;$$) {
669    #my ($self, $s, $doc, $onerror, $get_wrapper) = @_;    #my ($self, $s, $doc, $onerror, $get_wrapper) = @_;
670    my $self = shift;    my $self = shift;
   require utf8;  
671    my $s = ref $_[0] ? $_[0] : \($_[0]);    my $s = ref $_[0] ? $_[0] : \($_[0]);
672    open my $input, '<' . (utf8::is_utf8 ($$s) ? ':utf8' : ''), $s;    require Whatpm::Charset::DecodeHandle;
673    if ($_[3]) {    my $input = Whatpm::Charset::DecodeHandle::CharString->new ($s);
     $input = $_[3]->($input);  
   }  
674    return $self->parse_char_stream ($input, @_[1..$#_]);    return $self->parse_char_stream ($input, @_[1..$#_]);
675  } # parse_char_string  } # parse_char_string
676  *parse_string = \&parse_char_string; ## NOTE: Alias for backward compatibility.  *parse_string = \&parse_char_string; ## NOTE: Alias for backward compatibility.
677    
678  sub parse_char_stream ($$$;$) {  sub parse_char_stream ($$$;$$) {
679    my $self = ref $_[0] ? shift : shift->new;    my $self = ref $_[0] ? shift : shift->new;
680    my $input = $_[0];    my $input = $_[0];
681    $self->{document} = $_[1];    $self->{document} = $_[1];
# Line 639  sub parse_char_stream ($$$;$) { Line 686  sub parse_char_stream ($$$;$) {
686    $self->{confident} = 1 unless exists $self->{confident};    $self->{confident} = 1 unless exists $self->{confident};
687    $self->{document}->input_encoding ($self->{input_encoding})    $self->{document}->input_encoding ($self->{input_encoding})
688        if defined $self->{input_encoding};        if defined $self->{input_encoding};
689    ## TODO: |{input_encoding}| is needless?
690    
   my $i = 0;  
691    $self->{line_prev} = $self->{line} = 1;    $self->{line_prev} = $self->{line} = 1;
692    $self->{column_prev} = $self->{column} = 0;    $self->{column_prev} = -1;
693    $self->{set_next_char} = sub {    $self->{column} = 0;
694      $self->{set_nc} = sub {
695      my $self = shift;      my $self = shift;
696    
697      pop @{$self->{prev_char}};      my $char = '';
698      unshift @{$self->{prev_char}}, $self->{next_char};      if (defined $self->{next_nc}) {
699          $char = $self->{next_nc};
700      my $char;        delete $self->{next_nc};
701      if (defined $self->{next_next_char}) {        $self->{nc} = ord $char;
       $char = $self->{next_next_char};  
       delete $self->{next_next_char};  
702      } else {      } else {
703        $char = $input->getc;        $self->{char_buffer} = '';
704          $self->{char_buffer_pos} = 0;
705    
706          my $count = $input->manakai_read_until
707             ($self->{char_buffer}, qr/[^\x00\x0A\x0D]/, $self->{char_buffer_pos});
708          if ($count) {
709            $self->{line_prev} = $self->{line};
710            $self->{column_prev} = $self->{column};
711            $self->{column}++;
712            $self->{nc}
713                = ord substr ($self->{char_buffer}, $self->{char_buffer_pos}++, 1);
714            return;
715          }
716    
717          if ($input->read ($char, 1)) {
718            $self->{nc} = ord $char;
719          } else {
720            $self->{nc} = -1;
721            return;
722          }
723      }      }
     $self->{next_char} = -1 and return unless defined $char;  
     $self->{next_char} = ord $char;  
724    
725      ($self->{line_prev}, $self->{column_prev})      ($self->{line_prev}, $self->{column_prev})
726          = ($self->{line}, $self->{column});          = ($self->{line}, $self->{column});
727      $self->{column}++;      $self->{column}++;
728            
729      if ($self->{next_char} == 0x000A) { # LF      if ($self->{nc} == 0x000A) { # LF
730        !!!cp ('j1');        !!!cp ('j1');
731        $self->{line}++;        $self->{line}++;
732        $self->{column} = 0;        $self->{column} = 0;
733      } elsif ($self->{next_char} == 0x000D) { # CR      } elsif ($self->{nc} == 0x000D) { # CR
734        !!!cp ('j2');        !!!cp ('j2');
735        my $next = $input->getc;  ## TODO: support for abort/streaming
736        if (defined $next and $next ne "\x0A") {        my $next = '';
737          $self->{next_next_char} = $next;        if ($input->read ($next, 1) and $next ne "\x0A") {
738            $self->{next_nc} = $next;
739        }        }
740        $self->{next_char} = 0x000A; # LF # MUST        $self->{nc} = 0x000A; # LF # MUST
741        $self->{line}++;        $self->{line}++;
742        $self->{column} = 0;        $self->{column} = 0;
743      } elsif ($self->{next_char} > 0x10FFFF) {      } elsif ($self->{nc} == 0x0000) { # NULL
       !!!cp ('j3');  
       $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST  
     } elsif ($self->{next_char} == 0x0000) { # NULL  
744        !!!cp ('j4');        !!!cp ('j4');
745        !!!parse-error (type => 'NULL');        !!!parse-error (type => 'NULL');
746        $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST        $self->{nc} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
747      } elsif ($self->{next_char} <= 0x0008 or      }
748               (0x000E <= $self->{next_char} and $self->{next_char} <= 0x001F) or    };
749               (0x007F <= $self->{next_char} and $self->{next_char} <= 0x009F) or  
750               (0xD800 <= $self->{next_char} and $self->{next_char} <= 0xDFFF) or    $self->{read_until} = sub {
751               (0xFDD0 <= $self->{next_char} and $self->{next_char} <= 0xFDDF) or      #my ($scalar, $specials_range, $offset) = @_;
752               {      return 0 if defined $self->{next_nc};
753                0xFFFE => 1, 0xFFFF => 1, 0x1FFFE => 1, 0x1FFFF => 1,  
754                0x2FFFE => 1, 0x2FFFF => 1, 0x3FFFE => 1, 0x3FFFF => 1,      my $pattern = qr/[^$_[1]\x00\x0A\x0D]/;
755                0x4FFFE => 1, 0x4FFFF => 1, 0x5FFFE => 1, 0x5FFFF => 1,      my $offset = $_[2] || 0;
756                0x6FFFE => 1, 0x6FFFF => 1, 0x7FFFE => 1, 0x7FFFF => 1,  
757                0x8FFFE => 1, 0x8FFFF => 1, 0x9FFFE => 1, 0x9FFFF => 1,      if ($self->{char_buffer_pos} < length $self->{char_buffer}) {
758                0xAFFFE => 1, 0xAFFFF => 1, 0xBFFFE => 1, 0xBFFFF => 1,        pos ($self->{char_buffer}) = $self->{char_buffer_pos};
759                0xCFFFE => 1, 0xCFFFF => 1, 0xDFFFE => 1, 0xDFFFF => 1,        if ($self->{char_buffer} =~ /\G(?>$pattern)+/) {
760                0xEFFFE => 1, 0xEFFFF => 1, 0xFFFFE => 1, 0xFFFFF => 1,          substr ($_[0], $offset)
761                0x10FFFE => 1, 0x10FFFF => 1,              = substr ($self->{char_buffer}, $-[0], $+[0] - $-[0]);
762               }->{$self->{next_char}}) {          my $count = $+[0] - $-[0];
763        !!!cp ('j5');          if ($count) {
764        if ($self->{next_char} < 0x10000) {            $self->{column} += $count;
765          !!!parse-error (type => 'control char',            $self->{char_buffer_pos} += $count;
766                          text => (sprintf 'U+%04X', $self->{next_char}));            $self->{line_prev} = $self->{line};
767              $self->{column_prev} = $self->{column} - 1;
768              $self->{nc} = -1;
769            }
770            return $count;
771        } else {        } else {
772          !!!parse-error (type => 'control char',          return 0;
                         text => (sprintf 'U-%08X', $self->{next_char}));  
773        }        }
774        } else {
775          my $count = $input->manakai_read_until ($_[0], $pattern, $_[2]);
776          if ($count) {
777            $self->{column} += $count;
778            $self->{line_prev} = $self->{line};
779            $self->{column_prev} = $self->{column} - 1;
780            $self->{nc} = -1;
781          }
782          return $count;
783      }      }
784    };    }; # $self->{read_until}
   $self->{prev_char} = [-1, -1, -1];  
   $self->{next_char} = -1;  
785    
786    my $onerror = $_[2] || sub {    my $onerror = $_[2] || sub {
787      my (%opt) = @_;      my (%opt) = @_;
# Line 722  sub parse_char_stream ($$$;$) { Line 793  sub parse_char_stream ($$$;$) {
793      $onerror->(line => $self->{line}, column => $self->{column}, @_);      $onerror->(line => $self->{line}, column => $self->{column}, @_);
794    };    };
795    
796      my $char_onerror = sub {
797        my (undef, $type, %opt) = @_;
798        !!!parse-error (layer => 'encode',
799                        line => $self->{line}, column => $self->{column} + 1,
800                        %opt, type => $type);
801      }; # $char_onerror
802    
803      if ($_[3]) {
804        $input = $_[3]->($input);
805        $input->onerror ($char_onerror);
806      } else {
807        $input->onerror ($char_onerror) unless defined $input->onerror;
808      }
809    
810    $self->_initialize_tokenizer;    $self->_initialize_tokenizer;
811    $self->_initialize_tree_constructor;    $self->_initialize_tree_constructor;
812    $self->_construct_tree;    $self->_construct_tree;
# Line 741  sub new ($) { Line 826  sub new ($) {
826                info => 'i',                info => 'i',
827                uncertain => 'u'},                uncertain => 'u'},
828    }, $class;    }, $class;
829    $self->{set_next_char} = sub {    $self->{set_nc} = sub {
830      $self->{next_char} = -1;      $self->{nc} = -1;
831    };    };
832    $self->{parse_error} = sub {    $self->{parse_error} = sub {
833      #      #
# Line 769  sub RCDATA_CONTENT_MODEL () { CM_ENTITY Line 854  sub RCDATA_CONTENT_MODEL () { CM_ENTITY
854  sub PCDATA_CONTENT_MODEL () { CM_ENTITY | CM_FULL_MARKUP }  sub PCDATA_CONTENT_MODEL () { CM_ENTITY | CM_FULL_MARKUP }
855    
856  sub DATA_STATE () { 0 }  sub DATA_STATE () { 0 }
857  sub ENTITY_DATA_STATE () { 1 }  #sub ENTITY_DATA_STATE () { 1 }
858  sub TAG_OPEN_STATE () { 2 }  sub TAG_OPEN_STATE () { 2 }
859  sub CLOSE_TAG_OPEN_STATE () { 3 }  sub CLOSE_TAG_OPEN_STATE () { 3 }
860  sub TAG_NAME_STATE () { 4 }  sub TAG_NAME_STATE () { 4 }
# Line 780  sub BEFORE_ATTRIBUTE_VALUE_STATE () { 8 Line 865  sub BEFORE_ATTRIBUTE_VALUE_STATE () { 8
865  sub ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE () { 9 }  sub ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE () { 9 }
866  sub ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE () { 10 }  sub ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE () { 10 }
867  sub ATTRIBUTE_VALUE_UNQUOTED_STATE () { 11 }  sub ATTRIBUTE_VALUE_UNQUOTED_STATE () { 11 }
868  sub ENTITY_IN_ATTRIBUTE_VALUE_STATE () { 12 }  #sub ENTITY_IN_ATTRIBUTE_VALUE_STATE () { 12 }
869  sub MARKUP_DECLARATION_OPEN_STATE () { 13 }  sub MARKUP_DECLARATION_OPEN_STATE () { 13 }
870  sub COMMENT_START_STATE () { 14 }  sub COMMENT_START_STATE () { 14 }
871  sub COMMENT_START_DASH_STATE () { 15 }  sub COMMENT_START_DASH_STATE () { 15 }
# Line 807  sub CDATA_SECTION_STATE () { 35 } Line 892  sub CDATA_SECTION_STATE () { 35 }
892  sub MD_HYPHEN_STATE () { 36 } # "markup declaration open state" in the spec  sub MD_HYPHEN_STATE () { 36 } # "markup declaration open state" in the spec
893  sub MD_DOCTYPE_STATE () { 37 } # "markup declaration open state" in the spec  sub MD_DOCTYPE_STATE () { 37 } # "markup declaration open state" in the spec
894  sub MD_CDATA_STATE () { 38 } # "markup declaration open state" in the spec  sub MD_CDATA_STATE () { 38 } # "markup declaration open state" in the spec
895  sub CDATA_PCDATA_CLOSE_TAG_STATE () { 39 } # "close tag open state" in the spec  sub CDATA_RCDATA_CLOSE_TAG_STATE () { 39 } # "close tag open state" in the spec
896  sub CDATA_SECTION_MSE1_STATE () { 40 } # "CDATA section state" in the spec  sub CDATA_SECTION_MSE1_STATE () { 40 } # "CDATA section state" in the spec
897  sub CDATA_SECTION_MSE2_STATE () { 41 } # "CDATA section state" in the spec  sub CDATA_SECTION_MSE2_STATE () { 41 } # "CDATA section state" in the spec
898  sub PUBLIC_STATE () { 42 } # "after DOCTYPE name state" in the spec  sub PUBLIC_STATE () { 42 } # "after DOCTYPE name state" in the spec
899  sub SYSTEM_STATE () { 43 } # "after DOCTYPE name state" in the spec  sub SYSTEM_STATE () { 43 } # "after DOCTYPE name state" in the spec
900  sub ENTITY_STATE () { 44 } # "consume a character reference" in the spec  ## NOTE: "Entity data state", "entity in attribute value state", and
901    ## "consume a character reference" algorithm are jointly implemented
902    ## using the following six states:
903    sub ENTITY_STATE () { 44 }
904    sub ENTITY_HASH_STATE () { 45 }
905    sub NCR_NUM_STATE () { 46 }
906    sub HEXREF_X_STATE () { 47 }
907    sub HEXREF_HEX_STATE () { 48 }
908    sub ENTITY_NAME_STATE () { 49 }
909    sub PCDATA_STATE () { 50 } # "data state" in the spec
910    
911  sub DOCTYPE_TOKEN () { 1 }  sub DOCTYPE_TOKEN () { 1 }
912  sub COMMENT_TOKEN () { 2 }  sub COMMENT_TOKEN () { 2 }
# Line 834  sub IN_FOREIGN_CONTENT_IM () { 0b1000000 Line 928  sub IN_FOREIGN_CONTENT_IM () { 0b1000000
928      ## NOTE: "in foreign content" insertion mode is special; it is combined      ## NOTE: "in foreign content" insertion mode is special; it is combined
929      ## with the secondary insertion mode.  In this parser, they are stored      ## with the secondary insertion mode.  In this parser, they are stored
930      ## together in the bit-or'ed form.      ## together in the bit-or'ed form.
931    sub IN_CDATA_RCDATA_IM () { 0b1000000000000 }
932        ## NOTE: "in CDATA/RCDATA" insertion mode is also special; it is
933        ## combined with the original insertion mode.  In thie parser,
934        ## they are stored together in the bit-or'ed form.
935    
936  ## NOTE: "initial" and "before html" insertion modes have no constants.  ## NOTE: "initial" and "before html" insertion modes have no constants.
937    
# Line 865  sub IN_COLUMN_GROUP_IM () { 0b10 } Line 963  sub IN_COLUMN_GROUP_IM () { 0b10 }
963  sub _initialize_tokenizer ($) {  sub _initialize_tokenizer ($) {
964    my $self = shift;    my $self = shift;
965    $self->{state} = DATA_STATE; # MUST    $self->{state} = DATA_STATE; # MUST
966    #$self->{state_keyword}; # initialized when used    #$self->{s_kwd}; # state keyword - initialized when used
967      #$self->{entity__value}; # initialized when used
968      #$self->{entity__match}; # initialized when used
969    $self->{content_model} = PCDATA_CONTENT_MODEL; # be    $self->{content_model} = PCDATA_CONTENT_MODEL; # be
970    undef $self->{current_token};    undef $self->{ct}; # current token
971    undef $self->{current_attribute};    undef $self->{ca}; # current attribute
972    undef $self->{last_emitted_start_tag_name};    undef $self->{last_stag_name}; # last emitted start tag name
973    undef $self->{last_attribute_value_state};    #$self->{prev_state}; # initialized when used
974    delete $self->{self_closing};    delete $self->{self_closing};
975    $self->{char} = [];    $self->{char_buffer} = '';
976    # $self->{next_char}    $self->{char_buffer_pos} = 0;
977      $self->{nc} = -1; # next input character
978      #$self->{next_nc}
979    !!!next-input-character;    !!!next-input-character;
980    $self->{token} = [];    $self->{token} = [];
981    # $self->{escape}    # $self->{escape}
# Line 884  sub _initialize_tokenizer ($) { Line 986  sub _initialize_tokenizer ($) {
986  ##       CHARACTER_TOKEN, or END_OF_FILE_TOKEN  ##       CHARACTER_TOKEN, or END_OF_FILE_TOKEN
987  ##   ->{name} (DOCTYPE_TOKEN)  ##   ->{name} (DOCTYPE_TOKEN)
988  ##   ->{tag_name} (START_TAG_TOKEN, END_TAG_TOKEN)  ##   ->{tag_name} (START_TAG_TOKEN, END_TAG_TOKEN)
989  ##   ->{public_identifier} (DOCTYPE_TOKEN)  ##   ->{pubid} (DOCTYPE_TOKEN)
990  ##   ->{system_identifier} (DOCTYPE_TOKEN)  ##   ->{sysid} (DOCTYPE_TOKEN)
991  ##   ->{quirks} == 1 or 0 (DOCTYPE_TOKEN): "force-quirks" flag  ##   ->{quirks} == 1 or 0 (DOCTYPE_TOKEN): "force-quirks" flag
992  ##   ->{attributes} isa HASH (START_TAG_TOKEN, END_TAG_TOKEN)  ##   ->{attributes} isa HASH (START_TAG_TOKEN, END_TAG_TOKEN)
993  ##        ->{name}  ##        ->{name}
# Line 904  sub _initialize_tokenizer ($) { Line 1006  sub _initialize_tokenizer ($) {
1006  ## has completed loading.  If one has, then it MUST be executed  ## has completed loading.  If one has, then it MUST be executed
1007  ## and removed from the list.  ## and removed from the list.
1008    
1009  ## NOTE: HTML5 "Writing HTML documents" section, applied to  ## TODO: Polytheistic slash SHOULD NOT be used. (Applied only to atheists.)
1010  ## documents and not to user agents and conformance checkers,  ## (This requirement was dropped from HTML5 spec, unfortunately.)
1011  ## contains some requirements that are not detected by the  
1012  ## parsing algorithm:  my $is_space = {
1013  ## - Some requirements on character encoding declarations. ## TODO    0x0009 => 1, # CHARACTER TABULATION (HT)
1014  ## - "Elements MUST NOT contain content that their content model disallows."    0x000A => 1, # LINE FEED (LF)
1015  ##   ... Some are parse error, some are not (will be reported by c.c.).    #0x000B => 0, # LINE TABULATION (VT)
1016  ## - Polytheistic slash SHOULD NOT be used. (Applied only to atheists.) ## TODO    0x000C => 1, # FORM FEED (FF)
1017  ## - Text (in elements, attributes, and comments) SHOULD NOT contain    #0x000D => 1, # CARRIAGE RETURN (CR)
1018  ##   control characters other than space characters. ## TODO: (what is control character? C0, C1 and DEL?  Unicode control character?)    0x0020 => 1, # SPACE (SP)
1019    };
 ## TODO: HTML5 poses authors two SHOULD-level requirements that cannot  
 ## be detected by the HTML5 parsing algorithm:  
 ## - Text,  
1020    
1021  sub _get_next_token ($) {  sub _get_next_token ($) {
1022    my $self = shift;    my $self = shift;
1023    
1024    if ($self->{self_closing}) {    if ($self->{self_closing}) {
1025      !!!parse-error (type => 'nestc', token => $self->{current_token});      !!!parse-error (type => 'nestc', token => $self->{ct});
1026      ## NOTE: The |self_closing| flag is only set by start tag token.      ## NOTE: The |self_closing| flag is only set by start tag token.
1027      ## In addition, when a start tag token is emitted, it is always set to      ## In addition, when a start tag token is emitted, it is always set to
1028      ## |current_token|.      ## |ct|.
1029      delete $self->{self_closing};      delete $self->{self_closing};
1030    }    }
1031    
# Line 936  sub _get_next_token ($) { Line 1035  sub _get_next_token ($) {
1035    }    }
1036    
1037    A: {    A: {
1038      if ($self->{state} == DATA_STATE) {      if ($self->{state} == PCDATA_STATE) {
1039        if ($self->{next_char} == 0x0026) { # &        ## NOTE: Same as |DATA_STATE|, but only for |PCDATA| content model.
1040    
1041          if ($self->{nc} == 0x0026) { # &
1042            !!!cp (0.1);
1043            ## NOTE: In the spec, the tokenizer is switched to the
1044            ## "entity data state".  In this implementation, the tokenizer
1045            ## is switched to the |ENTITY_STATE|, which is an implementation
1046            ## of the "consume a character reference" algorithm.
1047            $self->{entity_add} = -1;
1048            $self->{prev_state} = DATA_STATE;
1049            $self->{state} = ENTITY_STATE;
1050            !!!next-input-character;
1051            redo A;
1052          } elsif ($self->{nc} == 0x003C) { # <
1053            !!!cp (0.2);
1054            $self->{state} = TAG_OPEN_STATE;
1055            !!!next-input-character;
1056            redo A;
1057          } elsif ($self->{nc} == -1) {
1058            !!!cp (0.3);
1059            !!!emit ({type => END_OF_FILE_TOKEN,
1060                      line => $self->{line}, column => $self->{column}});
1061            last A; ## TODO: ok?
1062          } else {
1063            !!!cp (0.4);
1064            #
1065          }
1066    
1067          # Anything else
1068          my $token = {type => CHARACTER_TOKEN,
1069                       data => chr $self->{nc},
1070                       line => $self->{line}, column => $self->{column},
1071                      };
1072          $self->{read_until}->($token->{data}, q[<&], length $token->{data});
1073    
1074          ## Stay in the state.
1075          !!!next-input-character;
1076          !!!emit ($token);
1077          redo A;
1078        } elsif ($self->{state} == DATA_STATE) {
1079          $self->{s_kwd} = '' unless defined $self->{s_kwd};
1080          if ($self->{nc} == 0x0026) { # &
1081            $self->{s_kwd} = '';
1082          if ($self->{content_model} & CM_ENTITY and # PCDATA | RCDATA          if ($self->{content_model} & CM_ENTITY and # PCDATA | RCDATA
1083              not $self->{escape}) {              not $self->{escape}) {
1084            !!!cp (1);            !!!cp (1);
# Line 945  sub _get_next_token ($) { Line 1086  sub _get_next_token ($) {
1086            ## "entity data state".  In this implementation, the tokenizer            ## "entity data state".  In this implementation, the tokenizer
1087            ## is switched to the |ENTITY_STATE|, which is an implementation            ## is switched to the |ENTITY_STATE|, which is an implementation
1088            ## of the "consume a character reference" algorithm.            ## of the "consume a character reference" algorithm.
1089            #$self->{state} = ENTITY_DATA_STATE;            $self->{entity_add} = -1;
1090            $self->{entity_in_attr} = 0;            $self->{prev_state} = DATA_STATE;
           $self->{entity_additional} = -1;  
1091            $self->{state} = ENTITY_STATE;            $self->{state} = ENTITY_STATE;
1092            !!!next-input-character;            !!!next-input-character;
1093            redo A;            redo A;
# Line 955  sub _get_next_token ($) { Line 1095  sub _get_next_token ($) {
1095            !!!cp (2);            !!!cp (2);
1096            #            #
1097          }          }
1098        } elsif ($self->{next_char} == 0x002D) { # -        } elsif ($self->{nc} == 0x002D) { # -
1099          if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA          if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
1100            unless ($self->{escape}) {            $self->{s_kwd} .= '-';
1101              if ($self->{prev_char}->[0] == 0x002D and # -            
1102                  $self->{prev_char}->[1] == 0x0021 and # !            if ($self->{s_kwd} eq '<!--') {
1103                  $self->{prev_char}->[2] == 0x003C) { # <              !!!cp (3);
1104                !!!cp (3);              $self->{escape} = 1; # unless $self->{escape};
1105                $self->{escape} = 1;              $self->{s_kwd} = '--';
1106              } else {              #
1107                !!!cp (4);            } elsif ($self->{s_kwd} eq '---') {
1108              }              !!!cp (4);
1109                $self->{s_kwd} = '--';
1110                #
1111            } else {            } else {
1112              !!!cp (5);              !!!cp (5);
1113                #
1114            }            }
1115          }          }
1116                    
1117          #          #
1118        } elsif ($self->{next_char} == 0x003C) { # <        } elsif ($self->{nc} == 0x0021) { # !
1119            if (length $self->{s_kwd}) {
1120              !!!cp (5.1);
1121              $self->{s_kwd} .= '!';
1122              #
1123            } else {
1124              !!!cp (5.2);
1125              #$self->{s_kwd} = '';
1126              #
1127            }
1128            #
1129          } elsif ($self->{nc} == 0x003C) { # <
1130          if ($self->{content_model} & CM_FULL_MARKUP or # PCDATA          if ($self->{content_model} & CM_FULL_MARKUP or # PCDATA
1131              (($self->{content_model} & CM_LIMITED_MARKUP) and # CDATA | RCDATA              (($self->{content_model} & CM_LIMITED_MARKUP) and # CDATA | RCDATA
1132               not $self->{escape})) {               not $self->{escape})) {
# Line 982  sub _get_next_token ($) { Line 1136  sub _get_next_token ($) {
1136            redo A;            redo A;
1137          } else {          } else {
1138            !!!cp (7);            !!!cp (7);
1139              $self->{s_kwd} = '';
1140            #            #
1141          }          }
1142        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1143          if ($self->{escape} and          if ($self->{escape} and
1144              ($self->{content_model} & CM_LIMITED_MARKUP)) { # RCDATA | CDATA              ($self->{content_model} & CM_LIMITED_MARKUP)) { # RCDATA | CDATA
1145            if ($self->{prev_char}->[0] == 0x002D and # -            if ($self->{s_kwd} eq '--') {
               $self->{prev_char}->[1] == 0x002D) { # -  
1146              !!!cp (8);              !!!cp (8);
1147              delete $self->{escape};              delete $self->{escape};
1148            } else {            } else {
# Line 998  sub _get_next_token ($) { Line 1152  sub _get_next_token ($) {
1152            !!!cp (10);            !!!cp (10);
1153          }          }
1154                    
1155            $self->{s_kwd} = '';
1156          #          #
1157        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1158          !!!cp (11);          !!!cp (11);
1159            $self->{s_kwd} = '';
1160          !!!emit ({type => END_OF_FILE_TOKEN,          !!!emit ({type => END_OF_FILE_TOKEN,
1161                    line => $self->{line}, column => $self->{column}});                    line => $self->{line}, column => $self->{column}});
1162          last A; ## TODO: ok?          last A; ## TODO: ok?
1163        } else {        } else {
1164          !!!cp (12);          !!!cp (12);
1165            $self->{s_kwd} = '';
1166            #
1167        }        }
1168    
1169        # Anything else        # Anything else
1170        my $token = {type => CHARACTER_TOKEN,        my $token = {type => CHARACTER_TOKEN,
1171                     data => chr $self->{next_char},                     data => chr $self->{nc},
1172                     line => $self->{line}, column => $self->{column},                     line => $self->{line}, column => $self->{column},
1173                    };                    };
1174        ## Stay in the data state        if ($self->{read_until}->($token->{data}, q[-!<>&],
1175        !!!next-input-character;                                  length $token->{data})) {
1176            $self->{s_kwd} = '';
1177        !!!emit ($token);        }
   
       redo A;  
     } elsif ($self->{state} == ENTITY_DATA_STATE) {  
       my ($l, $c) = ($self->{line_prev}, $self->{column_prev});  
   
       my $token = $self->{entity_return};  
   
       $self->{state} = DATA_STATE;  
       # next-input-character is already done  
1178    
1179        unless (defined $token) {        ## Stay in the data state.
1180          if ($self->{content_model} == PCDATA_CONTENT_MODEL) {
1181          !!!cp (13);          !!!cp (13);
1182          !!!emit ({type => CHARACTER_TOKEN, data => '&',          $self->{state} = PCDATA_STATE;
                   line => $l, column => $c,  
                  });  
1183        } else {        } else {
1184          !!!cp (14);          !!!cp (14);
1185          !!!emit ($token);          ## Stay in the state.
1186        }        }
1187          !!!next-input-character;
1188          !!!emit ($token);
1189        redo A;        redo A;
1190      } elsif ($self->{state} == TAG_OPEN_STATE) {      } elsif ($self->{state} == TAG_OPEN_STATE) {
1191        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
1192          if ($self->{next_char} == 0x002F) { # /          if ($self->{nc} == 0x002F) { # /
1193            !!!cp (15);            !!!cp (15);
1194            !!!next-input-character;            !!!next-input-character;
1195            $self->{state} = CLOSE_TAG_OPEN_STATE;            $self->{state} = CLOSE_TAG_OPEN_STATE;
1196            redo A;            redo A;
1197            } elsif ($self->{nc} == 0x0021) { # !
1198              !!!cp (15.1);
1199              $self->{s_kwd} = '<' unless $self->{escape};
1200              #
1201          } else {          } else {
1202            !!!cp (16);            !!!cp (16);
1203            ## reconsume            #
           $self->{state} = DATA_STATE;  
   
           !!!emit ({type => CHARACTER_TOKEN, data => '<',  
                     line => $self->{line_prev},  
                     column => $self->{column_prev},  
                    });  
   
           redo A;  
1204          }          }
1205    
1206            ## reconsume
1207            $self->{state} = DATA_STATE;
1208            !!!emit ({type => CHARACTER_TOKEN, data => '<',
1209                      line => $self->{line_prev},
1210                      column => $self->{column_prev},
1211                     });
1212            redo A;
1213        } elsif ($self->{content_model} & CM_FULL_MARKUP) { # PCDATA        } elsif ($self->{content_model} & CM_FULL_MARKUP) { # PCDATA
1214          if ($self->{next_char} == 0x0021) { # !          if ($self->{nc} == 0x0021) { # !
1215            !!!cp (17);            !!!cp (17);
1216            $self->{state} = MARKUP_DECLARATION_OPEN_STATE;            $self->{state} = MARKUP_DECLARATION_OPEN_STATE;
1217            !!!next-input-character;            !!!next-input-character;
1218            redo A;            redo A;
1219          } elsif ($self->{next_char} == 0x002F) { # /          } elsif ($self->{nc} == 0x002F) { # /
1220            !!!cp (18);            !!!cp (18);
1221            $self->{state} = CLOSE_TAG_OPEN_STATE;            $self->{state} = CLOSE_TAG_OPEN_STATE;
1222            !!!next-input-character;            !!!next-input-character;
1223            redo A;            redo A;
1224          } elsif (0x0041 <= $self->{next_char} and          } elsif (0x0041 <= $self->{nc} and
1225                   $self->{next_char} <= 0x005A) { # A..Z                   $self->{nc} <= 0x005A) { # A..Z
1226            !!!cp (19);            !!!cp (19);
1227            $self->{current_token}            $self->{ct}
1228              = {type => START_TAG_TOKEN,              = {type => START_TAG_TOKEN,
1229                 tag_name => chr ($self->{next_char} + 0x0020),                 tag_name => chr ($self->{nc} + 0x0020),
1230                 line => $self->{line_prev},                 line => $self->{line_prev},
1231                 column => $self->{column_prev}};                 column => $self->{column_prev}};
1232            $self->{state} = TAG_NAME_STATE;            $self->{state} = TAG_NAME_STATE;
1233            !!!next-input-character;            !!!next-input-character;
1234            redo A;            redo A;
1235          } elsif (0x0061 <= $self->{next_char} and          } elsif (0x0061 <= $self->{nc} and
1236                   $self->{next_char} <= 0x007A) { # a..z                   $self->{nc} <= 0x007A) { # a..z
1237            !!!cp (20);            !!!cp (20);
1238            $self->{current_token} = {type => START_TAG_TOKEN,            $self->{ct} = {type => START_TAG_TOKEN,
1239                                      tag_name => chr ($self->{next_char}),                                      tag_name => chr ($self->{nc}),
1240                                      line => $self->{line_prev},                                      line => $self->{line_prev},
1241                                      column => $self->{column_prev}};                                      column => $self->{column_prev}};
1242            $self->{state} = TAG_NAME_STATE;            $self->{state} = TAG_NAME_STATE;
1243            !!!next-input-character;            !!!next-input-character;
1244            redo A;            redo A;
1245          } elsif ($self->{next_char} == 0x003E) { # >          } elsif ($self->{nc} == 0x003E) { # >
1246            !!!cp (21);            !!!cp (21);
1247            !!!parse-error (type => 'empty start tag',            !!!parse-error (type => 'empty start tag',
1248                            line => $self->{line_prev},                            line => $self->{line_prev},
# Line 1102  sub _get_next_token ($) { Line 1256  sub _get_next_token ($) {
1256                     });                     });
1257    
1258            redo A;            redo A;
1259          } elsif ($self->{next_char} == 0x003F) { # ?          } elsif ($self->{nc} == 0x003F) { # ?
1260            !!!cp (22);            !!!cp (22);
1261            !!!parse-error (type => 'pio',            !!!parse-error (type => 'pio',
1262                            line => $self->{line_prev},                            line => $self->{line_prev},
1263                            column => $self->{column_prev});                            column => $self->{column_prev});
1264            $self->{state} = BOGUS_COMMENT_STATE;            $self->{state} = BOGUS_COMMENT_STATE;
1265            $self->{current_token} = {type => COMMENT_TOKEN, data => '',            $self->{ct} = {type => COMMENT_TOKEN, data => '',
1266                                      line => $self->{line_prev},                                      line => $self->{line_prev},
1267                                      column => $self->{column_prev},                                      column => $self->{column_prev},
1268                                     };                                     };
1269            ## $self->{next_char} is intentionally left as is            ## $self->{nc} is intentionally left as is
1270            redo A;            redo A;
1271          } else {          } else {
1272            !!!cp (23);            !!!cp (23);
# Line 1134  sub _get_next_token ($) { Line 1288  sub _get_next_token ($) {
1288        }        }
1289      } elsif ($self->{state} == CLOSE_TAG_OPEN_STATE) {      } elsif ($self->{state} == CLOSE_TAG_OPEN_STATE) {
1290        ## NOTE: The "close tag open state" in the spec is implemented as        ## NOTE: The "close tag open state" in the spec is implemented as
1291        ## |CLOSE_TAG_OPEN_STATE| and |CDATA_PCDATA_CLOSE_TAG_STATE|.        ## |CLOSE_TAG_OPEN_STATE| and |CDATA_RCDATA_CLOSE_TAG_STATE|.
1292    
1293        my ($l, $c) = ($self->{line_prev}, $self->{column_prev} - 1); # "<"of"</"        my ($l, $c) = ($self->{line_prev}, $self->{column_prev} - 1); # "<"of"</"
1294        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
1295          if (defined $self->{last_emitted_start_tag_name}) {          if (defined $self->{last_stag_name}) {
1296            $self->{state} = CDATA_PCDATA_CLOSE_TAG_STATE;            $self->{state} = CDATA_RCDATA_CLOSE_TAG_STATE;
1297            $self->{state_keyword} = '';            $self->{s_kwd} = '';
1298            ## Reconsume.            ## Reconsume.
1299            redo A;            redo A;
1300          } else {          } else {
# Line 1156  sub _get_next_token ($) { Line 1310  sub _get_next_token ($) {
1310          }          }
1311        }        }
1312    
1313        if (0x0041 <= $self->{next_char} and        if (0x0041 <= $self->{nc} and
1314            $self->{next_char} <= 0x005A) { # A..Z            $self->{nc} <= 0x005A) { # A..Z
1315          !!!cp (29);          !!!cp (29);
1316          $self->{current_token}          $self->{ct}
1317              = {type => END_TAG_TOKEN,              = {type => END_TAG_TOKEN,
1318                 tag_name => chr ($self->{next_char} + 0x0020),                 tag_name => chr ($self->{nc} + 0x0020),
1319                 line => $l, column => $c};                 line => $l, column => $c};
1320          $self->{state} = TAG_NAME_STATE;          $self->{state} = TAG_NAME_STATE;
1321          !!!next-input-character;          !!!next-input-character;
1322          redo A;          redo A;
1323        } elsif (0x0061 <= $self->{next_char} and        } elsif (0x0061 <= $self->{nc} and
1324                 $self->{next_char} <= 0x007A) { # a..z                 $self->{nc} <= 0x007A) { # a..z
1325          !!!cp (30);          !!!cp (30);
1326          $self->{current_token} = {type => END_TAG_TOKEN,          $self->{ct} = {type => END_TAG_TOKEN,
1327                                    tag_name => chr ($self->{next_char}),                                    tag_name => chr ($self->{nc}),
1328                                    line => $l, column => $c};                                    line => $l, column => $c};
1329          $self->{state} = TAG_NAME_STATE;          $self->{state} = TAG_NAME_STATE;
1330          !!!next-input-character;          !!!next-input-character;
1331          redo A;          redo A;
1332        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1333          !!!cp (31);          !!!cp (31);
1334          !!!parse-error (type => 'empty end tag',          !!!parse-error (type => 'empty end tag',
1335                          line => $self->{line_prev}, ## "<" in "</>"                          line => $self->{line_prev}, ## "<" in "</>"
# Line 1183  sub _get_next_token ($) { Line 1337  sub _get_next_token ($) {
1337          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1338          !!!next-input-character;          !!!next-input-character;
1339          redo A;          redo A;
1340        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1341          !!!cp (32);          !!!cp (32);
1342          !!!parse-error (type => 'bare etago');          !!!parse-error (type => 'bare etago');
1343          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
# Line 1198  sub _get_next_token ($) { Line 1352  sub _get_next_token ($) {
1352          !!!cp (33);          !!!cp (33);
1353          !!!parse-error (type => 'bogus end tag');          !!!parse-error (type => 'bogus end tag');
1354          $self->{state} = BOGUS_COMMENT_STATE;          $self->{state} = BOGUS_COMMENT_STATE;
1355          $self->{current_token} = {type => COMMENT_TOKEN, data => '',          $self->{ct} = {type => COMMENT_TOKEN, data => '',
1356                                    line => $self->{line_prev}, # "<" of "</"                                    line => $self->{line_prev}, # "<" of "</"
1357                                    column => $self->{column_prev} - 1,                                    column => $self->{column_prev} - 1,
1358                                   };                                   };
1359          ## NOTE: $self->{next_char} is intentionally left as is.          ## NOTE: $self->{nc} is intentionally left as is.
1360          ## Although the "anything else" case of the spec not explicitly          ## Although the "anything else" case of the spec not explicitly
1361          ## states that the next input character is to be reconsumed,          ## states that the next input character is to be reconsumed,
1362          ## it will be included to the |data| of the comment token          ## it will be included to the |data| of the comment token
# Line 1210  sub _get_next_token ($) { Line 1364  sub _get_next_token ($) {
1364          ## "bogus comment state" entry.          ## "bogus comment state" entry.
1365          redo A;          redo A;
1366        }        }
1367      } elsif ($self->{state} == CDATA_PCDATA_CLOSE_TAG_STATE) {      } elsif ($self->{state} == CDATA_RCDATA_CLOSE_TAG_STATE) {
1368        my $ch = substr $self->{last_emitted_start_tag_name}, length $self->{state_keyword}, 1;        my $ch = substr $self->{last_stag_name}, length $self->{s_kwd}, 1;
1369        if (length $ch) {        if (length $ch) {
1370          my $CH = $ch;          my $CH = $ch;
1371          $ch =~ tr/a-z/A-Z/;          $ch =~ tr/a-z/A-Z/;
1372          my $nch = chr $self->{next_char};          my $nch = chr $self->{nc};
1373          if ($nch eq $ch or $nch eq $CH) {          if ($nch eq $ch or $nch eq $CH) {
1374            !!!cp (24);            !!!cp (24);
1375            ## Stay in the state.            ## Stay in the state.
1376            $self->{state_keyword} .= $nch;            $self->{s_kwd} .= $nch;
1377            !!!next-input-character;            !!!next-input-character;
1378            redo A;            redo A;
1379          } else {          } else {
# Line 1227  sub _get_next_token ($) { Line 1381  sub _get_next_token ($) {
1381            $self->{state} = DATA_STATE;            $self->{state} = DATA_STATE;
1382            ## Reconsume.            ## Reconsume.
1383            !!!emit ({type => CHARACTER_TOKEN,            !!!emit ({type => CHARACTER_TOKEN,
1384                      data => '</' . $self->{state_keyword},                      data => '</' . $self->{s_kwd},
1385                      line => $self->{line_prev},                      line => $self->{line_prev},
1386                      column => $self->{column_prev} - 1 - length $self->{state_keyword},                      column => $self->{column_prev} - 1 - length $self->{s_kwd},
1387                     });                     });
1388            redo A;            redo A;
1389          }          }
1390        } else { # after "<{tag-name}"        } else { # after "<{tag-name}"
1391          unless ({          unless ($is_space->{$self->{nc}} or
1392                   0x0009 => 1, # HT                  {
                  0x000A => 1, # LF  
                  0x000B => 1, # VT  
                  0x000C => 1, # FF  
                  0x0020 => 1, # SP  
1393                   0x003E => 1, # >                   0x003E => 1, # >
1394                   0x002F => 1, # /                   0x002F => 1, # /
1395                   -1 => 1, # EOF                   -1 => 1, # EOF
1396                  }->{$self->{next_char}}) {                  }->{$self->{nc}}) {
1397            !!!cp (26);            !!!cp (26);
1398            ## Reconsume.            ## Reconsume.
1399            $self->{state} = DATA_STATE;            $self->{state} = DATA_STATE;
1400            !!!emit ({type => CHARACTER_TOKEN,            !!!emit ({type => CHARACTER_TOKEN,
1401                      data => '</' . $self->{state_keyword},                      data => '</' . $self->{s_kwd},
1402                      line => $self->{line_prev},                      line => $self->{line_prev},
1403                      column => $self->{column_prev} - 1 - length $self->{state_keyword},                      column => $self->{column_prev} - 1 - length $self->{s_kwd},
1404                     });                     });
1405            redo A;            redo A;
1406          } else {          } else {
1407            !!!cp (27);            !!!cp (27);
1408            $self->{current_token}            $self->{ct}
1409                = {type => END_TAG_TOKEN,                = {type => END_TAG_TOKEN,
1410                   tag_name => $self->{last_emitted_start_tag_name},                   tag_name => $self->{last_stag_name},
1411                   line => $self->{line_prev},                   line => $self->{line_prev},
1412                   column => $self->{column_prev} - 1 - length $self->{state_keyword}};                   column => $self->{column_prev} - 1 - length $self->{s_kwd}};
1413            $self->{state} = TAG_NAME_STATE;            $self->{state} = TAG_NAME_STATE;
1414            ## Reconsume.            ## Reconsume.
1415            redo A;            redo A;
1416          }          }
1417        }        }
1418      } elsif ($self->{state} == TAG_NAME_STATE) {      } elsif ($self->{state} == TAG_NAME_STATE) {
1419        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
1420          !!!cp (34);          !!!cp (34);
1421          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1422          !!!next-input-character;          !!!next-input-character;
1423          redo A;          redo A;
1424        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1425          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1426            !!!cp (35);            !!!cp (35);
1427            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1428          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1429            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1430            #if ($self->{current_token}->{attributes}) {            #if ($self->{ct}->{attributes}) {
1431            #  ## NOTE: This should never be reached.            #  ## NOTE: This should never be reached.
1432            #  !!! cp (36);            #  !!! cp (36);
1433            #  !!! parse-error (type => 'end tag attribute');            #  !!! parse-error (type => 'end tag attribute');
# Line 1289  sub _get_next_token ($) { Line 1435  sub _get_next_token ($) {
1435              !!!cp (37);              !!!cp (37);
1436            #}            #}
1437          } else {          } else {
1438            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1439          }          }
1440          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1441          !!!next-input-character;          !!!next-input-character;
1442    
1443          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1444    
1445          redo A;          redo A;
1446        } elsif (0x0041 <= $self->{next_char} and        } elsif (0x0041 <= $self->{nc} and
1447                 $self->{next_char} <= 0x005A) { # A..Z                 $self->{nc} <= 0x005A) { # A..Z
1448          !!!cp (38);          !!!cp (38);
1449          $self->{current_token}->{tag_name} .= chr ($self->{next_char} + 0x0020);          $self->{ct}->{tag_name} .= chr ($self->{nc} + 0x0020);
1450            # start tag or end tag            # start tag or end tag
1451          ## Stay in this state          ## Stay in this state
1452          !!!next-input-character;          !!!next-input-character;
1453          redo A;          redo A;
1454        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1455          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1456          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1457            !!!cp (39);            !!!cp (39);
1458            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1459          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1460            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1461            #if ($self->{current_token}->{attributes}) {            #if ($self->{ct}->{attributes}) {
1462            #  ## NOTE: This state should never be reached.            #  ## NOTE: This state should never be reached.
1463            #  !!! cp (40);            #  !!! cp (40);
1464            #  !!! parse-error (type => 'end tag attribute');            #  !!! parse-error (type => 'end tag attribute');
# Line 1320  sub _get_next_token ($) { Line 1466  sub _get_next_token ($) {
1466              !!!cp (41);              !!!cp (41);
1467            #}            #}
1468          } else {          } else {
1469            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1470          }          }
1471          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1472          # reconsume          # reconsume
1473    
1474          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1475    
1476          redo A;          redo A;
1477        } elsif ($self->{next_char} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
1478          !!!cp (42);          !!!cp (42);
1479          $self->{state} = SELF_CLOSING_START_TAG_STATE;          $self->{state} = SELF_CLOSING_START_TAG_STATE;
1480          !!!next-input-character;          !!!next-input-character;
1481          redo A;          redo A;
1482        } else {        } else {
1483          !!!cp (44);          !!!cp (44);
1484          $self->{current_token}->{tag_name} .= chr $self->{next_char};          $self->{ct}->{tag_name} .= chr $self->{nc};
1485            # start tag or end tag            # start tag or end tag
1486          ## Stay in the state          ## Stay in the state
1487          !!!next-input-character;          !!!next-input-character;
1488          redo A;          redo A;
1489        }        }
1490      } elsif ($self->{state} == BEFORE_ATTRIBUTE_NAME_STATE) {      } elsif ($self->{state} == BEFORE_ATTRIBUTE_NAME_STATE) {
1491        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
1492          !!!cp (45);          !!!cp (45);
1493          ## Stay in the state          ## Stay in the state
1494          !!!next-input-character;          !!!next-input-character;
1495          redo A;          redo A;
1496        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1497          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1498            !!!cp (46);            !!!cp (46);
1499            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1500          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1501            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1502            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1503              !!!cp (47);              !!!cp (47);
1504              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1505            } else {            } else {
1506              !!!cp (48);              !!!cp (48);
1507            }            }
1508          } else {          } else {
1509            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1510          }          }
1511          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1512          !!!next-input-character;          !!!next-input-character;
1513    
1514          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1515    
1516          redo A;          redo A;
1517        } elsif (0x0041 <= $self->{next_char} and        } elsif (0x0041 <= $self->{nc} and
1518                 $self->{next_char} <= 0x005A) { # A..Z                 $self->{nc} <= 0x005A) { # A..Z
1519          !!!cp (49);          !!!cp (49);
1520          $self->{current_attribute}          $self->{ca}
1521              = {name => chr ($self->{next_char} + 0x0020),              = {name => chr ($self->{nc} + 0x0020),
1522                 value => '',                 value => '',
1523                 line => $self->{line}, column => $self->{column}};                 line => $self->{line}, column => $self->{column}};
1524          $self->{state} = ATTRIBUTE_NAME_STATE;          $self->{state} = ATTRIBUTE_NAME_STATE;
1525          !!!next-input-character;          !!!next-input-character;
1526          redo A;          redo A;
1527        } elsif ($self->{next_char} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
1528          !!!cp (50);          !!!cp (50);
1529          $self->{state} = SELF_CLOSING_START_TAG_STATE;          $self->{state} = SELF_CLOSING_START_TAG_STATE;
1530          !!!next-input-character;          !!!next-input-character;
1531          redo A;          redo A;
1532        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1533          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1534          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1535            !!!cp (52);            !!!cp (52);
1536            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1537          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1538            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1539            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1540              !!!cp (53);              !!!cp (53);
1541              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1542            } else {            } else {
1543              !!!cp (54);              !!!cp (54);
1544            }            }
1545          } else {          } else {
1546            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1547          }          }
1548          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1549          # reconsume          # reconsume
1550    
1551          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1552    
1553          redo A;          redo A;
1554        } else {        } else {
# Line 1414  sub _get_next_token ($) { Line 1556  sub _get_next_token ($) {
1556               0x0022 => 1, # "               0x0022 => 1, # "
1557               0x0027 => 1, # '               0x0027 => 1, # '
1558               0x003D => 1, # =               0x003D => 1, # =
1559              }->{$self->{next_char}}) {              }->{$self->{nc}}) {
1560            !!!cp (55);            !!!cp (55);
1561            !!!parse-error (type => 'bad attribute name');            !!!parse-error (type => 'bad attribute name');
1562          } else {          } else {
1563            !!!cp (56);            !!!cp (56);
1564          }          }
1565          $self->{current_attribute}          $self->{ca}
1566              = {name => chr ($self->{next_char}),              = {name => chr ($self->{nc}),
1567                 value => '',                 value => '',
1568                 line => $self->{line}, column => $self->{column}};                 line => $self->{line}, column => $self->{column}};
1569          $self->{state} = ATTRIBUTE_NAME_STATE;          $self->{state} = ATTRIBUTE_NAME_STATE;
# Line 1430  sub _get_next_token ($) { Line 1572  sub _get_next_token ($) {
1572        }        }
1573      } elsif ($self->{state} == ATTRIBUTE_NAME_STATE) {      } elsif ($self->{state} == ATTRIBUTE_NAME_STATE) {
1574        my $before_leave = sub {        my $before_leave = sub {
1575          if (exists $self->{current_token}->{attributes} # start tag or end tag          if (exists $self->{ct}->{attributes} # start tag or end tag
1576              ->{$self->{current_attribute}->{name}}) { # MUST              ->{$self->{ca}->{name}}) { # MUST
1577            !!!cp (57);            !!!cp (57);
1578            !!!parse-error (type => 'duplicate attribute', text => $self->{current_attribute}->{name}, line => $self->{current_attribute}->{line}, column => $self->{current_attribute}->{column});            !!!parse-error (type => 'duplicate attribute', text => $self->{ca}->{name}, line => $self->{ca}->{line}, column => $self->{ca}->{column});
1579            ## Discard $self->{current_attribute} # MUST            ## Discard $self->{ca} # MUST
1580          } else {          } else {
1581            !!!cp (58);            !!!cp (58);
1582            $self->{current_token}->{attributes}->{$self->{current_attribute}->{name}}            $self->{ct}->{attributes}->{$self->{ca}->{name}}
1583              = $self->{current_attribute};              = $self->{ca};
1584          }          }
1585        }; # $before_leave        }; # $before_leave
1586    
1587        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
1588          !!!cp (59);          !!!cp (59);
1589          $before_leave->();          $before_leave->();
1590          $self->{state} = AFTER_ATTRIBUTE_NAME_STATE;          $self->{state} = AFTER_ATTRIBUTE_NAME_STATE;
1591          !!!next-input-character;          !!!next-input-character;
1592          redo A;          redo A;
1593        } elsif ($self->{next_char} == 0x003D) { # =        } elsif ($self->{nc} == 0x003D) { # =
1594          !!!cp (60);          !!!cp (60);
1595          $before_leave->();          $before_leave->();
1596          $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;          $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;
1597          !!!next-input-character;          !!!next-input-character;
1598          redo A;          redo A;
1599        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1600          $before_leave->();          $before_leave->();
1601          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1602            !!!cp (61);            !!!cp (61);
1603            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1604          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1605            !!!cp (62);            !!!cp (62);
1606            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1607            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1608              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1609            }            }
1610          } else {          } else {
1611            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1612          }          }
1613          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1614          !!!next-input-character;          !!!next-input-character;
1615    
1616          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1617    
1618          redo A;          redo A;
1619        } elsif (0x0041 <= $self->{next_char} and        } elsif (0x0041 <= $self->{nc} and
1620                 $self->{next_char} <= 0x005A) { # A..Z                 $self->{nc} <= 0x005A) { # A..Z
1621          !!!cp (63);          !!!cp (63);
1622          $self->{current_attribute}->{name} .= chr ($self->{next_char} + 0x0020);          $self->{ca}->{name} .= chr ($self->{nc} + 0x0020);
1623          ## Stay in the state          ## Stay in the state
1624          !!!next-input-character;          !!!next-input-character;
1625          redo A;          redo A;
1626        } elsif ($self->{next_char} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
1627          !!!cp (64);          !!!cp (64);
1628          $before_leave->();          $before_leave->();
1629          $self->{state} = SELF_CLOSING_START_TAG_STATE;          $self->{state} = SELF_CLOSING_START_TAG_STATE;
1630          !!!next-input-character;          !!!next-input-character;
1631          redo A;          redo A;
1632        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1633          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1634          $before_leave->();          $before_leave->();
1635          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1636            !!!cp (66);            !!!cp (66);
1637            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1638          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1639            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1640            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1641              !!!cp (67);              !!!cp (67);
1642              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1643            } else {            } else {
# Line 1507  sub _get_next_token ($) { Line 1645  sub _get_next_token ($) {
1645              !!!cp (68);              !!!cp (68);
1646            }            }
1647          } else {          } else {
1648            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1649          }          }
1650          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1651          # reconsume          # reconsume
1652    
1653          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1654    
1655          redo A;          redo A;
1656        } else {        } else {
1657          if ($self->{next_char} == 0x0022 or # "          if ($self->{nc} == 0x0022 or # "
1658              $self->{next_char} == 0x0027) { # '              $self->{nc} == 0x0027) { # '
1659            !!!cp (69);            !!!cp (69);
1660            !!!parse-error (type => 'bad attribute name');            !!!parse-error (type => 'bad attribute name');
1661          } else {          } else {
1662            !!!cp (70);            !!!cp (70);
1663          }          }
1664          $self->{current_attribute}->{name} .= chr ($self->{next_char});          $self->{ca}->{name} .= chr ($self->{nc});
1665          ## Stay in the state          ## Stay in the state
1666          !!!next-input-character;          !!!next-input-character;
1667          redo A;          redo A;
1668        }        }
1669      } elsif ($self->{state} == AFTER_ATTRIBUTE_NAME_STATE) {      } elsif ($self->{state} == AFTER_ATTRIBUTE_NAME_STATE) {
1670        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
1671          !!!cp (71);          !!!cp (71);
1672          ## Stay in the state          ## Stay in the state
1673          !!!next-input-character;          !!!next-input-character;
1674          redo A;          redo A;
1675        } elsif ($self->{next_char} == 0x003D) { # =        } elsif ($self->{nc} == 0x003D) { # =
1676          !!!cp (72);          !!!cp (72);
1677          $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;          $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;
1678          !!!next-input-character;          !!!next-input-character;
1679          redo A;          redo A;
1680        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1681          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1682            !!!cp (73);            !!!cp (73);
1683            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1684          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1685            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1686            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1687              !!!cp (74);              !!!cp (74);
1688              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1689            } else {            } else {
# Line 1557  sub _get_next_token ($) { Line 1691  sub _get_next_token ($) {
1691              !!!cp (75);              !!!cp (75);
1692            }            }
1693          } else {          } else {
1694            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1695          }          }
1696          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1697          !!!next-input-character;          !!!next-input-character;
1698    
1699          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1700    
1701          redo A;          redo A;
1702        } elsif (0x0041 <= $self->{next_char} and        } elsif (0x0041 <= $self->{nc} and
1703                 $self->{next_char} <= 0x005A) { # A..Z                 $self->{nc} <= 0x005A) { # A..Z
1704          !!!cp (76);          !!!cp (76);
1705          $self->{current_attribute}          $self->{ca}
1706              = {name => chr ($self->{next_char} + 0x0020),              = {name => chr ($self->{nc} + 0x0020),
1707                 value => '',                 value => '',
1708                 line => $self->{line}, column => $self->{column}};                 line => $self->{line}, column => $self->{column}};
1709          $self->{state} = ATTRIBUTE_NAME_STATE;          $self->{state} = ATTRIBUTE_NAME_STATE;
1710          !!!next-input-character;          !!!next-input-character;
1711          redo A;          redo A;
1712        } elsif ($self->{next_char} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
1713          !!!cp (77);          !!!cp (77);
1714          $self->{state} = SELF_CLOSING_START_TAG_STATE;          $self->{state} = SELF_CLOSING_START_TAG_STATE;
1715          !!!next-input-character;          !!!next-input-character;
1716          redo A;          redo A;
1717        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1718          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1719          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1720            !!!cp (79);            !!!cp (79);
1721            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1722          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1723            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1724            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1725              !!!cp (80);              !!!cp (80);
1726              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1727            } else {            } else {
# Line 1595  sub _get_next_token ($) { Line 1729  sub _get_next_token ($) {
1729              !!!cp (81);              !!!cp (81);
1730            }            }
1731          } else {          } else {
1732            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1733          }          }
1734          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1735          # reconsume          # reconsume
1736    
1737          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1738    
1739          redo A;          redo A;
1740        } else {        } else {
1741          if ($self->{next_char} == 0x0022 or # "          if ($self->{nc} == 0x0022 or # "
1742              $self->{next_char} == 0x0027) { # '              $self->{nc} == 0x0027) { # '
1743            !!!cp (78);            !!!cp (78);
1744            !!!parse-error (type => 'bad attribute name');            !!!parse-error (type => 'bad attribute name');
1745          } else {          } else {
1746            !!!cp (82);            !!!cp (82);
1747          }          }
1748          $self->{current_attribute}          $self->{ca}
1749              = {name => chr ($self->{next_char}),              = {name => chr ($self->{nc}),
1750                 value => '',                 value => '',
1751                 line => $self->{line}, column => $self->{column}};                 line => $self->{line}, column => $self->{column}};
1752          $self->{state} = ATTRIBUTE_NAME_STATE;          $self->{state} = ATTRIBUTE_NAME_STATE;
# Line 1620  sub _get_next_token ($) { Line 1754  sub _get_next_token ($) {
1754          redo A;                  redo A;        
1755        }        }
1756      } elsif ($self->{state} == BEFORE_ATTRIBUTE_VALUE_STATE) {      } elsif ($self->{state} == BEFORE_ATTRIBUTE_VALUE_STATE) {
1757        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP        
1758          !!!cp (83);          !!!cp (83);
1759          ## Stay in the state          ## Stay in the state
1760          !!!next-input-character;          !!!next-input-character;
1761          redo A;          redo A;
1762        } elsif ($self->{next_char} == 0x0022) { # "        } elsif ($self->{nc} == 0x0022) { # "
1763          !!!cp (84);          !!!cp (84);
1764          $self->{state} = ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE;          $self->{state} = ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE;
1765          !!!next-input-character;          !!!next-input-character;
1766          redo A;          redo A;
1767        } elsif ($self->{next_char} == 0x0026) { # &        } elsif ($self->{nc} == 0x0026) { # &
1768          !!!cp (85);          !!!cp (85);
1769          $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;          $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;
1770          ## reconsume          ## reconsume
1771          redo A;          redo A;
1772        } elsif ($self->{next_char} == 0x0027) { # '        } elsif ($self->{nc} == 0x0027) { # '
1773          !!!cp (86);          !!!cp (86);
1774          $self->{state} = ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE;          $self->{state} = ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE;
1775          !!!next-input-character;          !!!next-input-character;
1776          redo A;          redo A;
1777        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1778          !!!parse-error (type => 'empty unquoted attribute value');          !!!parse-error (type => 'empty unquoted attribute value');
1779          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1780            !!!cp (87);            !!!cp (87);
1781            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1782          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1783            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1784            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1785              !!!cp (88);              !!!cp (88);
1786              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1787            } else {            } else {
# Line 1659  sub _get_next_token ($) { Line 1789  sub _get_next_token ($) {
1789              !!!cp (89);              !!!cp (89);
1790            }            }
1791          } else {          } else {
1792            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1793          }          }
1794          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1795          !!!next-input-character;          !!!next-input-character;
1796    
1797          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1798    
1799          redo A;          redo A;
1800        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1801          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1802          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1803            !!!cp (90);            !!!cp (90);
1804            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1805          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1806            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1807            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1808              !!!cp (91);              !!!cp (91);
1809              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1810            } else {            } else {
# Line 1682  sub _get_next_token ($) { Line 1812  sub _get_next_token ($) {
1812              !!!cp (92);              !!!cp (92);
1813            }            }
1814          } else {          } else {
1815            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1816          }          }
1817          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1818          ## reconsume          ## reconsume
1819    
1820          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1821    
1822          redo A;          redo A;
1823        } else {        } else {
1824          if ($self->{next_char} == 0x003D) { # =          if ($self->{nc} == 0x003D) { # =
1825            !!!cp (93);            !!!cp (93);
1826            !!!parse-error (type => 'bad attribute value');            !!!parse-error (type => 'bad attribute value');
1827          } else {          } else {
1828            !!!cp (94);            !!!cp (94);
1829          }          }
1830          $self->{current_attribute}->{value} .= chr ($self->{next_char});          $self->{ca}->{value} .= chr ($self->{nc});
1831          $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;          $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;
1832          !!!next-input-character;          !!!next-input-character;
1833          redo A;          redo A;
1834        }        }
1835      } elsif ($self->{state} == ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE) {      } elsif ($self->{state} == ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE) {
1836        if ($self->{next_char} == 0x0022) { # "        if ($self->{nc} == 0x0022) { # "
1837          !!!cp (95);          !!!cp (95);
1838          $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;          $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;
1839          !!!next-input-character;          !!!next-input-character;
1840          redo A;          redo A;
1841        } elsif ($self->{next_char} == 0x0026) { # &        } elsif ($self->{nc} == 0x0026) { # &
1842          !!!cp (96);          !!!cp (96);
         $self->{last_attribute_value_state} = $self->{state};  
1843          ## NOTE: In the spec, the tokenizer is switched to the          ## NOTE: In the spec, the tokenizer is switched to the
1844          ## "entity in attribute value state".  In this implementation, the          ## "entity in attribute value state".  In this implementation, the
1845          ## tokenizer is switched to the |ENTITY_STATE|, which is an          ## tokenizer is switched to the |ENTITY_STATE|, which is an
1846          ## implementation of the "consume a character reference" algorithm.          ## implementation of the "consume a character reference" algorithm.
1847          #$self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE;          $self->{prev_state} = $self->{state};
1848          $self->{entity_in_attr} = 1;          $self->{entity_add} = 0x0022; # "
         $self->{entity_additional} = 0x0022; # "  
1849          $self->{state} = ENTITY_STATE;          $self->{state} = ENTITY_STATE;
1850          !!!next-input-character;          !!!next-input-character;
1851          redo A;          redo A;
1852        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1853          !!!parse-error (type => 'unclosed attribute value');          !!!parse-error (type => 'unclosed attribute value');
1854          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1855            !!!cp (97);            !!!cp (97);
1856            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1857          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1858            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1859            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1860              !!!cp (98);              !!!cp (98);
1861              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1862            } else {            } else {
# Line 1736  sub _get_next_token ($) { Line 1864  sub _get_next_token ($) {
1864              !!!cp (99);              !!!cp (99);
1865            }            }
1866          } else {          } else {
1867            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1868          }          }
1869          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1870          ## reconsume          ## reconsume
1871    
1872          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1873    
1874          redo A;          redo A;
1875        } else {        } else {
1876          !!!cp (100);          !!!cp (100);
1877          $self->{current_attribute}->{value} .= chr ($self->{next_char});          $self->{ca}->{value} .= chr ($self->{nc});
1878            $self->{read_until}->($self->{ca}->{value},
1879                                  q["&],
1880                                  length $self->{ca}->{value});
1881    
1882          ## Stay in the state          ## Stay in the state
1883          !!!next-input-character;          !!!next-input-character;
1884          redo A;          redo A;
1885        }        }
1886      } elsif ($self->{state} == ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE) {      } elsif ($self->{state} == ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE) {
1887        if ($self->{next_char} == 0x0027) { # '        if ($self->{nc} == 0x0027) { # '
1888          !!!cp (101);          !!!cp (101);
1889          $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;          $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;
1890          !!!next-input-character;          !!!next-input-character;
1891          redo A;          redo A;
1892        } elsif ($self->{next_char} == 0x0026) { # &        } elsif ($self->{nc} == 0x0026) { # &
1893          !!!cp (102);          !!!cp (102);
         $self->{last_attribute_value_state} = $self->{state};  
1894          ## NOTE: In the spec, the tokenizer is switched to the          ## NOTE: In the spec, the tokenizer is switched to the
1895          ## "entity in attribute value state".  In this implementation, the          ## "entity in attribute value state".  In this implementation, the
1896          ## tokenizer is switched to the |ENTITY_STATE|, which is an          ## tokenizer is switched to the |ENTITY_STATE|, which is an
1897          ## implementation of the "consume a character reference" algorithm.          ## implementation of the "consume a character reference" algorithm.
1898          #$self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE;          $self->{entity_add} = 0x0027; # '
1899          $self->{entity_in_attr} = 1;          $self->{prev_state} = $self->{state};
         $self->{entity_additional} = 0x0027; # '  
1900          $self->{state} = ENTITY_STATE;          $self->{state} = ENTITY_STATE;
1901          !!!next-input-character;          !!!next-input-character;
1902          redo A;          redo A;
1903        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1904          !!!parse-error (type => 'unclosed attribute value');          !!!parse-error (type => 'unclosed attribute value');
1905          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1906            !!!cp (103);            !!!cp (103);
1907            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1908          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1909            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1910            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1911              !!!cp (104);              !!!cp (104);
1912              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1913            } else {            } else {
# Line 1785  sub _get_next_token ($) { Line 1915  sub _get_next_token ($) {
1915              !!!cp (105);              !!!cp (105);
1916            }            }
1917          } else {          } else {
1918            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1919          }          }
1920          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1921          ## reconsume          ## reconsume
1922    
1923          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1924    
1925          redo A;          redo A;
1926        } else {        } else {
1927          !!!cp (106);          !!!cp (106);
1928          $self->{current_attribute}->{value} .= chr ($self->{next_char});          $self->{ca}->{value} .= chr ($self->{nc});
1929            $self->{read_until}->($self->{ca}->{value},
1930                                  q['&],
1931                                  length $self->{ca}->{value});
1932    
1933          ## Stay in the state          ## Stay in the state
1934          !!!next-input-character;          !!!next-input-character;
1935          redo A;          redo A;
1936        }        }
1937      } elsif ($self->{state} == ATTRIBUTE_VALUE_UNQUOTED_STATE) {      } elsif ($self->{state} == ATTRIBUTE_VALUE_UNQUOTED_STATE) {
1938        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # HT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
1939          !!!cp (107);          !!!cp (107);
1940          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1941          !!!next-input-character;          !!!next-input-character;
1942          redo A;          redo A;
1943        } elsif ($self->{next_char} == 0x0026) { # &        } elsif ($self->{nc} == 0x0026) { # &
1944          !!!cp (108);          !!!cp (108);
         $self->{last_attribute_value_state} = $self->{state};  
1945          ## NOTE: In the spec, the tokenizer is switched to the          ## NOTE: In the spec, the tokenizer is switched to the
1946          ## "entity in attribute value state".  In this implementation, the          ## "entity in attribute value state".  In this implementation, the
1947          ## tokenizer is switched to the |ENTITY_STATE|, which is an          ## tokenizer is switched to the |ENTITY_STATE|, which is an
1948          ## implementation of the "consume a character reference" algorithm.          ## implementation of the "consume a character reference" algorithm.
1949          #$self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE;          $self->{entity_add} = -1;
1950          $self->{entity_in_attr} = 1;          $self->{prev_state} = $self->{state};
         $self->{entity_additional} = -1;  
1951          $self->{state} = ENTITY_STATE;          $self->{state} = ENTITY_STATE;
1952          !!!next-input-character;          !!!next-input-character;
1953          redo A;          redo A;
1954        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1955          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1956            !!!cp (109);            !!!cp (109);
1957            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1958          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1959            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1960            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1961              !!!cp (110);              !!!cp (110);
1962              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1963            } else {            } else {
# Line 1837  sub _get_next_token ($) { Line 1965  sub _get_next_token ($) {
1965              !!!cp (111);              !!!cp (111);
1966            }            }
1967          } else {          } else {
1968            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1969          }          }
1970          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1971          !!!next-input-character;          !!!next-input-character;
1972    
1973          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1974    
1975          redo A;          redo A;
1976        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1977          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1978          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1979            !!!cp (112);            !!!cp (112);
1980            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1981          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1982            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1983            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1984              !!!cp (113);              !!!cp (113);
1985              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1986            } else {            } else {
# Line 1860  sub _get_next_token ($) { Line 1988  sub _get_next_token ($) {
1988              !!!cp (114);              !!!cp (114);
1989            }            }
1990          } else {          } else {
1991            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1992          }          }
1993          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1994          ## reconsume          ## reconsume
1995    
1996          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1997    
1998          redo A;          redo A;
1999        } else {        } else {
# Line 1873  sub _get_next_token ($) { Line 2001  sub _get_next_token ($) {
2001               0x0022 => 1, # "               0x0022 => 1, # "
2002               0x0027 => 1, # '               0x0027 => 1, # '
2003               0x003D => 1, # =               0x003D => 1, # =
2004              }->{$self->{next_char}}) {              }->{$self->{nc}}) {
2005            !!!cp (115);            !!!cp (115);
2006            !!!parse-error (type => 'bad attribute value');            !!!parse-error (type => 'bad attribute value');
2007          } else {          } else {
2008            !!!cp (116);            !!!cp (116);
2009          }          }
2010          $self->{current_attribute}->{value} .= chr ($self->{next_char});          $self->{ca}->{value} .= chr ($self->{nc});
2011            $self->{read_until}->($self->{ca}->{value},
2012                                  q["'=& >],
2013                                  length $self->{ca}->{value});
2014    
2015          ## Stay in the state          ## Stay in the state
2016          !!!next-input-character;          !!!next-input-character;
2017          redo A;          redo A;
2018        }        }
     } elsif ($self->{state} == ENTITY_IN_ATTRIBUTE_VALUE_STATE) {  
       my $token = $self->{entity_return};  
   
       unless (defined $token) {  
         !!!cp (117);  
         $self->{current_attribute}->{value} .= '&';  
       } else {  
         !!!cp (118);  
         $self->{current_attribute}->{value} .= $token->{data};  
         $self->{current_attribute}->{has_reference} = $token->{has_reference};  
         ## ISSUE: spec says "append the returned character token to the current attribute's value"  
       }  
   
       $self->{state} = $self->{last_attribute_value_state};  
       # next-input-character is already done  
       redo A;  
2019      } elsif ($self->{state} == AFTER_ATTRIBUTE_VALUE_QUOTED_STATE) {      } elsif ($self->{state} == AFTER_ATTRIBUTE_VALUE_QUOTED_STATE) {
2020        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
2021          !!!cp (118);          !!!cp (118);
2022          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
2023          !!!next-input-character;          !!!next-input-character;
2024          redo A;          redo A;
2025        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2026          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
2027            !!!cp (119);            !!!cp (119);
2028            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
2029          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
2030            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
2031            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
2032              !!!cp (120);              !!!cp (120);
2033              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
2034            } else {            } else {
# Line 1924  sub _get_next_token ($) { Line 2036  sub _get_next_token ($) {
2036              !!!cp (121);              !!!cp (121);
2037            }            }
2038          } else {          } else {
2039            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
2040          }          }
2041          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2042          !!!next-input-character;          !!!next-input-character;
2043    
2044          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
2045    
2046          redo A;          redo A;
2047        } elsif ($self->{next_char} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
2048          !!!cp (122);          !!!cp (122);
2049          $self->{state} = SELF_CLOSING_START_TAG_STATE;          $self->{state} = SELF_CLOSING_START_TAG_STATE;
2050          !!!next-input-character;          !!!next-input-character;
2051          redo A;          redo A;
2052        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2053          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
2054          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
2055            !!!cp (122.3);            !!!cp (122.3);
2056            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
2057          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
2058            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
2059              !!!cp (122.1);              !!!cp (122.1);
2060              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
2061            } else {            } else {
# Line 1951  sub _get_next_token ($) { Line 2063  sub _get_next_token ($) {
2063              !!!cp (122.2);              !!!cp (122.2);
2064            }            }
2065          } else {          } else {
2066            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
2067          }          }
2068          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2069          ## Reconsume.          ## Reconsume.
2070          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
2071          redo A;          redo A;
2072        } else {        } else {
2073          !!!cp ('124.1');          !!!cp ('124.1');
# Line 1965  sub _get_next_token ($) { Line 2077  sub _get_next_token ($) {
2077          redo A;          redo A;
2078        }        }
2079      } elsif ($self->{state} == SELF_CLOSING_START_TAG_STATE) {      } elsif ($self->{state} == SELF_CLOSING_START_TAG_STATE) {
2080        if ($self->{next_char} == 0x003E) { # >        if ($self->{nc} == 0x003E) { # >
2081          if ($self->{current_token}->{type} == END_TAG_TOKEN) {          if ($self->{ct}->{type} == END_TAG_TOKEN) {
2082            !!!cp ('124.2');            !!!cp ('124.2');
2083            !!!parse-error (type => 'nestc', token => $self->{current_token});            !!!parse-error (type => 'nestc', token => $self->{ct});
2084            ## TODO: Different type than slash in start tag            ## TODO: Different type than slash in start tag
2085            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
2086            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
2087              !!!cp ('124.4');              !!!cp ('124.4');
2088              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
2089            } else {            } else {
# Line 1986  sub _get_next_token ($) { Line 2098  sub _get_next_token ($) {
2098          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2099          !!!next-input-character;          !!!next-input-character;
2100    
2101          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
2102    
2103          redo A;          redo A;
2104        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2105          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
2106          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
2107            !!!cp (124.7);            !!!cp (124.7);
2108            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
2109          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
2110            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
2111              !!!cp (124.5);              !!!cp (124.5);
2112              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
2113            } else {            } else {
# Line 2003  sub _get_next_token ($) { Line 2115  sub _get_next_token ($) {
2115              !!!cp (124.6);              !!!cp (124.6);
2116            }            }
2117          } else {          } else {
2118            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
2119          }          }
2120          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2121          ## Reconsume.          ## Reconsume.
2122          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
2123          redo A;          redo A;
2124        } else {        } else {
2125          !!!cp ('124.4');          !!!cp ('124.4');
# Line 2023  sub _get_next_token ($) { Line 2135  sub _get_next_token ($) {
2135        ## NOTE: Unlike spec's "bogus comment state", this implementation        ## NOTE: Unlike spec's "bogus comment state", this implementation
2136        ## consumes characters one-by-one basis.        ## consumes characters one-by-one basis.
2137                
2138        if ($self->{next_char} == 0x003E) { # >        if ($self->{nc} == 0x003E) { # >
2139          !!!cp (124);          !!!cp (124);
2140          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2141          !!!next-input-character;          !!!next-input-character;
2142    
2143          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2144          redo A;          redo A;
2145        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2146          !!!cp (125);          !!!cp (125);
2147          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2148          ## reconsume          ## reconsume
2149    
2150          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2151          redo A;          redo A;
2152        } else {        } else {
2153          !!!cp (126);          !!!cp (126);
2154          $self->{current_token}->{data} .= chr ($self->{next_char}); # comment          $self->{ct}->{data} .= chr ($self->{nc}); # comment
2155            $self->{read_until}->($self->{ct}->{data},
2156                                  q[>],
2157                                  length $self->{ct}->{data});
2158    
2159          ## Stay in the state.          ## Stay in the state.
2160          !!!next-input-character;          !!!next-input-character;
2161          redo A;          redo A;
# Line 2047  sub _get_next_token ($) { Line 2163  sub _get_next_token ($) {
2163      } elsif ($self->{state} == MARKUP_DECLARATION_OPEN_STATE) {      } elsif ($self->{state} == MARKUP_DECLARATION_OPEN_STATE) {
2164        ## (only happen if PCDATA state)        ## (only happen if PCDATA state)
2165                
2166        if ($self->{next_char} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2167          !!!cp (133);          !!!cp (133);
2168          $self->{state} = MD_HYPHEN_STATE;          $self->{state} = MD_HYPHEN_STATE;
2169          !!!next-input-character;          !!!next-input-character;
2170          redo A;          redo A;
2171        } elsif ($self->{next_char} == 0x0044 or # D        } elsif ($self->{nc} == 0x0044 or # D
2172                 $self->{next_char} == 0x0064) { # d                 $self->{nc} == 0x0064) { # d
2173          ## ASCII case-insensitive.          ## ASCII case-insensitive.
2174          !!!cp (130);          !!!cp (130);
2175          $self->{state} = MD_DOCTYPE_STATE;          $self->{state} = MD_DOCTYPE_STATE;
2176          $self->{state_keyword} = chr $self->{next_char};          $self->{s_kwd} = chr $self->{nc};
2177          !!!next-input-character;          !!!next-input-character;
2178          redo A;          redo A;
2179        } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM and        } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM and
2180                 $self->{open_elements}->[-1]->[1] & FOREIGN_EL and                 $self->{open_elements}->[-1]->[1] & FOREIGN_EL and
2181                 $self->{next_char} == 0x005B) { # [                 $self->{nc} == 0x005B) { # [
2182          !!!cp (135.4);                          !!!cp (135.4);                
2183          $self->{state} = MD_CDATA_STATE;          $self->{state} = MD_CDATA_STATE;
2184          $self->{state_keyword} = '[';          $self->{s_kwd} = '[';
2185          !!!next-input-character;          !!!next-input-character;
2186          redo A;          redo A;
2187        } else {        } else {
# Line 2077  sub _get_next_token ($) { Line 2193  sub _get_next_token ($) {
2193                        column => $self->{column_prev} - 1);                        column => $self->{column_prev} - 1);
2194        ## Reconsume.        ## Reconsume.
2195        $self->{state} = BOGUS_COMMENT_STATE;        $self->{state} = BOGUS_COMMENT_STATE;
2196        $self->{current_token} = {type => COMMENT_TOKEN, data => '',        $self->{ct} = {type => COMMENT_TOKEN, data => '',
2197                                  line => $self->{line_prev},                                  line => $self->{line_prev},
2198                                  column => $self->{column_prev} - 1,                                  column => $self->{column_prev} - 1,
2199                                 };                                 };
2200        redo A;        redo A;
2201      } elsif ($self->{state} == MD_HYPHEN_STATE) {      } elsif ($self->{state} == MD_HYPHEN_STATE) {
2202        if ($self->{next_char} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2203          !!!cp (127);          !!!cp (127);
2204          $self->{current_token} = {type => COMMENT_TOKEN, data => '',          $self->{ct} = {type => COMMENT_TOKEN, data => '',
2205                                    line => $self->{line_prev},                                    line => $self->{line_prev},
2206                                    column => $self->{column_prev} - 2,                                    column => $self->{column_prev} - 2,
2207                                   };                                   };
# Line 2099  sub _get_next_token ($) { Line 2215  sub _get_next_token ($) {
2215                          column => $self->{column_prev} - 2);                          column => $self->{column_prev} - 2);
2216          $self->{state} = BOGUS_COMMENT_STATE;          $self->{state} = BOGUS_COMMENT_STATE;
2217          ## Reconsume.          ## Reconsume.
2218          $self->{current_token} = {type => COMMENT_TOKEN,          $self->{ct} = {type => COMMENT_TOKEN,
2219                                    data => '-',                                    data => '-',
2220                                    line => $self->{line_prev},                                    line => $self->{line_prev},
2221                                    column => $self->{column_prev} - 2,                                    column => $self->{column_prev} - 2,
# Line 2108  sub _get_next_token ($) { Line 2224  sub _get_next_token ($) {
2224        }        }
2225      } elsif ($self->{state} == MD_DOCTYPE_STATE) {      } elsif ($self->{state} == MD_DOCTYPE_STATE) {
2226        ## ASCII case-insensitive.        ## ASCII case-insensitive.
2227        if ($self->{next_char} == [        if ($self->{nc} == [
2228              undef,              undef,
2229              0x004F, # O              0x004F, # O
2230              0x0043, # C              0x0043, # C
2231              0x0054, # T              0x0054, # T
2232              0x0059, # Y              0x0059, # Y
2233              0x0050, # P              0x0050, # P
2234            ]->[length $self->{state_keyword}] or            ]->[length $self->{s_kwd}] or
2235            $self->{next_char} == [            $self->{nc} == [
2236              undef,              undef,
2237              0x006F, # o              0x006F, # o
2238              0x0063, # c              0x0063, # c
2239              0x0074, # t              0x0074, # t
2240              0x0079, # y              0x0079, # y
2241              0x0070, # p              0x0070, # p
2242            ]->[length $self->{state_keyword}]) {            ]->[length $self->{s_kwd}]) {
2243          !!!cp (131);          !!!cp (131);
2244          ## Stay in the state.          ## Stay in the state.
2245          $self->{state_keyword} .= chr $self->{next_char};          $self->{s_kwd} .= chr $self->{nc};
2246          !!!next-input-character;          !!!next-input-character;
2247          redo A;          redo A;
2248        } elsif ((length $self->{state_keyword}) == 6 and        } elsif ((length $self->{s_kwd}) == 6 and
2249                 ($self->{next_char} == 0x0045 or # E                 ($self->{nc} == 0x0045 or # E
2250                  $self->{next_char} == 0x0065)) { # e                  $self->{nc} == 0x0065)) { # e
2251          !!!cp (129);          !!!cp (129);
2252          $self->{state} = DOCTYPE_STATE;          $self->{state} = DOCTYPE_STATE;
2253          $self->{current_token} = {type => DOCTYPE_TOKEN,          $self->{ct} = {type => DOCTYPE_TOKEN,
2254                                    quirks => 1,                                    quirks => 1,
2255                                    line => $self->{line_prev},                                    line => $self->{line_prev},
2256                                    column => $self->{column_prev} - 7,                                    column => $self->{column_prev} - 7,
# Line 2145  sub _get_next_token ($) { Line 2261  sub _get_next_token ($) {
2261          !!!cp (132);                  !!!cp (132);        
2262          !!!parse-error (type => 'bogus comment',          !!!parse-error (type => 'bogus comment',
2263                          line => $self->{line_prev},                          line => $self->{line_prev},
2264                          column => $self->{column_prev} - 1 - length $self->{state_keyword});                          column => $self->{column_prev} - 1 - length $self->{s_kwd});
2265          $self->{state} = BOGUS_COMMENT_STATE;          $self->{state} = BOGUS_COMMENT_STATE;
2266          ## Reconsume.          ## Reconsume.
2267          $self->{current_token} = {type => COMMENT_TOKEN,          $self->{ct} = {type => COMMENT_TOKEN,
2268                                    data => $self->{state_keyword},                                    data => $self->{s_kwd},
2269                                    line => $self->{line_prev},                                    line => $self->{line_prev},
2270                                    column => $self->{column_prev} - 1 - length $self->{state_keyword},                                    column => $self->{column_prev} - 1 - length $self->{s_kwd},
2271                                   };                                   };
2272          redo A;          redo A;
2273        }        }
2274      } elsif ($self->{state} == MD_CDATA_STATE) {      } elsif ($self->{state} == MD_CDATA_STATE) {
2275        if ($self->{next_char} == {        if ($self->{nc} == {
2276              '[' => 0x0043, # C              '[' => 0x0043, # C
2277              '[C' => 0x0044, # D              '[C' => 0x0044, # D
2278              '[CD' => 0x0041, # A              '[CD' => 0x0041, # A
2279              '[CDA' => 0x0054, # T              '[CDA' => 0x0054, # T
2280              '[CDAT' => 0x0041, # A              '[CDAT' => 0x0041, # A
2281            }->{$self->{state_keyword}}) {            }->{$self->{s_kwd}}) {
2282          !!!cp (135.1);          !!!cp (135.1);
2283          ## Stay in the state.          ## Stay in the state.
2284          $self->{state_keyword} .= chr $self->{next_char};          $self->{s_kwd} .= chr $self->{nc};
2285          !!!next-input-character;          !!!next-input-character;
2286          redo A;          redo A;
2287        } elsif ($self->{state_keyword} eq '[CDATA' and        } elsif ($self->{s_kwd} eq '[CDATA' and
2288                 $self->{next_char} == 0x005B) { # [                 $self->{nc} == 0x005B) { # [
2289          !!!cp (135.2);          !!!cp (135.2);
2290          $self->{current_token} = {type => CHARACTER_TOKEN,          $self->{ct} = {type => CHARACTER_TOKEN,
2291                                    data => '',                                    data => '',
2292                                    line => $self->{line_prev},                                    line => $self->{line_prev},
2293                                    column => $self->{column_prev} - 7};                                    column => $self->{column_prev} - 7};
# Line 2182  sub _get_next_token ($) { Line 2298  sub _get_next_token ($) {
2298          !!!cp (135.3);          !!!cp (135.3);
2299          !!!parse-error (type => 'bogus comment',          !!!parse-error (type => 'bogus comment',
2300                          line => $self->{line_prev},                          line => $self->{line_prev},
2301                          column => $self->{column_prev} - 1 - length $self->{state_keyword});                          column => $self->{column_prev} - 1 - length $self->{s_kwd});
2302          $self->{state} = BOGUS_COMMENT_STATE;          $self->{state} = BOGUS_COMMENT_STATE;
2303          ## Reconsume.          ## Reconsume.
2304          $self->{current_token} = {type => COMMENT_TOKEN,          $self->{ct} = {type => COMMENT_TOKEN,
2305                                    data => $self->{state_keyword},                                    data => $self->{s_kwd},
2306                                    line => $self->{line_prev},                                    line => $self->{line_prev},
2307                                    column => $self->{column_prev} - 1 - length $self->{state_keyword},                                    column => $self->{column_prev} - 1 - length $self->{s_kwd},
2308                                   };                                   };
2309          redo A;          redo A;
2310        }        }
2311      } elsif ($self->{state} == COMMENT_START_STATE) {      } elsif ($self->{state} == COMMENT_START_STATE) {
2312        if ($self->{next_char} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2313          !!!cp (137);          !!!cp (137);
2314          $self->{state} = COMMENT_START_DASH_STATE;          $self->{state} = COMMENT_START_DASH_STATE;
2315          !!!next-input-character;          !!!next-input-character;
2316          redo A;          redo A;
2317        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2318          !!!cp (138);          !!!cp (138);
2319          !!!parse-error (type => 'bogus comment');          !!!parse-error (type => 'bogus comment');
2320          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2321          !!!next-input-character;          !!!next-input-character;
2322    
2323          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2324    
2325          redo A;          redo A;
2326        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2327          !!!cp (139);          !!!cp (139);
2328          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2329          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2330          ## reconsume          ## reconsume
2331    
2332          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2333    
2334          redo A;          redo A;
2335        } else {        } else {
2336          !!!cp (140);          !!!cp (140);
2337          $self->{current_token}->{data} # comment          $self->{ct}->{data} # comment
2338              .= chr ($self->{next_char});              .= chr ($self->{nc});
2339          $self->{state} = COMMENT_STATE;          $self->{state} = COMMENT_STATE;
2340          !!!next-input-character;          !!!next-input-character;
2341          redo A;          redo A;
2342        }        }
2343      } elsif ($self->{state} == COMMENT_START_DASH_STATE) {      } elsif ($self->{state} == COMMENT_START_DASH_STATE) {
2344        if ($self->{next_char} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2345          !!!cp (141);          !!!cp (141);
2346          $self->{state} = COMMENT_END_STATE;          $self->{state} = COMMENT_END_STATE;
2347          !!!next-input-character;          !!!next-input-character;
2348          redo A;          redo A;
2349        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2350          !!!cp (142);          !!!cp (142);
2351          !!!parse-error (type => 'bogus comment');          !!!parse-error (type => 'bogus comment');
2352          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2353          !!!next-input-character;          !!!next-input-character;
2354    
2355          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2356    
2357          redo A;          redo A;
2358        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2359          !!!cp (143);          !!!cp (143);
2360          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2361          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2362          ## reconsume          ## reconsume
2363    
2364          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2365    
2366          redo A;          redo A;
2367        } else {        } else {
2368          !!!cp (144);          !!!cp (144);
2369          $self->{current_token}->{data} # comment          $self->{ct}->{data} # comment
2370              .= '-' . chr ($self->{next_char});              .= '-' . chr ($self->{nc});
2371          $self->{state} = COMMENT_STATE;          $self->{state} = COMMENT_STATE;
2372          !!!next-input-character;          !!!next-input-character;
2373          redo A;          redo A;
2374        }        }
2375      } elsif ($self->{state} == COMMENT_STATE) {      } elsif ($self->{state} == COMMENT_STATE) {
2376        if ($self->{next_char} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2377          !!!cp (145);          !!!cp (145);
2378          $self->{state} = COMMENT_END_DASH_STATE;          $self->{state} = COMMENT_END_DASH_STATE;
2379          !!!next-input-character;          !!!next-input-character;
2380          redo A;          redo A;
2381        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2382          !!!cp (146);          !!!cp (146);
2383          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2384          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2385          ## reconsume          ## reconsume
2386    
2387          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2388    
2389          redo A;          redo A;
2390        } else {        } else {
2391          !!!cp (147);          !!!cp (147);
2392          $self->{current_token}->{data} .= chr ($self->{next_char}); # comment          $self->{ct}->{data} .= chr ($self->{nc}); # comment
2393            $self->{read_until}->($self->{ct}->{data},
2394                                  q[-],
2395                                  length $self->{ct}->{data});
2396    
2397          ## Stay in the state          ## Stay in the state
2398          !!!next-input-character;          !!!next-input-character;
2399          redo A;          redo A;
2400        }        }
2401      } elsif ($self->{state} == COMMENT_END_DASH_STATE) {      } elsif ($self->{state} == COMMENT_END_DASH_STATE) {
2402        if ($self->{next_char} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2403          !!!cp (148);          !!!cp (148);
2404          $self->{state} = COMMENT_END_STATE;          $self->{state} = COMMENT_END_STATE;
2405          !!!next-input-character;          !!!next-input-character;
2406          redo A;          redo A;
2407        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2408          !!!cp (149);          !!!cp (149);
2409          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2410          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2411          ## reconsume          ## reconsume
2412    
2413          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2414    
2415          redo A;          redo A;
2416        } else {        } else {
2417          !!!cp (150);          !!!cp (150);
2418          $self->{current_token}->{data} .= '-' . chr ($self->{next_char}); # comment          $self->{ct}->{data} .= '-' . chr ($self->{nc}); # comment
2419          $self->{state} = COMMENT_STATE;          $self->{state} = COMMENT_STATE;
2420          !!!next-input-character;          !!!next-input-character;
2421          redo A;          redo A;
2422        }        }
2423      } elsif ($self->{state} == COMMENT_END_STATE) {      } elsif ($self->{state} == COMMENT_END_STATE) {
2424        if ($self->{next_char} == 0x003E) { # >        if ($self->{nc} == 0x003E) { # >
2425          !!!cp (151);          !!!cp (151);
2426          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2427          !!!next-input-character;          !!!next-input-character;
2428    
2429          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2430    
2431          redo A;          redo A;
2432        } elsif ($self->{next_char} == 0x002D) { # -        } elsif ($self->{nc} == 0x002D) { # -
2433          !!!cp (152);          !!!cp (152);
2434          !!!parse-error (type => 'dash in comment',          !!!parse-error (type => 'dash in comment',
2435                          line => $self->{line_prev},                          line => $self->{line_prev},
2436                          column => $self->{column_prev});                          column => $self->{column_prev});
2437          $self->{current_token}->{data} .= '-'; # comment          $self->{ct}->{data} .= '-'; # comment
2438          ## Stay in the state          ## Stay in the state
2439          !!!next-input-character;          !!!next-input-character;
2440          redo A;          redo A;
2441        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2442          !!!cp (153);          !!!cp (153);
2443          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2444          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2445          ## reconsume          ## reconsume
2446    
2447          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2448    
2449          redo A;          redo A;
2450        } else {        } else {
# Line 2332  sub _get_next_token ($) { Line 2452  sub _get_next_token ($) {
2452          !!!parse-error (type => 'dash in comment',          !!!parse-error (type => 'dash in comment',
2453                          line => $self->{line_prev},                          line => $self->{line_prev},
2454                          column => $self->{column_prev});                          column => $self->{column_prev});
2455          $self->{current_token}->{data} .= '--' . chr ($self->{next_char}); # comment          $self->{ct}->{data} .= '--' . chr ($self->{nc}); # comment
2456          $self->{state} = COMMENT_STATE;          $self->{state} = COMMENT_STATE;
2457          !!!next-input-character;          !!!next-input-character;
2458          redo A;          redo A;
2459        }        }
2460      } elsif ($self->{state} == DOCTYPE_STATE) {      } elsif ($self->{state} == DOCTYPE_STATE) {
2461        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
2462          !!!cp (155);          !!!cp (155);
2463          $self->{state} = BEFORE_DOCTYPE_NAME_STATE;          $self->{state} = BEFORE_DOCTYPE_NAME_STATE;
2464          !!!next-input-character;          !!!next-input-character;
# Line 2355  sub _get_next_token ($) { Line 2471  sub _get_next_token ($) {
2471          redo A;          redo A;
2472        }        }
2473      } elsif ($self->{state} == BEFORE_DOCTYPE_NAME_STATE) {      } elsif ($self->{state} == BEFORE_DOCTYPE_NAME_STATE) {
2474        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
2475          !!!cp (157);          !!!cp (157);
2476          ## Stay in the state          ## Stay in the state
2477          !!!next-input-character;          !!!next-input-character;
2478          redo A;          redo A;
2479        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2480          !!!cp (158);          !!!cp (158);
2481          !!!parse-error (type => 'no DOCTYPE name');          !!!parse-error (type => 'no DOCTYPE name');
2482          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2483          !!!next-input-character;          !!!next-input-character;
2484    
2485          !!!emit ($self->{current_token}); # DOCTYPE (quirks)          !!!emit ($self->{ct}); # DOCTYPE (quirks)
2486    
2487          redo A;          redo A;
2488        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2489          !!!cp (159);          !!!cp (159);
2490          !!!parse-error (type => 'no DOCTYPE name');          !!!parse-error (type => 'no DOCTYPE name');
2491          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2492          ## reconsume          ## reconsume
2493    
2494          !!!emit ($self->{current_token}); # DOCTYPE (quirks)          !!!emit ($self->{ct}); # DOCTYPE (quirks)
2495    
2496          redo A;          redo A;
2497        } else {        } else {
2498          !!!cp (160);          !!!cp (160);
2499          $self->{current_token}->{name} = chr $self->{next_char};          $self->{ct}->{name} = chr $self->{nc};
2500          delete $self->{current_token}->{quirks};          delete $self->{ct}->{quirks};
 ## ISSUE: "Set the token's name name to the" in the spec  
2501          $self->{state} = DOCTYPE_NAME_STATE;          $self->{state} = DOCTYPE_NAME_STATE;
2502          !!!next-input-character;          !!!next-input-character;
2503          redo A;          redo A;
2504        }        }
2505      } elsif ($self->{state} == DOCTYPE_NAME_STATE) {      } elsif ($self->{state} == DOCTYPE_NAME_STATE) {
2506  ## ISSUE: Redundant "First," in the spec.  ## ISSUE: Redundant "First," in the spec.
2507        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
2508          !!!cp (161);          !!!cp (161);
2509          $self->{state} = AFTER_DOCTYPE_NAME_STATE;          $self->{state} = AFTER_DOCTYPE_NAME_STATE;
2510          !!!next-input-character;          !!!next-input-character;
2511          redo A;          redo A;
2512        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2513          !!!cp (162);          !!!cp (162);
2514          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2515          !!!next-input-character;          !!!next-input-character;
2516    
2517          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2518    
2519          redo A;          redo A;
2520        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2521          !!!cp (163);          !!!cp (163);
2522          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2523          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2524          ## reconsume          ## reconsume
2525    
2526          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2527          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2528    
2529          redo A;          redo A;
2530        } else {        } else {
2531          !!!cp (164);          !!!cp (164);
2532          $self->{current_token}->{name}          $self->{ct}->{name}
2533            .= chr ($self->{next_char}); # DOCTYPE            .= chr ($self->{nc}); # DOCTYPE
2534          ## Stay in the state          ## Stay in the state
2535          !!!next-input-character;          !!!next-input-character;
2536          redo A;          redo A;
2537        }        }
2538      } elsif ($self->{state} == AFTER_DOCTYPE_NAME_STATE) {      } elsif ($self->{state} == AFTER_DOCTYPE_NAME_STATE) {
2539        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
2540          !!!cp (165);          !!!cp (165);
2541          ## Stay in the state          ## Stay in the state
2542          !!!next-input-character;          !!!next-input-character;
2543          redo A;          redo A;
2544        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2545          !!!cp (166);          !!!cp (166);
2546          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2547          !!!next-input-character;          !!!next-input-character;
2548    
2549          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2550    
2551          redo A;          redo A;
2552        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2553          !!!cp (167);          !!!cp (167);
2554          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2555          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2556          ## reconsume          ## reconsume
2557    
2558          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2559          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2560    
2561          redo A;          redo A;
2562        } elsif ($self->{next_char} == 0x0050 or # P        } elsif ($self->{nc} == 0x0050 or # P
2563                 $self->{next_char} == 0x0070) { # p                 $self->{nc} == 0x0070) { # p
2564          $self->{state} = PUBLIC_STATE;          $self->{state} = PUBLIC_STATE;
2565          $self->{state_keyword} = chr $self->{next_char};          $self->{s_kwd} = chr $self->{nc};
2566          !!!next-input-character;          !!!next-input-character;
2567          redo A;          redo A;
2568        } elsif ($self->{next_char} == 0x0053 or # S        } elsif ($self->{nc} == 0x0053 or # S
2569                 $self->{next_char} == 0x0073) { # s                 $self->{nc} == 0x0073) { # s
2570          $self->{state} = SYSTEM_STATE;          $self->{state} = SYSTEM_STATE;
2571          $self->{state_keyword} = chr $self->{next_char};          $self->{s_kwd} = chr $self->{nc};
2572          !!!next-input-character;          !!!next-input-character;
2573          redo A;          redo A;
2574        } else {        } else {
2575          !!!cp (180);          !!!cp (180);
2576          !!!parse-error (type => 'string after DOCTYPE name');          !!!parse-error (type => 'string after DOCTYPE name');
2577          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2578    
2579          $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2580          !!!next-input-character;          !!!next-input-character;
# Line 2479  sub _get_next_token ($) { Line 2582  sub _get_next_token ($) {
2582        }        }
2583      } elsif ($self->{state} == PUBLIC_STATE) {      } elsif ($self->{state} == PUBLIC_STATE) {
2584        ## ASCII case-insensitive        ## ASCII case-insensitive
2585        if ($self->{next_char} == [        if ($self->{nc} == [
2586              undef,              undef,
2587              0x0055, # U              0x0055, # U
2588              0x0042, # B              0x0042, # B
2589              0x004C, # L              0x004C, # L
2590              0x0049, # I              0x0049, # I
2591            ]->[length $self->{state_keyword}] or            ]->[length $self->{s_kwd}] or
2592            $self->{next_char} == [            $self->{nc} == [
2593              undef,              undef,
2594              0x0075, # u              0x0075, # u
2595              0x0062, # b              0x0062, # b
2596              0x006C, # l              0x006C, # l
2597              0x0069, # i              0x0069, # i
2598            ]->[length $self->{state_keyword}]) {            ]->[length $self->{s_kwd}]) {
2599          !!!cp (175);          !!!cp (175);
2600          ## Stay in the state.          ## Stay in the state.
2601          $self->{state_keyword} .= chr $self->{next_char};          $self->{s_kwd} .= chr $self->{nc};
2602          !!!next-input-character;          !!!next-input-character;
2603          redo A;          redo A;
2604        } elsif ((length $self->{state_keyword}) == 5 and        } elsif ((length $self->{s_kwd}) == 5 and
2605                 ($self->{next_char} == 0x0043 or # C                 ($self->{nc} == 0x0043 or # C
2606                  $self->{next_char} == 0x0063)) { # c                  $self->{nc} == 0x0063)) { # c
2607          !!!cp (168);          !!!cp (168);
2608          $self->{state} = BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE;          $self->{state} = BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2609          !!!next-input-character;          !!!next-input-character;
# Line 2509  sub _get_next_token ($) { Line 2612  sub _get_next_token ($) {
2612          !!!cp (169);          !!!cp (169);
2613          !!!parse-error (type => 'string after DOCTYPE name',          !!!parse-error (type => 'string after DOCTYPE name',
2614                          line => $self->{line_prev},                          line => $self->{line_prev},
2615                          column => $self->{column_prev} + 1 - length $self->{state_keyword});                          column => $self->{column_prev} + 1 - length $self->{s_kwd});
2616          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2617    
2618          $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2619          ## Reconsume.          ## Reconsume.
# Line 2518  sub _get_next_token ($) { Line 2621  sub _get_next_token ($) {
2621        }        }
2622      } elsif ($self->{state} == SYSTEM_STATE) {      } elsif ($self->{state} == SYSTEM_STATE) {
2623        ## ASCII case-insensitive        ## ASCII case-insensitive
2624        if ($self->{next_char} == [        if ($self->{nc} == [
2625              undef,              undef,
2626              0x0059, # Y              0x0059, # Y
2627              0x0053, # S              0x0053, # S
2628              0x0054, # T              0x0054, # T
2629              0x0045, # E              0x0045, # E
2630            ]->[length $self->{state_keyword}] or            ]->[length $self->{s_kwd}] or
2631            $self->{next_char} == [            $self->{nc} == [
2632              undef,              undef,
2633              0x0079, # y              0x0079, # y
2634              0x0073, # s              0x0073, # s
2635              0x0074, # t              0x0074, # t
2636              0x0065, # e              0x0065, # e
2637            ]->[length $self->{state_keyword}]) {            ]->[length $self->{s_kwd}]) {
2638          !!!cp (170);          !!!cp (170);
2639          ## Stay in the state.          ## Stay in the state.
2640          $self->{state_keyword} .= chr $self->{next_char};          $self->{s_kwd} .= chr $self->{nc};
2641          !!!next-input-character;          !!!next-input-character;
2642          redo A;          redo A;
2643        } elsif ((length $self->{state_keyword}) == 5 and        } elsif ((length $self->{s_kwd}) == 5 and
2644                 ($self->{next_char} == 0x004D or # M                 ($self->{nc} == 0x004D or # M
2645                  $self->{next_char} == 0x006D)) { # m                  $self->{nc} == 0x006D)) { # m
2646          !!!cp (171);          !!!cp (171);
2647          $self->{state} = BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE;          $self->{state} = BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2648          !!!next-input-character;          !!!next-input-character;
# Line 2548  sub _get_next_token ($) { Line 2651  sub _get_next_token ($) {
2651          !!!cp (172);          !!!cp (172);
2652          !!!parse-error (type => 'string after DOCTYPE name',          !!!parse-error (type => 'string after DOCTYPE name',
2653                          line => $self->{line_prev},                          line => $self->{line_prev},
2654                          column => $self->{column_prev} + 1 - length $self->{state_keyword});                          column => $self->{column_prev} + 1 - length $self->{s_kwd});
2655          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2656    
2657          $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2658          ## Reconsume.          ## Reconsume.
2659          redo A;          redo A;
2660        }        }
2661      } elsif ($self->{state} == BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {      } elsif ($self->{state} == BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {
2662        if ({        if ($is_space->{$self->{nc}}) {
             0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,  
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_char}}) {  
2663          !!!cp (181);          !!!cp (181);
2664          ## Stay in the state          ## Stay in the state
2665          !!!next-input-character;          !!!next-input-character;
2666          redo A;          redo A;
2667        } elsif ($self->{next_char} eq 0x0022) { # "        } elsif ($self->{nc} eq 0x0022) { # "
2668          !!!cp (182);          !!!cp (182);
2669          $self->{current_token}->{public_identifier} = ''; # DOCTYPE          $self->{ct}->{pubid} = ''; # DOCTYPE
2670          $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE;          $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE;
2671          !!!next-input-character;          !!!next-input-character;
2672          redo A;          redo A;
2673        } elsif ($self->{next_char} eq 0x0027) { # '        } elsif ($self->{nc} eq 0x0027) { # '
2674          !!!cp (183);          !!!cp (183);
2675          $self->{current_token}->{public_identifier} = ''; # DOCTYPE          $self->{ct}->{pubid} = ''; # DOCTYPE
2676          $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE;          $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE;
2677          !!!next-input-character;          !!!next-input-character;
2678          redo A;          redo A;
2679        } elsif ($self->{next_char} eq 0x003E) { # >        } elsif ($self->{nc} eq 0x003E) { # >
2680          !!!cp (184);          !!!cp (184);
2681          !!!parse-error (type => 'no PUBLIC literal');          !!!parse-error (type => 'no PUBLIC literal');
2682    
2683          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2684          !!!next-input-character;          !!!next-input-character;
2685    
2686          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2687          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2688    
2689          redo A;          redo A;
2690        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2691          !!!cp (185);          !!!cp (185);
2692          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2693    
2694          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2695          ## reconsume          ## reconsume
2696    
2697          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2698          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2699    
2700          redo A;          redo A;
2701        } else {        } else {
2702          !!!cp (186);          !!!cp (186);
2703          !!!parse-error (type => 'string after PUBLIC');          !!!parse-error (type => 'string after PUBLIC');
2704          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2705    
2706          $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2707          !!!next-input-character;          !!!next-input-character;
2708          redo A;          redo A;
2709        }        }
2710      } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE) {      } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE) {
2711        if ($self->{next_char} == 0x0022) { # "        if ($self->{nc} == 0x0022) { # "
2712          !!!cp (187);          !!!cp (187);
2713          $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;          $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2714          !!!next-input-character;          !!!next-input-character;
2715          redo A;          redo A;
2716        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2717          !!!cp (188);          !!!cp (188);
2718          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed PUBLIC literal');
2719    
2720          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2721          !!!next-input-character;          !!!next-input-character;
2722    
2723          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2724          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2725    
2726          redo A;          redo A;
2727        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2728          !!!cp (189);          !!!cp (189);
2729          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed PUBLIC literal');
2730    
2731          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2732          ## reconsume          ## reconsume
2733    
2734          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2735          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2736    
2737          redo A;          redo A;
2738        } else {        } else {
2739          !!!cp (190);          !!!cp (190);
2740          $self->{current_token}->{public_identifier} # DOCTYPE          $self->{ct}->{pubid} # DOCTYPE
2741              .= chr $self->{next_char};              .= chr $self->{nc};
2742            $self->{read_until}->($self->{ct}->{pubid}, q[">],
2743                                  length $self->{ct}->{pubid});
2744    
2745          ## Stay in the state          ## Stay in the state
2746          !!!next-input-character;          !!!next-input-character;
2747          redo A;          redo A;
2748        }        }
2749      } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE) {      } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE) {
2750        if ($self->{next_char} == 0x0027) { # '        if ($self->{nc} == 0x0027) { # '
2751          !!!cp (191);          !!!cp (191);
2752          $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;          $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2753          !!!next-input-character;          !!!next-input-character;
2754          redo A;          redo A;
2755        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2756          !!!cp (192);          !!!cp (192);
2757          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed PUBLIC literal');
2758    
2759          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2760          !!!next-input-character;          !!!next-input-character;
2761    
2762          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2763          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2764    
2765          redo A;          redo A;
2766        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2767          !!!cp (193);          !!!cp (193);
2768          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed PUBLIC literal');
2769    
2770          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2771          ## reconsume          ## reconsume
2772    
2773          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2774          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2775    
2776          redo A;          redo A;
2777        } else {        } else {
2778          !!!cp (194);          !!!cp (194);
2779          $self->{current_token}->{public_identifier} # DOCTYPE          $self->{ct}->{pubid} # DOCTYPE
2780              .= chr $self->{next_char};              .= chr $self->{nc};
2781            $self->{read_until}->($self->{ct}->{pubid}, q['>],
2782                                  length $self->{ct}->{pubid});
2783    
2784          ## Stay in the state          ## Stay in the state
2785          !!!next-input-character;          !!!next-input-character;
2786          redo A;          redo A;
2787        }        }
2788      } elsif ($self->{state} == AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {      } elsif ($self->{state} == AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {
2789        if ({        if ($is_space->{$self->{nc}}) {
             0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,  
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_char}}) {  
2790          !!!cp (195);          !!!cp (195);
2791          ## Stay in the state          ## Stay in the state
2792          !!!next-input-character;          !!!next-input-character;
2793          redo A;          redo A;
2794        } elsif ($self->{next_char} == 0x0022) { # "        } elsif ($self->{nc} == 0x0022) { # "
2795          !!!cp (196);          !!!cp (196);
2796          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          $self->{ct}->{sysid} = ''; # DOCTYPE
2797          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;
2798          !!!next-input-character;          !!!next-input-character;
2799          redo A;          redo A;
2800        } elsif ($self->{next_char} == 0x0027) { # '        } elsif ($self->{nc} == 0x0027) { # '
2801          !!!cp (197);          !!!cp (197);
2802          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          $self->{ct}->{sysid} = ''; # DOCTYPE
2803          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;
2804          !!!next-input-character;          !!!next-input-character;
2805          redo A;          redo A;
2806        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2807          !!!cp (198);          !!!cp (198);
2808          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2809          !!!next-input-character;          !!!next-input-character;
2810    
2811          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2812    
2813          redo A;          redo A;
2814        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2815          !!!cp (199);          !!!cp (199);
2816          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2817    
2818          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2819          ## reconsume          ## reconsume
2820    
2821          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2822          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2823    
2824          redo A;          redo A;
2825        } else {        } else {
2826          !!!cp (200);          !!!cp (200);
2827          !!!parse-error (type => 'string after PUBLIC literal');          !!!parse-error (type => 'string after PUBLIC literal');
2828          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2829    
2830          $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2831          !!!next-input-character;          !!!next-input-character;
2832          redo A;          redo A;
2833        }        }
2834      } elsif ($self->{state} == BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {      } elsif ($self->{state} == BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {
2835        if ({        if ($is_space->{$self->{nc}}) {
             0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,  
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_char}}) {  
2836          !!!cp (201);          !!!cp (201);
2837          ## Stay in the state          ## Stay in the state
2838          !!!next-input-character;          !!!next-input-character;
2839          redo A;          redo A;
2840        } elsif ($self->{next_char} == 0x0022) { # "        } elsif ($self->{nc} == 0x0022) { # "
2841          !!!cp (202);          !!!cp (202);
2842          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          $self->{ct}->{sysid} = ''; # DOCTYPE
2843          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;
2844          !!!next-input-character;          !!!next-input-character;
2845          redo A;          redo A;
2846        } elsif ($self->{next_char} == 0x0027) { # '        } elsif ($self->{nc} == 0x0027) { # '
2847          !!!cp (203);          !!!cp (203);
2848          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          $self->{ct}->{sysid} = ''; # DOCTYPE
2849          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;
2850          !!!next-input-character;          !!!next-input-character;
2851          redo A;          redo A;
2852        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2853          !!!cp (204);          !!!cp (204);
2854          !!!parse-error (type => 'no SYSTEM literal');          !!!parse-error (type => 'no SYSTEM literal');
2855          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2856          !!!next-input-character;          !!!next-input-character;
2857    
2858          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2859          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2860    
2861          redo A;          redo A;
2862        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2863          !!!cp (205);          !!!cp (205);
2864          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2865    
2866          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2867          ## reconsume          ## reconsume
2868    
2869          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2870          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2871    
2872          redo A;          redo A;
2873        } else {        } else {
2874          !!!cp (206);          !!!cp (206);
2875          !!!parse-error (type => 'string after SYSTEM');          !!!parse-error (type => 'string after SYSTEM');
2876          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2877    
2878          $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2879          !!!next-input-character;          !!!next-input-character;
2880          redo A;          redo A;
2881        }        }
2882      } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE) {      } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE) {
2883        if ($self->{next_char} == 0x0022) { # "        if ($self->{nc} == 0x0022) { # "
2884          !!!cp (207);          !!!cp (207);
2885          $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;          $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2886          !!!next-input-character;          !!!next-input-character;
2887          redo A;          redo A;
2888        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2889          !!!cp (208);          !!!cp (208);
2890          !!!parse-error (type => 'unclosed SYSTEM literal');          !!!parse-error (type => 'unclosed SYSTEM literal');
2891    
2892          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2893          !!!next-input-character;          !!!next-input-character;
2894    
2895          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2896          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2897    
2898          redo A;          redo A;
2899        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2900          !!!cp (209);          !!!cp (209);
2901          !!!parse-error (type => 'unclosed SYSTEM literal');          !!!parse-error (type => 'unclosed SYSTEM literal');
2902    
2903          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2904          ## reconsume          ## reconsume
2905    
2906          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2907          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2908    
2909          redo A;          redo A;
2910        } else {        } else {
2911          !!!cp (210);          !!!cp (210);
2912          $self->{current_token}->{system_identifier} # DOCTYPE          $self->{ct}->{sysid} # DOCTYPE
2913              .= chr $self->{next_char};              .= chr $self->{nc};
2914            $self->{read_until}->($self->{ct}->{sysid}, q[">],
2915                                  length $self->{ct}->{sysid});
2916    
2917          ## Stay in the state          ## Stay in the state
2918          !!!next-input-character;          !!!next-input-character;
2919          redo A;          redo A;
2920        }        }
2921      } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE) {      } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE) {
2922        if ($self->{next_char} == 0x0027) { # '        if ($self->{nc} == 0x0027) { # '
2923          !!!cp (211);          !!!cp (211);
2924          $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;          $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2925          !!!next-input-character;          !!!next-input-character;
2926          redo A;          redo A;
2927        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2928          !!!cp (212);          !!!cp (212);
2929          !!!parse-error (type => 'unclosed SYSTEM literal');          !!!parse-error (type => 'unclosed SYSTEM literal');
2930    
2931          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2932          !!!next-input-character;          !!!next-input-character;
2933    
2934          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2935          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2936    
2937          redo A;          redo A;
2938        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2939          !!!cp (213);          !!!cp (213);
2940          !!!parse-error (type => 'unclosed SYSTEM literal');          !!!parse-error (type => 'unclosed SYSTEM literal');
2941    
2942          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2943          ## reconsume          ## reconsume
2944    
2945          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2946          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2947    
2948          redo A;          redo A;
2949        } else {        } else {
2950          !!!cp (214);          !!!cp (214);
2951          $self->{current_token}->{system_identifier} # DOCTYPE          $self->{ct}->{sysid} # DOCTYPE
2952              .= chr $self->{next_char};              .= chr $self->{nc};
2953            $self->{read_until}->($self->{ct}->{sysid}, q['>],
2954                                  length $self->{ct}->{sysid});
2955    
2956          ## Stay in the state          ## Stay in the state
2957          !!!next-input-character;          !!!next-input-character;
2958          redo A;          redo A;
2959        }        }
2960      } elsif ($self->{state} == AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {      } elsif ($self->{state} == AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {
2961        if ({        if ($is_space->{$self->{nc}}) {
             0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,  
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_char}}) {  
2962          !!!cp (215);          !!!cp (215);
2963          ## Stay in the state          ## Stay in the state
2964          !!!next-input-character;          !!!next-input-character;
2965          redo A;          redo A;
2966        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2967          !!!cp (216);          !!!cp (216);
2968          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2969          !!!next-input-character;          !!!next-input-character;
2970    
2971          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2972    
2973          redo A;          redo A;
2974        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2975          !!!cp (217);          !!!cp (217);
2976          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2977          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2978          ## reconsume          ## reconsume
2979    
2980          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2981          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2982    
2983          redo A;          redo A;
2984        } else {        } else {
2985          !!!cp (218);          !!!cp (218);
2986          !!!parse-error (type => 'string after SYSTEM literal');          !!!parse-error (type => 'string after SYSTEM literal');
2987          #$self->{current_token}->{quirks} = 1;          #$self->{ct}->{quirks} = 1;
2988    
2989          $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2990          !!!next-input-character;          !!!next-input-character;
2991          redo A;          redo A;
2992        }        }
2993      } elsif ($self->{state} == BOGUS_DOCTYPE_STATE) {      } elsif ($self->{state} == BOGUS_DOCTYPE_STATE) {
2994        if ($self->{next_char} == 0x003E) { # >        if ($self->{nc} == 0x003E) { # >
2995          !!!cp (219);          !!!cp (219);
2996          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2997          !!!next-input-character;          !!!next-input-character;
2998    
2999          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
3000    
3001          redo A;          redo A;
3002        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
3003          !!!cp (220);          !!!cp (220);
         !!!parse-error (type => 'unclosed DOCTYPE');  
3004          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
3005          ## reconsume          ## reconsume
3006    
3007          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
3008    
3009          redo A;          redo A;
3010        } else {        } else {
3011          !!!cp (221);          !!!cp (221);
3012            my $s = '';
3013            $self->{read_until}->($s, q[>], 0);
3014    
3015          ## Stay in the state          ## Stay in the state
3016          !!!next-input-character;          !!!next-input-character;
3017          redo A;          redo A;
# Line 2916  sub _get_next_token ($) { Line 3021  sub _get_next_token ($) {
3021        ## by three states, |CDATA_SECTION_STATE|, |CDATA_SECTION_MSE1_STATE|,        ## by three states, |CDATA_SECTION_STATE|, |CDATA_SECTION_MSE1_STATE|,
3022        ## and |CDATA_SECTION_MSE2_STATE|.        ## and |CDATA_SECTION_MSE2_STATE|.
3023                
3024        if ($self->{next_char} == 0x005D) { # ]        if ($self->{nc} == 0x005D) { # ]
3025          !!!cp (221.1);          !!!cp (221.1);
3026          $self->{state} = CDATA_SECTION_MSE1_STATE;          $self->{state} = CDATA_SECTION_MSE1_STATE;
3027          !!!next-input-character;          !!!next-input-character;
3028          redo A;          redo A;
3029        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
3030          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
3031          !!!next-input-character;          !!!next-input-character;
3032          if (length $self->{current_token}->{data}) { # character          if (length $self->{ct}->{data}) { # character
3033            !!!cp (221.2);            !!!cp (221.2);
3034            !!!emit ($self->{current_token}); # character            !!!emit ($self->{ct}); # character
3035          } else {          } else {
3036            !!!cp (221.3);            !!!cp (221.3);
3037            ## No token to emit. $self->{current_token} is discarded.            ## No token to emit. $self->{ct} is discarded.
3038          }                  }        
3039          redo A;          redo A;
3040        } else {        } else {
3041          !!!cp (221.4);          !!!cp (221.4);
3042          $self->{current_token}->{data} .= chr $self->{next_char};          $self->{ct}->{data} .= chr $self->{nc};
3043            $self->{read_until}->($self->{ct}->{data},
3044                                  q<]>,
3045                                  length $self->{ct}->{data});
3046    
3047          ## Stay in the state.          ## Stay in the state.
3048          !!!next-input-character;          !!!next-input-character;
3049          redo A;          redo A;
# Line 2942  sub _get_next_token ($) { Line 3051  sub _get_next_token ($) {
3051    
3052        ## ISSUE: "text tokens" in spec.        ## ISSUE: "text tokens" in spec.
3053      } elsif ($self->{state} == CDATA_SECTION_MSE1_STATE) {      } elsif ($self->{state} == CDATA_SECTION_MSE1_STATE) {
3054        if ($self->{next_char} == 0x005D) { # ]        if ($self->{nc} == 0x005D) { # ]
3055          !!!cp (221.5);          !!!cp (221.5);
3056          $self->{state} = CDATA_SECTION_MSE2_STATE;          $self->{state} = CDATA_SECTION_MSE2_STATE;
3057          !!!next-input-character;          !!!next-input-character;
3058          redo A;          redo A;
3059        } else {        } else {
3060          !!!cp (221.6);          !!!cp (221.6);
3061          $self->{current_token}->{data} .= ']';          $self->{ct}->{data} .= ']';
3062          $self->{state} = CDATA_SECTION_STATE;          $self->{state} = CDATA_SECTION_STATE;
3063          ## Reconsume.          ## Reconsume.
3064          redo A;          redo A;
3065        }        }
3066      } elsif ($self->{state} == CDATA_SECTION_MSE2_STATE) {      } elsif ($self->{state} == CDATA_SECTION_MSE2_STATE) {
3067        if ($self->{next_char} == 0x003E) { # >        if ($self->{nc} == 0x003E) { # >
3068          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
3069          !!!next-input-character;          !!!next-input-character;
3070          if (length $self->{current_token}->{data}) { # character          if (length $self->{ct}->{data}) { # character
3071            !!!cp (221.7);            !!!cp (221.7);
3072            !!!emit ($self->{current_token}); # character            !!!emit ($self->{ct}); # character
3073          } else {          } else {
3074            !!!cp (221.8);            !!!cp (221.8);
3075            ## No token to emit. $self->{current_token} is discarded.            ## No token to emit. $self->{ct} is discarded.
3076          }          }
3077          redo A;          redo A;
3078        } elsif ($self->{next_char} == 0x005D) { # ]        } elsif ($self->{nc} == 0x005D) { # ]
3079          !!!cp (221.9); # character          !!!cp (221.9); # character
3080          $self->{current_token}->{data} .= ']'; ## Add first "]" of "]]]".          $self->{ct}->{data} .= ']'; ## Add first "]" of "]]]".
3081          ## Stay in the state.          ## Stay in the state.
3082          !!!next-input-character;          !!!next-input-character;
3083          redo A;          redo A;
3084        } else {        } else {
3085          !!!cp (221.11);          !!!cp (221.11);
3086          $self->{current_token}->{data} .= ']]'; # character          $self->{ct}->{data} .= ']]'; # character
3087          $self->{state} = CDATA_SECTION_STATE;          $self->{state} = CDATA_SECTION_STATE;
3088          ## Reconsume.          ## Reconsume.
3089          redo A;          redo A;
3090        }        }
   
3091      } elsif ($self->{state} == ENTITY_STATE) {      } elsif ($self->{state} == ENTITY_STATE) {
3092        my $in_attr = $self->{entity_in_attr};        if ($is_space->{$self->{nc}} or
3093        my $additional = $self->{entity_additional};            {
3094                0x003C => 1, 0x0026 => 1, -1 => 1, # <, &
3095                $self->{entity_add} => 1,
3096              }->{$self->{nc}}) {
3097            !!!cp (1001);
3098            ## Don't consume
3099            ## No error
3100            ## Return nothing.
3101            #
3102          } elsif ($self->{nc} == 0x0023) { # #
3103            !!!cp (999);
3104            $self->{state} = ENTITY_HASH_STATE;
3105            $self->{s_kwd} = '#';
3106            !!!next-input-character;
3107            redo A;
3108          } elsif ((0x0041 <= $self->{nc} and
3109                    $self->{nc} <= 0x005A) or # A..Z
3110                   (0x0061 <= $self->{nc} and
3111                    $self->{nc} <= 0x007A)) { # a..z
3112            !!!cp (998);
3113            require Whatpm::_NamedEntityList;
3114            $self->{state} = ENTITY_NAME_STATE;
3115            $self->{s_kwd} = chr $self->{nc};
3116            $self->{entity__value} = $self->{s_kwd};
3117            $self->{entity__match} = 0;
3118            !!!next-input-character;
3119            redo A;
3120          } else {
3121            !!!cp (1027);
3122            !!!parse-error (type => 'bare ero');
3123            ## Return nothing.
3124            #
3125          }
3126    
3127    my ($l, $c) = ($self->{line_prev}, $self->{column_prev});        ## NOTE: No character is consumed by the "consume a character
3128          ## reference" algorithm.  In other word, there is an "&" character
3129          ## that does not introduce a character reference, which would be
3130          ## appended to the parent element or the attribute value in later
3131          ## process of the tokenizer.
3132    
3133          if ($self->{prev_state} == DATA_STATE) {
3134            !!!cp (997);
3135            $self->{state} = $self->{prev_state};
3136            ## Reconsume.
3137            !!!emit ({type => CHARACTER_TOKEN, data => '&',
3138                      line => $self->{line_prev},
3139                      column => $self->{column_prev},
3140                     });
3141            redo A;
3142          } else {
3143            !!!cp (996);
3144            $self->{ca}->{value} .= '&';
3145            $self->{state} = $self->{prev_state};
3146            ## Reconsume.
3147            redo A;
3148          }
3149        } elsif ($self->{state} == ENTITY_HASH_STATE) {
3150          if ($self->{nc} == 0x0078 or # x
3151              $self->{nc} == 0x0058) { # X
3152            !!!cp (995);
3153            $self->{state} = HEXREF_X_STATE;
3154            $self->{s_kwd} .= chr $self->{nc};
3155            !!!next-input-character;
3156            redo A;
3157          } elsif (0x0030 <= $self->{nc} and
3158                   $self->{nc} <= 0x0039) { # 0..9
3159            !!!cp (994);
3160            $self->{state} = NCR_NUM_STATE;
3161            $self->{s_kwd} = $self->{nc} - 0x0030;
3162            !!!next-input-character;
3163            redo A;
3164          } else {
3165            !!!parse-error (type => 'bare nero',
3166                            line => $self->{line_prev},
3167                            column => $self->{column_prev} - 1);
3168    
3169    if ({          ## NOTE: According to the spec algorithm, nothing is returned,
3170         0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, # HT, LF, VT, FF,          ## and then "&#" is appended to the parent element or the attribute
3171         0x0020 => 1, 0x003C => 1, 0x0026 => 1, -1 => 1, # SP, <, & # 0x000D # CR          ## value in the later processing.
3172         $additional => 1,  
3173        }->{$self->{next_char}}) {          if ($self->{prev_state} == DATA_STATE) {
3174      !!!cp (1001);            !!!cp (1019);
3175      ## Don't consume            $self->{state} = $self->{prev_state};
3176      ## No error            ## Reconsume.
3177      $self->{entity_return} = undef;            !!!emit ({type => CHARACTER_TOKEN,
3178      $self->{state} = $self->{entity_in_attr} ? ENTITY_IN_ATTRIBUTE_VALUE_STATE : ENTITY_DATA_STATE;                      data => '&#',
3179      redo A;                      line => $self->{line_prev},
3180    } elsif ($self->{next_char} == 0x0023) { # #                      column => $self->{column_prev} - 1,
3181      !!!next-input-character;                     });
     if ($self->{next_char} == 0x0078 or # x  
         $self->{next_char} == 0x0058) { # X  
       my $code;  
       X: {  
         my $x_char = $self->{next_char};  
         !!!next-input-character;  
         if (0x0030 <= $self->{next_char} and  
             $self->{next_char} <= 0x0039) { # 0..9  
           !!!cp (1002);  
           $code ||= 0;  
           $code *= 0x10;  
           $code += $self->{next_char} - 0x0030;  
           redo X;  
         } elsif (0x0061 <= $self->{next_char} and  
                  $self->{next_char} <= 0x0066) { # a..f  
           !!!cp (1003);  
           $code ||= 0;  
           $code *= 0x10;  
           $code += $self->{next_char} - 0x0060 + 9;  
           redo X;  
         } elsif (0x0041 <= $self->{next_char} and  
                  $self->{next_char} <= 0x0046) { # A..F  
           !!!cp (1004);  
           $code ||= 0;  
           $code *= 0x10;  
           $code += $self->{next_char} - 0x0040 + 9;  
           redo X;  
         } elsif (not defined $code) { # no hexadecimal digit  
           !!!cp (1005);  
           !!!parse-error (type => 'bare hcro', line => $l, column => $c);  
           !!!back-next-input-character ($x_char, $self->{next_char});  
           $self->{next_char} = 0x0023; # #  
           $self->{entity_return} = undef;  
           $self->{state} = $self->{entity_in_attr} ? ENTITY_IN_ATTRIBUTE_VALUE_STATE : ENTITY_DATA_STATE;  
3182            redo A;            redo A;
         } elsif ($self->{next_char} == 0x003B) { # ;  
           !!!cp (1006);  
           !!!next-input-character;  
3183          } else {          } else {
3184            !!!cp (1007);            !!!cp (993);
3185            !!!parse-error (type => 'no refc', line => $l, column => $c);            $self->{ca}->{value} .= '&#';
3186              $self->{state} = $self->{prev_state};
3187              ## Reconsume.
3188              redo A;
3189          }          }
3190          }
3191          if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) {      } elsif ($self->{state} == NCR_NUM_STATE) {
3192            !!!cp (1008);        if (0x0030 <= $self->{nc} and
3193            !!!parse-error (type => 'invalid character reference',            $self->{nc} <= 0x0039) { # 0..9
                           text => (sprintf 'U+%04X', $code),  
                           line => $l, column => $c);  
           $code = 0xFFFD;  
         } elsif ($code > 0x10FFFF) {  
           !!!cp (1009);  
           !!!parse-error (type => 'invalid character reference',  
                           text => (sprintf 'U-%08X', $code),  
                           line => $l, column => $c);  
           $code = 0xFFFD;  
         } elsif ($code == 0x000D) {  
           !!!cp (1010);  
           !!!parse-error (type => 'CR character reference', line => $l, column => $c);  
           $code = 0x000A;  
         } elsif (0x80 <= $code and $code <= 0x9F) {  
           !!!cp (1011);  
           !!!parse-error (type => 'C1 character reference', text => (sprintf 'U+%04X', $code), line => $l, column => $c);  
           $code = $c1_entity_char->{$code};  
         }  
   
         $self->{entity_return} = {type => CHARACTER_TOKEN, data => chr $code,  
                 has_reference => 1,  
                 line => $l, column => $c,  
                };  
         $self->{state} = $self->{entity_in_attr} ? ENTITY_IN_ATTRIBUTE_VALUE_STATE : ENTITY_DATA_STATE;  
         redo A;  
       } # X  
     } elsif (0x0030 <= $self->{next_char} and  
              $self->{next_char} <= 0x0039) { # 0..9  
       my $code = $self->{next_char} - 0x0030;  
       !!!next-input-character;  
         
       while (0x0030 <= $self->{next_char} and  
                 $self->{next_char} <= 0x0039) { # 0..9  
3194          !!!cp (1012);          !!!cp (1012);
3195          $code *= 10;          $self->{s_kwd} *= 10;
3196          $code += $self->{next_char} - 0x0030;          $self->{s_kwd} += $self->{nc} - 0x0030;
3197                    
3198            ## Stay in the state.
3199          !!!next-input-character;          !!!next-input-character;
3200        }          redo A;
3201          } elsif ($self->{nc} == 0x003B) { # ;
       if ($self->{next_char} == 0x003B) { # ;  
3202          !!!cp (1013);          !!!cp (1013);
3203          !!!next-input-character;          !!!next-input-character;
3204            #
3205        } else {        } else {
3206          !!!cp (1014);          !!!cp (1014);
3207          !!!parse-error (type => 'no refc', line => $l, column => $c);          !!!parse-error (type => 'no refc');
3208            ## Reconsume.
3209            #
3210        }        }
3211    
3212        if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) {        my $code = $self->{s_kwd};
3213          my $l = $self->{line_prev};
3214          my $c = $self->{column_prev};
3215          if ($charref_map->{$code}) {
3216          !!!cp (1015);          !!!cp (1015);
3217          !!!parse-error (type => 'invalid character reference',          !!!parse-error (type => 'invalid character reference',
3218                          text => (sprintf 'U+%04X', $code),                          text => (sprintf 'U+%04X', $code),
3219                          line => $l, column => $c);                          line => $l, column => $c);
3220          $code = 0xFFFD;          $code = $charref_map->{$code};
3221        } elsif ($code > 0x10FFFF) {        } elsif ($code > 0x10FFFF) {
3222          !!!cp (1016);          !!!cp (1016);
3223          !!!parse-error (type => 'invalid character reference',          !!!parse-error (type => 'invalid character reference',
3224                          text => (sprintf 'U-%08X', $code),                          text => (sprintf 'U-%08X', $code),
3225                          line => $l, column => $c);                          line => $l, column => $c);
3226          $code = 0xFFFD;          $code = 0xFFFD;
3227        } elsif ($code == 0x000D) {        }
3228          !!!cp (1017);  
3229          !!!parse-error (type => 'CR character reference',        if ($self->{prev_state} == DATA_STATE) {
3230                          line => $l, column => $c);          !!!cp (992);
3231          $code = 0x000A;          $self->{state} = $self->{prev_state};
3232        } elsif (0x80 <= $code and $code <= 0x9F) {          ## Reconsume.
3233          !!!cp (1018);          !!!emit ({type => CHARACTER_TOKEN, data => chr $code,
3234          !!!parse-error (type => 'C1 character reference',                    line => $l, column => $c,
3235                     });
3236            redo A;
3237          } else {
3238            !!!cp (991);
3239            $self->{ca}->{value} .= chr $code;
3240            $self->{ca}->{has_reference} = 1;
3241            $self->{state} = $self->{prev_state};
3242            ## Reconsume.
3243            redo A;
3244          }
3245        } elsif ($self->{state} == HEXREF_X_STATE) {
3246          if ((0x0030 <= $self->{nc} and $self->{nc} <= 0x0039) or
3247              (0x0041 <= $self->{nc} and $self->{nc} <= 0x0046) or
3248              (0x0061 <= $self->{nc} and $self->{nc} <= 0x0066)) {
3249            # 0..9, A..F, a..f
3250            !!!cp (990);
3251            $self->{state} = HEXREF_HEX_STATE;
3252            $self->{s_kwd} = 0;
3253            ## Reconsume.
3254            redo A;
3255          } else {
3256            !!!parse-error (type => 'bare hcro',
3257                            line => $self->{line_prev},
3258                            column => $self->{column_prev} - 2);
3259    
3260            ## NOTE: According to the spec algorithm, nothing is returned,
3261            ## and then "&#" followed by "X" or "x" is appended to the parent
3262            ## element or the attribute value in the later processing.
3263    
3264            if ($self->{prev_state} == DATA_STATE) {
3265              !!!cp (1005);
3266              $self->{state} = $self->{prev_state};
3267              ## Reconsume.
3268              !!!emit ({type => CHARACTER_TOKEN,
3269                        data => '&' . $self->{s_kwd},
3270                        line => $self->{line_prev},
3271                        column => $self->{column_prev} - length $self->{s_kwd},
3272                       });
3273              redo A;
3274            } else {
3275              !!!cp (989);
3276              $self->{ca}->{value} .= '&' . $self->{s_kwd};
3277              $self->{state} = $self->{prev_state};
3278              ## Reconsume.
3279              redo A;
3280            }
3281          }
3282        } elsif ($self->{state} == HEXREF_HEX_STATE) {
3283          if (0x0030 <= $self->{nc} and $self->{nc} <= 0x0039) {
3284            # 0..9
3285            !!!cp (1002);
3286            $self->{s_kwd} *= 0x10;
3287            $self->{s_kwd} += $self->{nc} - 0x0030;
3288            ## Stay in the state.
3289            !!!next-input-character;
3290            redo A;
3291          } elsif (0x0061 <= $self->{nc} and
3292                   $self->{nc} <= 0x0066) { # a..f
3293            !!!cp (1003);
3294            $self->{s_kwd} *= 0x10;
3295            $self->{s_kwd} += $self->{nc} - 0x0060 + 9;
3296            ## Stay in the state.
3297            !!!next-input-character;
3298            redo A;
3299          } elsif (0x0041 <= $self->{nc} and
3300                   $self->{nc} <= 0x0046) { # A..F
3301            !!!cp (1004);
3302            $self->{s_kwd} *= 0x10;
3303            $self->{s_kwd} += $self->{nc} - 0x0040 + 9;
3304            ## Stay in the state.
3305            !!!next-input-character;
3306            redo A;
3307          } elsif ($self->{nc} == 0x003B) { # ;
3308            !!!cp (1006);
3309            !!!next-input-character;
3310            #
3311          } else {
3312            !!!cp (1007);
3313            !!!parse-error (type => 'no refc',
3314                            line => $self->{line},
3315                            column => $self->{column});
3316            ## Reconsume.
3317            #
3318          }
3319    
3320          my $code = $self->{s_kwd};
3321          my $l = $self->{line_prev};
3322          my $c = $self->{column_prev};
3323          if ($charref_map->{$code}) {
3324            !!!cp (1008);
3325            !!!parse-error (type => 'invalid character reference',
3326                          text => (sprintf 'U+%04X', $code),                          text => (sprintf 'U+%04X', $code),
3327                          line => $l, column => $c);                          line => $l, column => $c);
3328          $code = $c1_entity_char->{$code};          $code = $charref_map->{$code};
3329          } elsif ($code > 0x10FFFF) {
3330            !!!cp (1009);
3331            !!!parse-error (type => 'invalid character reference',
3332                            text => (sprintf 'U-%08X', $code),
3333                            line => $l, column => $c);
3334            $code = 0xFFFD;
3335        }        }
3336          
3337        $self->{entity_return} = {type => CHARACTER_TOKEN, data => chr $code, has_reference => 1,        if ($self->{prev_state} == DATA_STATE) {
3338                line => $l, column => $c,          !!!cp (988);
3339               };          $self->{state} = $self->{prev_state};
3340        $self->{state} = $self->{entity_in_attr} ? ENTITY_IN_ATTRIBUTE_VALUE_STATE : ENTITY_DATA_STATE;          ## Reconsume.
3341        redo A;          !!!emit ({type => CHARACTER_TOKEN, data => chr $code,
3342      } else {                    line => $l, column => $c,
3343        !!!cp (1019);                   });
3344        !!!parse-error (type => 'bare nero', line => $l, column => $c);          redo A;
3345        !!!back-next-input-character ($self->{next_char});        } else {
3346        $self->{next_char} = 0x0023; # #          !!!cp (987);
3347        $self->{entity_return} = undef;          $self->{ca}->{value} .= chr $code;
3348        $self->{state} = $self->{entity_in_attr} ? ENTITY_IN_ATTRIBUTE_VALUE_STATE : ENTITY_DATA_STATE;          $self->{ca}->{has_reference} = 1;
3349        redo A;          $self->{state} = $self->{prev_state};
3350      }          ## Reconsume.
3351    } elsif ((0x0041 <= $self->{next_char} and          redo A;
3352              $self->{next_char} <= 0x005A) or        }
3353             (0x0061 <= $self->{next_char} and      } elsif ($self->{state} == ENTITY_NAME_STATE) {
3354              $self->{next_char} <= 0x007A)) {        if (length $self->{s_kwd} < 30 and
3355      my $entity_name = chr $self->{next_char};            ## NOTE: Some number greater than the maximum length of entity name
3356      !!!next-input-character;            ((0x0041 <= $self->{nc} and # a
3357                $self->{nc} <= 0x005A) or # x
3358      my $value = $entity_name;             (0x0061 <= $self->{nc} and # a
3359      my $match = 0;              $self->{nc} <= 0x007A) or # z
3360      require Whatpm::_NamedEntityList;             (0x0030 <= $self->{nc} and # 0
3361      our $EntityChar;              $self->{nc} <= 0x0039) or # 9
3362               $self->{nc} == 0x003B)) { # ;
3363      while (length $entity_name < 30 and          our $EntityChar;
3364             ## NOTE: Some number greater than the maximum length of entity name          $self->{s_kwd} .= chr $self->{nc};
3365             ((0x0041 <= $self->{next_char} and # a          if (defined $EntityChar->{$self->{s_kwd}}) {
3366               $self->{next_char} <= 0x005A) or # x            if ($self->{nc} == 0x003B) { # ;
3367              (0x0061 <= $self->{next_char} and # a              !!!cp (1020);
3368               $self->{next_char} <= 0x007A) or # z              $self->{entity__value} = $EntityChar->{$self->{s_kwd}};
3369              (0x0030 <= $self->{next_char} and # 0              $self->{entity__match} = 1;
3370               $self->{next_char} <= 0x0039) or # 9              !!!next-input-character;
3371              $self->{next_char} == 0x003B)) { # ;              #
3372        $entity_name .= chr $self->{next_char};            } else {
3373        if (defined $EntityChar->{$entity_name}) {              !!!cp (1021);
3374          if ($self->{next_char} == 0x003B) { # ;              $self->{entity__value} = $EntityChar->{$self->{s_kwd}};
3375            !!!cp (1020);              $self->{entity__match} = -1;
3376            $value = $EntityChar->{$entity_name};              ## Stay in the state.
3377            $match = 1;              !!!next-input-character;
3378            !!!next-input-character;              redo A;
3379            last;            }
3380          } else {          } else {
3381            !!!cp (1021);            !!!cp (1022);
3382            $value = $EntityChar->{$entity_name};            $self->{entity__value} .= chr $self->{nc};
3383            $match = -1;            $self->{entity__match} *= 2;
3384              ## Stay in the state.
3385            !!!next-input-character;            !!!next-input-character;
3386              redo A;
3387            }
3388          }
3389    
3390          my $data;
3391          my $has_ref;
3392          if ($self->{entity__match} > 0) {
3393            !!!cp (1023);
3394            $data = $self->{entity__value};
3395            $has_ref = 1;
3396            #
3397          } elsif ($self->{entity__match} < 0) {
3398            !!!parse-error (type => 'no refc');
3399            if ($self->{prev_state} != DATA_STATE and # in attribute
3400                $self->{entity__match} < -1) {
3401              !!!cp (1024);
3402              $data = '&' . $self->{s_kwd};
3403              #
3404            } else {
3405              !!!cp (1025);
3406              $data = $self->{entity__value};
3407              $has_ref = 1;
3408              #
3409          }          }
3410        } else {        } else {
3411          !!!cp (1022);          !!!cp (1026);
3412          $value .= chr $self->{next_char};          !!!parse-error (type => 'bare ero',
3413          $match *= 2;                          line => $self->{line_prev},
3414          !!!next-input-character;                          column => $self->{column_prev} - length $self->{s_kwd});
3415            $data = '&' . $self->{s_kwd};
3416            #
3417        }        }
3418      }    
3419              ## NOTE: In these cases, when a character reference is found,
3420      if ($match > 0) {        ## it is consumed and a character token is returned, or, otherwise,
3421        !!!cp (1023);        ## nothing is consumed and returned, according to the spec algorithm.
3422        $self->{entity_return} = {type => CHARACTER_TOKEN, data => $value, has_reference => 1,        ## In this implementation, anything that has been examined by the
3423                line => $l, column => $c,        ## tokenizer is appended to the parent element or the attribute value
3424               };        ## as string, either literal string when no character reference or
3425        $self->{state} = $self->{entity_in_attr} ? ENTITY_IN_ATTRIBUTE_VALUE_STATE : ENTITY_DATA_STATE;        ## entity-replaced string otherwise, in this stage, since any characters
3426        redo A;        ## that would not be consumed are appended in the data state or in an
3427      } elsif ($match < 0) {        ## appropriate attribute value state anyway.
3428        !!!parse-error (type => 'no refc', line => $l, column => $c);  
3429        if ($in_attr and $match < -1) {        if ($self->{prev_state} == DATA_STATE) {
3430          !!!cp (1024);          !!!cp (986);
3431          $self->{entity_return} = {type => CHARACTER_TOKEN, data => '&'.$entity_name,          $self->{state} = $self->{prev_state};
3432                  line => $l, column => $c,          ## Reconsume.
3433                 };          !!!emit ({type => CHARACTER_TOKEN,
3434          $self->{state} = $self->{entity_in_attr} ? ENTITY_IN_ATTRIBUTE_VALUE_STATE : ENTITY_DATA_STATE;                    data => $data,
3435                      line => $self->{line_prev},
3436                      column => $self->{column_prev} + 1 - length $self->{s_kwd},
3437                     });
3438          redo A;          redo A;
3439        } else {        } else {
3440          !!!cp (1025);          !!!cp (985);
3441          $self->{entity_return} = {type => CHARACTER_TOKEN, data => $value, has_reference => 1,          $self->{ca}->{value} .= $data;
3442                  line => $l, column => $c,          $self->{ca}->{has_reference} = 1 if $has_ref;
3443                 };          $self->{state} = $self->{prev_state};
3444          $self->{state} = $self->{entity_in_attr} ? ENTITY_IN_ATTRIBUTE_VALUE_STATE : ENTITY_DATA_STATE;          ## Reconsume.
3445          redo A;          redo A;
3446        }        }
3447      } else {      } else {
       !!!cp (1026);  
       !!!parse-error (type => 'bare ero', line => $l, column => $c);  
       ## NOTE: "No characters are consumed" in the spec.  
       $self->{entity_return} = {type => CHARACTER_TOKEN, data => '&'.$value,  
               line => $l, column => $c,  
              };  
       $self->{state} = $self->{entity_in_attr} ? ENTITY_IN_ATTRIBUTE_VALUE_STATE : ENTITY_DATA_STATE;  
       redo A;  
     }  
   } else {  
     !!!cp (1027);  
     ## no characters are consumed  
     !!!parse-error (type => 'bare ero', line => $l, column => $c);  
     $self->{entity_return} = undef;  
     $self->{state} = $self->{entity_in_attr} ? ENTITY_IN_ATTRIBUTE_VALUE_STATE : ENTITY_DATA_STATE;  
     redo A;  
   }  
   
     } else {  
3448        die "$0: $self->{state}: Unknown state";        die "$0: $self->{state}: Unknown state";
3449      }      }
3450    } # A      } # A  
# Line 3254  sub _construct_tree ($) { Line 3480  sub _construct_tree ($) {
3480    ## When an interactive UA render the $self->{document} available    ## When an interactive UA render the $self->{document} available
3481    ## to the user, or when it begin accepting user input, are    ## to the user, or when it begin accepting user input, are
3482    ## not defined.    ## not defined.
   
   ## Append a character: collect it and all subsequent consecutive  
   ## characters and insert one Text node whose data is concatenation  
   ## of all those characters. # MUST  
3483        
3484    !!!next-token;    !!!next-token;
3485    
3486    undef $self->{form_element};    undef $self->{form_element};
3487    undef $self->{head_element};    undef $self->{head_element};
3488      undef $self->{head_element_inserted};
3489    $self->{open_elements} = [];    $self->{open_elements} = [];
3490    undef $self->{inner_html_node};    undef $self->{inner_html_node};
3491      undef $self->{ignore_newline};
3492    
3493    ## NOTE: The "initial" insertion mode.    ## NOTE: The "initial" insertion mode.
3494    $self->_tree_construction_initial; # MUST    $self->_tree_construction_initial; # MUST
# Line 3291  sub _tree_construction_initial ($) { Line 3515  sub _tree_construction_initial ($) {
3515        $doctype_name = '' unless defined $doctype_name;        $doctype_name = '' unless defined $doctype_name;
3516        $doctype_name =~ tr/a-z/A-Z/; # ASCII case-insensitive        $doctype_name =~ tr/a-z/A-Z/; # ASCII case-insensitive
3517        if (not defined $token->{name} or # <!DOCTYPE>        if (not defined $token->{name} or # <!DOCTYPE>
3518            defined $token->{system_identifier}) {            defined $token->{sysid}) {
3519          !!!cp ('t1');          !!!cp ('t1');
3520          !!!parse-error (type => 'not HTML5', token => $token);          !!!parse-error (type => 'not HTML5', token => $token);
3521        } elsif ($doctype_name ne 'HTML') {        } elsif ($doctype_name ne 'HTML') {
3522          !!!cp ('t2');          !!!cp ('t2');
3523          !!!parse-error (type => 'not HTML5', token => $token);          !!!parse-error (type => 'not HTML5', token => $token);
3524        } elsif (defined $token->{public_identifier}) {        } elsif (defined $token->{pubid}) {
3525          if ($token->{public_identifier} eq 'XSLT-compat') {          if ($token->{pubid} eq 'XSLT-compat') {
3526            !!!cp ('t1.2');            !!!cp ('t1.2');
3527            !!!parse-error (type => 'XSLT-compat', token => $token,            !!!parse-error (type => 'XSLT-compat', token => $token,
3528                            level => $self->{level}->{should});                            level => $self->{level}->{should});
# Line 3314  sub _tree_construction_initial ($) { Line 3538  sub _tree_construction_initial ($) {
3538          ($token->{name}); ## ISSUE: If name is missing (e.g. <!DOCTYPE>)?          ($token->{name}); ## ISSUE: If name is missing (e.g. <!DOCTYPE>)?
3539        ## NOTE: Default value for both |public_id| and |system_id| attributes        ## NOTE: Default value for both |public_id| and |system_id| attributes
3540        ## are empty strings, so that we don't set any value in missing cases.        ## are empty strings, so that we don't set any value in missing cases.
3541        $doctype->public_id ($token->{public_identifier})        $doctype->public_id ($token->{pubid}) if defined $token->{pubid};
3542            if defined $token->{public_identifier};        $doctype->system_id ($token->{sysid}) if defined $token->{sysid};
       $doctype->system_id ($token->{system_identifier})  
           if defined $token->{system_identifier};  
3543        ## NOTE: Other DocumentType attributes are null or empty lists.        ## NOTE: Other DocumentType attributes are null or empty lists.
3544        ## ISSUE: internalSubset = null??        ## ISSUE: internalSubset = null??
3545        $self->{document}->append_child ($doctype);        $self->{document}->append_child ($doctype);
# Line 3325  sub _tree_construction_initial ($) { Line 3547  sub _tree_construction_initial ($) {
3547        if ($token->{quirks} or $doctype_name ne 'HTML') {        if ($token->{quirks} or $doctype_name ne 'HTML') {
3548          !!!cp ('t4');          !!!cp ('t4');
3549          $self->{document}->manakai_compat_mode ('quirks');          $self->{document}->manakai_compat_mode ('quirks');
3550        } elsif (defined $token->{public_identifier}) {        } elsif (defined $token->{pubid}) {
3551          my $pubid = $token->{public_identifier};          my $pubid = $token->{pubid};
3552          $pubid =~ tr/a-z/A-z/;          $pubid =~ tr/a-z/A-z/;
3553          my $prefix = [          my $prefix = [
3554            "+//SILMARIL//DTD HTML PRO V0R11 19970101//",            "+//SILMARIL//DTD HTML PRO V0R11 19970101//",
# Line 3400  sub _tree_construction_initial ($) { Line 3622  sub _tree_construction_initial ($) {
3622            $self->{document}->manakai_compat_mode ('quirks');            $self->{document}->manakai_compat_mode ('quirks');
3623          } elsif ($pubid =~ m[^-//W3C//DTD HTML 4.01 FRAMESET//] or          } elsif ($pubid =~ m[^-//W3C//DTD HTML 4.01 FRAMESET//] or
3624                   $pubid =~ m[^-//W3C//DTD HTML 4.01 TRANSITIONAL//]) {                   $pubid =~ m[^-//W3C//DTD HTML 4.01 TRANSITIONAL//]) {
3625            if (defined $token->{system_identifier}) {            if (defined $token->{sysid}) {
3626              !!!cp ('t6');              !!!cp ('t6');
3627              $self->{document}->manakai_compat_mode ('quirks');              $self->{document}->manakai_compat_mode ('quirks');
3628            } else {            } else {
# Line 3417  sub _tree_construction_initial ($) { Line 3639  sub _tree_construction_initial ($) {
3639        } else {        } else {
3640          !!!cp ('t10');          !!!cp ('t10');
3641        }        }
3642        if (defined $token->{system_identifier}) {        if (defined $token->{sysid}) {
3643          my $sysid = $token->{system_identifier};          my $sysid = $token->{sysid};
3644          $sysid =~ tr/A-Z/a-z/;          $sysid =~ tr/A-Z/a-z/;
3645          if ($sysid eq "http://www.ibm.com/data/dtd/v11/ibmxhtml1-transitional.dtd") {          if ($sysid eq "http://www.ibm.com/data/dtd/v11/ibmxhtml1-transitional.dtd") {
3646            ## NOTE: Ensure that |PUBLIC "(limited quirks)" "(quirks)"| is            ## NOTE: Ensure that |PUBLIC "(limited quirks)" "(quirks)"| is
# Line 3448  sub _tree_construction_initial ($) { Line 3670  sub _tree_construction_initial ($) {
3670        !!!ack-later;        !!!ack-later;
3671        return;        return;
3672      } elsif ($token->{type} == CHARACTER_TOKEN) {      } elsif ($token->{type} == CHARACTER_TOKEN) {
3673        if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D        if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
3674          ## Ignore the token          ## Ignore the token
3675    
3676          unless (length $token->{data}) {          unless (length $token->{data}) {
# Line 3505  sub _tree_construction_root_element ($) Line 3727  sub _tree_construction_root_element ($)
3727          !!!next-token;          !!!next-token;
3728          redo B;          redo B;
3729        } elsif ($token->{type} == CHARACTER_TOKEN) {        } elsif ($token->{type} == CHARACTER_TOKEN) {
3730          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D          if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
3731            ## Ignore the token.            ## Ignore the token.
3732    
3733            unless (length $token->{data}) {            unless (length $token->{data}) {
# Line 3572  sub _tree_construction_root_element ($) Line 3794  sub _tree_construction_root_element ($)
3794      ## NOTE: Reprocess the token.      ## NOTE: Reprocess the token.
3795      !!!ack-later;      !!!ack-later;
3796      return; ## Go to the "before head" insertion mode.      return; ## Go to the "before head" insertion mode.
   
     ## ISSUE: There is an issue in the spec  
3797    } # B    } # B
3798    
3799    die "$0: _tree_construction_root_element: This should never be reached";    die "$0: _tree_construction_root_element: This should never be reached";
# Line 3609  sub _reset_insertion_mode ($) { Line 3829  sub _reset_insertion_mode ($) {
3829          ## SVG elements.  Currently the HTML syntax supports only MathML and          ## SVG elements.  Currently the HTML syntax supports only MathML and
3830          ## SVG elements as foreigners.          ## SVG elements as foreigners.
3831          $new_mode = IN_BODY_IM | IN_FOREIGN_CONTENT_IM;          $new_mode = IN_BODY_IM | IN_FOREIGN_CONTENT_IM;
3832        } elsif ($node->[1] & TABLE_CELL_EL) {        } elsif ($node->[1] == TABLE_CELL_EL) {
3833          if ($last) {          if ($last) {
3834            !!!cp ('t28.2');            !!!cp ('t28.2');
3835            #            #
# Line 3638  sub _reset_insertion_mode ($) { Line 3858  sub _reset_insertion_mode ($) {
3858        $self->{insertion_mode} = $new_mode and return if defined $new_mode;        $self->{insertion_mode} = $new_mode and return if defined $new_mode;
3859                
3860        ## Step 15        ## Step 15
3861        if ($node->[1] & HTML_EL) {        if ($node->[1] == HTML_EL) {
3862          unless (defined $self->{head_element}) {          unless (defined $self->{head_element}) {
3863            !!!cp ('t29');            !!!cp ('t29');
3864            $self->{insertion_mode} = BEFORE_HEAD_IM;            $self->{insertion_mode} = BEFORE_HEAD_IM;
# Line 3770  sub _tree_construction_main ($) { Line 3990  sub _tree_construction_main ($) {
3990    
3991      ## Step 1      ## Step 1
3992      my $start_tag_name = $token->{tag_name};      my $start_tag_name = $token->{tag_name};
3993      my $el;      !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
     !!!create-element ($el, $HTML_NS, $start_tag_name, $token->{attributes}, $token);  
3994    
3995      ## Step 2      ## Step 2
     $insert->($el);  
   
     ## Step 3  
3996      $self->{content_model} = $content_model_flag; # CDATA or RCDATA      $self->{content_model} = $content_model_flag; # CDATA or RCDATA
3997      delete $self->{escape}; # MUST      delete $self->{escape}; # MUST
3998    
3999      ## Step 4      ## Step 3, 4
4000      my $text = '';      $self->{insertion_mode} |= IN_CDATA_RCDATA_IM;
     !!!nack ('t40.1');  
     !!!next-token;  
     while ($token->{type} == CHARACTER_TOKEN) { # or until stop tokenizing  
       !!!cp ('t40');  
       $text .= $token->{data};  
       !!!next-token;  
     }  
   
     ## Step 5  
     if (length $text) {  
       !!!cp ('t41');  
       my $text = $self->{document}->create_text_node ($text);  
       $el->append_child ($text);  
     }  
   
     ## Step 6  
     $self->{content_model} = PCDATA_CONTENT_MODEL;  
4001    
4002      ## Step 7      !!!nack ('t40.1');
     if ($token->{type} == END_TAG_TOKEN and  
         $token->{tag_name} eq $start_tag_name) {  
       !!!cp ('t42');  
       ## Ignore the token  
     } else {  
       ## NOTE: An end-of-file token.  
       if ($content_model_flag == CDATA_CONTENT_MODEL) {  
         !!!cp ('t43');  
         !!!parse-error (type => 'in CDATA:#eof', token => $token);  
       } elsif ($content_model_flag == RCDATA_CONTENT_MODEL) {  
         !!!cp ('t44');  
         !!!parse-error (type => 'in RCDATA:#eof', token => $token);  
       } else {  
         die "$0: $content_model_flag in parse_rcdata";  
       }  
     }  
4003      !!!next-token;      !!!next-token;
4004    }; # $parse_rcdata    }; # $parse_rcdata
4005    
4006    my $script_start_tag = sub () {    my $script_start_tag = sub () {
4007        ## Step 1
4008      my $script_el;      my $script_el;
4009      !!!create-element ($script_el, $HTML_NS, 'script', $token->{attributes}, $token);      !!!create-element ($script_el, $HTML_NS, 'script', $token->{attributes}, $token);
4010    
4011        ## Step 2
4012      ## TODO: mark as "parser-inserted"      ## TODO: mark as "parser-inserted"
4013    
4014        ## Step 3
4015        ## TODO: Mark as "already executed", if ...
4016    
4017        ## Step 4
4018        $insert->($script_el);
4019    
4020        ## ISSUE: $script_el is not put into the stack
4021        push @{$self->{open_elements}}, [$script_el, $el_category->{script}];
4022    
4023        ## Step 5
4024      $self->{content_model} = CDATA_CONTENT_MODEL;      $self->{content_model} = CDATA_CONTENT_MODEL;
4025      delete $self->{escape}; # MUST      delete $self->{escape}; # MUST
       
     my $text = '';  
     !!!nack ('t45.1');  
     !!!next-token;  
     while ($token->{type} == CHARACTER_TOKEN) {  
       !!!cp ('t45');  
       $text .= $token->{data};  
       !!!next-token;  
     } # stop if non-character token or tokenizer stops tokenising  
     if (length $text) {  
       !!!cp ('t46');  
       $script_el->manakai_append_text ($text);  
     }  
                 
     $self->{content_model} = PCDATA_CONTENT_MODEL;  
4026    
4027      if ($token->{type} == END_TAG_TOKEN and      ## Step 6-7
4028          $token->{tag_name} eq 'script') {      $self->{insertion_mode} |= IN_CDATA_RCDATA_IM;
       !!!cp ('t47');  
       ## Ignore the token  
     } else {  
       !!!cp ('t48');  
       !!!parse-error (type => 'in CDATA:#eof', token => $token);  
       ## ISSUE: And ignore?  
       ## TODO: mark as "already executed"  
     }  
       
     if (defined $self->{inner_html_node}) {  
       !!!cp ('t49');  
       ## TODO: mark as "already executed"  
     } else {  
       !!!cp ('t50');  
       ## TODO: $old_insertion_point = current insertion point  
       ## TODO: insertion point = just before the next input character  
4029    
4030        $insert->($script_el);      !!!nack ('t40.2');
         
       ## TODO: insertion point = $old_insertion_point (might be "undefined")  
         
       ## TODO: if there is a script that will execute as soon as the parser resume, then...  
     }  
       
4031      !!!next-token;      !!!next-token;
4032    }; # $script_start_tag    }; # $script_start_tag
4033    
4034    ## NOTE: $open_tables->[-1]->[0] is the "current table" element node.    ## NOTE: $open_tables->[-1]->[0] is the "current table" element node.
4035    ## NOTE: $open_tables->[-1]->[1] is the "tainted" flag.    ## NOTE: $open_tables->[-1]->[1] is the "tainted" flag.
4036      ## NOTE: $open_tables->[-1]->[2] is set false when non-Text node inserted.
4037    my $open_tables = [[$self->{open_elements}->[0]->[0]]];    my $open_tables = [[$self->{open_elements}->[0]->[0]]];
4038    
4039    my $formatting_end_tag = sub {    my $formatting_end_tag = sub {
# Line 3958  sub _tree_construction_main ($) { Line 4118  sub _tree_construction_main ($) {
4118            !!!cp ('t59');            !!!cp ('t59');
4119            $furthest_block = $node;            $furthest_block = $node;
4120            $furthest_block_i_in_open = $_;            $furthest_block_i_in_open = $_;
4121              ## NOTE: The topmost (eldest) node.
4122          } elsif ($node->[0] eq $formatting_element->[0]) {          } elsif ($node->[0] eq $formatting_element->[0]) {
4123            !!!cp ('t60');            !!!cp ('t60');
4124            last OE;            last OE;
# Line 4044  sub _tree_construction_main ($) { Line 4205  sub _tree_construction_main ($) {
4205          my $foster_parent_element;          my $foster_parent_element;
4206          my $next_sibling;          my $next_sibling;
4207          OE: for (reverse 0..$#{$self->{open_elements}}) {          OE: for (reverse 0..$#{$self->{open_elements}}) {
4208            if ($self->{open_elements}->[$_]->[1] & TABLE_EL) {            if ($self->{open_elements}->[$_]->[1] == TABLE_EL) {
4209                               my $parent = $self->{open_elements}->[$_]->[0]->parent_node;                               my $parent = $self->{open_elements}->[$_]->[0]->parent_node;
4210                               if (defined $parent and $parent->node_type == 1) {                               if (defined $parent and $parent->node_type == 1) {
4211                                 !!!cp ('t65.1');                                 !!!cp ('t65.1');
# Line 4104  sub _tree_construction_main ($) { Line 4265  sub _tree_construction_main ($) {
4265            $i = $_;            $i = $_;
4266          }          }
4267        } # OE        } # OE
4268        splice @{$self->{open_elements}}, $i + 1, 1, $clone;        splice @{$self->{open_elements}}, $i + 1, 0, $clone;
4269                
4270        ## Step 14        ## Step 14
4271        redo FET;        redo FET;
# Line 4122  sub _tree_construction_main ($) { Line 4283  sub _tree_construction_main ($) {
4283        my $foster_parent_element;        my $foster_parent_element;
4284        my $next_sibling;        my $next_sibling;
4285        OE: for (reverse 0..$#{$self->{open_elements}}) {        OE: for (reverse 0..$#{$self->{open_elements}}) {
4286          if ($self->{open_elements}->[$_]->[1] & TABLE_EL) {          if ($self->{open_elements}->[$_]->[1] == TABLE_EL) {
4287                               my $parent = $self->{open_elements}->[$_]->[0]->parent_node;                               my $parent = $self->{open_elements}->[$_]->[0]->parent_node;
4288                               if (defined $parent and $parent->node_type == 1) {                               if (defined $parent and $parent->node_type == 1) {
4289                                 !!!cp ('t70');                                 !!!cp ('t70');
# Line 4147  sub _tree_construction_main ($) { Line 4308  sub _tree_construction_main ($) {
4308      }      }
4309    }; # $insert_to_foster    }; # $insert_to_foster
4310    
4311      ## NOTE: Insert a character (MUST): When a character is inserted, if
4312      ## the last node that was inserted by the parser is a Text node and
4313      ## the character has to be inserted after that node, then the
4314      ## character is appended to the Text node.  However, if any other
4315      ## node is inserted by the parser, then a new Text node is created
4316      ## and the character is appended as that Text node.  If I'm not
4317      ## wrong, for a parser with scripting disabled, there are only two
4318      ## cases where this occurs.  One is the case where an element node
4319      ## is inserted to the |head| element.  This is covered by using the
4320      ## |$self->{head_element_inserted}| flag.  Another is the case where
4321      ## an element or comment is inserted into the |table| subtree while
4322      ## foster parenting happens.  This is covered by using the [2] flag
4323      ## of the |$open_tables| structure.  All other cases are handled
4324      ## simply by calling |manakai_append_text| method.
4325    
4326      ## TODO: |<body><script>document.write("a<br>");
4327      ## document.body.removeChild (document.body.lastChild);
4328      ## document.write ("b")</script>|
4329    
4330    B: while (1) {    B: while (1) {
4331      if ($token->{type} == DOCTYPE_TOKEN) {      if ($token->{type} == DOCTYPE_TOKEN) {
4332        !!!cp ('t73');        !!!cp ('t73');
# Line 4194  sub _tree_construction_main ($) { Line 4374  sub _tree_construction_main ($) {
4374        } else {        } else {
4375          !!!cp ('t87');          !!!cp ('t87');
4376          $self->{open_elements}->[-1]->[0]->append_child ($comment);          $self->{open_elements}->[-1]->[0]->append_child ($comment);
4377            $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
4378        }        }
4379        !!!next-token;        !!!next-token;
4380        next B;        next B;
4381        } elsif ($self->{insertion_mode} & IN_CDATA_RCDATA_IM) {
4382          if ($token->{type} == CHARACTER_TOKEN) {
4383            $token->{data} =~ s/^\x0A// if $self->{ignore_newline};
4384            delete $self->{ignore_newline};
4385    
4386            if (length $token->{data}) {
4387              !!!cp ('t43');
4388              $self->{open_elements}->[-1]->[0]->manakai_append_text
4389                  ($token->{data});
4390            } else {
4391              !!!cp ('t43.1');
4392            }
4393            !!!next-token;
4394            next B;
4395          } elsif ($token->{type} == END_TAG_TOKEN) {
4396            delete $self->{ignore_newline};
4397    
4398            if ($token->{tag_name} eq 'script') {
4399              !!!cp ('t50');
4400              
4401              ## Para 1-2
4402              my $script = pop @{$self->{open_elements}};
4403              
4404              ## Para 3
4405              $self->{insertion_mode} &= ~ IN_CDATA_RCDATA_IM;
4406    
4407              ## Para 4
4408              ## TODO: $old_insertion_point = $current_insertion_point;
4409              ## TODO: $current_insertion_point = just before $self->{nc};
4410    
4411              ## Para 5
4412              ## TODO: Run the $script->[0].
4413    
4414              ## Para 6
4415              ## TODO: $current_insertion_point = $old_insertion_point;
4416    
4417              ## Para 7
4418              ## TODO: if ($pending_external_script) {
4419                ## TODO: ...
4420              ## TODO: }
4421    
4422              !!!next-token;
4423              next B;
4424            } else {
4425              !!!cp ('t42');
4426    
4427              pop @{$self->{open_elements}};
4428    
4429              $self->{insertion_mode} &= ~ IN_CDATA_RCDATA_IM;
4430              !!!next-token;
4431              next B;
4432            }
4433          } elsif ($token->{type} == END_OF_FILE_TOKEN) {
4434            delete $self->{ignore_newline};
4435    
4436            !!!cp ('t44');
4437            !!!parse-error (type => 'not closed',
4438                            text => $self->{open_elements}->[-1]->[0]
4439                                ->manakai_local_name,
4440                            token => $token);
4441    
4442            #if ($self->{open_elements}->[-1]->[1] == SCRIPT_EL) {
4443            #  ## TODO: Mark as "already executed"
4444            #}
4445    
4446            pop @{$self->{open_elements}};
4447    
4448            $self->{insertion_mode} &= ~ IN_CDATA_RCDATA_IM;
4449            ## Reprocess.
4450            next B;
4451          } else {
4452            die "$0: $token->{type}: In CDATA/RCDATA: Unknown token type";        
4453          }
4454      } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM) {      } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM) {
4455        if ($token->{type} == CHARACTER_TOKEN) {        if ($token->{type} == CHARACTER_TOKEN) {
4456          !!!cp ('t87.1');          !!!cp ('t87.1');
# Line 4208  sub _tree_construction_main ($) { Line 4462  sub _tree_construction_main ($) {
4462               $self->{open_elements}->[-1]->[1] & FOREIGN_FLOW_CONTENT_EL) or               $self->{open_elements}->[-1]->[1] & FOREIGN_FLOW_CONTENT_EL) or
4463              not ($self->{open_elements}->[-1]->[1] & FOREIGN_EL) or              not ($self->{open_elements}->[-1]->[1] & FOREIGN_EL) or
4464              ($token->{tag_name} eq 'svg' and              ($token->{tag_name} eq 'svg' and
4465               $self->{open_elements}->[-1]->[1] & MML_AXML_EL)) {               $self->{open_elements}->[-1]->[1] == MML_AXML_EL)) {
4466            ## NOTE: "using the rules for secondary insertion mode"then"continue"            ## NOTE: "using the rules for secondary insertion mode"then"continue"
4467            !!!cp ('t87.2');            !!!cp ('t87.2');
4468            #            #
# Line 4309  sub _tree_construction_main ($) { Line 4563  sub _tree_construction_main ($) {
4563          pop @{$self->{open_elements}}          pop @{$self->{open_elements}}
4564              while $self->{open_elements}->[-1]->[1] & FOREIGN_EL;              while $self->{open_elements}->[-1]->[1] & FOREIGN_EL;
4565    
4566            ## NOTE: |<span><svg>| ... two parse errors, |<svg>| ... a parse error.
4567    
4568          $self->{insertion_mode} &= ~ IN_FOREIGN_CONTENT_IM;          $self->{insertion_mode} &= ~ IN_FOREIGN_CONTENT_IM;
4569          ## Reprocess.          ## Reprocess.
4570          next B;          next B;
# Line 4319  sub _tree_construction_main ($) { Line 4575  sub _tree_construction_main ($) {
4575    
4576      if ($self->{insertion_mode} & HEAD_IMS) {      if ($self->{insertion_mode} & HEAD_IMS) {
4577        if ($token->{type} == CHARACTER_TOKEN) {        if ($token->{type} == CHARACTER_TOKEN) {
4578          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {          if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
4579            unless ($self->{insertion_mode} == BEFORE_HEAD_IM) {            unless ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4580              !!!cp ('t88.2');              if ($self->{head_element_inserted}) {
4581              $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);                !!!cp ('t88.3');
4582                  $self->{open_elements}->[-1]->[0]->append_child
4583                    ($self->{document}->create_text_node ($1));
4584                  delete $self->{head_element_inserted};
4585                  ## NOTE: |</head> <link> |
4586                  #
4587                } else {
4588                  !!!cp ('t88.2');
4589                  $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
4590                  ## NOTE: |</head> &#x20;|
4591                  #
4592                }
4593            } else {            } else {
4594              !!!cp ('t88.1');              !!!cp ('t88.1');
4595              ## Ignore the token.              ## Ignore the token.
4596              !!!next-token;              #
             next B;  
4597            }            }
4598            unless (length $token->{data}) {            unless (length $token->{data}) {
4599              !!!cp ('t88');              !!!cp ('t88');
4600              !!!next-token;              !!!next-token;
4601              next B;              next B;
4602            }            }
4603    ## TODO: set $token->{column} appropriately
4604          }          }
4605    
4606          if ($self->{insertion_mode} == BEFORE_HEAD_IM) {          if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
# Line 4418  sub _tree_construction_main ($) { Line 4685  sub _tree_construction_main ($) {
4685            !!!cp ('t97');            !!!cp ('t97');
4686          }          }
4687    
4688              if ($token->{tag_name} eq 'base') {          if ($token->{tag_name} eq 'base') {
4689                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {            if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4690                  !!!cp ('t98');              !!!cp ('t98');
4691                  ## As if </noscript>              ## As if </noscript>
4692                  pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
4693                  !!!parse-error (type => 'in noscript', text => 'base',              !!!parse-error (type => 'in noscript', text => 'base',
4694                                  token => $token);                              token => $token);
4695                            
4696                  $self->{insertion_mode} = IN_HEAD_IM;              $self->{insertion_mode} = IN_HEAD_IM;
4697                  ## Reprocess in the "in head" insertion mode...              ## Reprocess in the "in head" insertion mode...
4698                } else {            } else {
4699                  !!!cp ('t99');              !!!cp ('t99');
4700                }            }
4701    
4702                ## NOTE: There is a "as if in head" code clone.            ## NOTE: There is a "as if in head" code clone.
4703                if ($self->{insertion_mode} == AFTER_HEAD_IM) {            if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4704                  !!!cp ('t100');              !!!cp ('t100');
4705                  !!!parse-error (type => 'after head',              !!!parse-error (type => 'after head',
4706                                  text => $token->{tag_name}, token => $token);                              text => $token->{tag_name}, token => $token);
4707                  push @{$self->{open_elements}},              push @{$self->{open_elements}},
4708                      [$self->{head_element}, $el_category->{head}];                  [$self->{head_element}, $el_category->{head}];
4709                } else {              $self->{head_element_inserted} = 1;
4710                  !!!cp ('t101');            } else {
4711                }              !!!cp ('t101');
4712                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);            }
4713                pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.            !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4714                pop @{$self->{open_elements}} # <head>            pop @{$self->{open_elements}};
4715                    if $self->{insertion_mode} == AFTER_HEAD_IM;            pop @{$self->{open_elements}} # <head>
4716                !!!nack ('t101.1');                if $self->{insertion_mode} == AFTER_HEAD_IM;
4717                !!!next-token;            !!!nack ('t101.1');
4718                next B;            !!!next-token;
4719              } elsif ($token->{tag_name} eq 'link') {            next B;
4720                ## NOTE: There is a "as if in head" code clone.          } elsif ($token->{tag_name} eq 'link') {
4721                if ($self->{insertion_mode} == AFTER_HEAD_IM) {            ## NOTE: There is a "as if in head" code clone.
4722                  !!!cp ('t102');            if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4723                  !!!parse-error (type => 'after head',              !!!cp ('t102');
4724                                  text => $token->{tag_name}, token => $token);              !!!parse-error (type => 'after head',
4725                  push @{$self->{open_elements}},                              text => $token->{tag_name}, token => $token);
4726                      [$self->{head_element}, $el_category->{head}];              push @{$self->{open_elements}},
4727                } else {                  [$self->{head_element}, $el_category->{head}];
4728                  !!!cp ('t103');              $self->{head_element_inserted} = 1;
4729                }            } else {
4730                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);              !!!cp ('t103');
4731                pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.            }
4732                pop @{$self->{open_elements}} # <head>            !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4733                    if $self->{insertion_mode} == AFTER_HEAD_IM;            pop @{$self->{open_elements}};
4734                !!!ack ('t103.1');            pop @{$self->{open_elements}} # <head>
4735                !!!next-token;                if $self->{insertion_mode} == AFTER_HEAD_IM;
4736                next B;            !!!ack ('t103.1');
4737              } elsif ($token->{tag_name} eq 'meta') {            !!!next-token;
4738                ## NOTE: There is a "as if in head" code clone.            next B;
4739                if ($self->{insertion_mode} == AFTER_HEAD_IM) {          } elsif ($token->{tag_name} eq 'command' or
4740                  !!!cp ('t104');                   $token->{tag_name} eq 'eventsource') {
4741                  !!!parse-error (type => 'after head',            if ($self->{insertion_mode} == IN_HEAD_IM) {
4742                                  text => $token->{tag_name}, token => $token);              ## NOTE: If the insertion mode at the time of the emission
4743                  push @{$self->{open_elements}},              ## of the token was "before head", $self->{insertion_mode}
4744                      [$self->{head_element}, $el_category->{head}];              ## is already changed to |IN_HEAD_IM|.
4745                } else {  
4746                  !!!cp ('t105');              ## NOTE: There is a "as if in head" code clone.
4747                }              !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4748                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);              pop @{$self->{open_elements}};
4749                my $meta_el = pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.              pop @{$self->{open_elements}} # <head>
4750                    if $self->{insertion_mode} == AFTER_HEAD_IM;
4751                !!!ack ('t103.2');
4752                !!!next-token;
4753                next B;
4754              } else {
4755                ## NOTE: "in head noscript" or "after head" insertion mode
4756                ## - in these cases, these tags are treated as same as
4757                ## normal in-body tags.
4758                !!!cp ('t103.3');
4759                #
4760              }
4761            } elsif ($token->{tag_name} eq 'meta') {
4762              ## NOTE: There is a "as if in head" code clone.
4763              if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4764                !!!cp ('t104');
4765                !!!parse-error (type => 'after head',
4766                                text => $token->{tag_name}, token => $token);
4767                push @{$self->{open_elements}},
4768                    [$self->{head_element}, $el_category->{head}];
4769                $self->{head_element_inserted} = 1;
4770              } else {
4771                !!!cp ('t105');
4772              }
4773              !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4774              my $meta_el = pop @{$self->{open_elements}};
4775    
4776                unless ($self->{confident}) {                unless ($self->{confident}) {
4777                  if ($token->{attributes}->{charset}) {                  if ($token->{attributes}->{charset}) {
# Line 4497  sub _tree_construction_main ($) { Line 4789  sub _tree_construction_main ($) {
4789                  } elsif ($token->{attributes}->{content}) {                  } elsif ($token->{attributes}->{content}) {
4790                    if ($token->{attributes}->{content}->{value}                    if ($token->{attributes}->{content}->{value}
4791                        =~ /[Cc][Hh][Aa][Rr][Ss][Ee][Tt]                        =~ /[Cc][Hh][Aa][Rr][Ss][Ee][Tt]
4792                            [\x09-\x0D\x20]*=                            [\x09\x0A\x0C\x0D\x20]*=
4793                            [\x09-\x0D\x20]*(?>"([^"]*)"|'([^']*)'|                            [\x09\x0A\x0C\x0D\x20]*(?>"([^"]*)"|'([^']*)'|
4794                            ([^"'\x09-\x0D\x20][^\x09-\x0D\x20\x3B]*))/x) {                            ([^"'\x09\x0A\x0C\x0D\x20]
4795                               [^\x09\x0A\x0C\x0D\x20\x3B]*))/x) {
4796                      !!!cp ('t107');                      !!!cp ('t107');
4797                      ## NOTE: Whether the encoding is supported or not is handled                      ## NOTE: Whether the encoding is supported or not is handled
4798                      ## in the {change_encoding} callback.                      ## in the {change_encoding} callback.
# Line 4536  sub _tree_construction_main ($) { Line 4829  sub _tree_construction_main ($) {
4829                !!!ack ('t110.1');                !!!ack ('t110.1');
4830                !!!next-token;                !!!next-token;
4831                next B;                next B;
4832              } elsif ($token->{tag_name} eq 'title') {          } elsif ($token->{tag_name} eq 'title') {
4833                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {            if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4834                  !!!cp ('t111');              !!!cp ('t111');
4835                  ## As if </noscript>              ## As if </noscript>
4836                  pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
4837                  !!!parse-error (type => 'in noscript', text => 'title',              !!!parse-error (type => 'in noscript', text => 'title',
4838                                  token => $token);                              token => $token);
4839                            
4840                  $self->{insertion_mode} = IN_HEAD_IM;              $self->{insertion_mode} = IN_HEAD_IM;
4841                  ## Reprocess in the "in head" insertion mode...              ## Reprocess in the "in head" insertion mode...
4842                } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {            } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4843                  !!!cp ('t112');              !!!cp ('t112');
4844                  !!!parse-error (type => 'after head',              !!!parse-error (type => 'after head',
4845                                  text => $token->{tag_name}, token => $token);                              text => $token->{tag_name}, token => $token);
4846                  push @{$self->{open_elements}},              push @{$self->{open_elements}},
4847                      [$self->{head_element}, $el_category->{head}];                  [$self->{head_element}, $el_category->{head}];
4848                } else {              $self->{head_element_inserted} = 1;
4849                  !!!cp ('t113');            } else {
4850                }              !!!cp ('t113');
4851              }
4852    
4853                ## NOTE: There is a "as if in head" code clone.            ## NOTE: There is a "as if in head" code clone.
4854                my $parent = defined $self->{head_element} ? $self->{head_element}            $parse_rcdata->(RCDATA_CONTENT_MODEL);
4855                    : $self->{open_elements}->[-1]->[0];            ## ISSUE: A spec bug [Bug 6038]
4856                $parse_rcdata->(RCDATA_CONTENT_MODEL);            splice @{$self->{open_elements}}, -2, 1, () # <head>
4857                pop @{$self->{open_elements}} # <head>                if ($self->{insertion_mode} & AFTER_HEAD_IM) == AFTER_HEAD_IM;
4858                    if $self->{insertion_mode} == AFTER_HEAD_IM;            next B;
4859                next B;          } elsif ($token->{tag_name} eq 'style' or
4860              } elsif ($token->{tag_name} eq 'style' or                   $token->{tag_name} eq 'noframes') {
4861                       $token->{tag_name} eq 'noframes') {            ## NOTE: Or (scripting is enabled and tag_name eq 'noscript' and
4862                ## NOTE: Or (scripting is enabled and tag_name eq 'noscript' and            ## insertion mode IN_HEAD_IM)
4863                ## insertion mode IN_HEAD_IM)            ## NOTE: There is a "as if in head" code clone.
4864                ## NOTE: There is a "as if in head" code clone.            if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4865                if ($self->{insertion_mode} == AFTER_HEAD_IM) {              !!!cp ('t114');
4866                  !!!cp ('t114');              !!!parse-error (type => 'after head',
4867                  !!!parse-error (type => 'after head',                              text => $token->{tag_name}, token => $token);
4868                                  text => $token->{tag_name}, token => $token);              push @{$self->{open_elements}},
4869                  push @{$self->{open_elements}},                  [$self->{head_element}, $el_category->{head}];
4870                      [$self->{head_element}, $el_category->{head}];              $self->{head_element_inserted} = 1;
4871                } else {            } else {
4872                  !!!cp ('t115');              !!!cp ('t115');
4873                }            }
4874                $parse_rcdata->(CDATA_CONTENT_MODEL);            $parse_rcdata->(CDATA_CONTENT_MODEL);
4875                pop @{$self->{open_elements}} # <head>            ## ISSUE: A spec bug [Bug 6038]
4876                    if $self->{insertion_mode} == AFTER_HEAD_IM;            splice @{$self->{open_elements}}, -2, 1, () # <head>
4877                next B;                if ($self->{insertion_mode} & AFTER_HEAD_IM) == AFTER_HEAD_IM;
4878              } elsif ($token->{tag_name} eq 'noscript') {            next B;
4879            } elsif ($token->{tag_name} eq 'noscript') {
4880                if ($self->{insertion_mode} == IN_HEAD_IM) {                if ($self->{insertion_mode} == IN_HEAD_IM) {
4881                  !!!cp ('t116');                  !!!cp ('t116');
4882                  ## NOTE: and scripting is disalbed                  ## NOTE: and scripting is disalbed
# Line 4602  sub _tree_construction_main ($) { Line 4897  sub _tree_construction_main ($) {
4897                  !!!cp ('t118');                  !!!cp ('t118');
4898                  #                  #
4899                }                }
4900              } elsif ($token->{tag_name} eq 'script') {          } elsif ($token->{tag_name} eq 'script') {
4901                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {            if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4902                  !!!cp ('t119');              !!!cp ('t119');
4903                  ## As if </noscript>              ## As if </noscript>
4904                  pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
4905                  !!!parse-error (type => 'in noscript', text => 'script',              !!!parse-error (type => 'in noscript', text => 'script',
4906                                  token => $token);                              token => $token);
4907                            
4908                  $self->{insertion_mode} = IN_HEAD_IM;              $self->{insertion_mode} = IN_HEAD_IM;
4909                  ## Reprocess in the "in head" insertion mode...              ## Reprocess in the "in head" insertion mode...
4910                } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {            } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4911                  !!!cp ('t120');              !!!cp ('t120');
4912                  !!!parse-error (type => 'after head',              !!!parse-error (type => 'after head',
4913                                  text => $token->{tag_name}, token => $token);                              text => $token->{tag_name}, token => $token);
4914                  push @{$self->{open_elements}},              push @{$self->{open_elements}},
4915                      [$self->{head_element}, $el_category->{head}];                  [$self->{head_element}, $el_category->{head}];
4916                } else {              $self->{head_element_inserted} = 1;
4917                  !!!cp ('t121');            } else {
4918                }              !!!cp ('t121');
4919              }
4920    
4921                ## NOTE: There is a "as if in head" code clone.            ## NOTE: There is a "as if in head" code clone.
4922                $script_start_tag->();            $script_start_tag->();
4923                pop @{$self->{open_elements}} # <head>            ## ISSUE: A spec bug  [Bug 6038]
4924                    if $self->{insertion_mode} == AFTER_HEAD_IM;            splice @{$self->{open_elements}}, -2, 1 # <head>
4925                next B;                if ($self->{insertion_mode} & AFTER_HEAD_IM) == AFTER_HEAD_IM;
4926              } elsif ($token->{tag_name} eq 'body' or            next B;
4927                       $token->{tag_name} eq 'frameset') {          } elsif ($token->{tag_name} eq 'body' or
4928                     $token->{tag_name} eq 'frameset') {
4929                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4930                  !!!cp ('t122');                  !!!cp ('t122');
4931                  ## As if </noscript>                  ## As if </noscript>
# Line 4763  sub _tree_construction_main ($) { Line 5060  sub _tree_construction_main ($) {
5060              } elsif ({              } elsif ({
5061                        body => 1, html => 1,                        body => 1, html => 1,
5062                       }->{$token->{tag_name}}) {                       }->{$token->{tag_name}}) {
5063                if ($self->{insertion_mode} == BEFORE_HEAD_IM or                ## TODO: This branch is entirely redundant.
5064                  if ($self->{insertion_mode} == BEFORE_HEAD_IM or
5065                    $self->{insertion_mode} == IN_HEAD_IM or                    $self->{insertion_mode} == IN_HEAD_IM or
5066                    $self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {                    $self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
5067                  !!!cp ('t140');                  !!!cp ('t140');
# Line 4935  sub _tree_construction_main ($) { Line 5233  sub _tree_construction_main ($) {
5233        } else {        } else {
5234          die "$0: $token->{type}: Unknown token type";          die "$0: $token->{type}: Unknown token type";
5235        }        }
   
           ## ISSUE: An issue in the spec.  
5236      } elsif ($self->{insertion_mode} & BODY_IMS) {      } elsif ($self->{insertion_mode} & BODY_IMS) {
5237            if ($token->{type} == CHARACTER_TOKEN) {            if ($token->{type} == CHARACTER_TOKEN) {
5238              !!!cp ('t150');              !!!cp ('t150');
# Line 4956  sub _tree_construction_main ($) { Line 5252  sub _tree_construction_main ($) {
5252                  ## have an element in table scope                  ## have an element in table scope
5253                  for (reverse 0..$#{$self->{open_elements}}) {                  for (reverse 0..$#{$self->{open_elements}}) {
5254                    my $node = $self->{open_elements}->[$_];                    my $node = $self->{open_elements}->[$_];
5255                    if ($node->[1] & TABLE_CELL_EL) {                    if ($node->[1] == TABLE_CELL_EL) {
5256                      !!!cp ('t151');                      !!!cp ('t151');
5257    
5258                      ## Close the cell                      ## Close the cell
# Line 4990  sub _tree_construction_main ($) { Line 5286  sub _tree_construction_main ($) {
5286                  INSCOPE: {                  INSCOPE: {
5287                    for (reverse 0..$#{$self->{open_elements}}) {                    for (reverse 0..$#{$self->{open_elements}}) {
5288                      my $node = $self->{open_elements}->[$_];                      my $node = $self->{open_elements}->[$_];
5289                      if ($node->[1] & CAPTION_EL) {                      if ($node->[1] == CAPTION_EL) {
5290                        !!!cp ('t155');                        !!!cp ('t155');
5291                        $i = $_;                        $i = $_;
5292                        last INSCOPE;                        last INSCOPE;
# Line 5016  sub _tree_construction_main ($) { Line 5312  sub _tree_construction_main ($) {
5312                    pop @{$self->{open_elements}};                    pop @{$self->{open_elements}};
5313                  }                  }
5314    
5315                  unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {                  unless ($self->{open_elements}->[-1]->[1] == CAPTION_EL) {
5316                    !!!cp ('t159');                    !!!cp ('t159');
5317                    !!!parse-error (type => 'not closed',                    !!!parse-error (type => 'not closed',
5318                                    text => $self->{open_elements}->[-1]->[0]                                    text => $self->{open_elements}->[-1]->[0]
# Line 5113  sub _tree_construction_main ($) { Line 5409  sub _tree_construction_main ($) {
5409                  INSCOPE: {                  INSCOPE: {
5410                    for (reverse 0..$#{$self->{open_elements}}) {                    for (reverse 0..$#{$self->{open_elements}}) {
5411                      my $node = $self->{open_elements}->[$_];                      my $node = $self->{open_elements}->[$_];
5412                      if ($node->[1] & CAPTION_EL) {                      if ($node->[1] == CAPTION_EL) {
5413                        !!!cp ('t171');                        !!!cp ('t171');
5414                        $i = $_;                        $i = $_;
5415                        last INSCOPE;                        last INSCOPE;
# Line 5138  sub _tree_construction_main ($) { Line 5434  sub _tree_construction_main ($) {
5434                    pop @{$self->{open_elements}};                    pop @{$self->{open_elements}};
5435                  }                  }
5436                                    
5437                  unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {                  unless ($self->{open_elements}->[-1]->[1] == CAPTION_EL) {
5438                    !!!cp ('t175');                    !!!cp ('t175');
5439                    !!!parse-error (type => 'not closed',                    !!!parse-error (type => 'not closed',
5440                                    text => $self->{open_elements}->[-1]->[0]                                    text => $self->{open_elements}->[-1]->[0]
# Line 5188  sub _tree_construction_main ($) { Line 5484  sub _tree_construction_main ($) {
5484                                line => $token->{line},                                line => $token->{line},
5485                                column => $token->{column}};                                column => $token->{column}};
5486                      next B;                      next B;
5487                    } elsif ($node->[1] & TABLE_CELL_EL) {                    } elsif ($node->[1] == TABLE_CELL_EL) {
5488                      !!!cp ('t180');                      !!!cp ('t180');
5489                      $tn = $node->[0]->manakai_local_name;                      $tn = $node->[0]->manakai_local_name;
5490                      ## NOTE: There is exactly one |td| or |th| element                      ## NOTE: There is exactly one |td| or |th| element
# Line 5217  sub _tree_construction_main ($) { Line 5513  sub _tree_construction_main ($) {
5513                my $i;                my $i;
5514                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5515                  my $node = $self->{open_elements}->[$_];                  my $node = $self->{open_elements}->[$_];
5516                  if ($node->[1] & CAPTION_EL) {                  if ($node->[1] == CAPTION_EL) {
5517                    !!!cp ('t184');                    !!!cp ('t184');
5518                    $i = $_;                    $i = $_;
5519                    last INSCOPE;                    last INSCOPE;
# Line 5241  sub _tree_construction_main ($) { Line 5537  sub _tree_construction_main ($) {
5537                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
5538                }                }
5539    
5540                unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {                unless ($self->{open_elements}->[-1]->[1] == CAPTION_EL) {
5541                  !!!cp ('t188');                  !!!cp ('t188');
5542                  !!!parse-error (type => 'not closed',                  !!!parse-error (type => 'not closed',
5543                                  text => $self->{open_elements}->[-1]->[0]                                  text => $self->{open_elements}->[-1]->[0]
# Line 5308  sub _tree_construction_main ($) { Line 5604  sub _tree_construction_main ($) {
5604      } elsif ($self->{insertion_mode} & TABLE_IMS) {      } elsif ($self->{insertion_mode} & TABLE_IMS) {
5605        if ($token->{type} == CHARACTER_TOKEN) {        if ($token->{type} == CHARACTER_TOKEN) {
5606          if (not $open_tables->[-1]->[1] and # tainted          if (not $open_tables->[-1]->[1] and # tainted
5607              $token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {              $token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
5608            $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);            $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
5609                                
5610            unless (length $token->{data}) {            unless (length $token->{data}) {
# Line 5322  sub _tree_construction_main ($) { Line 5618  sub _tree_construction_main ($) {
5618    
5619          !!!parse-error (type => 'in table:#text', token => $token);          !!!parse-error (type => 'in table:#text', token => $token);
5620    
5621              ## As if in body, but insert into foster parent element          ## NOTE: As if in body, but insert into the foster parent element.
5622              ## ISSUE: Spec says that "whenever a node would be inserted          $reconstruct_active_formatting_elements->($insert_to_foster);
             ## into the current node" while characters might not be  
             ## result in a new Text node.  
             $reconstruct_active_formatting_elements->($insert_to_foster);  
5623                            
5624              if ($self->{open_elements}->[-1]->[1] & TABLE_ROWS_EL) {          if ($self->{open_elements}->[-1]->[1] & TABLE_ROWS_EL) {
5625                # MUST            # MUST
5626                my $foster_parent_element;            my $foster_parent_element;
5627                my $next_sibling;            my $next_sibling;
5628                my $prev_sibling;            my $prev_sibling;
5629                OE: for (reverse 0..$#{$self->{open_elements}}) {            OE: for (reverse 0..$#{$self->{open_elements}}) {
5630                  if ($self->{open_elements}->[$_]->[1] & TABLE_EL) {              if ($self->{open_elements}->[$_]->[1] == TABLE_EL) {
5631                    my $parent = $self->{open_elements}->[$_]->[0]->parent_node;                my $parent = $self->{open_elements}->[$_]->[0]->parent_node;
5632                    if (defined $parent and $parent->node_type == 1) {                if (defined $parent and $parent->node_type == 1) {
5633                      !!!cp ('t196');                  $foster_parent_element = $parent;
5634                      $foster_parent_element = $parent;                  !!!cp ('t196');
5635                      $next_sibling = $self->{open_elements}->[$_]->[0];                  $next_sibling = $self->{open_elements}->[$_]->[0];
5636                      $prev_sibling = $next_sibling->previous_sibling;                  $prev_sibling = $next_sibling->previous_sibling;
5637                    } else {                  #
                     !!!cp ('t197');  
                     $foster_parent_element = $self->{open_elements}->[$_ - 1]->[0];  
                     $prev_sibling = $foster_parent_element->last_child;  
                   }  
                   last OE;  
                 }  
               } # OE  
               $foster_parent_element = $self->{open_elements}->[0]->[0] and  
               $prev_sibling = $foster_parent_element->last_child  
                 unless defined $foster_parent_element;  
               if (defined $prev_sibling and  
                   $prev_sibling->node_type == 3) {  
                 !!!cp ('t198');  
                 $prev_sibling->manakai_append_text ($token->{data});  
5638                } else {                } else {
5639                  !!!cp ('t199');                  !!!cp ('t197');
5640                  $foster_parent_element->insert_before                  $foster_parent_element = $self->{open_elements}->[$_ - 1]->[0];
5641                    ($self->{document}->create_text_node ($token->{data}),                  $prev_sibling = $foster_parent_element->last_child;
5642                     $next_sibling);                  #
5643                }                }
5644                  last OE;
5645                }
5646              } # OE
5647              $foster_parent_element = $self->{open_elements}->[0]->[0] and
5648              $prev_sibling = $foster_parent_element->last_child
5649                  unless defined $foster_parent_element;
5650              undef $prev_sibling unless $open_tables->[-1]->[2]; # ~node inserted
5651              if (defined $prev_sibling and
5652                  $prev_sibling->node_type == 3) {
5653                !!!cp ('t198');
5654                $prev_sibling->manakai_append_text ($token->{data});
5655              } else {
5656                !!!cp ('t199');
5657                $foster_parent_element->insert_before
5658                    ($self->{document}->create_text_node ($token->{data}),
5659                     $next_sibling);
5660              }
5661            $open_tables->[-1]->[1] = 1; # tainted            $open_tables->[-1]->[1] = 1; # tainted
5662              $open_tables->[-1]->[2] = 1; # ~node inserted
5663          } else {          } else {
5664              ## NOTE: Fragment case or in a foster parent'ed element
5665              ## (e.g. |<table><span>a|).  In fragment case, whether the
5666              ## character is appended to existing node or a new node is
5667              ## created is irrelevant, since the foster parent'ed nodes
5668              ## are discarded and fragment parsing does not invoke any
5669              ## script.
5670            !!!cp ('t200');            !!!cp ('t200');
5671            $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});            $self->{open_elements}->[-1]->[0]->manakai_append_text
5672                  ($token->{data});
5673          }          }
5674                            
5675          !!!next-token;          !!!next-token;
# Line 5402  sub _tree_construction_main ($) { Line 5706  sub _tree_construction_main ($) {
5706                pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
5707              }              }
5708                                    
5709                  $self->{insertion_mode} = IN_ROW_IM;              $self->{insertion_mode} = IN_ROW_IM;
5710                  if ($token->{tag_name} eq 'tr') {              if ($token->{tag_name} eq 'tr') {
5711                    !!!cp ('t204');                !!!cp ('t204');
5712                    !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5713                    !!!nack ('t204');                $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5714                    !!!next-token;                !!!nack ('t204');
5715                    next B;                !!!next-token;
5716                  } else {                next B;
5717                    !!!cp ('t205');              } else {
5718                    !!!insert-element ('tr',, $token);                !!!cp ('t205');
5719                    ## reprocess in the "in row" insertion mode                !!!insert-element ('tr',, $token);
5720                  }                ## reprocess in the "in row" insertion mode
5721                } else {              }
5722                  !!!cp ('t206');            } else {
5723                }              !!!cp ('t206');
5724              }
5725    
5726                ## Clear back to table row context                ## Clear back to table row context
5727                while (not ($self->{open_elements}->[-1]->[1]                while (not ($self->{open_elements}->[-1]->[1]
# Line 5425  sub _tree_construction_main ($) { Line 5730  sub _tree_construction_main ($) {
5730                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
5731                }                }
5732                                
5733                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);            !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5734                $self->{insertion_mode} = IN_CELL_IM;            $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5735              $self->{insertion_mode} = IN_CELL_IM;
5736    
5737                push @$active_formatting_elements, ['#marker', ''];            push @$active_formatting_elements, ['#marker', ''];
5738                                
5739                !!!nack ('t207.1');            !!!nack ('t207.1');
5740              !!!next-token;
5741              next B;
5742            } elsif ({
5743                      caption => 1, col => 1, colgroup => 1,
5744                      tbody => 1, tfoot => 1, thead => 1,
5745                      tr => 1, # $self->{insertion_mode} == IN_ROW_IM
5746                     }->{$token->{tag_name}}) {
5747              if ($self->{insertion_mode} == IN_ROW_IM) {
5748                ## As if </tr>
5749                ## have an element in table scope
5750                my $i;
5751                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5752                  my $node = $self->{open_elements}->[$_];
5753                  if ($node->[1] == TABLE_ROW_EL) {
5754                    !!!cp ('t208');
5755                    $i = $_;
5756                    last INSCOPE;
5757                  } elsif ($node->[1] & TABLE_SCOPING_EL) {
5758                    !!!cp ('t209');
5759                    last INSCOPE;
5760                  }
5761                } # INSCOPE
5762                unless (defined $i) {
5763                  !!!cp ('t210');
5764                  ## TODO: This type is wrong.
5765                  !!!parse-error (type => 'unmacthed end tag',
5766                                  text => $token->{tag_name}, token => $token);
5767                  ## Ignore the token
5768                  !!!nack ('t210.1');
5769                !!!next-token;                !!!next-token;
5770                next B;                next B;
5771              } elsif ({              }
                       caption => 1, col => 1, colgroup => 1,  
                       tbody => 1, tfoot => 1, thead => 1,  
                       tr => 1, # $self->{insertion_mode} == IN_ROW_IM  
                      }->{$token->{tag_name}}) {  
               if ($self->{insertion_mode} == IN_ROW_IM) {  
                 ## As if </tr>  
                 ## have an element in table scope  
                 my $i;  
                 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
                   my $node = $self->{open_elements}->[$_];  
                   if ($node->[1] & TABLE_ROW_EL) {  
                     !!!cp ('t208');  
                     $i = $_;  
                     last INSCOPE;  
                   } elsif ($node->[1] & TABLE_SCOPING_EL) {  
                     !!!cp ('t209');  
                     last INSCOPE;  
                   }  
                 } # INSCOPE  
                 unless (defined $i) {  
                   !!!cp ('t210');  
 ## TODO: This type is wrong.  
                   !!!parse-error (type => 'unmacthed end tag',  
                                   text => $token->{tag_name}, token => $token);  
                   ## Ignore the token  
                   !!!nack ('t210.1');  
                   !!!next-token;  
                   next B;  
                 }  
5772                                    
5773                  ## Clear back to table row context                  ## Clear back to table row context
5774                  while (not ($self->{open_elements}->[-1]->[1]                  while (not ($self->{open_elements}->[-1]->[1]
# Line 5490  sub _tree_construction_main ($) { Line 5796  sub _tree_construction_main ($) {
5796                  my $i;                  my $i;
5797                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5798                    my $node = $self->{open_elements}->[$_];                    my $node = $self->{open_elements}->[$_];
5799                    if ($node->[1] & TABLE_ROW_GROUP_EL) {                    if ($node->[1] == TABLE_ROW_GROUP_EL) {
5800                      !!!cp ('t214');                      !!!cp ('t214');
5801                      $i = $_;                      $i = $_;
5802                      last INSCOPE;                      last INSCOPE;
# Line 5532  sub _tree_construction_main ($) { Line 5838  sub _tree_construction_main ($) {
5838                  !!!cp ('t218');                  !!!cp ('t218');
5839                }                }
5840    
5841                if ($token->{tag_name} eq 'col') {            if ($token->{tag_name} eq 'col') {
5842                  ## Clear back to table context              ## Clear back to table context
5843                  while (not ($self->{open_elements}->[-1]->[1]              while (not ($self->{open_elements}->[-1]->[1]
5844                                  & TABLE_SCOPING_EL)) {                              & TABLE_SCOPING_EL)) {
5845                    !!!cp ('t219');                !!!cp ('t219');
5846                    ## ISSUE: Can this state be reached?                ## ISSUE: Can this state be reached?
5847                    pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
5848                  }              }
5849                                
5850                  !!!insert-element ('colgroup',, $token);              !!!insert-element ('colgroup',, $token);
5851                  $self->{insertion_mode} = IN_COLUMN_GROUP_IM;              $self->{insertion_mode} = IN_COLUMN_GROUP_IM;
5852                  ## reprocess              ## reprocess
5853                  !!!ack-later;              $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5854                  next B;              !!!ack-later;
5855                } elsif ({              next B;
5856                          caption => 1,            } elsif ({
5857                          colgroup => 1,                      caption => 1,
5858                          tbody => 1, tfoot => 1, thead => 1,                      colgroup => 1,
5859                         }->{$token->{tag_name}}) {                      tbody => 1, tfoot => 1, thead => 1,
5860                  ## Clear back to table context                     }->{$token->{tag_name}}) {
5861                ## Clear back to table context
5862                  while (not ($self->{open_elements}->[-1]->[1]                  while (not ($self->{open_elements}->[-1]->[1]
5863                                  & TABLE_SCOPING_EL)) {                                  & TABLE_SCOPING_EL)) {
5864                    !!!cp ('t220');                    !!!cp ('t220');
# Line 5559  sub _tree_construction_main ($) { Line 5866  sub _tree_construction_main ($) {
5866                    pop @{$self->{open_elements}};                    pop @{$self->{open_elements}};
5867                  }                  }
5868                                    
5869                  push @$active_formatting_elements, ['#marker', '']              push @$active_formatting_elements, ['#marker', '']
5870                      if $token->{tag_name} eq 'caption';                  if $token->{tag_name} eq 'caption';
5871                                    
5872                  !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);              !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5873                  $self->{insertion_mode} = {              $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5874                                             caption => IN_CAPTION_IM,              $self->{insertion_mode} = {
5875                                             colgroup => IN_COLUMN_GROUP_IM,                                         caption => IN_CAPTION_IM,
5876                                             tbody => IN_TABLE_BODY_IM,                                         colgroup => IN_COLUMN_GROUP_IM,
5877                                             tfoot => IN_TABLE_BODY_IM,                                         tbody => IN_TABLE_BODY_IM,
5878                                             thead => IN_TABLE_BODY_IM,                                         tfoot => IN_TABLE_BODY_IM,
5879                                            }->{$token->{tag_name}};                                         thead => IN_TABLE_BODY_IM,
5880                  !!!next-token;                                        }->{$token->{tag_name}};
5881                  !!!nack ('t220.1');              !!!next-token;
5882                  next B;              !!!nack ('t220.1');
5883                } else {              next B;
5884                  die "$0: in table: <>: $token->{tag_name}";            } else {
5885                }              die "$0: in table: <>: $token->{tag_name}";
5886              }
5887              } elsif ($token->{tag_name} eq 'table') {              } elsif ($token->{tag_name} eq 'table') {
5888                !!!parse-error (type => 'not closed',                !!!parse-error (type => 'not closed',
5889                                text => $self->{open_elements}->[-1]->[0]                                text => $self->{open_elements}->[-1]->[0]
# Line 5587  sub _tree_construction_main ($) { Line 5895  sub _tree_construction_main ($) {
5895                my $i;                my $i;
5896                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5897                  my $node = $self->{open_elements}->[$_];                  my $node = $self->{open_elements}->[$_];
5898                  if ($node->[1] & TABLE_EL) {                  if ($node->[1] == TABLE_EL) {
5899                    !!!cp ('t221');                    !!!cp ('t221');
5900                    $i = $_;                    $i = $_;
5901                    last INSCOPE;                    last INSCOPE;
# Line 5614  sub _tree_construction_main ($) { Line 5922  sub _tree_construction_main ($) {
5922                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
5923                }                }
5924    
5925                unless ($self->{open_elements}->[-1]->[1] & TABLE_EL) {                unless ($self->{open_elements}->[-1]->[1] == TABLE_EL) {
5926                  !!!cp ('t225');                  !!!cp ('t225');
5927                  ## NOTE: |<table><tr><table>|                  ## NOTE: |<table><tr><table>|
5928                  !!!parse-error (type => 'not closed',                  !!!parse-error (type => 'not closed',
# Line 5638  sub _tree_construction_main ($) { Line 5946  sub _tree_construction_main ($) {
5946              !!!cp ('t227.8');              !!!cp ('t227.8');
5947              ## NOTE: This is a "as if in head" code clone.              ## NOTE: This is a "as if in head" code clone.
5948              $parse_rcdata->(CDATA_CONTENT_MODEL);              $parse_rcdata->(CDATA_CONTENT_MODEL);
5949                $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5950              next B;              next B;
5951            } else {            } else {
5952              !!!cp ('t227.7');              !!!cp ('t227.7');
# Line 5648  sub _tree_construction_main ($) { Line 5957  sub _tree_construction_main ($) {
5957              !!!cp ('t227.6');              !!!cp ('t227.6');
5958              ## NOTE: This is a "as if in head" code clone.              ## NOTE: This is a "as if in head" code clone.
5959              $script_start_tag->();              $script_start_tag->();
5960                $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5961              next B;              next B;
5962            } else {            } else {
5963              !!!cp ('t227.5');              !!!cp ('t227.5');
# Line 5663  sub _tree_construction_main ($) { Line 5973  sub _tree_construction_main ($) {
5973                                  text => $token->{tag_name}, token => $token);                                  text => $token->{tag_name}, token => $token);
5974    
5975                  !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);                  !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5976                    $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5977    
5978                  ## TODO: form element pointer                  ## TODO: form element pointer
5979    
# Line 5700  sub _tree_construction_main ($) { Line 6011  sub _tree_construction_main ($) {
6011                my $i;                my $i;
6012                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6013                  my $node = $self->{open_elements}->[$_];                  my $node = $self->{open_elements}->[$_];
6014                  if ($node->[1] & TABLE_ROW_EL) {                  if ($node->[1] == TABLE_ROW_EL) {
6015                    !!!cp ('t228');                    !!!cp ('t228');
6016                    $i = $_;                    $i = $_;
6017                    last INSCOPE;                    last INSCOPE;
# Line 5741  sub _tree_construction_main ($) { Line 6052  sub _tree_construction_main ($) {
6052                  my $i;                  my $i;
6053                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6054                    my $node = $self->{open_elements}->[$_];                    my $node = $self->{open_elements}->[$_];
6055                    if ($node->[1] & TABLE_ROW_EL) {                    if ($node->[1] == TABLE_ROW_EL) {
6056                      !!!cp ('t233');                      !!!cp ('t233');
6057                      $i = $_;                      $i = $_;
6058                      last INSCOPE;                      last INSCOPE;
# Line 5779  sub _tree_construction_main ($) { Line 6090  sub _tree_construction_main ($) {
6090                  my $i;                  my $i;
6091                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6092                    my $node = $self->{open_elements}->[$_];                    my $node = $self->{open_elements}->[$_];
6093                    if ($node->[1] & TABLE_ROW_GROUP_EL) {                    if ($node->[1] == TABLE_ROW_GROUP_EL) {
6094                      !!!cp ('t237');                      !!!cp ('t237');
6095                      $i = $_;                      $i = $_;
6096                      last INSCOPE;                      last INSCOPE;
# Line 5826  sub _tree_construction_main ($) { Line 6137  sub _tree_construction_main ($) {
6137                my $i;                my $i;
6138                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6139                  my $node = $self->{open_elements}->[$_];                  my $node = $self->{open_elements}->[$_];
6140                  if ($node->[1] & TABLE_EL) {                  if ($node->[1] == TABLE_EL) {
6141                    !!!cp ('t241');                    !!!cp ('t241');
6142                    $i = $_;                    $i = $_;
6143                    last INSCOPE;                    last INSCOPE;
# Line 5885  sub _tree_construction_main ($) { Line 6196  sub _tree_construction_main ($) {
6196                  my $i;                  my $i;
6197                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6198                    my $node = $self->{open_elements}->[$_];                    my $node = $self->{open_elements}->[$_];
6199                    if ($node->[1] & TABLE_ROW_EL) {                    if ($node->[1] == TABLE_ROW_EL) {
6200                      !!!cp ('t250');                      !!!cp ('t250');
6201                      $i = $_;                      $i = $_;
6202                      last INSCOPE;                      last INSCOPE;
# Line 5975  sub _tree_construction_main ($) { Line 6286  sub _tree_construction_main ($) {
6286            #            #
6287          }          }
6288        } elsif ($token->{type} == END_OF_FILE_TOKEN) {        } elsif ($token->{type} == END_OF_FILE_TOKEN) {
6289          unless ($self->{open_elements}->[-1]->[1] & HTML_EL and          unless ($self->{open_elements}->[-1]->[1] == HTML_EL and
6290                  @{$self->{open_elements}} == 1) { # redundant, maybe                  @{$self->{open_elements}} == 1) { # redundant, maybe
6291            !!!parse-error (type => 'in body:#eof', token => $token);            !!!parse-error (type => 'in body:#eof', token => $token);
6292            !!!cp ('t259.1');            !!!cp ('t259.1');
# Line 5992  sub _tree_construction_main ($) { Line 6303  sub _tree_construction_main ($) {
6303        }        }
6304      } elsif ($self->{insertion_mode} == IN_COLUMN_GROUP_IM) {      } elsif ($self->{insertion_mode} == IN_COLUMN_GROUP_IM) {
6305            if ($token->{type} == CHARACTER_TOKEN) {            if ($token->{type} == CHARACTER_TOKEN) {
6306              if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {              if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
6307                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
6308                unless (length $token->{data}) {                unless (length $token->{data}) {
6309                  !!!cp ('t260');                  !!!cp ('t260');
# Line 6017  sub _tree_construction_main ($) { Line 6328  sub _tree_construction_main ($) {
6328              }              }
6329            } elsif ($token->{type} == END_TAG_TOKEN) {            } elsif ($token->{type} == END_TAG_TOKEN) {
6330              if ($token->{tag_name} eq 'colgroup') {              if ($token->{tag_name} eq 'colgroup') {
6331                if ($self->{open_elements}->[-1]->[1] & HTML_EL) {                if ($self->{open_elements}->[-1]->[1] == HTML_EL) {
6332                  !!!cp ('t264');                  !!!cp ('t264');
6333                  !!!parse-error (type => 'unmatched end tag',                  !!!parse-error (type => 'unmatched end tag',
6334                                  text => 'colgroup', token => $token);                                  text => 'colgroup', token => $token);
# Line 6043  sub _tree_construction_main ($) { Line 6354  sub _tree_construction_main ($) {
6354                #                #
6355              }              }
6356        } elsif ($token->{type} == END_OF_FILE_TOKEN) {        } elsif ($token->{type} == END_OF_FILE_TOKEN) {
6357          if ($self->{open_elements}->[-1]->[1] & HTML_EL and          if ($self->{open_elements}->[-1]->[1] == HTML_EL and
6358              @{$self->{open_elements}} == 1) { # redundant, maybe              @{$self->{open_elements}} == 1) { # redundant, maybe
6359            !!!cp ('t270.2');            !!!cp ('t270.2');
6360            ## Stop parsing.            ## Stop parsing.
# Line 6061  sub _tree_construction_main ($) { Line 6372  sub _tree_construction_main ($) {
6372        }        }
6373    
6374            ## As if </colgroup>            ## As if </colgroup>
6375            if ($self->{open_elements}->[-1]->[1] & HTML_EL) {            if ($self->{open_elements}->[-1]->[1] == HTML_EL) {
6376              !!!cp ('t269');              !!!cp ('t269');
6377  ## TODO: Wrong error type?  ## TODO: Wrong error type?
6378              !!!parse-error (type => 'unmatched end tag',              !!!parse-error (type => 'unmatched end tag',
# Line 6086  sub _tree_construction_main ($) { Line 6397  sub _tree_construction_main ($) {
6397          next B;          next B;
6398        } elsif ($token->{type} == START_TAG_TOKEN) {        } elsif ($token->{type} == START_TAG_TOKEN) {
6399          if ($token->{tag_name} eq 'option') {          if ($token->{tag_name} eq 'option') {
6400            if ($self->{open_elements}->[-1]->[1] & OPTION_EL) {            if ($self->{open_elements}->[-1]->[1] == OPTION_EL) {
6401              !!!cp ('t272');              !!!cp ('t272');
6402              ## As if </option>              ## As if </option>
6403              pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
# Line 6099  sub _tree_construction_main ($) { Line 6410  sub _tree_construction_main ($) {
6410            !!!next-token;            !!!next-token;
6411            next B;            next B;
6412          } elsif ($token->{tag_name} eq 'optgroup') {          } elsif ($token->{tag_name} eq 'optgroup') {
6413            if ($self->{open_elements}->[-1]->[1] & OPTION_EL) {            if ($self->{open_elements}->[-1]->[1] == OPTION_EL) {
6414              !!!cp ('t274');              !!!cp ('t274');
6415              ## As if </option>              ## As if </option>
6416              pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
# Line 6107  sub _tree_construction_main ($) { Line 6418  sub _tree_construction_main ($) {
6418              !!!cp ('t275');              !!!cp ('t275');
6419            }            }
6420    
6421            if ($self->{open_elements}->[-1]->[1] & OPTGROUP_EL) {            if ($self->{open_elements}->[-1]->[1] == OPTGROUP_EL) {
6422              !!!cp ('t276');              !!!cp ('t276');
6423              ## As if </optgroup>              ## As if </optgroup>
6424              pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
# Line 6137  sub _tree_construction_main ($) { Line 6448  sub _tree_construction_main ($) {
6448            my $i;            my $i;
6449            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6450              my $node = $self->{open_elements}->[$_];              my $node = $self->{open_elements}->[$_];
6451              if ($node->[1] & SELECT_EL) {              if ($node->[1] == SELECT_EL) {
6452                !!!cp ('t278');                !!!cp ('t278');
6453                $i = $_;                $i = $_;
6454                last INSCOPE;                last INSCOPE;
# Line 6182  sub _tree_construction_main ($) { Line 6493  sub _tree_construction_main ($) {
6493          }          }
6494        } elsif ($token->{type} == END_TAG_TOKEN) {        } elsif ($token->{type} == END_TAG_TOKEN) {
6495          if ($token->{tag_name} eq 'optgroup') {          if ($token->{tag_name} eq 'optgroup') {
6496            if ($self->{open_elements}->[-1]->[1] & OPTION_EL and            if ($self->{open_elements}->[-1]->[1] == OPTION_EL and
6497                $self->{open_elements}->[-2]->[1] & OPTGROUP_EL) {                $self->{open_elements}->[-2]->[1] == OPTGROUP_EL) {
6498              !!!cp ('t283');              !!!cp ('t283');
6499              ## As if </option>              ## As if </option>
6500              splice @{$self->{open_elements}}, -2;              splice @{$self->{open_elements}}, -2;
6501            } elsif ($self->{open_elements}->[-1]->[1] & OPTGROUP_EL) {            } elsif ($self->{open_elements}->[-1]->[1] == OPTGROUP_EL) {
6502              !!!cp ('t284');              !!!cp ('t284');
6503              pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
6504            } else {            } else {
# Line 6200  sub _tree_construction_main ($) { Line 6511  sub _tree_construction_main ($) {
6511            !!!next-token;            !!!next-token;
6512            next B;            next B;
6513          } elsif ($token->{tag_name} eq 'option') {          } elsif ($token->{tag_name} eq 'option') {
6514            if ($self->{open_elements}->[-1]->[1] & OPTION_EL) {            if ($self->{open_elements}->[-1]->[1] == OPTION_EL) {
6515              !!!cp ('t286');              !!!cp ('t286');
6516              pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
6517            } else {            } else {
# Line 6217  sub _tree_construction_main ($) { Line 6528  sub _tree_construction_main ($) {
6528            my $i;            my $i;
6529            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6530              my $node = $self->{open_elements}->[$_];              my $node = $self->{open_elements}->[$_];
6531              if ($node->[1] & SELECT_EL) {              if ($node->[1] == SELECT_EL) {
6532                !!!cp ('t288');                !!!cp ('t288');
6533                $i = $_;                $i = $_;
6534                last INSCOPE;                last INSCOPE;
# Line 6279  sub _tree_construction_main ($) { Line 6590  sub _tree_construction_main ($) {
6590            undef $i;            undef $i;
6591            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6592              my $node = $self->{open_elements}->[$_];              my $node = $self->{open_elements}->[$_];
6593              if ($node->[1] & SELECT_EL) {              if ($node->[1] == SELECT_EL) {
6594                !!!cp ('t295');                !!!cp ('t295');
6595                $i = $_;                $i = $_;
6596                last INSCOPE;                last INSCOPE;
# Line 6318  sub _tree_construction_main ($) { Line 6629  sub _tree_construction_main ($) {
6629            next B;            next B;
6630          }          }
6631        } elsif ($token->{type} == END_OF_FILE_TOKEN) {        } elsif ($token->{type} == END_OF_FILE_TOKEN) {
6632          unless ($self->{open_elements}->[-1]->[1] & HTML_EL and          unless ($self->{open_elements}->[-1]->[1] == HTML_EL and
6633                  @{$self->{open_elements}} == 1) { # redundant, maybe                  @{$self->{open_elements}} == 1) { # redundant, maybe
6634            !!!cp ('t299.1');            !!!cp ('t299.1');
6635            !!!parse-error (type => 'in body:#eof', token => $token);            !!!parse-error (type => 'in body:#eof', token => $token);
# Line 6333  sub _tree_construction_main ($) { Line 6644  sub _tree_construction_main ($) {
6644        }        }
6645      } elsif ($self->{insertion_mode} & BODY_AFTER_IMS) {      } elsif ($self->{insertion_mode} & BODY_AFTER_IMS) {
6646        if ($token->{type} == CHARACTER_TOKEN) {        if ($token->{type} == CHARACTER_TOKEN) {
6647          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {          if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
6648            my $data = $1;            my $data = $1;
6649            ## As if in body            ## As if in body
6650            $reconstruct_active_formatting_elements->($insert_to_current);            $reconstruct_active_formatting_elements->($insert_to_current);
# Line 6350  sub _tree_construction_main ($) { Line 6661  sub _tree_construction_main ($) {
6661          if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {          if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
6662            !!!cp ('t301');            !!!cp ('t301');
6663            !!!parse-error (type => 'after html:#text', token => $token);            !!!parse-error (type => 'after html:#text', token => $token);
6664              #
           ## Reprocess in the "after body" insertion mode.  
6665          } else {          } else {
6666            !!!cp ('t302');            !!!cp ('t302');
6667              ## "after body" insertion mode
6668              !!!parse-error (type => 'after body:#text', token => $token);
6669              #
6670          }          }
           
         ## "after body" insertion mode  
         !!!parse-error (type => 'after body:#text', token => $token);  
6671    
6672          $self->{insertion_mode} = IN_BODY_IM;          $self->{insertion_mode} = IN_BODY_IM;
6673          ## reprocess          ## reprocess
# Line 6367  sub _tree_construction_main ($) { Line 6677  sub _tree_construction_main ($) {
6677            !!!cp ('t303');            !!!cp ('t303');
6678            !!!parse-error (type => 'after html',            !!!parse-error (type => 'after html',
6679                            text => $token->{tag_name}, token => $token);                            text => $token->{tag_name}, token => $token);
6680                        #
           ## Reprocess in the "after body" insertion mode.  
6681          } else {          } else {
6682            !!!cp ('t304');            !!!cp ('t304');
6683              ## "after body" insertion mode
6684              !!!parse-error (type => 'after body',
6685                              text => $token->{tag_name}, token => $token);
6686              #
6687          }          }
6688    
         ## "after body" insertion mode  
         !!!parse-error (type => 'after body',  
                         text => $token->{tag_name}, token => $token);  
   
6689          $self->{insertion_mode} = IN_BODY_IM;          $self->{insertion_mode} = IN_BODY_IM;
6690          !!!ack-later;          !!!ack-later;
6691          ## reprocess          ## reprocess
# Line 6387  sub _tree_construction_main ($) { Line 6696  sub _tree_construction_main ($) {
6696            !!!parse-error (type => 'after html:/',            !!!parse-error (type => 'after html:/',
6697                            text => $token->{tag_name}, token => $token);                            text => $token->{tag_name}, token => $token);
6698                        
6699            $self->{insertion_mode} = AFTER_BODY_IM;            $self->{insertion_mode} = IN_BODY_IM;
6700            ## Reprocess in the "after body" insertion mode.            ## Reprocess.
6701              next B;
6702          } else {          } else {
6703            !!!cp ('t306');            !!!cp ('t306');
6704          }          }
# Line 6426  sub _tree_construction_main ($) { Line 6736  sub _tree_construction_main ($) {
6736        }        }
6737      } elsif ($self->{insertion_mode} & FRAME_IMS) {      } elsif ($self->{insertion_mode} & FRAME_IMS) {
6738        if ($token->{type} == CHARACTER_TOKEN) {        if ($token->{type} == CHARACTER_TOKEN) {
6739          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {          if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
6740            $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);            $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
6741                        
6742            unless (length $token->{data}) {            unless (length $token->{data}) {
# Line 6436  sub _tree_construction_main ($) { Line 6746  sub _tree_construction_main ($) {
6746            }            }
6747          }          }
6748                    
6749          if ($token->{data} =~ s/^[^\x09\x0A\x0B\x0C\x20]+//) {          if ($token->{data} =~ s/^[^\x09\x0A\x0C\x20]+//) {
6750            if ($self->{insertion_mode} == IN_FRAMESET_IM) {            if ($self->{insertion_mode} == IN_FRAMESET_IM) {
6751              !!!cp ('t311');              !!!cp ('t311');
6752              !!!parse-error (type => 'in frameset:#text', token => $token);              !!!parse-error (type => 'in frameset:#text', token => $token);
# Line 6506  sub _tree_construction_main ($) { Line 6816  sub _tree_construction_main ($) {
6816        } elsif ($token->{type} == END_TAG_TOKEN) {        } elsif ($token->{type} == END_TAG_TOKEN) {
6817          if ($token->{tag_name} eq 'frameset' and          if ($token->{tag_name} eq 'frameset' and
6818              $self->{insertion_mode} == IN_FRAMESET_IM) {              $self->{insertion_mode} == IN_FRAMESET_IM) {
6819            if ($self->{open_elements}->[-1]->[1] & HTML_EL and            if ($self->{open_elements}->[-1]->[1] == HTML_EL and
6820                @{$self->{open_elements}} == 1) {                @{$self->{open_elements}} == 1) {
6821              !!!cp ('t325');              !!!cp ('t325');
6822              !!!parse-error (type => 'unmatched end tag',              !!!parse-error (type => 'unmatched end tag',
# Line 6520  sub _tree_construction_main ($) { Line 6830  sub _tree_construction_main ($) {
6830            }            }
6831    
6832            if (not defined $self->{inner_html_node} and            if (not defined $self->{inner_html_node} and
6833                not ($self->{open_elements}->[-1]->[1] & FRAMESET_EL)) {                not ($self->{open_elements}->[-1]->[1] == FRAMESET_EL)) {
6834              !!!cp ('t327');              !!!cp ('t327');
6835              $self->{insertion_mode} = AFTER_FRAMESET_IM;              $self->{insertion_mode} = AFTER_FRAMESET_IM;
6836            } else {            } else {
# Line 6552  sub _tree_construction_main ($) { Line 6862  sub _tree_construction_main ($) {
6862            next B;            next B;
6863          }          }
6864        } elsif ($token->{type} == END_OF_FILE_TOKEN) {        } elsif ($token->{type} == END_OF_FILE_TOKEN) {
6865          unless ($self->{open_elements}->[-1]->[1] & HTML_EL and          unless ($self->{open_elements}->[-1]->[1] == HTML_EL and
6866                  @{$self->{open_elements}} == 1) { # redundant, maybe                  @{$self->{open_elements}} == 1) { # redundant, maybe
6867            !!!cp ('t331.1');            !!!cp ('t331.1');
6868            !!!parse-error (type => 'in body:#eof', token => $token);            !!!parse-error (type => 'in body:#eof', token => $token);
# Line 6565  sub _tree_construction_main ($) { Line 6875  sub _tree_construction_main ($) {
6875        } else {        } else {
6876          die "$0: $token->{type}: Unknown token type";          die "$0: $token->{type}: Unknown token type";
6877        }        }
   
       ## ISSUE: An issue in spec here  
6878      } else {      } else {
6879        die "$0: $self->{insertion_mode}: Unknown insertion mode";        die "$0: $self->{insertion_mode}: Unknown insertion mode";
6880      }      }
# Line 6584  sub _tree_construction_main ($) { Line 6892  sub _tree_construction_main ($) {
6892          $parse_rcdata->(CDATA_CONTENT_MODEL);          $parse_rcdata->(CDATA_CONTENT_MODEL);
6893          next B;          next B;
6894        } elsif ({        } elsif ({
6895                  base => 1, link => 1,                  base => 1, command => 1, eventsource => 1, link => 1,
6896                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
6897          !!!cp ('t334');          !!!cp ('t334');
6898          ## NOTE: This is an "as if in head" code clone, only "-t" differs          ## NOTE: This is an "as if in head" code clone, only "-t" differs
6899          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
6900          pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.          pop @{$self->{open_elements}};
6901          !!!ack ('t334.1');          !!!ack ('t334.1');
6902          !!!next-token;          !!!next-token;
6903          next B;          next B;
6904        } elsif ($token->{tag_name} eq 'meta') {        } elsif ($token->{tag_name} eq 'meta') {
6905          ## NOTE: This is an "as if in head" code clone, only "-t" differs          ## NOTE: This is an "as if in head" code clone, only "-t" differs
6906          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
6907          my $meta_el = pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.          my $meta_el = pop @{$self->{open_elements}};
6908    
6909          unless ($self->{confident}) {          unless ($self->{confident}) {
6910            if ($token->{attributes}->{charset}) {            if ($token->{attributes}->{charset}) {
# Line 6613  sub _tree_construction_main ($) { Line 6921  sub _tree_construction_main ($) {
6921            } elsif ($token->{attributes}->{content}) {            } elsif ($token->{attributes}->{content}) {
6922              if ($token->{attributes}->{content}->{value}              if ($token->{attributes}->{content}->{value}
6923                  =~ /[Cc][Hh][Aa][Rr][Ss][Ee][Tt]                  =~ /[Cc][Hh][Aa][Rr][Ss][Ee][Tt]
6924                      [\x09-\x0D\x20]*=                      [\x09\x0A\x0C\x0D\x20]*=
6925                      [\x09-\x0D\x20]*(?>"([^"]*)"|'([^']*)'|                      [\x09\x0A\x0C\x0D\x20]*(?>"([^"]*)"|'([^']*)'|
6926                      ([^"'\x09-\x0D\x20][^\x09-\x0D\x20\x3B]*))/x) {                      ([^"'\x09\x0A\x0C\x0D\x20][^\x09\x0A\x0C\x0D\x20\x3B]*))
6927                       /x) {
6928                !!!cp ('t336');                !!!cp ('t336');
6929                ## NOTE: Whether the encoding is supported or not is handled                ## NOTE: Whether the encoding is supported or not is handled
6930                ## in the {change_encoding} callback.                ## in the {change_encoding} callback.
# Line 6656  sub _tree_construction_main ($) { Line 6965  sub _tree_construction_main ($) {
6965          !!!parse-error (type => 'in body', text => 'body', token => $token);          !!!parse-error (type => 'in body', text => 'body', token => $token);
6966                                
6967          if (@{$self->{open_elements}} == 1 or          if (@{$self->{open_elements}} == 1 or
6968              not ($self->{open_elements}->[1]->[1] & BODY_EL)) {              not ($self->{open_elements}->[1]->[1] == BODY_EL)) {
6969            !!!cp ('t342');            !!!cp ('t342');
6970            ## Ignore the token            ## Ignore the token
6971          } else {          } else {
# Line 6674  sub _tree_construction_main ($) { Line 6983  sub _tree_construction_main ($) {
6983          !!!next-token;          !!!next-token;
6984          next B;          next B;
6985        } elsif ({        } elsif ({
6986                  address => 1, blockquote => 1, center => 1, dir => 1,                  ## NOTE: Start tags for non-phrasing flow content elements
6987                  div => 1, dl => 1, fieldset => 1,  
6988                  h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,                  ## NOTE: The normal one
6989                  menu => 1, ol => 1, p => 1, ul => 1,                  address => 1, article => 1, aside => 1, blockquote => 1,
6990                    center => 1, datagrid => 1, details => 1, dialog => 1,
6991                    dir => 1, div => 1, dl => 1, fieldset => 1, figure => 1,
6992                    footer => 1, h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1,
6993                    h6 => 1, header => 1, menu => 1, nav => 1, ol => 1, p => 1,
6994                    section => 1, ul => 1,
6995                    ## NOTE: As normal, but drops leading newline
6996                  pre => 1, listing => 1,                  pre => 1, listing => 1,
6997                    ## NOTE: As normal, but interacts with the form element pointer
6998                  form => 1,                  form => 1,
6999                    
7000                  table => 1,                  table => 1,
7001                  hr => 1,                  hr => 1,
7002                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
# Line 6694  sub _tree_construction_main ($) { Line 7011  sub _tree_construction_main ($) {
7011    
7012          ## has a p element in scope          ## has a p element in scope
7013          INSCOPE: for (reverse @{$self->{open_elements}}) {          INSCOPE: for (reverse @{$self->{open_elements}}) {
7014            if ($_->[1] & P_EL) {            if ($_->[1] == P_EL) {
7015              !!!cp ('t344');              !!!cp ('t344');
7016              !!!back-token; # <form>              !!!back-token; # <form>
7017              $token = {type => END_TAG_TOKEN, tag_name => 'p',              $token = {type => END_TAG_TOKEN, tag_name => 'p',
# Line 6746  sub _tree_construction_main ($) { Line 7063  sub _tree_construction_main ($) {
7063            !!!next-token;            !!!next-token;
7064          }          }
7065          next B;          next B;
7066        } elsif ({li => 1, dt => 1, dd => 1}->{$token->{tag_name}}) {        } elsif ($token->{tag_name} eq 'li') {
7067          ## has a p element in scope          ## NOTE: As normal, but imply </li> when there's another <li> ...
7068    
7069            ## NOTE: Special, Scope (<li><foo><li> == <li><foo><li/></foo></li>)
7070              ## Interpreted as <li><foo/></li><li/> (non-conforming)
7071              ## blockquote (O9.27), center (O), dd (Fx3, O, S3.1.2, IE7),
7072              ## dt (Fx, O, S, IE), dl (O), fieldset (O, S, IE), form (Fx, O, S),
7073              ## hn (O), pre (O), applet (O, S), button (O, S), marquee (Fx, O, S),
7074              ## object (Fx)
7075              ## Generate non-tree (non-conforming)
7076              ## basefont (IE7 (where basefont is non-void)), center (IE),
7077              ## form (IE), hn (IE)
7078            ## address, div, p (<li><foo><li> == <li><foo/></li><li/>)
7079              ## Interpreted as <li><foo><li/></foo></li> (non-conforming)
7080              ## div (Fx, S)
7081    
7082            my $non_optional;
7083            my $i = -1;
7084    
7085            ## 1.
7086            for my $node (reverse @{$self->{open_elements}}) {
7087              if ($node->[1] == LI_EL) {
7088                ## 2. (a) As if </li>
7089                {
7090                  ## If no </li> - not applied
7091                  #
7092    
7093                  ## Otherwise
7094    
7095                  ## 1. generate implied end tags, except for </li>
7096                  #
7097    
7098                  ## 2. If current node != "li", parse error
7099                  if ($non_optional) {
7100                    !!!parse-error (type => 'not closed',
7101                                    text => $non_optional->[0]->manakai_local_name,
7102                                    token => $token);
7103                    !!!cp ('t355');
7104                  } else {
7105                    !!!cp ('t356');
7106                  }
7107    
7108                  ## 3. Pop
7109                  splice @{$self->{open_elements}}, $i;
7110                }
7111    
7112                last; ## 2. (b) goto 5.
7113              } elsif (
7114                       ## NOTE: not "formatting" and not "phrasing"
7115                       ($node->[1] & SPECIAL_EL or
7116                        $node->[1] & SCOPING_EL) and
7117                       ## NOTE: "li", "dt", and "dd" are in |SPECIAL_EL|.
7118                       (not $node->[1] & ADDRESS_DIV_P_EL)
7119                      ) {
7120                ## 3.
7121                !!!cp ('t357');
7122                last; ## goto 5.
7123              } elsif ($node->[1] & END_TAG_OPTIONAL_EL) {
7124                !!!cp ('t358');
7125                #
7126              } else {
7127                !!!cp ('t359');
7128                $non_optional ||= $node;
7129                #
7130              }
7131              ## 4.
7132              ## goto 2.
7133              $i--;
7134            }
7135    
7136            ## 5. (a) has a |p| element in scope
7137          INSCOPE: for (reverse @{$self->{open_elements}}) {          INSCOPE: for (reverse @{$self->{open_elements}}) {
7138            if ($_->[1] & P_EL) {            if ($_->[1] == P_EL) {
7139              !!!cp ('t353');              !!!cp ('t353');
7140    
7141                ## NOTE: |<p><li>|, for example.
7142    
7143              !!!back-token; # <x>              !!!back-token; # <x>
7144              $token = {type => END_TAG_TOKEN, tag_name => 'p',              $token = {type => END_TAG_TOKEN, tag_name => 'p',
7145                        line => $token->{line}, column => $token->{column}};                        line => $token->{line}, column => $token->{column}};
# Line 6760  sub _tree_construction_main ($) { Line 7149  sub _tree_construction_main ($) {
7149              last INSCOPE;              last INSCOPE;
7150            }            }
7151          } # INSCOPE          } # INSCOPE
7152              
7153          ## Step 1          ## 5. (b) insert
7154            !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7155            !!!nack ('t359.1');
7156            !!!next-token;
7157            next B;
7158          } elsif ($token->{tag_name} eq 'dt' or
7159                   $token->{tag_name} eq 'dd') {
7160            ## NOTE: As normal, but imply </dt> or </dd> when ...
7161    
7162            my $non_optional;
7163          my $i = -1;          my $i = -1;
7164          my $node = $self->{open_elements}->[$i];  
7165          my $li_or_dtdd = {li => {li => 1},          ## 1.
7166                            dt => {dt => 1, dd => 1},          for my $node (reverse @{$self->{open_elements}}) {
7167                            dd => {dt => 1, dd => 1}}->{$token->{tag_name}};            if ($node->[1] == DT_EL or $node->[1] == DD_EL) {
7168          LI: {              ## 2. (a) As if </li>
7169            ## Step 2              {
7170            if ($li_or_dtdd->{$node->[0]->manakai_local_name}) {                ## If no </li> - not applied
7171              if ($i != -1) {                #
7172                !!!cp ('t355');  
7173                !!!parse-error (type => 'not closed',                ## Otherwise
7174                                text => $self->{open_elements}->[-1]->[0]  
7175                                    ->manakai_local_name,                ## 1. generate implied end tags, except for </dt> or </dd>
7176                                token => $token);                #
7177              } else {  
7178                !!!cp ('t356');                ## 2. If current node != "dt"|"dd", parse error
7179                  if ($non_optional) {
7180                    !!!parse-error (type => 'not closed',
7181                                    text => $non_optional->[0]->manakai_local_name,
7182                                    token => $token);
7183                    !!!cp ('t355.1');
7184                  } else {
7185                    !!!cp ('t356.1');
7186                  }
7187    
7188                  ## 3. Pop
7189                  splice @{$self->{open_elements}}, $i;
7190              }              }
7191              splice @{$self->{open_elements}}, $i;  
7192              last LI;              last; ## 2. (b) goto 5.
7193              } elsif (
7194                       ## NOTE: not "formatting" and not "phrasing"
7195                       ($node->[1] & SPECIAL_EL or
7196                        $node->[1] & SCOPING_EL) and
7197                       ## NOTE: "li", "dt", and "dd" are in |SPECIAL_EL|.
7198    
7199                       (not $node->[1] & ADDRESS_DIV_P_EL)
7200                      ) {
7201                ## 3.
7202                !!!cp ('t357.1');
7203                last; ## goto 5.
7204              } elsif ($node->[1] & END_TAG_OPTIONAL_EL) {
7205                !!!cp ('t358.1');
7206                #
7207            } else {            } else {
7208              !!!cp ('t357');              !!!cp ('t359.1');
7209            }              $non_optional ||= $node;
7210                          #
           ## Step 3  
           if (not ($node->[1] & FORMATTING_EL) and  
               #not $phrasing_category->{$node->[1]} and  
               ($node->[1] & SPECIAL_EL or  
                $node->[1] & SCOPING_EL) and  
               not ($node->[1] & ADDRESS_EL) and  
               not ($node->[1] & DIV_EL)) {  
             !!!cp ('t358');  
             last LI;  
7211            }            }
7212                        ## 4.
7213            !!!cp ('t359');            ## goto 2.
           ## Step 4  
7214            $i--;            $i--;
7215            $node = $self->{open_elements}->[$i];          }
7216            redo LI;  
7217          } # LI          ## 5. (a) has a |p| element in scope
7218                      INSCOPE: for (reverse @{$self->{open_elements}}) {
7219              if ($_->[1] == P_EL) {
7220                !!!cp ('t353.1');
7221                !!!back-token; # <x>
7222                $token = {type => END_TAG_TOKEN, tag_name => 'p',
7223                          line => $token->{line}, column => $token->{column}};
7224                next B;
7225              } elsif ($_->[1] & SCOPING_EL) {
7226                !!!cp ('t354.1');
7227                last INSCOPE;
7228              }
7229            } # INSCOPE
7230    
7231            ## 5. (b) insert
7232          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7233          !!!nack ('t359.1');          !!!nack ('t359.2');
7234          !!!next-token;          !!!next-token;
7235          next B;          next B;
7236        } elsif ($token->{tag_name} eq 'plaintext') {        } elsif ($token->{tag_name} eq 'plaintext') {
7237            ## NOTE: As normal, but effectively ends parsing
7238    
7239          ## has a p element in scope          ## has a p element in scope
7240          INSCOPE: for (reverse @{$self->{open_elements}}) {          INSCOPE: for (reverse @{$self->{open_elements}}) {
7241            if ($_->[1] & P_EL) {            if ($_->[1] == P_EL) {
7242              !!!cp ('t367');              !!!cp ('t367');
7243              !!!back-token; # <plaintext>              !!!back-token; # <plaintext>
7244              $token = {type => END_TAG_TOKEN, tag_name => 'p',              $token = {type => END_TAG_TOKEN, tag_name => 'p',
# Line 6832  sub _tree_construction_main ($) { Line 7260  sub _tree_construction_main ($) {
7260        } elsif ($token->{tag_name} eq 'a') {        } elsif ($token->{tag_name} eq 'a') {
7261          AFE: for my $i (reverse 0..$#$active_formatting_elements) {          AFE: for my $i (reverse 0..$#$active_formatting_elements) {
7262            my $node = $active_formatting_elements->[$i];            my $node = $active_formatting_elements->[$i];
7263            if ($node->[1] & A_EL) {            if ($node->[1] == A_EL) {
7264              !!!cp ('t371');              !!!cp ('t371');
7265              !!!parse-error (type => 'in a:a', token => $token);              !!!parse-error (type => 'in a:a', token => $token);
7266                            
# Line 6876  sub _tree_construction_main ($) { Line 7304  sub _tree_construction_main ($) {
7304          ## has a |nobr| element in scope          ## has a |nobr| element in scope
7305          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7306            my $node = $self->{open_elements}->[$_];            my $node = $self->{open_elements}->[$_];
7307            if ($node->[1] & NOBR_EL) {            if ($node->[1] == NOBR_EL) {
7308              !!!cp ('t376');              !!!cp ('t376');
7309              !!!parse-error (type => 'in nobr:nobr', token => $token);              !!!parse-error (type => 'in nobr:nobr', token => $token);
7310              !!!back-token; # <nobr>              !!!back-token; # <nobr>
# Line 6899  sub _tree_construction_main ($) { Line 7327  sub _tree_construction_main ($) {
7327          ## has a button element in scope          ## has a button element in scope
7328          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7329            my $node = $self->{open_elements}->[$_];            my $node = $self->{open_elements}->[$_];
7330            if ($node->[1] & BUTTON_EL) {            if ($node->[1] == BUTTON_EL) {
7331              !!!cp ('t378');              !!!cp ('t378');
7332              !!!parse-error (type => 'in button:button', token => $token);              !!!parse-error (type => 'in button:button', token => $token);
7333              !!!back-token; # <button>              !!!back-token; # <button>
# Line 6999  sub _tree_construction_main ($) { Line 7427  sub _tree_construction_main ($) {
7427            next B;            next B;
7428          }          }
7429        } elsif ($token->{tag_name} eq 'textarea') {        } elsif ($token->{tag_name} eq 'textarea') {
7430          my $tag_name = $token->{tag_name};          ## Step 1
7431          my $el;          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
         !!!create-element ($el, $HTML_NS, $token->{tag_name}, $token->{attributes}, $token);  
7432                    
7433            ## Step 2
7434          ## TODO: $self->{form_element} if defined          ## TODO: $self->{form_element} if defined
7435    
7436            ## Step 3
7437            $self->{ignore_newline} = 1;
7438    
7439            ## Step 4
7440            ## ISSUE: This step is wrong. (r2302 enbugged)
7441    
7442            ## Step 5
7443          $self->{content_model} = RCDATA_CONTENT_MODEL;          $self->{content_model} = RCDATA_CONTENT_MODEL;
7444          delete $self->{escape}; # MUST          delete $self->{escape}; # MUST
7445            
7446          $insert->($el);          ## Step 6-7
7447                    $self->{insertion_mode} |= IN_CDATA_RCDATA_IM;
7448          my $text = '';  
7449          !!!nack ('t392.1');          !!!nack ('t392.1');
7450          !!!next-token;          !!!next-token;
7451          if ($token->{type} == CHARACTER_TOKEN) {          next B;
7452            $token->{data} =~ s/^\x0A//;        } elsif ($token->{tag_name} eq 'optgroup' or
7453            unless (length $token->{data}) {                 $token->{tag_name} eq 'option') {
7454              !!!cp ('t392');          ## has an |option| element in scope
7455              !!!next-token;          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7456            } else {            my $node = $self->{open_elements}->[$_];
7457              !!!cp ('t393');            if ($node->[1] == OPTION_EL) {
7458                !!!cp ('t397.1');
7459                ## NOTE: As if </option>
7460                !!!back-token; # <option> or <optgroup>
7461                $token = {type => END_TAG_TOKEN, tag_name => 'option',
7462                          line => $token->{line}, column => $token->{column}};
7463                next B;
7464              } elsif ($node->[1] & SCOPING_EL) {
7465                !!!cp ('t397.2');
7466                last INSCOPE;
7467            }            }
7468          } else {          } # INSCOPE
7469            !!!cp ('t394');  
7470          }          $reconstruct_active_formatting_elements->($insert_to_current);
7471          while ($token->{type} == CHARACTER_TOKEN) {  
7472            !!!cp ('t395');          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7473            $text .= $token->{data};  
7474            !!!next-token;          !!!nack ('t397.3');
         }  
         if (length $text) {  
           !!!cp ('t396');  
           $el->manakai_append_text ($text);  
         }  
           
         $self->{content_model} = PCDATA_CONTENT_MODEL;  
           
         if ($token->{type} == END_TAG_TOKEN and  
             $token->{tag_name} eq $tag_name) {  
           !!!cp ('t397');  
           ## Ignore the token  
         } else {  
           !!!cp ('t398');  
           !!!parse-error (type => 'in RCDATA:#eof', token => $token);  
         }  
7475          !!!next-token;          !!!next-token;
7476          next B;          redo B;
7477        } elsif ($token->{tag_name} eq 'rt' or        } elsif ($token->{tag_name} eq 'rt' or
7478                 $token->{tag_name} eq 'rp') {                 $token->{tag_name} eq 'rp') {
7479          ## has a |ruby| element in scope          ## has a |ruby| element in scope
7480          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7481            my $node = $self->{open_elements}->[$_];            my $node = $self->{open_elements}->[$_];
7482            if ($node->[1] & RUBY_EL) {            if ($node->[1] == RUBY_EL) {
7483              !!!cp ('t398.1');              !!!cp ('t398.1');
7484              ## generate implied end tags              ## generate implied end tags
7485              while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {              while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
7486                !!!cp ('t398.2');                !!!cp ('t398.2');
7487                pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
7488              }              }
7489              unless ($self->{open_elements}->[-1]->[1] & RUBY_EL) {              unless ($self->{open_elements}->[-1]->[1] == RUBY_EL) {
7490                !!!cp ('t398.3');                !!!cp ('t398.3');
7491                !!!parse-error (type => 'not closed',                !!!parse-error (type => 'not closed',
7492                                text => $self->{open_elements}->[-1]->[0]                                text => $self->{open_elements}->[-1]->[0]
7493                                    ->manakai_local_name,                                    ->manakai_local_name,
7494                                token => $token);                                token => $token);
7495                pop @{$self->{open_elements}}                pop @{$self->{open_elements}}
7496                    while not $self->{open_elements}->[-1]->[1] & RUBY_EL;                    while not $self->{open_elements}->[-1]->[1] == RUBY_EL;
7497              }              }
7498              last INSCOPE;              last INSCOPE;
7499            } elsif ($node->[1] & SCOPING_EL) {            } elsif ($node->[1] & SCOPING_EL) {
# Line 7092  sub _tree_construction_main ($) { Line 7521  sub _tree_construction_main ($) {
7521                    
7522          if ($self->{self_closing}) {          if ($self->{self_closing}) {
7523            pop @{$self->{open_elements}};            pop @{$self->{open_elements}};
7524            !!!ack ('t398.1');            !!!ack ('t398.6');
7525          } else {          } else {
7526            !!!cp ('t398.2');            !!!cp ('t398.7');
7527            $self->{insertion_mode} |= IN_FOREIGN_CONTENT_IM;            $self->{insertion_mode} |= IN_FOREIGN_CONTENT_IM;
7528            ## NOTE: |<body><math><mi><svg>| -> "in foreign content" insertion            ## NOTE: |<body><math><mi><svg>| -> "in foreign content" insertion
7529            ## mode, "in body" (not "in foreign content") secondary insertion            ## mode, "in body" (not "in foreign content") secondary insertion
# Line 7105  sub _tree_construction_main ($) { Line 7534  sub _tree_construction_main ($) {
7534          next B;          next B;
7535        } elsif ({        } elsif ({
7536                  caption => 1, col => 1, colgroup => 1, frame => 1,                  caption => 1, col => 1, colgroup => 1, frame => 1,
7537                  frameset => 1, head => 1, option => 1, optgroup => 1,                  frameset => 1, head => 1,
7538                  tbody => 1, td => 1, tfoot => 1, th => 1,                  tbody => 1, td => 1, tfoot => 1, th => 1,
7539                  thead => 1, tr => 1,                  thead => 1, tr => 1,
7540                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
# Line 7116  sub _tree_construction_main ($) { Line 7545  sub _tree_construction_main ($) {
7545          !!!nack ('t401.1'); ## NOTE: |<col/>| or |<frame/>| here is an error.          !!!nack ('t401.1'); ## NOTE: |<col/>| or |<frame/>| here is an error.
7546          !!!next-token;          !!!next-token;
7547          next B;          next B;
7548                  } elsif ($token->{tag_name} eq 'param' or
7549          ## ISSUE: An issue on HTML5 new elements in the spec.                 $token->{tag_name} eq 'source') {
7550            !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7551            pop @{$self->{open_elements}};
7552    
7553            !!!ack ('t398.5');
7554            !!!next-token;
7555            redo B;
7556        } else {        } else {
7557          if ($token->{tag_name} eq 'image') {          if ($token->{tag_name} eq 'image') {
7558            !!!cp ('t384');            !!!cp ('t384');
# Line 7140  sub _tree_construction_main ($) { Line 7575  sub _tree_construction_main ($) {
7575            !!!nack ('t380.1');            !!!nack ('t380.1');
7576          } elsif ({          } elsif ({
7577                    b => 1, big => 1, em => 1, font => 1, i => 1,                    b => 1, big => 1, em => 1, font => 1, i => 1,
7578                    s => 1, small => 1, strile => 1,                    s => 1, small => 1, strike => 1,
7579                    strong => 1, tt => 1, u => 1,                    strong => 1, tt => 1, u => 1,
7580                   }->{$token->{tag_name}}) {                   }->{$token->{tag_name}}) {
7581            !!!cp ('t375');            !!!cp ('t375');
# Line 7153  sub _tree_construction_main ($) { Line 7588  sub _tree_construction_main ($) {
7588            !!!ack ('t388.2');            !!!ack ('t388.2');
7589          } elsif ({          } elsif ({
7590                    area => 1, basefont => 1, bgsound => 1, br => 1,                    area => 1, basefont => 1, bgsound => 1, br => 1,
7591                    embed => 1, img => 1, param => 1, spacer => 1, wbr => 1,                    embed => 1, img => 1, spacer => 1, wbr => 1,
                   #image => 1,  
7592                   }->{$token->{tag_name}}) {                   }->{$token->{tag_name}}) {
7593            !!!cp ('t388.1');            !!!cp ('t388.1');
7594            pop @{$self->{open_elements}};            pop @{$self->{open_elements}};
# Line 7185  sub _tree_construction_main ($) { Line 7619  sub _tree_construction_main ($) {
7619          my $i;          my $i;
7620          INSCOPE: {          INSCOPE: {
7621            for (reverse @{$self->{open_elements}}) {            for (reverse @{$self->{open_elements}}) {
7622              if ($_->[1] & BODY_EL) {              if ($_->[1] == BODY_EL) {
7623                !!!cp ('t405');                !!!cp ('t405');
7624                $i = $_;                $i = $_;
7625                last INSCOPE;                last INSCOPE;
# Line 7195  sub _tree_construction_main ($) { Line 7629  sub _tree_construction_main ($) {
7629              }              }
7630            }            }
7631    
7632            !!!parse-error (type => 'start tag not allowed',            ## NOTE: |<marquee></body>|, |<svg><foreignobject></body>|
7633    
7634              !!!parse-error (type => 'unmatched end tag',
7635                            text => $token->{tag_name}, token => $token);                            text => $token->{tag_name}, token => $token);
7636            ## NOTE: Ignore the token.            ## NOTE: Ignore the token.
7637            !!!next-token;            !!!next-token;
# Line 7221  sub _tree_construction_main ($) { Line 7657  sub _tree_construction_main ($) {
7657          ## TODO: Update this code.  It seems that the code below is not          ## TODO: Update this code.  It seems that the code below is not
7658          ## up-to-date, though it has same effect as speced.          ## up-to-date, though it has same effect as speced.
7659          if (@{$self->{open_elements}} > 1 and          if (@{$self->{open_elements}} > 1 and
7660              $self->{open_elements}->[1]->[1] & BODY_EL) {              $self->{open_elements}->[1]->[1] == BODY_EL) {
7661            ## ISSUE: There is an issue in the spec.            unless ($self->{open_elements}->[-1]->[1] == BODY_EL) {
           unless ($self->{open_elements}->[-1]->[1] & BODY_EL) {  
7662              !!!cp ('t406');              !!!cp ('t406');
7663              !!!parse-error (type => 'not closed',              !!!parse-error (type => 'not closed',
7664                              text => $self->{open_elements}->[1]->[0]                              text => $self->{open_elements}->[1]->[0]
# Line 7244  sub _tree_construction_main ($) { Line 7679  sub _tree_construction_main ($) {
7679            next B;            next B;
7680          }          }
7681        } elsif ({        } elsif ({
7682                  address => 1, blockquote => 1, center => 1, dir => 1,                  ## NOTE: End tags for non-phrasing flow content elements
7683                  div => 1, dl => 1, fieldset => 1, listing => 1,  
7684                  menu => 1, ol => 1, pre => 1, ul => 1,                  ## NOTE: The normal ones
7685                    address => 1, article => 1, aside => 1, blockquote => 1,
7686                    center => 1, datagrid => 1, details => 1, dialog => 1,
7687                    dir => 1, div => 1, dl => 1, fieldset => 1, figure => 1,
7688                    footer => 1, header => 1, listing => 1, menu => 1, nav => 1,
7689                    ol => 1, pre => 1, section => 1, ul => 1,
7690    
7691                    ## NOTE: As normal, but ... optional tags
7692                  dd => 1, dt => 1, li => 1,                  dd => 1, dt => 1, li => 1,
7693    
7694                  applet => 1, button => 1, marquee => 1, object => 1,                  applet => 1, button => 1, marquee => 1, object => 1,
7695                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
7696            ## NOTE: Code for <li> start tags includes "as if </li>" code.
7697            ## Code for <dt> or <dd> start tags includes "as if </dt> or
7698            ## </dd>" code.
7699    
7700          ## has an element in scope          ## has an element in scope
7701          my $i;          my $i;
7702          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
# Line 7276  sub _tree_construction_main ($) { Line 7723  sub _tree_construction_main ($) {
7723                    dd => ($token->{tag_name} ne 'dd'),                    dd => ($token->{tag_name} ne 'dd'),
7724                    dt => ($token->{tag_name} ne 'dt'),                    dt => ($token->{tag_name} ne 'dt'),
7725                    li => ($token->{tag_name} ne 'li'),                    li => ($token->{tag_name} ne 'li'),
7726                      option => 1,
7727                      optgroup => 1,
7728                    p => 1,                    p => 1,
7729                    rt => 1,                    rt => 1,
7730                    rp => 1,                    rp => 1,
# Line 7308  sub _tree_construction_main ($) { Line 7757  sub _tree_construction_main ($) {
7757          !!!next-token;          !!!next-token;
7758          next B;          next B;
7759        } elsif ($token->{tag_name} eq 'form') {        } elsif ($token->{tag_name} eq 'form') {
7760            ## NOTE: As normal, but interacts with the form element pointer
7761    
7762          undef $self->{form_element};          undef $self->{form_element};
7763    
7764          ## has an element in scope          ## has an element in scope
7765          my $i;          my $i;
7766          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7767            my $node = $self->{open_elements}->[$_];            my $node = $self->{open_elements}->[$_];
7768            if ($node->[1] & FORM_EL) {            if ($node->[1] == FORM_EL) {
7769              !!!cp ('t418');              !!!cp ('t418');
7770              $i = $_;              $i = $_;
7771              last INSCOPE;              last INSCOPE;
# Line 7355  sub _tree_construction_main ($) { Line 7806  sub _tree_construction_main ($) {
7806          !!!next-token;          !!!next-token;
7807          next B;          next B;
7808        } elsif ({        } elsif ({
7809                    ## NOTE: As normal, except acts as a closer for any ...
7810                  h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,                  h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,
7811                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
7812          ## has an element in scope          ## has an element in scope
7813          my $i;          my $i;
7814          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7815            my $node = $self->{open_elements}->[$_];            my $node = $self->{open_elements}->[$_];
7816            if ($node->[1] & HEADING_EL) {            if ($node->[1] == HEADING_EL) {
7817              !!!cp ('t423');              !!!cp ('t423');
7818              $i = $_;              $i = $_;
7819              last INSCOPE;              last INSCOPE;
# Line 7400  sub _tree_construction_main ($) { Line 7852  sub _tree_construction_main ($) {
7852          !!!next-token;          !!!next-token;
7853          next B;          next B;
7854        } elsif ($token->{tag_name} eq 'p') {        } elsif ($token->{tag_name} eq 'p') {
7855            ## NOTE: As normal, except </p> implies <p> and ...
7856    
7857          ## has an element in scope          ## has an element in scope
7858            my $non_optional;
7859          my $i;          my $i;
7860          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7861            my $node = $self->{open_elements}->[$_];            my $node = $self->{open_elements}->[$_];
7862            if ($node->[1] & P_EL) {            if ($node->[1] == P_EL) {
7863              !!!cp ('t410.1');              !!!cp ('t410.1');
7864              $i = $_;              $i = $_;
7865              last INSCOPE;              last INSCOPE;
7866            } elsif ($node->[1] & SCOPING_EL) {            } elsif ($node->[1] & SCOPING_EL) {
7867              !!!cp ('t411.1');              !!!cp ('t411.1');
7868              last INSCOPE;              last INSCOPE;
7869              } elsif ($node->[1] & END_TAG_OPTIONAL_EL) {
7870                ## NOTE: |END_TAG_OPTIONAL_EL| includes "p"
7871                !!!cp ('t411.2');
7872                #
7873              } else {
7874                !!!cp ('t411.3');
7875                $non_optional ||= $node;
7876                #
7877            }            }
7878          } # INSCOPE          } # INSCOPE
7879    
7880          if (defined $i) {          if (defined $i) {
7881            if ($self->{open_elements}->[-1]->[0]->manakai_local_name            ## 1. Generate implied end tags
7882                    ne $token->{tag_name}) {            #
7883    
7884              ## 2. If current node != "p", parse error
7885              if ($non_optional) {
7886              !!!cp ('t412.1');              !!!cp ('t412.1');
7887              !!!parse-error (type => 'not closed',              !!!parse-error (type => 'not closed',
7888                              text => $self->{open_elements}->[-1]->[0]                              text => $non_optional->[0]->manakai_local_name,
                                 ->manakai_local_name,  
7889                              token => $token);                              token => $token);
7890            } else {            } else {
7891              !!!cp ('t414.1');              !!!cp ('t414.1');
7892            }            }
7893    
7894              ## 3. Pop
7895            splice @{$self->{open_elements}}, $i;            splice @{$self->{open_elements}}, $i;
7896          } else {          } else {
7897            !!!cp ('t413.1');            !!!cp ('t413.1');
# Line 7445  sub _tree_construction_main ($) { Line 7911  sub _tree_construction_main ($) {
7911        } elsif ({        } elsif ({
7912                  a => 1,                  a => 1,
7913                  b => 1, big => 1, em => 1, font => 1, i => 1,                  b => 1, big => 1, em => 1, font => 1, i => 1,
7914                  nobr => 1, s => 1, small => 1, strile => 1,                  nobr => 1, s => 1, small => 1, strike => 1,
7915                  strong => 1, tt => 1, u => 1,                  strong => 1, tt => 1, u => 1,
7916                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
7917          !!!cp ('t427');          !!!cp ('t427');
# Line 7466  sub _tree_construction_main ($) { Line 7932  sub _tree_construction_main ($) {
7932          ## Ignore the token.          ## Ignore the token.
7933          !!!next-token;          !!!next-token;
7934          next B;          next B;
       } elsif ({  
                 caption => 1, col => 1, colgroup => 1, frame => 1,  
                 frameset => 1, head => 1, option => 1, optgroup => 1,  
                 tbody => 1, td => 1, tfoot => 1, th => 1,  
                 thead => 1, tr => 1,  
                 area => 1, basefont => 1, bgsound => 1,  
                 embed => 1, hr => 1, iframe => 1, image => 1,  
                 img => 1, input => 1, isindex => 1, noembed => 1,  
                 noframes => 1, param => 1, select => 1, spacer => 1,  
                 table => 1, textarea => 1, wbr => 1,  
                 noscript => 0, ## TODO: if scripting is enabled  
                }->{$token->{tag_name}}) {  
         !!!cp ('t429');  
         !!!parse-error (type => 'unmatched end tag',  
                         text => $token->{tag_name}, token => $token);  
         ## Ignore the token  
         !!!next-token;  
         next B;  
           
         ## ISSUE: Issue on HTML5 new elements in spec  
           
7935        } else {        } else {
7936            if ($token->{tag_name} eq 'sarcasm') {
7937              sleep 0.001; # take a deep breath
7938            }
7939    
7940          ## Step 1          ## Step 1
7941          my $node_i = -1;          my $node_i = -1;
7942          my $node = $self->{open_elements}->[$node_i];          my $node = $self->{open_elements}->[$node_i];
7943    
7944          ## Step 2          ## Step 2
7945          S2: {          S2: {
7946            if ($node->[0]->manakai_local_name eq $token->{tag_name}) {            my $node_tag_name = $node->[0]->manakai_local_name;
7947              $node_tag_name =~ tr/A-Z/a-z/; # for SVG camelCase tag names
7948              if ($node_tag_name eq $token->{tag_name}) {
7949              ## Step 1              ## Step 1
7950              ## generate implied end tags              ## generate implied end tags
7951              while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {              while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
# Line 7507  sub _tree_construction_main ($) { Line 7958  sub _tree_construction_main ($) {
7958              }              }
7959                    
7960              ## Step 2              ## Step 2
7961              if ($self->{open_elements}->[-1]->[0]->manakai_local_name              my $current_tag_name
7962                      ne $token->{tag_name}) {                  = $self->{open_elements}->[-1]->[0]->manakai_local_name;
7963                $current_tag_name =~ tr/A-Z/a-z/;
7964                if ($current_tag_name ne $token->{tag_name}) {
7965                !!!cp ('t431');                !!!cp ('t431');
7966                ## NOTE: <x><y></x>                ## NOTE: <x><y></x>
7967                !!!parse-error (type => 'not closed',                !!!parse-error (type => 'not closed',
# Line 7536  sub _tree_construction_main ($) { Line 7989  sub _tree_construction_main ($) {
7989                ## Ignore the token                ## Ignore the token
7990                !!!next-token;                !!!next-token;
7991                last S2;                last S2;
             }  
7992    
7993                  ## NOTE: |<span><dd></span>a|: In Safari 3.1.2 and Opera
7994                  ## 9.27, "a" is a child of <dd> (conforming).  In
7995                  ## Firefox 3.0.2, "a" is a child of <body>.  In WinIE 7,
7996                  ## "a" is a child of both <body> and <dd>.
7997                }
7998                
7999              !!!cp ('t434');              !!!cp ('t434');
8000            }            }
8001                        
# Line 7578  sub _tree_construction_main ($) { Line 8036  sub _tree_construction_main ($) {
8036    ## TODO: script stuffs    ## TODO: script stuffs
8037  } # _tree_construct_main  } # _tree_construct_main
8038    
8039  sub set_inner_html ($$$;$) {  sub set_inner_html ($$$$;$) {
8040    my $class = shift;    my $class = shift;
8041    my $node = shift;    my $node = shift;
8042    my $s = \$_[0];    #my $s = \$_[0];
8043    my $onerror = $_[1];    my $onerror = $_[1];
8044    my $get_wrapper = $_[2] || sub ($) { return $_[0] };    my $get_wrapper = $_[2] || sub ($) { return $_[0] };
8045    
# Line 7602  sub set_inner_html ($$$;$) { Line 8060  sub set_inner_html ($$$;$) {
8060      }      }
8061    
8062      ## Step 3, 4, 5 # MUST      ## Step 3, 4, 5 # MUST
8063      $class->parse_char_string ($$s => $node, $onerror, $get_wrapper);      $class->parse_char_string ($_[0] => $node, $onerror, $get_wrapper);
8064    } elsif ($nt == 1) {    } elsif ($nt == 1) {
8065      ## TODO: If non-html element      ## TODO: If non-html element
8066    
# Line 7621  sub set_inner_html ($$$;$) { Line 8079  sub set_inner_html ($$$;$) {
8079      my $i = 0;      my $i = 0;
8080      $p->{line_prev} = $p->{line} = 1;      $p->{line_prev} = $p->{line} = 1;
8081      $p->{column_prev} = $p->{column} = 0;      $p->{column_prev} = $p->{column} = 0;
8082      $p->{set_next_char} = sub {      require Whatpm::Charset::DecodeHandle;
8083        my $input = Whatpm::Charset::DecodeHandle::CharString->new (\($_[0]));
8084        $input = $get_wrapper->($input);
8085        $p->{set_nc} = sub {
8086        my $self = shift;        my $self = shift;
8087    
8088        pop @{$self->{prev_char}};        my $char = '';
8089        unshift @{$self->{prev_char}}, $self->{next_char};        if (defined $self->{next_nc}) {
8090            $char = $self->{next_nc};
8091        $self->{next_char} = -1 and return if $i >= length $$s;          delete $self->{next_nc};
8092        $self->{next_char} = ord substr $$s, $i++, 1;          $self->{nc} = ord $char;
8093          } else {
8094            $self->{char_buffer} = '';
8095            $self->{char_buffer_pos} = 0;
8096            
8097            my $count = $input->manakai_read_until
8098                ($self->{char_buffer}, qr/[^\x00\x0A\x0D]/,
8099                 $self->{char_buffer_pos});
8100            if ($count) {
8101              $self->{line_prev} = $self->{line};
8102              $self->{column_prev} = $self->{column};
8103              $self->{column}++;
8104              $self->{nc}
8105                  = ord substr ($self->{char_buffer},
8106                                $self->{char_buffer_pos}++, 1);
8107              return;
8108            }
8109            
8110            if ($input->read ($char, 1)) {
8111              $self->{nc} = ord $char;
8112            } else {
8113              $self->{nc} = -1;
8114              return;
8115            }
8116          }
8117    
8118        ($p->{line_prev}, $p->{column_prev}) = ($p->{line}, $p->{column});        ($p->{line_prev}, $p->{column_prev}) = ($p->{line}, $p->{column});
8119        $p->{column}++;        $p->{column}++;
8120    
8121        if ($self->{next_char} == 0x000A) { # LF        if ($self->{nc} == 0x000A) { # LF
8122          $p->{line}++;          $p->{line}++;
8123          $p->{column} = 0;          $p->{column} = 0;
8124          !!!cp ('i1');          !!!cp ('i1');
8125        } elsif ($self->{next_char} == 0x000D) { # CR        } elsif ($self->{nc} == 0x000D) { # CR
8126          $i++ if substr ($$s, $i, 1) eq "\x0A";  ## TODO: support for abort/streaming
8127          $self->{next_char} = 0x000A; # LF # MUST          my $next = '';
8128            if ($input->read ($next, 1) and $next ne "\x0A") {
8129              $self->{next_nc} = $next;
8130            }
8131            $self->{nc} = 0x000A; # LF # MUST
8132          $p->{line}++;          $p->{line}++;
8133          $p->{column} = 0;          $p->{column} = 0;
8134          !!!cp ('i2');          !!!cp ('i2');
8135        } elsif ($self->{next_char} > 0x10FFFF) {        } elsif ($self->{nc} == 0x0000) { # NULL
         $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST  
         !!!cp ('i3');  
       } elsif ($self->{next_char} == 0x0000) { # NULL  
8136          !!!cp ('i4');          !!!cp ('i4');
8137          !!!parse-error (type => 'NULL');          !!!parse-error (type => 'NULL');
8138          $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST          $self->{nc} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
       } elsif ($self->{next_char} <= 0x0008 or  
                (0x000E <= $self->{next_char} and  
                 $self->{next_char} <= 0x001F) or  
                (0x007F <= $self->{next_char} and  
                 $self->{next_char} <= 0x009F) or  
                (0xD800 <= $self->{next_char} and  
                 $self->{next_char} <= 0xDFFF) or  
                (0xFDD0 <= $self->{next_char} and  
                 $self->{next_char} <= 0xFDDF) or  
                {  
                 0xFFFE => 1, 0xFFFF => 1, 0x1FFFE => 1, 0x1FFFF => 1,  
                 0x2FFFE => 1, 0x2FFFF => 1, 0x3FFFE => 1, 0x3FFFF => 1,  
                 0x4FFFE => 1, 0x4FFFF => 1, 0x5FFFE => 1, 0x5FFFF => 1,  
                 0x6FFFE => 1, 0x6FFFF => 1, 0x7FFFE => 1, 0x7FFFF => 1,  
                 0x8FFFE => 1, 0x8FFFF => 1, 0x9FFFE => 1, 0x9FFFF => 1,  
                 0xAFFFE => 1, 0xAFFFF => 1, 0xBFFFE => 1, 0xBFFFF => 1,  
                 0xCFFFE => 1, 0xCFFFF => 1, 0xDFFFE => 1, 0xDFFFF => 1,  
                 0xEFFFE => 1, 0xEFFFF => 1, 0xFFFFE => 1, 0xFFFFF => 1,  
                 0x10FFFE => 1, 0x10FFFF => 1,  
                }->{$self->{next_char}}) {  
         !!!cp ('i4.1');  
         if ($self->{next_char} < 0x10000) {  
           !!!parse-error (type => 'control char',  
                           text => (sprintf 'U+%04X', $self->{next_char}));  
         } else {  
           !!!parse-error (type => 'control char',  
                           text => (sprintf 'U-%08X', $self->{next_char}));  
         }  
8139        }        }
8140      };      };
8141      $p->{prev_char} = [-1, -1, -1];  
8142      $p->{next_char} = -1;      $p->{read_until} = sub {
8143              #my ($scalar, $specials_range, $offset) = @_;
8144          return 0 if defined $p->{next_nc};
8145    
8146          my $pattern = qr/[^$_[1]\x00\x0A\x0D]/;
8147          my $offset = $_[2] || 0;
8148          
8149          if ($p->{char_buffer_pos} < length $p->{char_buffer}) {
8150            pos ($p->{char_buffer}) = $p->{char_buffer_pos};
8151            if ($p->{char_buffer} =~ /\G(?>$pattern)+/) {
8152              substr ($_[0], $offset)
8153                  = substr ($p->{char_buffer}, $-[0], $+[0] - $-[0]);
8154              my $count = $+[0] - $-[0];
8155              if ($count) {
8156                $p->{column} += $count;
8157                $p->{char_buffer_pos} += $count;
8158                $p->{line_prev} = $p->{line};
8159                $p->{column_prev} = $p->{column} - 1;
8160                $p->{nc} = -1;
8161              }
8162              return $count;
8163            } else {
8164              return 0;
8165            }
8166          } else {
8167            my $count = $input->manakai_read_until ($_[0], $pattern, $_[2]);
8168            if ($count) {
8169              $p->{column} += $count;
8170              $p->{column_prev} += $count;
8171              $p->{nc} = -1;
8172            }
8173            return $count;
8174          }
8175        }; # $p->{read_until}
8176    
8177      my $ponerror = $onerror || sub {      my $ponerror = $onerror || sub {
8178        my (%opt) = @_;        my (%opt) = @_;
8179        my $line = $opt{line};        my $line = $opt{line};
# Line 7697  sub set_inner_html ($$$;$) { Line 8188  sub set_inner_html ($$$;$) {
8188        $ponerror->(line => $p->{line}, column => $p->{column}, @_);        $ponerror->(line => $p->{line}, column => $p->{column}, @_);
8189      };      };
8190            
8191        my $char_onerror = sub {
8192          my (undef, $type, %opt) = @_;
8193          $ponerror->(layer => 'encode',
8194                      line => $p->{line}, column => $p->{column} + 1,
8195                      %opt, type => $type);
8196        }; # $char_onerror
8197        $input->onerror ($char_onerror);
8198    
8199      $p->_initialize_tokenizer;      $p->_initialize_tokenizer;
8200      $p->_initialize_tree_constructor;      $p->_initialize_tree_constructor;
8201    
# Line 7732  sub set_inner_html ($$$;$) { Line 8231  sub set_inner_html ($$$;$) {
8231      push @{$p->{open_elements}}, [$root, $el_category->{html}];      push @{$p->{open_elements}}, [$root, $el_category->{html}];
8232    
8233      undef $p->{head_element};      undef $p->{head_element};
8234        undef $p->{head_element_inserted};
8235    
8236      ## Step 6 # MUST      ## Step 6 # MUST
8237      $p->_reset_insertion_mode;      $p->_reset_insertion_mode;

Legend:
Removed from v.1.167  
changed lines
  Added in v.1.206

[email protected]
ViewVC Help
Powered by ViewVC 1.1.24